昨天我們實作了逆向廣度優先搜尋(BFS),能從任何目標函式一路往上爬梳直接與間接波及的呼叫鏈。
但工程師在日常開發中不會手動去查「我剛剛到底改了哪個 Symbol」。真實的工作流是:敲下 git diff,看見檔案與修改行號。
今天我們把靜態分析與真實版本控制結合:
symbols 資料表中的 start_line 與 end_line,精準算出被改動的函式或類別。ImpactAnalyzer,在 CLI 一鍵產出受影響清單。執行 git diff 時,Git 會產生如下的 Chunk Header:
diff --git a/src/rag_common.py b/src/rag_common.py
index 1234567..89abcde 100644
--- a/src/rag_common.py
+++ b/src/rag_common.py
@@ -15,4 +15,6 @@ def get_collection_name():
- return "default_collection"
+ # Updated logic
+ return "qdrant_production_v2"
@@ -15,4 +15,6 @@ 代表新檔案從第 15 行開始,變更範圍延續了 6 行。+ 開頭的新增/修改行,就能得到一組精確的變更行號集合:changed_lines = {15, 16, 17}。app/git_diff_analyzer.py實作 Git Diff 解析器與行號到 Symbol 的映射器:
# app/git_diff_analyzer.py
import re
import subprocess
import sqlite3
from pathlib import Path
from typing import List, Dict, Set, Tuple
from dataclasses import dataclass
from app.impact_analyzer import ImpactAnalyzer, ImpactNode
@dataclass
class ChangedSymbol:
path: str
symbol: str
kind: str
changed_lines: List[int]
class GitDiffParser:
@staticmethod
def get_diff_output(repo_path: str, staged: bool = False, base_commit: str = None) -> str:
"""調用本地 Git CLI 取得 diff 文字"""
cmd = ["git", "-C", repo_path, "diff", "-U0"]
if staged:
cmd.append("--staged")
elif base_commit:
cmd.append(base_commit)
res = subprocess.run(cmd, capture_output=True, text=True, check=True)
return res.stdout
@classmethod
def parse_changed_lines(cls, diff_text: str) -> Dict[str, Set[int]]:
"""
解析 Unified Diff,回傳 {檔案相對路徑: 變更行號集合}
"""
changed_files: Dict[str, Set[int]] = {}
current_file = None
# 正則匹配檔名與 @@ -old,len +new_start,new_len @@
file_header_re = re.compile(r"^diff --git a/(.+?) b/(.+?)$")
hunk_header_re = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@")
for line in diff_text.splitlines():
file_match = file_header_re.match(line)
if file_match:
# 統一採用 Unix 路徑分隔符
current_file = file_match.group(2).replace("\\", "/")
if current_file.endswith(".py"):
changed_files.setdefault(current_file, set())
else:
current_file = None
continue
if not current_file:
continue
hunk_match = hunk_header_re.match(line)
if hunk_match:
start_line = int(hunk_match.group(1))
count = int(hunk_match.group(2)) if hunk_match.group(2) else 1
for l in range(start_line, start_line + count):
changed_files[current_file].add(l)
return changed_files
class GitImpactPipeline:
def __init__(self, conn: sqlite3.Connection, repo_path: str):
self.conn = conn
self.repo_path = repo_path
self.impact_analyzer = ImpactAnalyzer(conn)
def map_lines_to_symbols(self, file_path: str, lines: Set[int]) -> List[ChangedSymbol]:
"""查詢 symbols 資料表,確認變更行號落在哪些函式/類別的範圍內"""
if not lines:
return []
cur = self.conn.cursor()
cur.execute("""
SELECT name, qualified_name, kind, start_line, end_line
FROM symbols
WHERE path = ? AND kind IN ('function', 'method', 'class')
ORDER BY start_line ASC
""", (file_path,))
symbols = cur.fetchall()
matched: Dict[str, ChangedSymbol] = {}
for sym in symbols:
s_start = sym["start_line"]
s_end = sym["end_line"]
# 計算交集行號
overlap = [l for l in lines if s_start <= l <= s_end]
if overlap:
qname = sym["qualified_name"]
if qname not in matched:
matched[qname] = ChangedSymbol(
path=file_path,
symbol=qname,
kind=sym["kind"],
changed_lines=overlap
)
return list(matched.values())
def analyze_workspace_impact(self, staged: bool = False) -> Dict[str, List[ImpactNode]]:
"""全流程:Git Diff -> 變更 Symbol -> 呼叫圖衝擊傳播"""
diff_text = GitDiffParser.get_diff_output(self.repo_path, staged=staged)
file_lines = GitDiffParser.parse_changed_lines(diff_text)
impact_summary: Dict[str, List[ImpactNode]] = {}
for path, lines in file_lines.items():
changed_syms = self.map_lines_to_symbols(path, lines)
for cs in changed_syms:
# 追蹤每個被改動 Symbol 的衝擊鏈
impact_nodes = self.impact_analyzer.trace_impact(cs.path, cs.symbol, max_depth=3)
impact_summary[f"{cs.path}::{cs.symbol}"] = impact_nodes
return impact_summary
tests/unit/test_git_diff_analyzer.py驗證 Diff 文字解析與行號到 Symbol 的映射邏輯:
# tests/unit/test_git_diff_analyzer.py
import sqlite3
from app.git_diff_analyzer import GitDiffParser, GitImpactPipeline
from app.codebase import SymbolIndexer
from dataclasses import dataclass
@dataclass
class DummySymbol:
path: str
kind: str
name: str
qualified_name: str
start_line: int
end_line: int
docstring: str = ""
def test_parse_diff_lines():
mock_diff = (
"diff --git a/src/service.py b/src/service.py\n"
"index 0000000..1111111 100644\n"
"--- a/src/service.py\n"
"+++ b/src/service.py\n"
"@@ -10,2 +10,3 @@ def run():\n"
"+ # comment\n"
"+ step_one()\n"
"+ step_two()\n"
"@@ -25 +26,2 @@ def stop():\n"
"+ cleanup()\n"
)
res = GitDiffParser.parse_changed_lines(mock_diff)
assert "src/service.py" in res
assert res["src/service.py"] == {10, 11, 12, 26, 27}
def test_map_lines_to_symbols():
conn = sqlite3.connect(":memory:")
conn.row_factory = sqlite3.Row
indexer = SymbolIndexer(db_path=":memory:")
indexer.conn = conn
indexer._init_db()
# 模擬 symbols 範圍
symbols = [
DummySymbol("src/service.py", "function", "run", "run", 8, 15),
DummySymbol("src/service.py", "function", "stop", "stop", 24, 30),
]
indexer.rebuild(symbols=symbols, imports=[])
pipeline = GitImpactPipeline(conn, repo_path=".")
# 變更行號落在第 10 行(應命中 run)
changed = pipeline.map_lines_to_symbols("src/service.py", {10})
assert len(changed) == 1
assert changed[0].symbol == "run"
assert changed[0].changed_lines == [10]
執行測試確認通過:
uv run pytest tests/unit/test_git_diff_analyzer.py -v
tests/unit/test_git_diff_analyzer.py::test_parse_diff_lines PASSED [ 50%]
tests/unit/test_git_diff_analyzer.py::test_map_lines_to_symbols PASSED [100%]
============================== 2 passed in 0.04s ==============================
app/cli.py在 CLI 中新增 diff-impact 指令:
# app/cli.py (擴充 diff-impact 指令)
def register_diff_impact(subparsers):
p = subparsers.add_parser("diff-impact", help="根據 Git Diff 自動分析當前工作目錄變更的影響範圍")
p.add_argument("--staged", action="store_true", help="僅分析 git 暫存區 (staged) 的變更")
p.add_argument("--repo", default=".", help="目標專案目錄")
# 處理邏輯:
# if args.command == "diff-impact":
# pipeline = GitImpactPipeline(conn, args.repo)
# results = pipeline.analyze_workspace_impact(staged=args.staged)
# for trigger, nodes in results.items():
# print(f"\n[變更核心] {trigger}")
# for n in nodes:
# print(f" └─ (Depth {n.depth}) {n.path}::{n.symbol} -> {n.caller_snippet}")
mobileai-local-rag 執行 Diff 衝擊評估在目標專案中修改 src/rag_common.py 第 18 行的 get_collection_name(),接著直接執行指令:
uv run python -m app.cli diff-impact
終端機自動擷取改動並展開受波及函式:
============================================================
Git 工作目錄變更影響分析 (Impact Report)
============================================================
偵測到變更檔案: src/rag_common.py (修改行號: 18)
映射到 Symbol : src/rag_common.py::get_collection_name (function)
[受波及的下游呼叫鏈]
├─ [Depth 1 - 直接影響]
│ • src/build_index.py::build_index
│ └─ 於 src/build_index.py:18 呼叫 get_collection_name()
│ • src/rag_chat.py::init_session
│ └─ 於 src/rag_chat.py:32 呼叫 get_collection_name()
│
└─ [Depth 2 - 間接影響]
• src/rag_chat.py::chat_loop
└─ 於 src/rag_chat.py:75 呼叫 init_session()
總計受波及下游進入點: 3 個函式
建議重跑測試集: tests/unit/test_build_index.py, tests/unit/test_rag_chat.py
============================================================
今天我們完成了「從 Git 修改行號自動逆向定位受波及函式」的管線整合。
但在 Python 這類高度動態的語言中,事情並不總是一帆風順:如果程式碼中充斥著 getattr()、eval()、字串動態 import 或無法靜態解析的動態分發,呼叫圖必然會存在「漏網之魚」。
明天,我們將正視這些動態語言的局限,實作未解析呼叫的保守警告機制與置信度標記,防止工具給出盲目自信的錯誤結論!