chore: sync local changes, add Chinese docs and opencode config
This commit is contained in:
@@ -0,0 +1,287 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Aggregate per-file deep-read quality (V2.1 three-part gate).
|
||||
|
||||
A-stage implementation run by the main agent after each sub-agent wave.
|
||||
Reads sub-agent output JSON files that were written to disk (not returned
|
||||
into the main context), computes the three-part gate per file:
|
||||
|
||||
① unit completeness : sub-agent semantic_units must cover every graph unit
|
||||
(Function/Class/Test node) of the file. Hard, 100%.
|
||||
Giant files (units<3 or max unit >80% of lines) are
|
||||
exempt from this gate (unit_exempt_files).
|
||||
② line coverage : |union(read_ranges)| / real line count >= 95%.
|
||||
Denominator is the REAL file line count read once
|
||||
(cached), NOT graph node line_end (verified +-1 skew).
|
||||
③ anti-fake budget : reported to the main agent; only a sampled re-read
|
||||
can detect fake reads (engine cannot).
|
||||
|
||||
Usage:
|
||||
python aggregate_deep_read.py <repo_root> <out_dir>
|
||||
<repo_root> repository root (for graph + real file reads)
|
||||
<out_dir> directory containing sub-agent result JSON files
|
||||
(each: {"path": rel, "total_lines": int,
|
||||
"read_ranges": [[s,e],...],
|
||||
"semantic_units": [{"range":[s,e],"kind","name"}],
|
||||
"findings": [...]})
|
||||
|
||||
Outputs to stdout:
|
||||
{status, line_gap_files[], unit_gap_files[], unit_exempt_files[],
|
||||
verified_files[], summary}
|
||||
"""
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _abs(root: Path, rel: str) -> Path:
|
||||
p = Path(rel)
|
||||
return p if p.is_absolute() else root / p
|
||||
|
||||
|
||||
def _real_line_count(root: Path, rel: str, cache: dict) -> int:
|
||||
"""Real file line count, cached. Independent of graph node line_end."""
|
||||
if rel in cache:
|
||||
return cache[rel]
|
||||
p = _abs(root, rel)
|
||||
try:
|
||||
n = len(p.read_text(encoding="utf-8", errors="replace").splitlines())
|
||||
except OSError:
|
||||
n = 0
|
||||
cache[rel] = n
|
||||
return n
|
||||
|
||||
|
||||
def _union_len(ranges: list) -> int:
|
||||
"""Covered line count of a list of inclusive [s,e] ranges."""
|
||||
if not ranges:
|
||||
return 0
|
||||
merged: list[list[int]] = []
|
||||
for s, e in sorted((int(a), int(b)) for a, b in ranges):
|
||||
if s < 1:
|
||||
s = 1
|
||||
if e < s:
|
||||
continue
|
||||
if merged and s <= merged[-1][1] + 1:
|
||||
merged[-1][1] = max(merged[-1][1], e)
|
||||
else:
|
||||
merged.append([s, e])
|
||||
return sum(e - s + 1 for s, e in merged)
|
||||
|
||||
|
||||
def _graph_units(store, root: Path, rel: str) -> list[dict]:
|
||||
"""Graph semantic-unit nodes (Function/Class/Test) with line ranges."""
|
||||
abs_path = _abs(root, rel).as_posix().replace("/", "\\")
|
||||
q = abs_path.replace("\\", "/")
|
||||
try:
|
||||
rows = store._conn.execute(
|
||||
"SELECT kind, name, line_start, line_end FROM nodes "
|
||||
"WHERE file_path = ? AND kind IN ('Function','Class','Test') "
|
||||
"ORDER BY line_start",
|
||||
(q,),
|
||||
).fetchall()
|
||||
except Exception:
|
||||
return []
|
||||
out = []
|
||||
for r in rows:
|
||||
try:
|
||||
out.append(
|
||||
{
|
||||
"kind": r["kind"],
|
||||
"name": r["name"],
|
||||
"line_start": int(r["line_start"]),
|
||||
"line_end": int(r["line_end"]),
|
||||
}
|
||||
)
|
||||
except (KeyError, TypeError):
|
||||
continue
|
||||
return out
|
||||
|
||||
|
||||
def _overlap(a: list, b: list) -> int:
|
||||
lo, hi = max(a[0], b[0]), min(a[1], b[1])
|
||||
return max(0, hi - lo + 1)
|
||||
|
||||
|
||||
def _unit_covered(
|
||||
graph_unit: dict,
|
||||
read_ranges: list,
|
||||
unit_ranges: list,
|
||||
matched_uids: set,
|
||||
) -> bool:
|
||||
"""A graph unit is covered iff one reported unit range (a) overlaps its
|
||||
range by >=80% of the graph unit's span, AND (b) is not already claimed by
|
||||
a higher-overlap graph unit (one-to-one matching), AND (c) >=80% of the
|
||||
graph unit's lines fall inside union(read_ranges).
|
||||
|
||||
The one-to-one rule stops a single wide range (e.g. the whole file) from
|
||||
covering every unit by pretending to be all of them."""
|
||||
gs, ge = graph_unit["line_start"], graph_unit["line_end"]
|
||||
gspan = max(1, ge - gs + 1)
|
||||
# exact-range match wins (handles nested/overlapping graph units)
|
||||
exact = [i for i, (s, e) in enumerate(unit_ranges) if s == gs and e == ge]
|
||||
if exact:
|
||||
best_idx = exact[0]
|
||||
if best_idx in matched_uids:
|
||||
return False
|
||||
matched_uids.add(best_idx)
|
||||
else:
|
||||
best_idx, best_overlap = None, 0
|
||||
for i, (s, e) in enumerate(unit_ranges):
|
||||
ov = _overlap([gs, ge], [s, e])
|
||||
if ov > best_overlap:
|
||||
best_overlap, best_idx = ov, i
|
||||
if best_idx is None or best_idx in matched_uids:
|
||||
return False
|
||||
if best_overlap / gspan < 0.8:
|
||||
return False
|
||||
matched_uids.add(best_idx)
|
||||
in_union = 0
|
||||
for s, e in _merge_ranges(read_ranges):
|
||||
lo, hi = max(gs, s), min(ge, e)
|
||||
if lo <= hi:
|
||||
in_union += hi - lo + 1
|
||||
if (in_union / gspan) < 0.8:
|
||||
return False
|
||||
matched_uids.add(best_idx)
|
||||
return True
|
||||
|
||||
|
||||
def _merge_ranges(ranges: list) -> list[list[int]]:
|
||||
merged: list[list[int]] = []
|
||||
for s, e in sorted((int(a), int(b)) for a, b in ranges or []):
|
||||
if s < 1:
|
||||
s = 1
|
||||
if e < s:
|
||||
continue
|
||||
if merged and s <= merged[-1][1] + 1:
|
||||
merged[-1][1] = max(merged[-1][1], e)
|
||||
else:
|
||||
merged.append([s, e])
|
||||
return merged
|
||||
|
||||
|
||||
def _is_giant_file(graph_units: list, real_lines: int) -> bool:
|
||||
"""unit-exempt when the largest unit spans >80% of the file's lines.
|
||||
|
||||
Aligned with the engine (scoring.py _is_giant_file): small files with a
|
||||
few ordinary units are NOT exempt - they must cover every unit. A single
|
||||
huge function (e.g. migrations.rs run_migrations = 98%) makes unit
|
||||
completeness meaningless, so such files are checked on line coverage only.
|
||||
"""
|
||||
if not graph_units or real_lines <= 0:
|
||||
return False
|
||||
largest = max(u["line_end"] - u["line_start"] + 1 for u in graph_units)
|
||||
return (largest / real_lines) > 0.8
|
||||
|
||||
|
||||
def _load_subagent_files(out_dir: Path) -> list[dict]:
|
||||
payloads = []
|
||||
for f in sorted(out_dir.glob("*.json")):
|
||||
try:
|
||||
payloads.append(json.loads(f.read_text(encoding="utf-8")))
|
||||
except (OSError, ValueError) as exc:
|
||||
print(json.dumps({"status": "error", "file": str(f), "error": str(exc)}))
|
||||
return payloads
|
||||
|
||||
|
||||
def aggregate(repo_root: str, out_dir: str):
|
||||
root = Path(repo_root)
|
||||
out = Path(out_dir)
|
||||
# engine import path: prefer CRG_ENGINE env, else known local checkout
|
||||
engine = (
|
||||
__import__("os").environ.get("CRG_ENGINE")
|
||||
or r"D:\code-review-graph\code-review-graph-main"
|
||||
)
|
||||
sys.path.insert(0, engine)
|
||||
from code_review_graph.tools._common import _get_store
|
||||
|
||||
store, root2 = _get_store(repo_root)
|
||||
cache: dict = {}
|
||||
verified: list[str] = []
|
||||
line_gap: list[dict] = []
|
||||
unit_gap: list[dict] = []
|
||||
exempt: list[dict] = []
|
||||
seen: set = set()
|
||||
|
||||
for payload in _load_subagent_files(out):
|
||||
rel = (payload.get("path") or "").replace("\\", "/")
|
||||
if not rel or rel in seen:
|
||||
continue
|
||||
seen.add(rel)
|
||||
ranges = payload.get("read_ranges") or []
|
||||
units = payload.get("semantic_units") or []
|
||||
reported_total = payload.get("total_lines")
|
||||
real = _real_line_count(root, rel, cache)
|
||||
if real == 0:
|
||||
continue
|
||||
|
||||
line_cov = _union_len(ranges) / real
|
||||
g_units = _graph_units(store, root, rel)
|
||||
giant = _is_giant_file(g_units, real)
|
||||
unit_ranges = [tuple(map(int, (u.get("range") or [0, 0])[:2])) for u in units]
|
||||
|
||||
# ① unit completeness
|
||||
if g_units and not giant:
|
||||
matched_uids: set = set()
|
||||
uncovered = [
|
||||
{"name": u["name"], "range": [u["line_start"], u["line_end"]]}
|
||||
for u in g_units
|
||||
if not _unit_covered(u, ranges, unit_ranges, matched_uids)
|
||||
]
|
||||
if uncovered:
|
||||
unit_gap.append(
|
||||
{
|
||||
"path": rel,
|
||||
"total_units": len(g_units),
|
||||
"covered_units": len(g_units) - len(uncovered),
|
||||
"uncovered": uncovered,
|
||||
}
|
||||
)
|
||||
# a unit gap also fails the file
|
||||
continue
|
||||
elif g_units and giant:
|
||||
exempt.append(
|
||||
{
|
||||
"path": rel,
|
||||
"reason": "giant-file: units=%d largest_span=%.0f%%"
|
||||
% (len(g_units), 100 * max(u["line_end"] - u["line_start"] + 1 for u in g_units) / real),
|
||||
}
|
||||
)
|
||||
|
||||
# ② line coverage (applies to every file)
|
||||
if line_cov < 0.95:
|
||||
line_gap.append(
|
||||
{
|
||||
"path": rel,
|
||||
"coverage_pct": round(line_cov * 100, 1),
|
||||
"total_lines": real,
|
||||
"covered_lines": _union_len(ranges),
|
||||
"reported_total": reported_total,
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
verified.append(rel)
|
||||
|
||||
store.close()
|
||||
|
||||
result = {
|
||||
"status": "ok",
|
||||
"verified_files": sorted(verified),
|
||||
"line_gap_files": sorted(line_gap, key=lambda x: x["coverage_pct"]),
|
||||
"unit_gap_files": sorted(unit_gap, key=lambda x: x["total_units"]),
|
||||
"unit_exempt_files": sorted(exempt, key=lambda x: x["path"]),
|
||||
"summary": (
|
||||
f"verified={len(verified)} line_gap={len(line_gap)} "
|
||||
f"unit_gap={len(unit_gap)} unit_exempt={len(exempt)} "
|
||||
f"(unit_exempt are checked on line coverage only)"
|
||||
),
|
||||
}
|
||||
print(json.dumps(result, ensure_ascii=False, indent=1))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) != 3:
|
||||
print(__doc__)
|
||||
sys.exit(2)
|
||||
aggregate(sys.argv[1], sys.argv[2])
|
||||
Reference in New Issue
Block a user