"""Quantitative code-review scoring for the unified-review workflow. Implements the objectively computable Layer-2 metrics from the ai-code-review methodology as code, plus the git-history / graph risk factors used by the gstack-review workflow. The report metric set consists solely of the five objective heuristic metrics computed here; ``llm_judged`` is returned empty for backward compatibility. The three public entry points are: * :func:`score_review` - all objective Layer-2 metrics for changed files * :func:`dedupe_findings` - fingerprint dedup + confidence merge * :func:`build_report_data` - data feed for the HTML report Every metric returns a dict with ``score``/``value``, ``grade`` (one of ``good``/``warn``/``fail``), ``thresholds`` and ``evidence`` so the agent can cite the numbers instead of asserting a vibe. """ from __future__ import annotations import re from pathlib import Path from typing import Any, Callable, Optional from .changes import ( compute_file_churn, map_changes_to_nodes, parse_diff_ranges, ) from .constants import SECURITY_KEYWORDS from .graph import GraphStore from .parser import normalize_file_path from collections import Counter # --------------------------------------------------------------------------- # Thresholds (aligned with the ai-code-review Layer-2 scoring rubrics) # --------------------------------------------------------------------------- # good/warn/fail boundaries per metric. THRESHOLDS: dict[str, dict[str, float]] = { "redundancy_rate": {"good_max": 10.0, "warn_max": 20.0}, # % duplicate lines "exception_coverage": {"warn_min": 50.0, "good_min": 30.0}, # exception/normal % "sql_risk": {"warn_at": 1, "fail_at": 3}, # risk count "high_risk_density": {"warn_min": 70.0, "good_min": 90.0}, # covered % (0..100) "vulnerability_risk": {"warn_at": 1, "fail_at": 2}, # risk count } # --------------------------------------------------------------------------- # Keyword tables # --------------------------------------------------------------------------- # Exception-path markers used by the exception-coverage heuristic. These are # deliberately language-agnostic and conservative: a hit is counted as an # exception/error path, not a happy path. _EXCEPTION_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"\btry\b"), re.compile(r"\bexcept\b"), re.compile(r"\bcatch\s*\("), re.compile(r"\braise\b"), re.compile(r"\bthrow\b"), re.compile(r"\?\.\.\."), re.compile(r"\bif\s+err\s*!=\s*nil\b"), re.compile(r"\bif\s+.*\berror\b"), re.compile(r"\.catch\s*\("), re.compile(r"\belse\s*\{?\s*return\s+(?:false|None|null|nil|0)\b"), ) # SQL injection / unparameterized-query markers. A hit is a warning, not a # proof: the agent must confirm the surrounding context before acting. _SQL_RISK_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"[\"'](?:SELECT|INSERT|UPDATE|DELETE)\s+.*?\+\s*[\"']", re.I | re.S), re.compile(r"\bf?\".*?(?:SELECT|INSERT|UPDATE|DELETE).*?(?:\$\{|\{)\w", re.I | re.S), re.compile(r"\bexec(?:ute)?\s*\(\s*[\"'].*?(?:SELECT|INSERT|UPDATE|DELETE)", re.I | re.S), re.compile(r"WHERE\s+.*?=\s*[\"']\s*\+\s*\w", re.I | re.S), re.compile(r"\.format\s*\(.*\).*?(?:SELECT|INSERT|UPDATE|DELETE)", re.I | re.S), re.compile(r"%\s*[\"']\s*%\s*\(?.*?(?:SELECT|INSERT|UPDATE|DELETE)", re.I | re.S), ) # Concurrency / transaction / data-consistency markers for high-risk density. _HIGH_RISK_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"\b(?:async|await)\b"), re.compile(r"\b(?:thread|goroutine|mutex|lock|unlock|semaphore)\b"), re.compile(r"\b(?:transaction|commit|rollback|begin)\b", re.I), re.compile(r"\b(?:atomic|concurrent|race)\b", re.I), re.compile(r"\b(?:SELECT|INSERT|UPDATE|DELETE)\b", re.I), re.compile(r"\b(?:lock|with_lock|select_for_update)\b", re.I), re.compile(r"\b(?:cache|invalidate|evict)\b", re.I), ) # OWASP / secret markers for the vulnerability heuristic. _VULNERABILITY_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"\b(?:password|passwd|pwd)\s*=\s*[\"'][^\"']{1,32}[\"']", re.I), re.compile(r"\b(?:api[_-]?key|secret|token)\s*=\s*[\"'][^\"']+[\"']", re.I), re.compile(r"\b(?:eval|exec)\s*\(", re.I), re.compile(r"\bsubprocess\s*\(.*shell\s*=\s*True", re.I | re.S), re.compile(r"\b(?:innerHTML|dangerouslySetInnerHTML|html\.safe|mark_safe|v-html)\b", re.I), re.compile(r"\b(?: str: """Classify a numeric value into good/warn/fail for one metric.""" t = THRESHOLDS[metric] if metric in ("sql_risk", "vulnerability_risk"): # Count-based: 0 = good, warn_at count = warn, fail_at count = fail. if value <= 0: return "good" if value < t["fail_at"]: return "warn" return "fail" if metric == "redundancy_rate": # percentage-based, lower is better return "good" if value <= t["good_max"] else ( "warn" if value <= t["warn_max"] else "fail" ) # percentage-based, higher is better if metric == "exception_coverage": return "good" if value >= t["warn_min"] else ( "warn" if value >= t["good_min"] else "fail" ) return "good" if value >= t["good_min"] else ( "warn" if value >= t["warn_min"] else "fail" ) def _iter_source_lines( changed_files: list[str], repo_root: Path, ) -> list[tuple[str, str, int]]: """Yield (rel_path, line, line_no) for each changed source line. Changed files that cannot be read are skipped silently; a missing file is not a scoring failure. ``changed_files`` is resolved relative to ``repo_root`` when not absolute. """ out: list[tuple[str, str, int]] = [] for rel in changed_files: raw = rel.replace("\\", "/") candidate = Path(rel) if not candidate.is_absolute(): candidate = repo_root / raw if not candidate.is_file(): continue try: lines = candidate.read_text( encoding="utf-8", errors="replace", ).splitlines() except OSError: continue for i, line in enumerate(lines, start=1): out.append((raw, line, i)) return out def _count_matching(lines: list[tuple[str, str, int]], patterns: tuple[re.Pattern[str], ...]) -> int: """Count lines matching any pattern (deduped per line).""" hits = 0 for _path, line, _no in lines: if any(p.search(line) for p in patterns): hits += 1 return hits def _normalized_signature(line: str) -> str: """Normalise a source line into a reusable duplicate signature.""" text = re.sub(r"\s+", " ", line).strip() text = re.sub(r"\b\d+\b", "N", text) text = re.sub(r"[\"'][^\"']*[\"']", '"s"', text) return text.lower() def _find_repeated_blocks( changed_files: list[str], repo_root: Path, ) -> list[dict[str, Any]]: """Detect near-duplicate source blocks (>=3 occurrences) across the diff. A coarse structural proxy for the redundancy metric: a line that appears with the same normalised signature in at least three places is counted as duplicate content. This is heuristic -- the agent confirms before acting. """ lines = _iter_source_lines(changed_files, repo_root) sig_count: dict[str, list[tuple[str, int]]] = {} for rel, line, no in lines: sig = _normalized_signature(line) if len(sig) < 24: continue # ignore trivially short lines sig_count.setdefault(sig, []).append((rel, no)) blocks: list[dict[str, Any]] = [] dup_lines = 0 for sig, occurrences in sig_count.items(): if len(occurrences) < 3: continue dup_lines += len(occurrences) blocks.append({ "signature": sig[:120], "occurrences": len(occurrences), "locations": [ {"file": rel, "line": no} for rel, no in occurrences[:8] ], }) blocks.sort(key=lambda b: b["occurrences"], reverse=True) total = len(lines) rate = (dup_lines / total * 100.0) if total else 0.0 return [blocks, rate] # --------------------------------------------------------------------------- # Public metrics # --------------------------------------------------------------------------- def compute_sql_risk( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Scan changed lines for string-interpolated / unparameterized SQL.""" lines = _iter_source_lines(changed_files, repo_root) locations: list[dict[str, Any]] = [] for rel, line, no in lines: if any(p.search(line) for p in _SQL_RISK_PATTERNS): locations.append({ "file": rel, "line": no, "snippet": line.strip()[:200], }) count = len(locations) return { "metric": "sql_risk", "value": count, "grade": _grade("sql_risk", float(count)), "thresholds": THRESHOLDS["sql_risk"], "evidence": locations[:20], "note": "字符串拼接 SQL 的启发式扫描。修复前请逐一确认每个位置;" "用 EXPLAIN 评估性能风险。", } def compute_exception_coverage( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Estimate exception-path coverage of changed code. Ratio = exception-path lines / (normal-path lines + exception-path lines). The ai-code-review rubric uses ~1 exception per 2 happy paths as the good baseline (>=50% of normal paths have a counterpart), hence the thresholds on this ratio. """ lines = _iter_source_lines(changed_files, repo_root) total = len(lines) exc = _count_matching(lines, _EXCEPTION_PATTERNS) normal = max(0, total - exc) ratio = (exc / (normal + exc) * 100.0) if (normal + exc) else 0.0 return { "metric": "exception_coverage", "value": round(ratio, 2), "grade": _grade("exception_coverage", ratio), "thresholds": THRESHOLDS["exception_coverage"], "evidence": { "total_lines": total, "exception_path_lines": exc, "normal_path_lines": normal, }, "note": "异常/错误路径行的启发式占比。请人工复核边界条件与错误处理。", } def compute_redundancy_rate( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Estimate code redundancy in the changed files.""" blocks, rate = _find_repeated_blocks(changed_files, repo_root) return { "metric": "redundancy_rate", "value": round(rate, 2), "grade": _grade("redundancy_rate", rate), "thresholds": THRESHOLDS["redundancy_rate"], "evidence": blocks[:20], "note": "重复代码块启发式占比(规范化行在 >=3 处出现)。" "抽取公共逻辑前请确认。", } def compute_high_risk_density( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Estimate high-risk scenario coverage of changed code. Density = (high-risk lines with matching concurrency/transaction/security markers) / (high-risk-relevant lines). When no high-risk lines exist the metric is N/A rather than a pass/fail. """ lines = _iter_source_lines(changed_files, repo_root) relevant = [ (rel, line, no) for rel, line, no in lines if any(p.search(line) for p in _HIGH_RISK_PATTERNS) ] if not relevant: return { "metric": "high_risk_density", "value": None, "grade": "na", "thresholds": THRESHOLDS["high_risk_density"], "evidence": {}, "note": "变更中未检出并发/事务/数据一致性模式——" "除非审查发现缺口,标记为 N/A。", } covered = 0 for _rel, line, _no in relevant: # Every relevant line is counted as "covered by review attention"; # the marker is a signal for the agent to verify, not a defect. if line.strip(): covered += 1 density = covered / len(relevant) * 100.0 return { "metric": "high_risk_density", "value": round(density, 2), "grade": _grade("high_risk_density", density), "thresholds": THRESHOLDS["high_risk_density"], "evidence": { "high_risk_lines": len(relevant), "covered_lines": covered, "locations": [ {"file": rel, "line": no, "snippet": line.strip()[:160]} for rel, line, no in relevant[:20] ], }, "note": "并发/事务/安全模式密度。属审查注意力信号,非正确性评分。", } def compute_vulnerability_heuristic( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Scan changed lines for OWASP / secret-like patterns (heuristic).""" lines = _iter_source_lines(changed_files, repo_root) locations: list[dict[str, Any]] = [] for rel, line, no in lines: if any(p.search(line) for p in _VULNERABILITY_PATTERNS): locations.append({ "file": rel, "line": no, "snippet": line.strip()[:200], }) count = len(locations) return { "metric": "vulnerability_risk", "value": count, "grade": _grade("vulnerability_risk", float(count)), "thresholds": THRESHOLDS["vulnerability_risk"], "evidence": locations[:20], "note": "OWASP/密钥模式的启发式扫描。真实漏洞需依赖扫描器" "(npm audit、pip-audit、govulncheck)确认。", } def compute_risk_factors( store: GraphStore, repo_root: Path, changed_files: list[str], include_churn: bool = True, ) -> dict[str, Any]: """Compute git-history + graph risk factors for changed files. Returns churn hotspots, cross-community coupling and hub dependencies -- the structural input the agent uses to prioritise review attention. """ abs_files = [normalize_file_path(repo_root / f) for f in changed_files] # git churn over the trailing window (reuses changes.compute_file_churn). churn: dict[str, Any] = {"enabled": include_churn, "hotspots": []} if include_churn: counts = compute_file_churn(str(repo_root)) hotspots = [ {"file": f, "commits": c} for f, c in sorted(counts.items(), key=lambda kv: kv[1], reverse=True) if c >= 3 ] churn["hotspots"] = hotspots[:20] # Cross-community callers of the changed functions. A caller from a # different community than the changed node signals coupling that deserves # extra review attention. changed_nodes = [] for fp in abs_files: changed_nodes.extend(store.get_nodes_by_file(fp)) node_qns = {n.qualified_name for n in changed_nodes} caller_qns: set[str] = set() incoming: dict[str, list[str]] = {} for qn in node_qns: edges = store.get_edges_by_target(qn) callers = [e.source_qualified for e in edges if e.kind == "CALLS"] incoming[qn] = callers caller_qns.update(callers) # Batch community lookups: changed nodes and their callers. all_qns = list(node_qns | caller_qns) cid_map = store.get_community_ids_by_qualified_names(all_qns) cross_community: list[dict[str, Any]] = [] hub_dependencies: list[dict[str, Any]] = [] for qn in node_qns: tgt_cid = cid_map.get(qn) for caller in incoming.get(qn, []): src_cid = cid_map.get(caller) if src_cid is not None and tgt_cid is not None and src_cid != tgt_cid: cross_community.append({ "caller": caller, "callee": qn, "edge": "CALLS", }) if len(incoming.get(qn, [])) >= 10: hub_dependencies.append({ "node": qn, "callers": len(incoming.get(qn, [])), }) return { "churn": churn, "cross_community_edges": cross_community[:20], "hub_dependencies": hub_dependencies[:20], "note": "结构性风险因子。高变更频率 + 跨社区耦合 + 中枢依赖" "意味着该改动需要额外审查关注。", } def score_review( store: GraphStore, repo_root: Path, changed_files: list[str], include_churn: bool = True, progress_cb: Callable[[float, Optional[str]], None] | None = None, ) -> dict[str, Any]: """Compute all objective Layer-2 metrics for a set of changed files. Args: store: Open graph store (caller owns and closes it). repo_root: Repository root. changed_files: Changed file paths relative to ``repo_root``. include_churn: Include git-churn risk factors. progress_cb: Optional ``(fraction, message)`` progress callback; invoked once per metric (fraction = k/5). Returns: Dict with ``metrics`` (per-metric score/grade/evidence), ``risk_factors``, ``llm_judged`` and ``summary``. """ metric_fns = { "sql_risk": compute_sql_risk, "exception_coverage": compute_exception_coverage, "redundancy_rate": compute_redundancy_rate, "high_risk_density": compute_high_risk_density, "vulnerability_risk": compute_vulnerability_heuristic, } metrics: dict[str, Any] = {} for idx, (name, fn) in enumerate(metric_fns.items()): if progress_cb is not None: progress_cb(idx / len(metric_fns), f"computing {name}") metrics[name] = fn(changed_files, repo_root) if progress_cb is not None: progress_cb(1.0, "metrics done") risk_factors = compute_risk_factors( store, repo_root, changed_files, include_churn=include_churn, ) grades = [m["grade"] for m in metrics.values() if m["grade"] != "na"] worst = "fail" if "fail" in grades else ( "warn" if "warn" in grades else "good" ) summary_parts = [ f"Scored {len(changed_files)} changed file(s) with {len(metrics)} objective metrics.", f"Overall objective grade: {worst.upper()}", ] for name, m in metrics.items(): if m["grade"] == "na": summary_parts.append(f" - {name}: N/A") else: summary_parts.append(f" - {name}: {m['grade']} (value={m['value']})") if risk_factors["churn"]["hotspots"]: summary_parts.append( f" - {len(risk_factors['churn']['hotspots'])} churn hotspot(s)" ) if risk_factors["cross_community_edges"]: summary_parts.append( f" - {len(risk_factors['cross_community_edges'])} cross-community edge(s)" ) return { "status": "ok", "summary": "\n".join(summary_parts), "metrics": metrics, "risk_factors": risk_factors, "llm_judged": [], "objective_grade": worst, } # --------------------------------------------------------------------------- # dedupe_findings # --------------------------------------------------------------------------- def _fingerprint(finding: dict[str, Any]) -> str: path = str(finding.get("path", "")).replace("\\", "/") line = finding.get("line") category = str(finding.get("category", "")).strip().lower() if line: return f"{path}:{line}:{category}" return f"{path}:{category}" def dedupe_findings( findings: list[dict[str, Any]], suppress_prior: list[dict[str, Any]] | None = None, ) -> dict[str, Any]: """Merge findings by fingerprint, boost multi-source confidence. Mirrors the gstack-review "collect and merge" step: * same fingerprint -> keep the highest-confidence finding * findings confirmed by multiple sources -> confidence +1 (cap 10) * confidence gates: >=7 normal, 5-6 caveat, 3-4 appendix, 1-2 suppressed * PR quality score = max(0, 10 - (critical*2 + informational*0.5)) Args: findings: Finding dicts with at least ``path``, ``category``, ``severity``, ``confidence`` and optional ``source``/``line``. suppress_prior: Findings previously skipped by the user (read from a prior review-log). Entries whose fingerprint matches AND whose file was not changed since then are dropped. Returns: Dict with ``merged``, ``suppressed``, ``quality_score`` and ``summary``. """ groups: dict[str, list[dict[str, Any]]] = {} for f in findings: fp = _fingerprint(f) groups.setdefault(fp, []).append(f) merged: list[dict[str, Any]] = [] suppressed_by_gate = 0 for fp, items in groups.items(): best = max(items, key=lambda x: x.get("confidence", 0)) sources = {str(i.get("source", "")).strip() for i in items if i.get("source")} confirmed = len(sources) > 1 conf = float(best.get("confidence", 0)) if confirmed: conf = min(10.0, conf + 1.0) best["multi_source_confirmed"] = True best["confirming_sources"] = sorted(sources) best["confidence"] = conf if conf < 2: suppressed_by_gate += 1 continue # suppressed entirely if conf < 4: best["display"] = "appendix" elif conf < 7: best["display"] = "normal-with-caveat" else: best["display"] = "normal" merged.append(best) # Prior-review suppression (user explicitly skipped, file unchanged). suppressed_prior = 0 if suppress_prior: prior_fps = {_fingerprint(f) for f in suppress_prior} kept: list[dict[str, Any]] = [] for f in merged: if _fingerprint(f) in prior_fps: suppressed_prior += 1 continue kept.append(f) merged = kept critical = sum(1 for f in merged if str(f.get("severity", "")).upper() in ("CRITICAL", "BLOCKER")) informational = sum( 1 for f in merged if str(f.get("severity", "")).upper() in ("INFORMATIONAL", "MINOR", "WARN") ) quality = max(0.0, 10.0 - (critical * 2.0 + informational * 0.5)) return { "status": "ok", "summary": ( f"Merged {len(findings)} raw finding(s) into {len(merged)} " f"unique issue(s); quality score {quality:.1f}/10." ), "merged": merged, "suppressed_by_confidence": suppressed_by_gate, "suppressed_by_prior": suppressed_prior, "quality_score": round(quality, 2), "counts": {"critical": critical, "informational": informational}, } # --------------------------------------------------------------------------- # Report data # --------------------------------------------------------------------------- def _normalise_coverage(coverage: Any) -> dict[str, Any] | None: """Normalise the ``coverage`` block passed into the HTML report feed. If ``coverage`` is not a dict (e.g. an AI agent passed ``True`` or a partial subset), return ``None`` so the report renders no coverage section rather than crashing. When a dict is given, keep it as-is so every field the agent chose to transmit survives; the HTML/Markdown templates provide ``N/A`` fallbacks for missing counts so a partial transmission never shows a misleading ``0/0``. """ if not isinstance(coverage, dict) or not coverage: return None return coverage #: The five objective metrics produced by score_review_tool. Any other key #: an agent passes in review_data.metrics (e.g. blast_radius, objective_grade, #: llm_judged leftovers) is filtered out so the report only renders the #: canonical five rows. _OBJECTIVE_METRIC_KEYS: frozenset[str] = frozenset({ "sql_risk", "exception_coverage", "redundancy_rate", "high_risk_density", "vulnerability_risk", }) def build_report_data( review_data: dict[str, Any], ) -> dict[str, Any]: """Normalise score_review/dedupe output into the HTML report feed. ``review_data`` may contain ``metrics`` (from :func:`score_review`), ``findings`` (from :func:`dedupe_findings`) and free-form ``verdict``, ``tier``, ``scope`` fields. The returned dict is JSON-serialisable and ready to be injected into ``report-template.html`` as ``{{REPORT_DATA}}``. """ # Normalise ``files``: agents sometimes pass a list instead of a # comma-separated string. Accept both; a list is joined so the report # never renders Python/JSON list syntax. raw_files = review_data.get("files", "") files_text = ( ", ".join(str(f) for f in raw_files) if isinstance(raw_files, list) else raw_files ) # ``reviewed_files`` drives the collapsible
list. Agents may # pass it as an array OR as a comma-separated string; accept both (a # string is split so the Markdown renderer never iterates char-by-char). # Fall back to the ``files`` array when nothing structured was passed. reviewed_files = review_data.get("reviewed_files") or [] if isinstance(reviewed_files, str): reviewed_files = [ p.strip() for p in reviewed_files.split(",") if p.strip() ] elif not reviewed_files and isinstance(raw_files, list): reviewed_files = [str(f) for f in raw_files] data: dict[str, Any] = { "scope": review_data.get("scope", "change-level"), "tier": review_data.get("tier", "standard"), "timestamp": review_data.get("timestamp", ""), "files": files_text, "reviewed_files": reviewed_files, "baseline": review_data.get("baseline", "generic"), "verdict": review_data.get("verdict", "❌ FAIL"), "quality_score": review_data.get("quality_score"), "counts": review_data.get("counts", {}), "metrics": {}, "issues": [], "manual_review": review_data.get("manual_review", []), "llm_judged": review_data.get("llm_judged", []), "coverage": _normalise_coverage(review_data.get("coverage")), "spot_check": _normalise_coverage(review_data.get("spot_check")), } # Objective metrics only: filter out stray keys (blast_radius, # objective_grade, ...) so the report always renders the canonical five. metrics = review_data.get("metrics") or {} for name, m in metrics.items(): if name not in _OBJECTIVE_METRIC_KEYS: continue if isinstance(m, dict): data["metrics"][name] = { "grade": m.get("grade"), "value": m.get("value"), "note": m.get("note", ""), "evidence": m.get("evidence", {}), } findings = review_data.get("findings") or [] for f in findings: data["issues"].append({ "severity": f.get("severity", "minor"), "category": f.get("category", ""), "message": f.get("summary", f.get("message", "")), "location": ( f"{f.get('path', '')}:{f.get('line', '')}" if f.get("line") else str(f.get("path", "")) ), "confidence": f.get("confidence"), "fix": f.get("fix", ""), }) return data # --------------------------------------------------------------------------- # Markdown report rendering (Chinese) # --------------------------------------------------------------------------- #: Human-readable labels for the Chinese Markdown report. _METRIC_LABELS: dict[str, str] = { "sql_risk": "SQL 注入风险", "exception_coverage": "异常分支覆盖", "redundancy_rate": "代码冗余率", "high_risk_density": "高风险场景密度", "vulnerability_risk": "漏洞风险", } _GRADE_LABELS: dict[str, str] = { "good": "良好", "warn": "警告", "fail": "不合格", "na": "不适用", } _SEVERITY_LABELS: dict[str, str] = { "blocker": "🔴 阻塞", "critical": "🔴 严重", "major": "🟡 主要", "warn": "🟡 主要", "minor": "🔵 次要", "informational": "🔵 次要", } def render_markdown_report(review_data: dict[str, Any]) -> str: """Render the review data as a Chinese, standalone Markdown report. Mirrors the HTML report content: verdict, tier, scope, objective metrics, issues, manual-review items and LLM-judged metrics. The output is a single Markdown document with no external dependencies. Args: review_data: Review data dict (see :func:`build_report_data`). Returns: The Markdown report text. """ data = build_report_data(review_data) lines: list[str] = [] lines.append("# 代码审查报告\n") verdict = str(data["verdict"]) verdict_line = f"- **结论**:{verdict}" tier = data.get("tier") or "standard" scope = data.get("scope") or "change-level" baseline = data.get("baseline") or "generic" lines.append(verdict_line) lines.append( f"- **档位**:{tier} · **范围**:{scope} · **基线**:{baseline}" ) if data.get("timestamp"): lines.append(f"- **生成时间**:{data['timestamp']}") if data.get("files") and not (data.get("reviewed_files") or []): lines.append(f"- **文件**:{data['files']}") qs = data.get("quality_score") if qs is not None: lines.append(f"- **PR 质量分**:{qs}/10") counts = data.get("counts") or {} if counts: lines.append( f"- **问题统计**:{counts.get('critical', 0)} 严重 · " f"{counts.get('informational', 0)} 次要" ) lines.append("") # Reviewed files (collapsible). Renders from the structured # ``reviewed_files`` array when present; otherwise the flat ``files`` # string is used (as the meta line above). Defensive: if a raw string # ever slips through (build_report_data already normalises it), split it # so we never iterate a string char-by-char. reviewed = data.get("reviewed_files") or [] if isinstance(reviewed, str): reviewed = [p.strip() for p in reviewed.split(",") if p.strip()] if reviewed: lines.append(f"**审查文件({len(reviewed)} 个)**") lines.append("") lines.append("
") lines.append(f"点击展开 / 收起({len(reviewed)} 个文件)") lines.append("") for f in reviewed: lines.append(f"- `{f}`") lines.append("") lines.append("
") lines.append("") # Objective metrics metrics = data.get("metrics") or {} if metrics: lines.append("## 客观指标\n") lines.append("| 指标 | 数值 | 评级 | 说明 |") lines.append("|---|---|---|---|") for name, m in metrics.items(): label = _METRIC_LABELS.get(name, name) value = m.get("value") value_text = str(value) if value is not None else "N/A" grade = m.get("grade") or "na" grade_text = _GRADE_LABELS.get(grade, grade) note = str(m.get("note", "") or "") note = note.replace("\n", " ") lines.append(f"| {label} | {value_text} | {grade_text} | {note} |") lines.append("") # Coverage. For gate="both+line" (whole-project) the file-count and # high-risk rows render; for gate="line+unit" (feature reviews) the # engine returns coverage_pct=None so only the line/unit rows plus the # fail-closed status line render. coverage = data.get("coverage") or {} if coverage: pct = coverage.get("coverage_pct") hr_pct = coverage.get("high_risk_coverage_pct") grade = coverage.get("grade") or "na" grade_text = _GRADE_LABELS.get(grade, grade) deep_read = coverage.get("deep_read_count", "N/A") total = coverage.get("total_files", "N/A") hr_total = coverage.get("high_risk_total_files", "N/A") hr_deep = coverage.get("high_risk_deep_count", "N/A") overall_target = coverage.get("overall_target", coverage.get("target", "N/A")) hr_target = coverage.get("high_risk_target", coverage.get("target", "N/A")) reached = coverage.get("target_reached", False) status = "✅ 达标" if reached else "🔴 覆盖不足" uncovered = coverage.get("uncovered_files") or [] silent = coverage.get("silent_files") or [] # Line / unit coverage (gate="both+line" / "line+unit"): fail-closed. # A missing value means the review never ran the line-coverage gate - # surface it explicitly instead of silently omitting the field. line_pct = coverage.get("line_coverage_pct") unit_pct = coverage.get("unit_coverage_pct") line_target = coverage.get("line_target", 95.0) unit_target = coverage.get("unit_target", 100.0) line_gap_n = len(coverage.get("line_gap_files") or []) unit_gap_n = len(coverage.get("unit_gap_files") or []) missing_n = len(coverage.get("missing_data_files") or []) lines.append("## 覆盖度\n") if pct is not None: lines.append( f"- **覆盖度(全库)**:{pct}% — 已深读 {deep_read}/{total} 个源文件" f"(目标 {overall_target}%)" ) lines.append( f"- **覆盖度(高风险)**:{hr_pct}% — 已深读 {hr_deep}/{hr_total} 个高风险文件" f"(目标 {hr_target}%){status}" ) else: lines.append( f"- **状态**:{status}(gate=\"line+unit\":仅行/单元覆盖,不做文件数覆盖检查)" ) if line_pct is None or unit_pct is None: lines.append("- **行覆盖**:未执行 🔴(coverage_tool 未用 gate=\"both+line\" 或未传三件套数据)") else: line_ok = "✅" if line_pct >= line_target and line_gap_n == 0 else "🔴" unit_ok = "✅" if unit_pct >= unit_target and unit_gap_n == 0 else "🔴" lines.append( f"- **行覆盖**:{line_pct}% — 目标 {line_target}%(缺口 {line_gap_n} 文件){line_ok}" ) lines.append( f"- **单元覆盖**:{unit_pct}% — 目标 {unit_target}%(缺口 {unit_gap_n} 文件){unit_ok}" ) if missing_n: lines.append( f"- **三件套数据缺失**:{missing_n} 个文件(缺 read_ranges/语义单元,已按 fail-closed 计为缺口)" ) if uncovered: lines.append( f"- **未深读文件**:{len(uncovered)} 个" f"(静默文件 {len(silent)} 个)" ) lines.append("") # Anti-fake spot check (three-piece suite item 3). Fail-closed: a # missing/incomplete spot_check renders "未执行 🔴" so reviews that # skipped the sampled re-read are visible instead of silently green. spot = data.get("spot_check") if spot: groups = spot.get("groups_sampled") files = spot.get("files_sampled") units = spot.get("units_sampled") fake = spot.get("fake_read_found", 0) rereread = spot.get("groups_rereread") or [] if units: mark = "🔴 发现假读" if (fake or rereread) else "✅" lines.append( f"- **防伪抽验**:抽样 {files} 文件 / {units} 单元 / {groups} 组," f"假读 {fake}{mark}" ) if rereread: lines.append( f" - 因假读重读组:{', '.join(rereread)}" ) else: lines.append("- **防伪抽验**:未执行 🔴(spot_check 已上报但单元数为 0)") else: lines.append( "- **防伪抽验**:未执行 🔴(主代理未回读任何语义单元;" "Step 5.5 应执行每组抽 2 文件 × 2-3 单元并落盘 spot_check)" ) lines.append("") # Issues issues = data.get("issues") or [] lines.append(f"## 问题清单({len(issues)})\n") if not issues: lines.append("未发现问题。\n") for i, issue in enumerate(issues, start=1): sev = str(issue.get("severity") or "minor").lower() sev_label = _SEVERITY_LABELS.get(sev, sev) category = str(issue.get("category") or "") message = str(issue.get("message") or "") location = str(issue.get("location") or "") conf = issue.get("confidence") fix = str(issue.get("fix") or "") lines.append( f"{i}. **{sev_label}** {message}" f"{f'(置信度 {conf}/10)' if conf is not None else ''}" ) if category: lines.append(f" - **类别**:{category}") if location: lines.append(f" - **位置**:`{location}`") if fix: lines.append(f" - **修复建议**:{fix}") lines.append("") # Manual review manual = data.get("manual_review") or [] if manual: lines.append("## 需要人工审查\n") for m in manual: lines.append(f"- {m}") lines.append("") # LLM-judged metrics judged = data.get("llm_judged") or [] if judged: lines.append("## 需 LLM 判断的指标\n") lines.append(", ".join(str(j) for j in judged)) lines.append("") return "\n".join(lines).strip() + "\n" # --------------------------------------------------------------------------- # Coverage computation (review coverage of the whole project) # --------------------------------------------------------------------------- #: Coverage targets (percentage of source files deep-read). Fixed to the #: single standard tier; the fast/strict tiers were removed. Per-gate #: targets: ``overall`` is the whole-project file-count target, ``high_risk`` #: the signal-flagged subset target. ``gate="both"`` requires both to pass. COVERAGE_TARGETS: dict[str, dict[str, float]] = { "standard": { "overall": 85.0, "high_risk": 95.0, }, } def _coverage_targets(tier: str = "standard") -> tuple[float, float]: """Resolve the ``(overall, high_risk)`` target pair for a tier. Backwards-compatible: a legacy flat float value (e.g. ``95.0``) is treated as applying to *both* gates. A per-gate dict is honoured as-is. """ cfg = COVERAGE_TARGETS.get(tier, COVERAGE_TARGETS.get("standard", {})) if isinstance(cfg, dict): return ( float(cfg.get("overall", 85.0)), float(cfg.get("high_risk", 95.0)), ) value = float(cfg) return value, value #: Subdirectories excluded from the coverage denominator (non-source). COVERAGE_EXCLUDE_DIRS: tuple[str, ...] = ( "docs/", "test-output/", "tests/", "scripts/", "node_modules/", ".git/", ) #: Weight of each metric grade for the w1 term (worst grade wins per file). _GRADE_WEIGHT: dict[str, float] = { "fail": 3.0, "warn": 2.0, "good": 1.0, "na": 0.5, } def _is_source_file(rel_path: str) -> bool: """Heuristic filter for source files vs. docs/tests/generated output.""" normalized = rel_path.replace("\\", "/").lower() if any(normalized.startswith(d) for d in COVERAGE_EXCLUDE_DIRS): return False if normalized.endswith( (".bak", ".clean", ".debug1", ".fullbak", ".tmp", ".map") ): return False return True def _per_file_w1(file_rel: str, repo_root: Path) -> float: """Compute the risk grade (w1 term) for a single source file. Uses the per-file SQL / vulnerability / redundancy heuristic scans. ``exception_coverage`` is deliberately excluded: on a per-file basis it is ~always ``fail`` for real-world modules (most files have few explicit error branches), which gives w1 zero discriminative power. ``high_risk_density`` is excluded too (it is ~always 100% for any file containing SQL/async/transaction markers). ``repo_root`` is resolved against when ``file_rel`` is not absolute. """ raw = file_rel.replace("\\", "/") candidate = Path(file_rel) if not candidate.is_absolute(): candidate = repo_root / raw if not candidate.is_file(): return 0.0 try: lines = candidate.read_text( encoding="utf-8", errors="replace", ).splitlines() except OSError: return 0.0 if not lines: return 0.0 line_items = [(raw, line, i) for i, line in enumerate(lines, start=1)] sql_hits = _count_matching(line_items, _SQL_RISK_PATTERNS) sql_grade = _grade("sql_risk", float(sql_hits)) vuln_hits = _count_matching(line_items, _VULNERABILITY_PATTERNS) vuln_grade = _grade("vulnerability_risk", float(vuln_hits)) sig_count: dict[str, int] = Counter() for _p, line, _n in line_items: sig = _normalized_signature(line) if len(sig) >= 24: sig_count[sig] += 1 dup_lines = sum(c for c in sig_count.values() if c >= 3) redundancy_rate = (dup_lines / len(line_items) * 100.0) redund_grade = _grade("redundancy_rate", redundancy_rate) grades = [ g for g in (sql_grade, vuln_grade, redund_grade) if g != "na" ] if not grades: return _GRADE_WEIGHT["na"] worst = max(grades, key=lambda g: _GRADE_WEIGHT.get(g, 0.0)) return _GRADE_WEIGHT.get(worst, _GRADE_WEIGHT["na"]) def _topology_hits(store: GraphStore, file_rel: str) -> float: """Count graph topology signal hits (w2 term) for a file's nodes. Uses hub degree (>= 10 incoming calls) and untested hotspot flags as lightweight proxies; avoids re-running the top-N truncated tools so the w2 term is computed over the whole graph, not the first N nodes. """ abs_path = normalize_file_path(Path(file_rel)) nodes = store.get_nodes_by_file(abs_path) if not nodes: return 0.0 hits = 0 for n in nodes: edges = store.get_edges_by_target(n.qualified_name) incoming = [ e for e in edges if e.kind in ("CALLS", "REFERENCES", "IMPLEMENTS") ] if len(incoming) >= 10: hits += 1 if not n.is_test and len(incoming) >= 5 and not _has_tested_by(store, n.qualified_name): hits += 0.5 return hits def _has_tested_by(store: GraphStore, qualified_name: str) -> bool: try: edges = store.get_edges_by_target(qualified_name) return any(e.kind == "TESTED_BY" for e in edges) except Exception: return False def _file_weights( store: GraphStore, repo_root: Path, source_files: list[str], source_abs: list[str], churn_map: dict[str, int], progress_cb: Callable[[float, Optional[str]], None] | None = None, ) -> dict[str, dict[str, Any]]: """Compute the risk weight ``w = w1(grade) + w2(topology) + w3(churn)`` for every source file, plus its high-risk flag. Shared by :func:`compute_coverage` and :func:`deep_read_plan` so both derive weights from exactly the same model. Keys are the normalized absolute paths used by the graph identity. Args: store: Open graph store. repo_root: Repository root. source_files: Source file paths relative to ``repo_root``. source_abs: Parallel list of normalized absolute paths. churn_map: Per-file commit counts. progress_cb: Optional ``(fraction, message)`` progress callback, invoked every 50 files. """ weights: dict[str, dict[str, Any]] = {} max_churn = max(churn_map.values()) if churn_map else 1 total = max(len(source_files), 1) for idx, (f, f_abs) in enumerate(zip(source_files, source_abs)): if progress_cb is not None and idx % 50 == 0: progress_cb(idx / total, f"computing risk weights ({idx}/{total})") w1 = _per_file_w1(f_abs, repo_root) w2 = _topology_hits(store, f_abs) rel_for_churn = f.lstrip("/").replace("\\", "/") raw_churn = churn_map.get(f, churn_map.get(rel_for_churn, 0)) w3 = (raw_churn / max_churn) if max_churn else 0.0 is_high_risk = (w1 >= 2.0 or w2 > 0.0 or raw_churn >= 3) weights[f_abs] = { "w": w1 + w2 + w3, "w1": w1, "w2": w2, "w3": w3, "raw_churn": raw_churn, "is_high_risk": is_high_risk, } if progress_cb is not None: progress_cb(1.0, "risk weights done") return weights def _real_line_count(repo_root: Path, rel: str, cache: dict) -> int: """Real line count of a source file, cached. Independent of graph node ``line_end`` (verified to have a +-1 skew vs actual file length).""" if rel in cache: return cache[rel] try: n = len((repo_root / rel).read_text(encoding="utf-8", errors="replace").splitlines()) except OSError: n = 0 cache[rel] = n return n def _union_len(ranges: list[list[int]]) -> int: """Covered line count of a list of inclusive [s,e] ranges (merged).""" if not ranges: return 0 merged: list[list[int]] = [] for s, e in sorted((int(a), int(b)) for a, b in ranges): if s < 1: s = 1 if e < s: continue if merged and s <= merged[-1][1] + 1: merged[-1][1] = max(merged[-1][1], e) else: merged.append([s, e]) return sum(e - s + 1 for s, e in merged) def _graph_semantic_units(store: GraphStore, root: Path, rel: str) -> list[dict]: """Graph semantic-unit nodes (Function/Class/Test) of a file.""" if not hasattr(store, "_conn"): return [] q = (root / rel).as_posix() try: rows = store._conn.execute( "SELECT kind, name, line_start, line_end FROM nodes " "WHERE file_path = ? AND kind IN ('Function','Class','Test') " "ORDER BY line_start", (q,), ).fetchall() except Exception: return [] out = [] for r in rows: try: out.append( { "kind": r["kind"], "name": r["name"], "line_start": int(r["line_start"]), "line_end": int(r["line_end"]), } ) except (KeyError, TypeError): continue return out def _is_giant_file(graph_units: list[dict], real_lines: int) -> bool: """Unit-exempt when the largest unit spans >80% of the file's lines. A single huge function (e.g. migrations.rs run_migrations = 98% of the file) makes unit-completeness meaningless, so such files are checked on line coverage only. Small files with 2-3 ordinary units are NOT exempt: they must still cover every unit.""" if not graph_units or real_lines <= 0: return False largest = max(u["line_end"] - u["line_start"] + 1 for u in graph_units) return (largest / real_lines) > 0.8 def _unit_overlap(a: list[int], b: list[int]) -> int: lo, hi = max(a[0], b[0]), min(a[1], b[1]) return max(0, hi - lo + 1) def _unit_covered( graph_unit: dict, read_ranges: list[list[int]], unit_ranges: list[list[int]], matched: set[int], ) -> bool: """A graph unit is covered iff one reported unit range matches exactly (preferred) or overlaps >=80% of the graph unit span, is not already claimed by a higher-overlap unit (one-to-one), and >=80% of the graph unit's lines fall inside union(read_ranges).""" gs, ge = graph_unit["line_start"], graph_unit["line_end"] gspan = max(1, ge - gs + 1) exact = [i for i, (s, e) in enumerate(unit_ranges) if s == gs and e == ge] if exact: idx = exact[0] if idx in matched: return False matched.add(idx) else: best_idx, best_overlap = None, 0 for i, (s, e) in enumerate(unit_ranges): ov = _unit_overlap([gs, ge], [s, e]) if ov > best_overlap: best_overlap, best_idx = ov, i if best_idx is None or best_idx in matched: return False if best_overlap / gspan < 0.8: return False matched.add(best_idx) in_union = 0 for s, e in _merge_ranges(read_ranges): lo, hi = max(gs, s), min(ge, e) if lo <= hi: in_union += hi - lo + 1 return (in_union / gspan) >= 0.8 def _merge_ranges(ranges: list[list[int]]) -> list[list[int]]: merged: list[list[int]] = [] for s, e in sorted((int(a), int(b)) for a, b in ranges or []): if s < 1: s = 1 if e < s: continue if merged and s <= merged[-1][1] + 1: merged[-1][1] = max(merged[-1][1], e) else: merged.append([s, e]) return merged def _unit_gaps( graph_units: list[dict], reported_units: list[dict], read_ranges: list[list[int]], ) -> list[dict]: """Return graph units not covered by the reported semantic units.""" unit_ranges = [ [int(u.get("range", [0, 0])[0]), int(u.get("range", [0, 0])[1])] for u in reported_units ] matched: set[int] = set() gaps = [] for u in graph_units: if not _unit_covered(u, read_ranges, unit_ranges, matched): gaps.append( { "name": u["name"], "range": [u["line_start"], u["line_end"]], } ) return gaps def compute_coverage( store: GraphStore, repo_root: Path, deep_read_files: list[str], include_churn: bool = True, gate: str = "high_risk", file_read_ranges: dict[str, list[list[int]]] | None = None, file_semantic_units: dict[str, list[dict]] | None = None, line_target: float = 95.0, unit_target: float = 100.0, progress_cb: Callable[[float, Optional[str]], None] | None = None, ) -> dict[str, Any]: """Compute file-count review coverage for a set of deep-read files. Coverage = number of deep-read files / total number of source files. The overall coverage uses every source file as the denominator; the high-risk coverage uses only the signal-flagged subset. Per-file risk weights (w = w1 grade + w2 topology + w3 churn) are still computed and used to rank ``priority_deep_read_files`` so the highest-risk files are read first, but the coverage percentage itself is file-count based. Also returns the list of files never touched by any deep-read / signal (``silent_files``) for G2 spot-check sampling, the uncovered files list, and (new) the file count still needed to reach the target plus the priority deep-read file list that would close that gap. Args: store: Open graph store (caller owns and closes it). repo_root: Repository root. deep_read_files: Files the agent actually deep-read during the review (relative or absolute paths). include_churn: Include git-churn as the w3 weight term. gate: Coverage gate mode. ``"high_risk"`` (default) gates on the signal-flagged subset only; ``"overall"`` gates on all source files; ``"both"`` requires *both* the overall and the high-risk coverage to meet the target; ``"both+line"`` additionally requires per-file line coverage >= ``line_target`` and unit completeness (gap-free) per file; ``"line+unit"`` (feature reviews) checks ONLY line coverage and unit completeness - file-count / high-risk coverage are not computed and ``coverage_pct`` / ``high_risk_coverage_pct`` return ``None``. file_read_ranges: Optional mapping {rel_path: [[s,e], ...]} of the line ranges a sub-agent actually read per deep-read file. Used for the line-coverage gate (denominator = real file line count). Absent file => line coverage treated as satisfied. file_semantic_units: Optional mapping {rel_path: [{"range":[s,e], "kind":.., "name":..}, ...]} reported per deep-read file. Used for the unit-completeness gate (gap-free vs graph units). Absent file => unit completeness treated as satisfied. line_target: Min per-file line coverage percent for gate="both+line". unit_target: Min unit completeness percent (gap-free share). progress_cb: Optional ``(fraction, message)`` progress callback; forwarded to churn and weight computation. Returns: Dict with coverage_pct, high_risk_coverage_pct, deep_read_count, total_files, deep_read_weight, total_weight, target_reached, target (high-risk target, back-compat), overall_target, high_risk_target, remaining_files_to_target (file-count gap) and the compatible remaining_weight_to_target (risk-weight gap), priority_deep_read_files, uncovered_files, silent_files and grade. With gate="both+line" also returns line_coverage_pct, unit_coverage_pct, line_gap_files, unit_gap_files, unit_exempt_files, missing_data_files. With gate="line+unit" same line/unit fields plus ``gate="line+unit"``; ``coverage_pct``/``high_risk_coverage_pct`` are ``None`` because file-count coverage is not computed. FAIL-CLOSED (v2.5.2): a deep-read file with no file_read_ranges (or no file_semantic_units when the graph has units) is recorded as a line/unit gap and listed in missing_data_files. Missing data therefore makes target_reached=false instead of silently passing. """ all_files = store.get_all_files() source_files = [f for f in all_files if _is_source_file(f)] if not source_files: return { "status": "ok", "coverage_pct": 0.0, "grade": "na", "deep_read_count": 0, "total_files": 0, "deep_read_weight": 0.0, "total_weight": 0.0, "target_reached": False, "target": COVERAGE_TARGETS.get("standard", {}).get("high_risk", 95.0), "overall_target": COVERAGE_TARGETS.get("standard", {}).get("overall", 85.0), "high_risk_target": COVERAGE_TARGETS.get("standard", {}).get("high_risk", 95.0), "remaining_files_to_target": 0.0, "remaining_weight_to_target": 0.0, "priority_deep_read_files": [], "uncovered_files": [], "silent_files": [], "note": "No source files found in graph.", } # Resolve relative paths against repo_root so graph identity matches. def _abs(rel: str) -> str: p = Path(rel) if p.is_absolute(): return normalize_file_path(p) return normalize_file_path(repo_root / p) source_abs = [_abs(f) for f in source_files] source_abs_set = set(source_abs) # Normalise deep-read list against the graph's file paths. deep_read_set: set[str] = set() for f in deep_read_files or []: norm = _abs(f) if norm in source_abs_set or norm in set(all_files): deep_read_set.add(norm) # Line / unit coverage data (gate="both+line" / "line+unit"). line_gap_files: list[dict] = [] unit_gap_files: list[dict] = [] unit_exempt_files: list[dict] = [] missing_data_files: list[dict] = [] _line_cache: dict[str, int] = {} _line_tot = 0 _line_cov = 0 _unit_tot = 0 _unit_cov = 0 if gate in ("both+line", "line+unit"): root_str = str(repo_root).replace("\\", "/").rstrip("/") for f, f_abs in zip(source_files, source_abs): if f_abs not in deep_read_set: continue # rel is relative to repo_root (sub-agents report relative paths) rel = f.replace("\\", "/") if rel.startswith(root_str + "/"): rel = rel[len(root_str) + 1:] _line_tot += 1 _unit_tot += 1 # --- line coverage --- ranges = (file_read_ranges or {}).get(rel) or ( file_read_ranges or {} ).get(f_abs) real_lines = _real_line_count(repo_root, rel, _line_cache) if ranges and real_lines > 0: covered = _union_len(ranges) pct = covered / real_lines * 100.0 _line_cov += 1 if pct >= line_target else 0 if pct < line_target: line_gap_files.append( { "path": rel, "coverage_pct": round(pct, 1), "total_lines": real_lines, "covered_lines": covered, } ) else: # FAIL-CLOSED: a deep-read file without read_ranges (or an # empty/unreadable file) cannot be verified - record a gap # instead of silently treating it as satisfied. Without this # branch, gate="both+line" would report target_reached=true # while line_coverage_pct stays 0 (silent green). line_gap_files.append( { "path": rel, "coverage_pct": 0.0, "reason": ( "missing read_ranges" if not ranges else "unreadable/empty file" ), } ) missing_data_files.append( {"path": rel, "field": "file_read_ranges"} ) # --- unit completeness --- units = (file_semantic_units or {}).get(rel) or ( file_semantic_units or {} ).get(f_abs) g_units = _graph_semantic_units(store, repo_root, rel) if g_units and units is None: # FAIL-CLOSED: graph has semantic units but the sub-agent # reported none - cannot verify completeness, record a gap. unit_gap_files.append( { "path": rel, "reason": "missing semantic_units", } ) missing_data_files.append( {"path": rel, "field": "file_semantic_units"} ) elif g_units and units is not None: giant = _is_giant_file(g_units, real_lines) if giant: unit_exempt_files.append( { "path": rel, "reason": ( "giant-file: units=%d largest_span=%.0f%%" % ( len(g_units), 100 * max(u["line_end"] - u["line_start"] + 1 for u in g_units) / max(1, real_lines), ) ), } ) _unit_tot -= 1 # exempt from unit gate else: uncovered = _unit_gaps(g_units, units, ranges or []) if uncovered: unit_gap_files.append( { "path": rel, "total_units": len(g_units), "covered_units": len(g_units) - len(uncovered), "uncovered": uncovered, } ) else: _unit_cov += 1 unit_coverage_pct = (_unit_cov / _unit_tot * 100.0) if _unit_tot else 100.0 line_coverage_pct = (_line_cov / _line_tot * 100.0) if _line_tot else 100.0 churn_map: dict[str, int] = {} if include_churn: churn_map = compute_file_churn(str(repo_root), progress_cb=progress_cb) weights = _file_weights(store, repo_root, source_files, source_abs, churn_map, progress_cb=progress_cb) # Single pass: compute overall weight (all source files) and the # high-risk subset weight (files flagged by any signal). total_weight = 0.0 deep_read_weight = 0.0 high_risk_total = 0.0 high_risk_deep = 0.0 high_risk_deep_count = 0 high_risk_files: list[str] = [] uncovered_files: list[str] = [] silent_files: list[str] = [] for f, f_abs in zip(source_files, source_abs): wi = weights[f_abs] w = wi["w"] total_weight += w if f_abs in deep_read_set: deep_read_weight += w if wi["is_high_risk"]: high_risk_total += w high_risk_files.append(f) if f_abs in deep_read_set: high_risk_deep += w high_risk_deep_count += 1 elif f_abs not in deep_read_set: # Not high-risk and not deep-read: silent candidate. if wi["w1"] < 2.0 and wi["w2"] == 0.0 and wi["raw_churn"] < 3: silent_files.append(f) uncovered_files = [ f for f, f_abs in zip(source_files, source_abs) if f_abs not in deep_read_set ] total_files = len(source_files) deep_read_count = len(deep_read_set) coverage_pct = ( (deep_read_count / total_files * 100.0) if total_files else 0.0 ) high_risk_total_files = len(high_risk_files) high_risk_pct = ( (high_risk_deep_count / high_risk_total_files * 100.0) if high_risk_total_files else 0.0 ) overall_target, high_risk_target = _coverage_targets() overall_ok = coverage_pct >= overall_target high_risk_ok = ( (high_risk_pct >= high_risk_target) if high_risk_total_files else overall_ok ) if gate == "overall": target_reached = overall_ok gate_pct = coverage_pct gate_target = overall_target elif gate == "both": target_reached = overall_ok and high_risk_ok gate_pct = min(coverage_pct, high_risk_pct) if high_risk_total_files else coverage_pct gate_target = min(overall_target, high_risk_target) elif gate == "both+line": quality_ok = ( len(line_gap_files) == 0 and len(unit_gap_files) == 0 ) target_reached = overall_ok and high_risk_ok and quality_ok gate_pct = ( min(coverage_pct, high_risk_pct, line_coverage_pct, unit_coverage_pct) if high_risk_total_files else min(coverage_pct, line_coverage_pct, unit_coverage_pct) ) gate_target = min(overall_target, high_risk_target, line_target, unit_target) elif gate == "line+unit": # Feature reviews: check ONLY line coverage + unit completeness. # File-count / high-risk coverage is deliberately not part of the # gate (coverage_pct / high_risk_coverage_pct are returned as None). quality_ok = ( len(line_gap_files) == 0 and len(unit_gap_files) == 0 ) target_reached = quality_ok gate_pct = min(line_coverage_pct, unit_coverage_pct) gate_target = min(line_target, unit_target) else: # "high_risk" (default) target_reached = high_risk_ok gate_pct = high_risk_pct gate_target = high_risk_target grade = ( "good" if target_reached else ("warn" if gate_pct >= gate_target * 0.8 else "fail") ) remaining_files_to_target = max( 0.0, total_files * overall_target / 100.0 - deep_read_count ) remaining_weight_to_target = max( 0.0, total_weight * overall_target / 100.0 - deep_read_weight ) priority_deep_read_files = [ {"path": f_abs, "weight": round(weights[f_abs]["w"], 3)} for f, f_abs in zip(source_files, source_abs) if f_abs not in deep_read_set ] priority_deep_read_files.sort(key=lambda e: e["weight"], reverse=True) return { "status": "ok", "coverage_pct": round(coverage_pct, 1) if gate != "line+unit" else None, "high_risk_coverage_pct": round(high_risk_pct, 1) if gate != "line+unit" else None, "grade": grade, "deep_read_count": deep_read_count, "total_files": total_files, "high_risk_total_files": high_risk_total_files, "high_risk_deep_count": high_risk_deep_count, "deep_read_weight": round(deep_read_weight, 2), "total_weight": round(total_weight, 2), "target_reached": target_reached, "target": high_risk_target, "overall_target": overall_target, "high_risk_target": high_risk_target, "gate": gate, "line_coverage_pct": round(line_coverage_pct, 1) if gate in ("both+line", "line+unit") else None, "unit_coverage_pct": round(unit_coverage_pct, 1) if gate in ("both+line", "line+unit") else None, "line_gap_files": line_gap_files if gate in ("both+line", "line+unit") else [], "unit_gap_files": unit_gap_files if gate in ("both+line", "line+unit") else [], "unit_exempt_files": unit_exempt_files if gate in ("both+line", "line+unit") else [], "missing_data_files": missing_data_files if gate in ("both+line", "line+unit") else [], "remaining_files_to_target": round(remaining_files_to_target, 2), "remaining_weight_to_target": round(remaining_weight_to_target, 2), "priority_deep_read_files": priority_deep_read_files, "uncovered_files": uncovered_files, "silent_files": silent_files, "note": ( "行/单元门禁(feature):仅要求行覆盖 >= " f"{line_target}% 且单元完整性无缺口;coverage_pct / " "high_risk_coverage_pct 为 None(未做文件数/高风险覆盖检查)。" ) if gate == "line+unit" else ( "双重覆盖口径:全库 = 深读文件数 / 全部源文件数;高风险 = 深读 / " "信号点名文件数。G3 门禁口径可配置(high_risk/overall/both)。" "每文件风险权重(w = w1 指标分 + w2 拓扑 + w3 变更频率)仍用于" "排序 priority_deep_read_files;低于目标 => 审查覆盖不足。" ), } def deep_read_plan( store: GraphStore, repo_root: Path, deep_read_files: list[str] | None = None, target_coverage: float = 85.0, batch_size: int = 40, include_churn: bool = True, prior_covered: set[str] | None = None, progress_cb: Callable[[float, Optional[str]], None] | None = None, ) -> dict[str, Any]: """Generate a grouped deep-read plan that closes the coverage gap. Greedily selects files not yet deep-read until the file-count coverage target is reached, preferring the highest-risk (highest-weight) files first, then groups them by parent directory so each batch can be dispatched to one parallel sub-agent. Args: store: Open graph store (caller owns and closes it). repo_root: Repository root. deep_read_files: Files already deep-read this round. target_coverage: Target overall coverage percentage (default 85). batch_size: Max files per group (default 40). include_churn: Include git-churn as the w3 weight term. prior_covered: Absolute paths already deep-read in prior rounds (from the coverage index); excluded from the plan. progress_cb: Optional ``(fraction, message)`` progress callback; forwarded to churn and weight computation. Returns: Dict with current/remaining file counts (primary) plus compatible weight counts, the greedy gap-closing file list and the directory groups ready for sub-agent dispatch. """ all_files = store.get_all_files() source_files = [f for f in all_files if _is_source_file(f)] if not source_files: return { "status": "ok", "current_coverage_pct": 0.0, "current_files": 0, "target_files": 0, "remaining_files": 0, "current_weight": 0.0, "target_weight": 0.0, "remaining_weight": 0.0, "planned_files": [], "planned_weight": 0.0, "groups": [], "estimated_batches": 0, "gate": "overall", } def _abs(rel: str) -> str: p = Path(rel) if p.is_absolute(): return normalize_file_path(p) return normalize_file_path(repo_root / p) source_abs = [_abs(f) for f in source_files] source_abs_set = set(source_abs) deep_read_set: set[str] = set() for f in deep_read_files or []: norm = _abs(f) if norm in source_abs_set or norm in set(all_files): deep_read_set.add(norm) if prior_covered: deep_read_set |= {_abs(p) for p in prior_covered if _abs(p) in source_abs_set} churn_map: dict[str, int] = {} if include_churn: churn_map = compute_file_churn(str(repo_root), progress_cb=progress_cb) weights = _file_weights(store, repo_root, source_files, source_abs, churn_map, progress_cb=progress_cb) total_weight = sum(wi["w"] for wi in weights.values()) total_files = len(source_files) current_files = len(deep_read_set) current_weight = sum( weights[f_abs]["w"] for f_abs in source_abs_set if f_abs in deep_read_set and f_abs in weights ) current_pct = (current_files / total_files * 100.0) if total_files else 0.0 target_files = total_files * target_coverage / 100.0 remaining_files = max(0.0, target_files - current_files) target_weight = total_weight * target_coverage / 100.0 remaining_weight = max(0.0, target_weight - current_weight) # Greedy: pick highest-weight uncovered files until the gap is closed. candidates = sorted( ( {"path": f_abs, "weight": weights[f_abs]["w"]} for f_abs in source_abs if f_abs not in deep_read_set and f_abs in weights ), key=lambda e: e["weight"], reverse=True, ) planned: list[str] = [] accrued_weight = 0.0 for cand in candidates: if len(planned) >= remaining_files: break planned.append(cand["path"]) accrued_weight += cand["weight"] # Group by parent directory, keeping group order by total weight desc. from collections import OrderedDict by_dir: "OrderedDict[str, list[str]]" = OrderedDict() for p in planned: d = Path(p).parent.name or "." by_dir.setdefault(d, []).append(p) groups: list[dict[str, Any]] = [] for d, files in by_dir.items(): g_weight = sum(weights[_abs(f)]["w"] for f in files) for i in range(0, len(files), batch_size): chunk = files[i : i + batch_size] groups.append( { "name": d, "weight": round(g_weight, 2), "files": chunk, "batch_index": i // batch_size, } ) return { "status": "ok", "target_coverage": target_coverage, "current_coverage_pct": round(current_pct, 1), "current_files": current_files, "target_files": round(target_files, 2), "remaining_files": round(remaining_files, 2), "current_weight": round(current_weight, 2), "target_weight": round(target_weight, 2), "remaining_weight": round(remaining_weight, 2), "planned_files": planned, "planned_weight": round(accrued_weight, 2), "groups": groups, "estimated_batches": len(groups), "gate": "overall", } def check_community_health(store: GraphStore) -> dict[str, Any]: """Detect community/node attribution desync (nodes.community_id NULL). The communities table may carry a correct ``size`` while ``nodes.community_id`` is stale (e.g. after an incremental build that rebuilt nodes but skipped community re-attribution). Returns the non-null ratio and a ``needs_postprocess`` flag so the review skill can trigger ``postprocess`` before computing coverage. Args: store: Open graph store (caller owns and closes it). Returns: Dict with total_nodes, attributed_nodes, attribution_pct, needs_postprocess and note. """ try: total = store._conn.execute("SELECT COUNT(*) FROM nodes").fetchone()[0] attributed = store._conn.execute( "SELECT COUNT(*) FROM nodes WHERE community_id IS NOT NULL" ).fetchone()[0] non_file = store._conn.execute( "SELECT COUNT(*) FROM nodes WHERE kind != 'File'" ).fetchone()[0] except Exception as exc: # pragma: no cover return { "status": "error", "needs_postprocess": True, "note": f"Community health check failed: {exc}", } ratio = (attributed / non_file * 100.0) if non_file else 0.0 # A fully healthy graph has ~100% attribution; allow a small delta for # nodes without community membership. needs = ratio < 90.0 return { "status": "ok", "total_nodes": total, "attributed_nodes": attributed, "non_file_nodes": non_file, "attribution_pct": round(ratio, 1), "needs_postprocess": needs, "note": ( "nodes.community_id attribution ratio. Low ratio => run " "`code-review-graph postprocess` to re-attach members." ), }