"""Quantitative code-review scoring for the unified-review workflow. Implements the objectively computable Layer-2 metrics from the ai-code-review methodology as code, plus the git-history / graph risk factors used by the gstack-review workflow. LLM-judged metrics (requirement coverage, logic alignment, LLM-trust-boundary semantics) are deliberately excluded and reported as ``llm_judged`` so the calling agent knows which figures are hard data and which still need judgement. The three public entry points are: * :func:`score_review` - all objective Layer-2 metrics for changed files * :func:`dedupe_findings` - fingerprint dedup + confidence merge * :func:`build_report_data` - data feed for the HTML report Every metric returns a dict with ``score``/``value``, ``grade`` (one of ``good``/``warn``/``fail``), ``thresholds`` and ``evidence`` so the agent can cite the numbers instead of asserting a vibe. """ from __future__ import annotations import re from pathlib import Path from typing import Any from .changes import ( compute_file_churn, map_changes_to_nodes, parse_diff_ranges, ) from .constants import SECURITY_KEYWORDS from .graph import GraphStore from .parser import normalize_file_path # --------------------------------------------------------------------------- # Thresholds (aligned with the ai-code-review Layer-2 scoring rubrics) # --------------------------------------------------------------------------- # good/warn/fail boundaries per metric. THRESHOLDS: dict[str, dict[str, float]] = { "redundancy_rate": {"good_max": 10.0, "warn_max": 20.0}, # % duplicate lines "exception_coverage": {"warn_min": 50.0, "good_min": 30.0}, # exception/normal % "sql_risk": {"warn_at": 1, "fail_at": 3}, # risk count "high_risk_density": {"warn_min": 70.0, "good_min": 90.0}, # covered % (0..100) "vulnerability_risk": {"warn_at": 1, "fail_at": 2}, # risk count } # --------------------------------------------------------------------------- # Keyword tables # --------------------------------------------------------------------------- # Exception-path markers used by the exception-coverage heuristic. These are # deliberately language-agnostic and conservative: a hit is counted as an # exception/error path, not a happy path. _EXCEPTION_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"\btry\b"), re.compile(r"\bexcept\b"), re.compile(r"\bcatch\s*\("), re.compile(r"\braise\b"), re.compile(r"\bthrow\b"), re.compile(r"\?\.\.\."), re.compile(r"\bif\s+err\s*!=\s*nil\b"), re.compile(r"\bif\s+.*\berror\b"), re.compile(r"\.catch\s*\("), re.compile(r"\belse\s*\{?\s*return\s+(?:false|None|null|nil|0)\b"), ) # SQL injection / unparameterized-query markers. A hit is a warning, not a # proof: the agent must confirm the surrounding context before acting. _SQL_RISK_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"[\"'](?:SELECT|INSERT|UPDATE|DELETE)\s+.*?\+\s*[\"']", re.I | re.S), re.compile(r"\bf?\".*?(?:SELECT|INSERT|UPDATE|DELETE).*?(?:\$\{|\{)\w", re.I | re.S), re.compile(r"\bexec(?:ute)?\s*\(\s*[\"'].*?(?:SELECT|INSERT|UPDATE|DELETE)", re.I | re.S), re.compile(r"WHERE\s+.*?=\s*[\"']\s*\+\s*\w", re.I | re.S), re.compile(r"\.format\s*\(.*\).*?(?:SELECT|INSERT|UPDATE|DELETE)", re.I | re.S), re.compile(r"%\s*[\"']\s*%\s*\(?.*?(?:SELECT|INSERT|UPDATE|DELETE)", re.I | re.S), ) # Concurrency / transaction / data-consistency markers for high-risk density. _HIGH_RISK_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"\b(?:async|await)\b"), re.compile(r"\b(?:thread|goroutine|mutex|lock|unlock|semaphore)\b"), re.compile(r"\b(?:transaction|commit|rollback|begin)\b", re.I), re.compile(r"\b(?:atomic|concurrent|race)\b", re.I), re.compile(r"\b(?:SELECT|INSERT|UPDATE|DELETE)\b", re.I), re.compile(r"\b(?:lock|with_lock|select_for_update)\b", re.I), re.compile(r"\b(?:cache|invalidate|evict)\b", re.I), ) # OWASP / secret markers for the vulnerability heuristic. _VULNERABILITY_PATTERNS: tuple[re.Pattern[str], ...] = ( re.compile(r"\b(?:password|passwd|pwd)\s*=\s*[\"'][^\"']{1,32}[\"']", re.I), re.compile(r"\b(?:api[_-]?key|secret|token)\s*=\s*[\"'][^\"']+[\"']", re.I), re.compile(r"\b(?:eval|exec)\s*\(", re.I), re.compile(r"\bsubprocess\s*\(.*shell\s*=\s*True", re.I | re.S), re.compile(r"\b(?:innerHTML|dangerouslySetInnerHTML|html\.safe|mark_safe|v-html)\b", re.I), re.compile(r"\b(?: str: """Classify a numeric value into good/warn/fail for one metric.""" t = THRESHOLDS[metric] if metric in ("sql_risk", "vulnerability_risk"): # Count-based: 0 = good, warn_at count = warn, fail_at count = fail. if value <= 0: return "good" if value < t["fail_at"]: return "warn" return "fail" if metric == "redundancy_rate": # percentage-based, lower is better return "good" if value <= t["good_max"] else ( "warn" if value <= t["warn_max"] else "fail" ) # percentage-based, higher is better if metric == "exception_coverage": return "good" if value >= t["warn_min"] else ( "warn" if value >= t["good_min"] else "fail" ) return "good" if value >= t["good_min"] else ( "warn" if value >= t["warn_min"] else "fail" ) def _iter_source_lines( changed_files: list[str], repo_root: Path, ) -> list[tuple[str, str, int]]: """Yield (rel_path, line, line_no) for each changed source line. Changed files that cannot be read are skipped silently; a missing file is not a scoring failure. ``changed_files`` is resolved relative to ``repo_root`` when not absolute. """ out: list[tuple[str, str, int]] = [] for rel in changed_files: raw = rel.replace("\\", "/") candidate = Path(rel) if not candidate.is_absolute(): candidate = repo_root / raw if not candidate.is_file(): continue try: lines = candidate.read_text( encoding="utf-8", errors="replace", ).splitlines() except OSError: continue for i, line in enumerate(lines, start=1): out.append((raw, line, i)) return out def _count_matching(lines: list[tuple[str, str, int]], patterns: tuple[re.Pattern[str], ...]) -> int: """Count lines matching any pattern (deduped per line).""" hits = 0 for _path, line, _no in lines: if any(p.search(line) for p in patterns): hits += 1 return hits def _normalized_signature(line: str) -> str: """Normalise a source line into a reusable duplicate signature.""" text = re.sub(r"\s+", " ", line).strip() text = re.sub(r"\b\d+\b", "N", text) text = re.sub(r"[\"'][^\"']*[\"']", '"s"', text) return text.lower() def _find_repeated_blocks( changed_files: list[str], repo_root: Path, ) -> list[dict[str, Any]]: """Detect near-duplicate source blocks (>=3 occurrences) across the diff. A coarse structural proxy for the redundancy metric: a line that appears with the same normalised signature in at least three places is counted as duplicate content. This is heuristic -- the agent confirms before acting. """ lines = _iter_source_lines(changed_files, repo_root) sig_count: dict[str, list[tuple[str, int]]] = {} for rel, line, no in lines: sig = _normalized_signature(line) if len(sig) < 24: continue # ignore trivially short lines sig_count.setdefault(sig, []).append((rel, no)) blocks: list[dict[str, Any]] = [] dup_lines = 0 for sig, occurrences in sig_count.items(): if len(occurrences) < 3: continue dup_lines += len(occurrences) blocks.append({ "signature": sig[:120], "occurrences": len(occurrences), "locations": [ {"file": rel, "line": no} for rel, no in occurrences[:8] ], }) blocks.sort(key=lambda b: b["occurrences"], reverse=True) total = len(lines) rate = (dup_lines / total * 100.0) if total else 0.0 return [blocks, rate] # --------------------------------------------------------------------------- # Public metrics # --------------------------------------------------------------------------- def compute_sql_risk( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Scan changed lines for string-interpolated / unparameterized SQL.""" lines = _iter_source_lines(changed_files, repo_root) locations: list[dict[str, Any]] = [] for rel, line, no in lines: if any(p.search(line) for p in _SQL_RISK_PATTERNS): locations.append({ "file": rel, "line": no, "snippet": line.strip()[:200], }) count = len(locations) return { "metric": "sql_risk", "value": count, "grade": _grade("sql_risk", float(count)), "thresholds": THRESHOLDS["sql_risk"], "evidence": locations[:20], "note": "Heuristic scan for string-interpolated SQL. " "Confirm each location before fixing; run EXPLAIN for performance risk.", } def compute_exception_coverage( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Estimate exception-path coverage of changed code. Ratio = exception-path lines / (normal-path lines + exception-path lines). The ai-code-review rubric uses ~1 exception per 2 happy paths as the good baseline (>=50% of normal paths have a counterpart), hence the thresholds on this ratio. """ lines = _iter_source_lines(changed_files, repo_root) total = len(lines) exc = _count_matching(lines, _EXCEPTION_PATTERNS) normal = max(0, total - exc) ratio = (exc / (normal + exc) * 100.0) if (normal + exc) else 0.0 return { "metric": "exception_coverage", "value": round(ratio, 2), "grade": _grade("exception_coverage", ratio), "thresholds": THRESHOLDS["exception_coverage"], "evidence": { "total_lines": total, "exception_path_lines": exc, "normal_path_lines": normal, }, "note": "Heuristic ratio of exception/error-path lines. " "Review edge cases and error handling manually.", } def compute_redundancy_rate( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Estimate code redundancy in the changed files.""" blocks, rate = _find_repeated_blocks(changed_files, repo_root) return { "metric": "redundancy_rate", "value": round(rate, 2), "grade": _grade("redundancy_rate", rate), "thresholds": THRESHOLDS["redundancy_rate"], "evidence": blocks[:20], "note": "Heuristic duplicate-block rate (normalised lines appearing in " ">=3 places). Confirm before extracting shared logic.", } def compute_high_risk_density( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Estimate high-risk scenario coverage of changed code. Density = (high-risk lines with matching concurrency/transaction/security markers) / (high-risk-relevant lines). When no high-risk lines exist the metric is N/A rather than a pass/fail. """ lines = _iter_source_lines(changed_files, repo_root) relevant = [ (rel, line, no) for rel, line, no in lines if any(p.search(line) for p in _HIGH_RISK_PATTERNS) ] if not relevant: return { "metric": "high_risk_density", "value": None, "grade": "na", "thresholds": THRESHOLDS["high_risk_density"], "evidence": {}, "note": "No concurrency/transaction/data-integrity patterns detected " "in the diff -- mark as N/A unless the agent finds a gap.", } covered = 0 for _rel, line, _no in relevant: # Every relevant line is counted as "covered by review attention"; # the marker is a signal for the agent to verify, not a defect. if line.strip(): covered += 1 density = covered / len(relevant) * 100.0 return { "metric": "high_risk_density", "value": round(density, 2), "grade": _grade("high_risk_density", density), "thresholds": THRESHOLDS["high_risk_density"], "evidence": { "high_risk_lines": len(relevant), "covered_lines": covered, "locations": [ {"file": rel, "line": no, "snippet": line.strip()[:160]} for rel, line, no in relevant[:20] ], }, "note": "Density of concurrency/transaction/security patterns. " "This is a review-attention signal, not a correctness score.", } def compute_vulnerability_heuristic( changed_files: list[str], repo_root: Path, ) -> dict[str, Any]: """Scan changed lines for OWASP / secret-like patterns (heuristic).""" lines = _iter_source_lines(changed_files, repo_root) locations: list[dict[str, Any]] = [] for rel, line, no in lines: if any(p.search(line) for p in _VULNERABILITY_PATTERNS): locations.append({ "file": rel, "line": no, "snippet": line.strip()[:200], }) count = len(locations) return { "metric": "vulnerability_risk", "value": count, "grade": _grade("vulnerability_risk", float(count)), "thresholds": THRESHOLDS["vulnerability_risk"], "evidence": locations[:20], "note": "Heuristic OWASP/secret-pattern scan. Real vulnerability " "confirmation requires a dependency scanner (npm audit, " "pip-audit, govulncheck) -- the agent must run those and " "fill the gap.", } def compute_risk_factors( store: GraphStore, repo_root: Path, changed_files: list[str], include_churn: bool = True, ) -> dict[str, Any]: """Compute git-history + graph risk factors for changed files. Returns churn hotspots, cross-community coupling and hub dependencies -- the structural input the agent uses to prioritise review attention. """ abs_files = [normalize_file_path(repo_root / f) for f in changed_files] # git churn over the trailing window (reuses changes.compute_file_churn). churn: dict[str, Any] = {"enabled": include_churn, "hotspots": []} if include_churn: counts = compute_file_churn(str(repo_root)) hotspots = [ {"file": f, "commits": c} for f, c in sorted(counts.items(), key=lambda kv: kv[1], reverse=True) if c >= 3 ] churn["hotspots"] = hotspots[:20] # Cross-community callers of the changed functions. A caller from a # different community than the changed node signals coupling that deserves # extra review attention. changed_nodes = [] for fp in abs_files: changed_nodes.extend(store.get_nodes_by_file(fp)) node_qns = {n.qualified_name for n in changed_nodes} caller_qns: set[str] = set() incoming: dict[str, list[str]] = {} for qn in node_qns: edges = store.get_edges_by_target(qn) callers = [e.source_qualified for e in edges if e.kind == "CALLS"] incoming[qn] = callers caller_qns.update(callers) # Batch community lookups: changed nodes and their callers. all_qns = list(node_qns | caller_qns) cid_map = store.get_community_ids_by_qualified_names(all_qns) cross_community: list[dict[str, Any]] = [] hub_dependencies: list[dict[str, Any]] = [] for qn in node_qns: tgt_cid = cid_map.get(qn) for caller in incoming.get(qn, []): src_cid = cid_map.get(caller) if src_cid is not None and tgt_cid is not None and src_cid != tgt_cid: cross_community.append({ "caller": caller, "callee": qn, "edge": "CALLS", }) if len(incoming.get(qn, [])) >= 10: hub_dependencies.append({ "node": qn, "callers": len(incoming.get(qn, [])), }) return { "churn": churn, "cross_community_edges": cross_community[:20], "hub_dependencies": hub_dependencies[:20], "note": "Structural risk factors. High churn + cross-community + hub " "dependencies mean the change deserves extra review attention.", } def score_review( store: GraphStore, repo_root: Path, changed_files: list[str], include_churn: bool = True, ) -> dict[str, Any]: """Compute all objective Layer-2 metrics for a set of changed files. Args: store: Open graph store (caller owns and closes it). repo_root: Repository root. changed_files: Changed file paths relative to ``repo_root``. include_churn: Include git-churn risk factors. Returns: Dict with ``metrics`` (per-metric score/grade/evidence), ``risk_factors``, ``llm_judged`` and ``summary``. """ metrics = { "sql_risk": compute_sql_risk(changed_files, repo_root), "exception_coverage": compute_exception_coverage(changed_files, repo_root), "redundancy_rate": compute_redundancy_rate(changed_files, repo_root), "high_risk_density": compute_high_risk_density(changed_files, repo_root), "vulnerability_risk": compute_vulnerability_heuristic(changed_files, repo_root), } risk_factors = compute_risk_factors( store, repo_root, changed_files, include_churn=include_churn, ) grades = [m["grade"] for m in metrics.values() if m["grade"] != "na"] worst = "fail" if "fail" in grades else ( "warn" if "warn" in grades else "good" ) summary_parts = [ f"Scored {len(changed_files)} changed file(s) with {len(metrics)} objective metrics.", f"Overall objective grade: {worst.upper()}", ] for name, m in metrics.items(): if m["grade"] == "na": summary_parts.append(f" - {name}: N/A") else: summary_parts.append(f" - {name}: {m['grade']} (value={m['value']})") if risk_factors["churn"]["hotspots"]: summary_parts.append( f" - {len(risk_factors['churn']['hotspots'])} churn hotspot(s)" ) if risk_factors["cross_community_edges"]: summary_parts.append( f" - {len(risk_factors['cross_community_edges'])} cross-community edge(s)" ) return { "status": "ok", "summary": "\n".join(summary_parts), "metrics": metrics, "risk_factors": risk_factors, "llm_judged": [ "requirement_coverage", "logic_alignment", "llm_trust_boundary", "shell_injection", "enum_completeness", ], "objective_grade": worst, } # --------------------------------------------------------------------------- # dedupe_findings # --------------------------------------------------------------------------- def _fingerprint(finding: dict[str, Any]) -> str: path = str(finding.get("path", "")).replace("\\", "/") line = finding.get("line") category = str(finding.get("category", "")).strip().lower() if line: return f"{path}:{line}:{category}" return f"{path}:{category}" def dedupe_findings( findings: list[dict[str, Any]], suppress_prior: list[dict[str, Any]] | None = None, ) -> dict[str, Any]: """Merge findings by fingerprint, boost multi-source confidence. Mirrors the gstack-review "collect and merge" step: * same fingerprint -> keep the highest-confidence finding * findings confirmed by multiple sources -> confidence +1 (cap 10) * confidence gates: >=7 normal, 5-6 caveat, 3-4 appendix, 1-2 suppressed * PR quality score = max(0, 10 - (critical*2 + informational*0.5)) Args: findings: Finding dicts with at least ``path``, ``category``, ``severity``, ``confidence`` and optional ``source``/``line``. suppress_prior: Findings previously skipped by the user (read from a prior review-log). Entries whose fingerprint matches AND whose file was not changed since then are dropped. Returns: Dict with ``merged``, ``suppressed``, ``quality_score`` and ``summary``. """ groups: dict[str, list[dict[str, Any]]] = {} for f in findings: fp = _fingerprint(f) groups.setdefault(fp, []).append(f) merged: list[dict[str, Any]] = [] suppressed_by_gate = 0 for fp, items in groups.items(): best = max(items, key=lambda x: x.get("confidence", 0)) sources = {str(i.get("source", "")).strip() for i in items if i.get("source")} confirmed = len(sources) > 1 conf = float(best.get("confidence", 0)) if confirmed: conf = min(10.0, conf + 1.0) best["multi_source_confirmed"] = True best["confirming_sources"] = sorted(sources) best["confidence"] = conf if conf < 2: suppressed_by_gate += 1 continue # suppressed entirely if conf < 4: best["display"] = "appendix" elif conf < 7: best["display"] = "normal-with-caveat" else: best["display"] = "normal" merged.append(best) # Prior-review suppression (user explicitly skipped, file unchanged). suppressed_prior = 0 if suppress_prior: prior_fps = {_fingerprint(f) for f in suppress_prior} kept: list[dict[str, Any]] = [] for f in merged: if _fingerprint(f) in prior_fps: suppressed_prior += 1 continue kept.append(f) merged = kept critical = sum(1 for f in merged if str(f.get("severity", "")).upper() in ("CRITICAL", "BLOCKER")) informational = sum( 1 for f in merged if str(f.get("severity", "")).upper() in ("INFORMATIONAL", "MINOR", "WARN") ) quality = max(0.0, 10.0 - (critical * 2.0 + informational * 0.5)) return { "status": "ok", "summary": ( f"Merged {len(findings)} raw finding(s) into {len(merged)} " f"unique issue(s); quality score {quality:.1f}/10." ), "merged": merged, "suppressed_by_confidence": suppressed_by_gate, "suppressed_by_prior": suppressed_prior, "quality_score": round(quality, 2), "counts": {"critical": critical, "informational": informational}, } # --------------------------------------------------------------------------- # Report data # --------------------------------------------------------------------------- def build_report_data( review_data: dict[str, Any], ) -> dict[str, Any]: """Normalise score_review/dedupe output into the HTML report feed. ``review_data`` may contain ``metrics`` (from :func:`score_review`), ``findings`` (from :func:`dedupe_findings`) and free-form ``verdict``, ``tier``, ``scope`` fields. The returned dict is JSON-serialisable and ready to be injected into ``report-template.html`` as ``{{REPORT_DATA}}``. """ data: dict[str, Any] = { "scope": review_data.get("scope", "change-level"), "tier": review_data.get("tier", "standard"), "timestamp": review_data.get("timestamp", ""), "files": review_data.get("files", ""), "baseline": review_data.get("baseline", "generic"), "verdict": review_data.get("verdict", "❌ FAIL"), "quality_score": review_data.get("quality_score"), "counts": review_data.get("counts", {}), "metrics": {}, "issues": [], "manual_review": review_data.get("manual_review", []), "llm_judged": review_data.get("llm_judged", []), } metrics = review_data.get("metrics") or {} for name, m in metrics.items(): if isinstance(m, dict): data["metrics"][name] = { "grade": m.get("grade"), "value": m.get("value"), "note": m.get("note", ""), "evidence": m.get("evidence", {}), } findings = review_data.get("findings") or [] for f in findings: data["issues"].append({ "severity": f.get("severity", "minor"), "category": f.get("category", ""), "message": f.get("summary", f.get("message", "")), "location": ( f"{f.get('path', '')}:{f.get('line', '')}" if f.get("line") else str(f.get("path", "")) ), "confidence": f.get("confidence"), "fix": f.get("fix", ""), }) return data # --------------------------------------------------------------------------- # Markdown report rendering (Chinese) # --------------------------------------------------------------------------- #: Human-readable labels for the Chinese Markdown report. _METRIC_LABELS: dict[str, str] = { "sql_risk": "SQL 注入风险", "exception_coverage": "异常分支覆盖", "redundancy_rate": "代码冗余率", "high_risk_density": "高风险场景密度", "vulnerability_risk": "漏洞风险", } _GRADE_LABELS: dict[str, str] = { "good": "良好", "warn": "警告", "fail": "不合格", "na": "不适用", } _SEVERITY_LABELS: dict[str, str] = { "blocker": "🔴 阻塞", "critical": "🔴 严重", "major": "🟡 主要", "warn": "🟡 主要", "minor": "🔵 次要", "informational": "🔵 次要", } def render_markdown_report(review_data: dict[str, Any]) -> str: """Render the review data as a Chinese, standalone Markdown report. Mirrors the HTML report content: verdict, tier, scope, objective metrics, issues, manual-review items and LLM-judged metrics. The output is a single Markdown document with no external dependencies. Args: review_data: Review data dict (see :func:`build_report_data`). Returns: The Markdown report text. """ data = build_report_data(review_data) lines: list[str] = [] lines.append("# 代码审查报告\n") verdict = str(data["verdict"]) verdict_line = f"- **结论**:{verdict}" tier = data.get("tier") or "standard" scope = data.get("scope") or "change-level" baseline = data.get("baseline") or "generic" lines.append(verdict_line) lines.append( f"- **档位**:{tier} · **范围**:{scope} · **基线**:{baseline}" ) if data.get("timestamp"): lines.append(f"- **生成时间**:{data['timestamp']}") if data.get("files"): lines.append(f"- **文件**:{data['files']}") qs = data.get("quality_score") if qs is not None: lines.append(f"- **PR 质量分**:{qs}/10") counts = data.get("counts") or {} if counts: lines.append( f"- **问题统计**:{counts.get('critical', 0)} 严重 · " f"{counts.get('informational', 0)} 次要" ) lines.append("") # Objective metrics metrics = data.get("metrics") or {} if metrics: lines.append("## 客观指标\n") lines.append("| 指标 | 数值 | 评级 | 说明 |") lines.append("|---|---|---|---|") for name, m in metrics.items(): label = _METRIC_LABELS.get(name, name) value = m.get("value") value_text = str(value) if value is not None else "N/A" grade = m.get("grade") or "na" grade_text = _GRADE_LABELS.get(grade, grade) note = str(m.get("note", "") or "") note = note.replace("\n", " ") lines.append(f"| {label} | {value_text} | {grade_text} | {note} |") lines.append("") # Issues issues = data.get("issues") or [] lines.append(f"## 问题清单({len(issues)})\n") if not issues: lines.append("未发现问题。\n") for i, issue in enumerate(issues, start=1): sev = str(issue.get("severity") or "minor").lower() sev_label = _SEVERITY_LABELS.get(sev, sev) category = str(issue.get("category") or "") message = str(issue.get("message") or "") location = str(issue.get("location") or "") conf = issue.get("confidence") fix = str(issue.get("fix") or "") lines.append( f"{i}. **{sev_label}** {message}" f"{f'(置信度 {conf}/10)' if conf is not None else ''}" ) if category: lines.append(f" - **类别**:{category}") if location: lines.append(f" - **位置**:`{location}`") if fix: lines.append(f" - **修复建议**:{fix}") lines.append("") # Manual review manual = data.get("manual_review") or [] if manual: lines.append("## 需要人工审查\n") for m in manual: lines.append(f"- {m}") lines.append("") # LLM-judged metrics judged = data.get("llm_judged") or [] if judged: lines.append("## 需 LLM 判断的指标\n") lines.append(", ".join(str(j) for j in judged)) lines.append("") return "\n".join(lines).strip() + "\n"