Files
2026Technology-Competition/tests/test_eval_scorer.py
T
lhl becd3e1f57 chore(assets): 参赛提交规范红线修复(ASCII 化 + 相对路径)
按《参赛成果物提交规范·赛道一》§6 红线:
- samples/ 目录改名 sample/(git mv,保留历史)
- 10 个中日文样本文件 + docs 参赛手册 PDF 重命名为 ASCII
  (requirements_*/template_*/rules_*/contestant-handbook.pdf)
- tests/test_zh_template.py 硬编码绝对路径 D:\00_project\Genesis 改为相对路径
- 全局更新 21 个活动文件引用;历史日志/审查文档不改(追加说明记录)
全量 pytest 431 passed / 99.15%
2026-08-26 14:15:52 +08:00

181 lines
5.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""评分器测试(T13OV4)。
OV4 裁定:成功标准无量度(无黄金集/评分器)→ 建立黄金集 + 评分器。
本文件测试确定性可机器验证维度(可追溯性/占位符残留/章节完整性),
LLM 语义维度预留钩子;并测试黄金集加载。
"""
from __future__ import annotations
import pytest
from genesis.data_models import (
CellValue,
ExcelTable,
ParsedTemplate,
Provenance,
SheetType,
StructuredSource,
)
from genesis.eval.scorer import (
ChapterArtifact,
ChapterScorer,
DimensionScore,
EvalReport,
)
from genesis.eval.golden_set import GoldenCase, GoldenSet
# ---------- 小源(供可追溯性维度定位) ----------
def _source() -> StructuredSource:
cell = CellValue(
value="登録",
provenance=Provenance(file_name="f.xlsx", sheet_name="機能一覧", row=3, column="C", column_header="x"),
)
table = ExcelTable(
name="機能一覧", detected_type=SheetType.FUNCTION,
extraction_method="structured", headers=["v"], rows=[{"v": cell}],
)
return StructuredSource(
tables=[table],
template=ParsedTemplate(file_name="t.docx", sections=[], placeholders={}, styles={}),
rule_docs=[], image_analyses=[], existing_system=None, comments=[],
)
# ---------- 可追溯性维度 ----------
def test_traceability_full_when_all_uris_resolvable():
source = _source()
artifact = ChapterArtifact(
chapter_id="ch3",
text="機能一覧(出典: f.xlsx#機能一覧!C3",
source_uris=["f.xlsx#機能一覧!C3"],
template_sections_expected=["ch3"],
)
report = ChapterScorer().score([artifact], source)
trace = _dim(report, "traceability")
assert trace.score == 1.0
assert trace.passed is True
def test_traceability_zero_when_uris_fake():
source = _source()
artifact = ChapterArtifact(
chapter_id="ch3",
text="機能(出典: fake.xlsx#X!Z9",
source_uris=["fake.xlsx#X!Z9"],
template_sections_expected=["ch3"],
)
report = ChapterScorer().score([artifact], source)
trace = _dim(report, "traceability")
assert trace.score == 0.0
assert trace.passed is False
# ---------- 占位符残留维度 ----------
def test_placeholder_residue_fails():
source = _source()
artifact = ChapterArtifact(
chapter_id="ch3",
text="未替换占位符 {{section:db_tables}}",
source_uris=["f.xlsx#機能一覧!C3"],
template_sections_expected=["ch3"],
)
report = ChapterScorer().score([artifact], source)
dim = _dim(report, "placeholder_residue")
assert dim.score == 0.0
assert dim.passed is False
def test_placeholder_residue_ok_when_clean():
source = _source()
artifact = ChapterArtifact(
chapter_id="ch3", text="正常生成内容", source_uris=[], template_sections_expected=["ch3"],
)
report = ChapterScorer().score([artifact], source)
assert _dim(report, "placeholder_residue").score == 1.0
# ---------- 章节完整性维度 ----------
def test_completeness_fails_when_section_missing():
source = _source()
# 期望 ch3/ch4 两章,但只生成 ch3
artifacts = [ChapterArtifact(
chapter_id="ch3", text="a", source_uris=[], template_sections_expected=["ch3", "ch4"],
)]
report = ChapterScorer().score(artifacts, source)
dim = _dim(report, "chapter_completeness")
assert dim.score == 0.5
assert dim.passed is False
# ---------- LLM 维度钩子 ----------
def test_llm_dimension_hook_invoked():
source = _source()
called = {}
def fake_llm(chapter: ChapterArtifact) -> DimensionScore:
called["hit"] = True
return DimensionScore(name="llm_accuracy", score=0.8, passed=True, detail="stub")
artifact = ChapterArtifact(
chapter_id="ch3", text="x", source_uris=[], template_sections_expected=["ch3"],
)
scorer = ChapterScorer(llm_evaluators={"llm_accuracy": fake_llm})
report = scorer.score([artifact], source)
assert called.get("hit") is True
assert _dim(report, "llm_accuracy").score == 0.8
# ---------- 总分聚合 + 通过判定 ----------
def test_total_score_aggregation():
source = _source()
artifact = ChapterArtifact(
chapter_id="ch3", text="正常(出典: f.xlsx#機能一覧!C3",
source_uris=["f.xlsx#機能一覧!C3"], template_sections_expected=["ch3"],
)
report = ChapterScorer().score([artifact], source)
assert isinstance(report, EvalReport)
assert 0.0 <= report.total_score <= 1.0
# 全部确定性维度满分 → 总分接近 1.0(仅 llm 维度默认中性 0.5)
assert report.total_score >= 0.8
def test_empty_chapters_does_not_crash():
"""空章节输入:无引用/无约束 → 确定性维度中性满分,不应抛错。"""
source = _source()
report = ChapterScorer().score([], source)
assert report.total_score == 1.0
assert report.passed is True
# ---------- 黄金集加载 ----------
def test_golden_set_load(tmp_path):
yaml_text = """
cases:
- id: g1
input_ref: sample/要件定義.xlsx
expected_min_score: 0.7
note: 脱敏真实样本回归基线
"""
p = tmp_path / "golden_set.yaml"
p.write_text(yaml_text, encoding="utf-8")
gs = GoldenSet.load(p)
assert len(gs.cases) == 1
assert gs.cases[0].id == "g1"
assert gs.cases[0].expected_min_score == 0.7
def _dim(report: EvalReport, name: str) -> DimensionScore:
for d in report.dimensions:
if d.name == name:
return d
raise AssertionError(f"维度未找到: {name}")