feat(eval): populate EvalReport.per_chapter + failed_chapters in score()
This commit is contained in:
@@ -36,12 +36,19 @@ class ChapterArtifact:
|
|||||||
template_sections_expected: list[str]
|
template_sections_expected: list[str]
|
||||||
|
|
||||||
|
|
||||||
|
# 逐章评估通过阈值(基于逐章总分)
|
||||||
|
PASS_THRESHOLD: float = 0.6
|
||||||
|
# 内容充分性维度:文本长度达到该值即视为充分
|
||||||
|
ADEQUACY_MIN_LEN: int = 15
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class EvalReport:
|
class EvalReport:
|
||||||
dimensions: list[DimensionScore]
|
dimensions: list[DimensionScore]
|
||||||
total_score: float
|
total_score: float
|
||||||
passed: bool
|
passed: bool
|
||||||
failed_chapters: list[str] = field(default_factory=list)
|
failed_chapters: list[str] = field(default_factory=list)
|
||||||
|
per_chapter: list["EvalReport"] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
# 维度默认通过阈值
|
# 维度默认通过阈值
|
||||||
@@ -75,7 +82,20 @@ class ChapterScorer:
|
|||||||
|
|
||||||
total = sum(d.score for d in dimensions) / len(dimensions) if dimensions else 0.0
|
total = sum(d.score for d in dimensions) / len(dimensions) if dimensions else 0.0
|
||||||
passed = all(d.passed for d in dimensions)
|
passed = all(d.passed for d in dimensions)
|
||||||
return EvalReport(dimensions=dimensions, total_score=round(total, 4), passed=passed)
|
|
||||||
|
# 逐章评估:对每一章节独立评分,收集 per_chapter,并标记未达标章节
|
||||||
|
per_chapter = [self._score_chapter(ch, source) for ch in chapters]
|
||||||
|
failed_chapters = [
|
||||||
|
ch.chapter_id for ch, r in zip(chapters, per_chapter) if r.failed_chapters
|
||||||
|
]
|
||||||
|
|
||||||
|
return EvalReport(
|
||||||
|
dimensions=dimensions,
|
||||||
|
total_score=round(total, 4),
|
||||||
|
passed=passed,
|
||||||
|
failed_chapters=failed_chapters,
|
||||||
|
per_chapter=per_chapter,
|
||||||
|
)
|
||||||
|
|
||||||
# ---------- 确定性维度 ----------
|
# ---------- 确定性维度 ----------
|
||||||
|
|
||||||
@@ -123,3 +143,35 @@ class ChapterScorer:
|
|||||||
avg = sum(s.score for s in scores) / len(scores) if scores else 0.5
|
avg = sum(s.score for s in scores) / len(scores) if scores else 0.5
|
||||||
detail = " | ".join(s.detail for s in scores) if scores else "no chapters"
|
detail = " | ".join(s.detail for s in scores) if scores else "no chapters"
|
||||||
return DimensionScore(name, round(avg, 4), all(s.passed for s in scores), detail)
|
return DimensionScore(name, round(avg, 4), all(s.passed for s in scores), detail)
|
||||||
|
|
||||||
|
# ---------- 逐章评估 ----------
|
||||||
|
|
||||||
|
def _score_chapter(self, chapter: ChapterArtifact, source: StructuredSource) -> EvalReport:
|
||||||
|
"""对单个章节独立评分,返回该章节的 EvalReport。
|
||||||
|
|
||||||
|
复用确定性维度逻辑(可追溯性/占位符残留)并计算内容充分性,
|
||||||
|
总分低于 PASS_THRESHOLD 即判该章节失败。
|
||||||
|
"""
|
||||||
|
dims = [
|
||||||
|
self._traceability([chapter], source),
|
||||||
|
self._placeholder_residue([chapter]),
|
||||||
|
self._adequacy(chapter),
|
||||||
|
]
|
||||||
|
total = sum(d.score for d in dims) / len(dims) if dims else 0.0
|
||||||
|
failed = (not all(d.passed for d in dims)) or total < PASS_THRESHOLD
|
||||||
|
return EvalReport(
|
||||||
|
dimensions=dims,
|
||||||
|
total_score=round(total, 4),
|
||||||
|
passed=not failed,
|
||||||
|
failed_chapters=[chapter.chapter_id] if failed else [],
|
||||||
|
per_chapter=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
def _adequacy(self, chapter: ChapterArtifact) -> DimensionScore:
|
||||||
|
"""内容充分性:基于正文字本长度启发式判定章节是否充分。"""
|
||||||
|
text = chapter.text or ""
|
||||||
|
score = 1.0 if len(text) >= ADEQUACY_MIN_LEN else 0.3
|
||||||
|
passed = score >= PASS_THRESHOLD
|
||||||
|
return DimensionScore(
|
||||||
|
"adequacy", score, passed, f"内容长度 {len(text)}(充分阈值 {ADEQUACY_MIN_LEN})"
|
||||||
|
)
|
||||||
|
|||||||
@@ -0,0 +1,82 @@
|
|||||||
|
"""Phase5-T11(缩减版):验证 EvalReport.per_chapter 与 failed_chapters 被正确填充。
|
||||||
|
|
||||||
|
真实 score() 签名为 score(chapters: list[ChapterArtifact], source: StructuredSource),
|
||||||
|
故测试按真实签名构造输入(使用 ChapterArtifact)。
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from genesis.data_models import (
|
||||||
|
CellValue,
|
||||||
|
ExcelTable,
|
||||||
|
ParsedTemplate,
|
||||||
|
Provenance,
|
||||||
|
SheetType,
|
||||||
|
StructuredSource,
|
||||||
|
)
|
||||||
|
from genesis.eval.scorer import ChapterArtifact, ChapterScorer, EvalReport
|
||||||
|
|
||||||
|
|
||||||
|
def _source() -> StructuredSource:
|
||||||
|
cell = CellValue(
|
||||||
|
value="登録",
|
||||||
|
provenance=Provenance(file_name="f.xlsx", sheet_name="機能一覧", row=3, column="C", column_header="x"),
|
||||||
|
)
|
||||||
|
table = ExcelTable(
|
||||||
|
name="機能一覧", detected_type=SheetType.FUNCTION,
|
||||||
|
extraction_method="structured", headers=["v"], rows=[{"v": cell}],
|
||||||
|
)
|
||||||
|
return StructuredSource(
|
||||||
|
tables=[table],
|
||||||
|
template=ParsedTemplate(file_name="t.docx", sections=[], placeholders={}, styles={}),
|
||||||
|
rule_docs=[], image_analyses=[], existing_system=None, comments=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _artifact(cid: str, text: str) -> ChapterArtifact:
|
||||||
|
return ChapterArtifact(
|
||||||
|
chapter_id=cid, text=text, source_uris=[], template_sections_expected=[cid],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_score_populates_per_chapter_and_failed():
|
||||||
|
source = _source()
|
||||||
|
good = _artifact("a", "充分且规范的说明内容,满足写入规则要求。")
|
||||||
|
bad = _artifact("b", "x") # 过短/不充分 -> 应判失败
|
||||||
|
report: EvalReport = ChapterScorer().score([good, bad], source)
|
||||||
|
assert isinstance(report.per_chapter, list)
|
||||||
|
assert len(report.per_chapter) == 2
|
||||||
|
assert all(isinstance(r, EvalReport) for r in report.per_chapter)
|
||||||
|
assert isinstance(report.failed_chapters, list)
|
||||||
|
assert "b" in report.failed_chapters # 不充分章节被标记
|
||||||
|
|
||||||
|
|
||||||
|
def test_per_chapter_reports_have_dimensions():
|
||||||
|
source = _source()
|
||||||
|
artifacts = [_artifact("a", "充分且规范的说明内容,满足写入规则要求。"), _artifact("b", "x")]
|
||||||
|
report = ChapterScorer().score(artifacts, source)
|
||||||
|
for r in report.per_chapter:
|
||||||
|
assert len(r.dimensions) > 0
|
||||||
|
assert 0.0 <= r.total_score <= 1.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_per_chapter_failed_chapters_aligns():
|
||||||
|
"""每个逐章报告自身的 failed_chapters 应与顶层 failed_chapters 一致。"""
|
||||||
|
source = _source()
|
||||||
|
artifacts = [_artifact("a", "充分且规范的说明内容,满足写入规则要求。"), _artifact("b", "x")]
|
||||||
|
report = ChapterScorer().score(artifacts, source)
|
||||||
|
failed_ids = {ch for ch in report.failed_chapters}
|
||||||
|
for ch, r in zip(artifacts, report.per_chapter):
|
||||||
|
if ch.chapter_id in failed_ids:
|
||||||
|
assert ch.chapter_id in r.failed_chapters
|
||||||
|
else:
|
||||||
|
assert r.failed_chapters == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_chapters_backward_compatible():
|
||||||
|
"""空章节输入:向后兼容,per_chapter/failed_chapters 为空列表。"""
|
||||||
|
source = _source()
|
||||||
|
report = ChapterScorer().score([], source)
|
||||||
|
assert report.per_chapter == []
|
||||||
|
assert report.failed_chapters == []
|
||||||
|
assert report.total_score == 1.0
|
||||||
Reference in New Issue
Block a user