feat(writer): 章级管道重试兜底 + 表格 caption 语言检查 + 文档收尾

#1/#2 真实 LLM 随机方差兜底:orchestrator.generate 新增 chapter_attempts(默认 3),
单章 WriterGenerationError 不连坐整次 run,重试耗尽才抛错(不吞错)。新增
tests/test_orchestrator_retry.py(前 N 次失败后恢复 / 耗尽仍抛错)。

#3 表格 caption 语言检查:find_language_violations 现检 table.caption(生成正文需跟随
输出语言),rows/headers 仍照抄源不检;QA 维度经 ChapterArtifact.blocks=(type,text,caption)
同步生效。修复真实 zh 输出中表格说明引用日文源表名绕过强制的问题。

文档收尾:README 新增 --output-language/中文模板用法;design.md §6.2.1 输出语言控制、
§7.2 校验清单 10→11 项;计划验收勾选;.gitignore 加 .opencode/。
全量 pytest 431 passed / 99.15%
This commit is contained in:
lhl
2026-08-26 00:49:36 +08:00
parent 84d3b87da3
commit 1925ef7239
8 changed files with 178 additions and 25 deletions
+77
View File
@@ -0,0 +1,77 @@
"""#1/#2orchestrator 章级重试兜底测试。
问题:真实 LLM 输出有随机方差(如 ja 跑 function_list 单次输出疑似纯汉字被判违规),
WriterAgent 内部 max_retries=2 耗尽后抛 WriterGenerationError → 整次 run_trial 死亡。
修复:orchestrator.generate 增加章级管道重试(chapter_attempts),单章失败不连坐整次运行。
"""
from __future__ import annotations
from pathlib import Path
from types import SimpleNamespace
import pytest
from genesis.data_models import StructuredSource
from genesis.parsers.word_template_parser import WordTemplateParser
from genesis.writer.exceptions import WriterGenerationError
from genesis.writer.orchestrator import WriteOrchestrator
class _FlakyEngine:
"""前 N 次调用抛错(模拟随机硬失败),之后成功。"""
def __init__(self, fail_calls: int):
self.fail_calls = fail_calls
self.calls = 0
def chat_structured(self, *, session_id, prompt, variables, schema, retry_count=2):
self.calls += 1
if self.calls <= self.fail_calls:
raise RuntimeError(f"模拟第 {self.calls} 次调用失败")
return SimpleNamespace(
data={
"title": variables["title"],
"blocks": [{"type": "paragraph",
"text": "本機能はFakeLLMにより生成された十分な説明内容であり、書込規則を満たす。"}],
},
status="ok",
)
def _ss() -> StructuredSource:
template = Path("samples/phase5-slice/template.docx")
if not template.exists():
pytest.skip("样本模板缺失,跳过")
parsed = WordTemplateParser().parse(str(template))
return StructuredSource(template=parsed, tables=[], rule_docs=[],
image_analyses=[], existing_system=None, comments=[])
def test_generate_recovers_stochastic_single_chapter_failure(tmp_path):
"""前 3 次调用失败(跨章/章内重试),后续成功 → 整次生成不崩溃。"""
out = tmp_path / "o.docx"
engine = _FlakyEngine(fail_calls=3)
# chapter_attempts=3,内部 max_retries=2 → 单章最多约 3×2 次调用
contents = WriteOrchestrator().generate(
_ss(), str(out), samples_dir="nonexistent_dir_xyz",
engine=engine, template_path=str(Path("samples/phase5-slice/template.docx")),
)
assert contents, "应成功产出章节"
assert out.is_file()
def test_generate_hard_fails_after_chapter_attempts_exhausted(tmp_path):
"""始终失败 → 章级重试耗尽后仍抛 WriterGenerationError(不吞错)。"""
out = tmp_path / "o.docx"
class AlwaysFail(_FlakyEngine):
def chat_structured(self, *, session_id, prompt, variables, schema, retry_count=2):
self.calls += 1
raise RuntimeError("always fail")
with pytest.raises(WriterGenerationError):
WriteOrchestrator().generate(
_ss(), str(out), samples_dir="nonexistent_dir_xyz",
engine=AlwaysFail(0), template_path=str(Path("samples/phase5-slice/template.docx")),
chapter_attempts=1,
)
+43 -2
View File
@@ -79,15 +79,56 @@ def test_no_violation_ja_expected_japanese_paragraph():
def test_heading_and_table_blocks_excluded():
# heading 与 table(照抄原文)即使含中文也不算违规
# heading 与 table 的 rows/text(照抄原文)即使含中文也不算违规
blocks = _blocks(
("heading", "機能一覧表"),
("table", "機能ID 機能名"), # 表格不检
("table", "機能ID 機能名"), # 表格数据不检
("paragraph", "本機能は注文処理を行う。"),
)
assert not find_language_violations(blocks, "ja")
def _blocks_with_caption(*specs):
"""(type, text, caption) 构造内容块。"""
out = []
for i, (t, text, caption) in enumerate(specs):
out.append(ContentBlock(block_id=str(i), type=t, text=text, caption=caption))
return out
def test_table_caption_checked_zh_expected():
# 表格 caption 是生成正文(非源数据):zh 期望下含日文假名 → 违规
blocks = _blocks_with_caption(
("table", "", "TB001(訂単テーブル)与 TB002(即時行情テーブル)为既有表,本次变更涉及列定义调整。"),
)
assert find_language_violations(blocks, "zh")
def test_table_caption_checked_ja_expected():
# ja 期望下 caption 为纯中文(无假名 ≥12)→ 违规
blocks = _blocks_with_caption(
("table", "", "这是一段纯中文的表格说明内容,应当被判定为语言违规。"),
)
assert find_language_violations(blocks, "ja")
def test_table_caption_matching_language_not_flagged():
# 与期望语言一致的 caption 不违规
blocks = _blocks_with_caption(
("table", "", "表5-1 功能一览表(展示既有与新规功能清单)"),
)
assert not find_language_violations(blocks, "zh")
def test_table_rows_still_excluded_even_with_caption():
# 表格 rows 照抄源数据:即使含假名也不因 rows 判违规;仅 caption 参与检查
blocks = _blocks_with_caption(
("table", "", "表5-1 功能一览"),
)
blocks[0].rows = [["機能ID", "機能名"], ["F001", "注文処理"]]
assert not find_language_violations(blocks, "zh")
def test_short_chinese_not_flagged_under_threshold():
# 长度<12 的短术语不误杀
blocks = _blocks(("paragraph", "中文术语"))