feat(writer): 章级管道重试兜底 + 表格 caption 语言检查 + 文档收尾
#1/#2 真实 LLM 随机方差兜底:orchestrator.generate 新增 chapter_attempts(默认 3), 单章 WriterGenerationError 不连坐整次 run,重试耗尽才抛错(不吞错)。新增 tests/test_orchestrator_retry.py(前 N 次失败后恢复 / 耗尽仍抛错)。 #3 表格 caption 语言检查:find_language_violations 现检 table.caption(生成正文需跟随 输出语言),rows/headers 仍照抄源不检;QA 维度经 ChapterArtifact.blocks=(type,text,caption) 同步生效。修复真实 zh 输出中表格说明引用日文源表名绕过强制的问题。 文档收尾:README 新增 --output-language/中文模板用法;design.md §6.2.1 输出语言控制、 §7.2 校验清单 10→11 项;计划验收勾选;.gitignore 加 .opencode/。 全量 pytest 431 passed / 99.15%
This commit is contained in:
@@ -35,8 +35,8 @@ class ChapterArtifact:
|
||||
source_uris: list[str]
|
||||
template_sections_expected: list[str]
|
||||
expected_language: str = "" # 期望输出语言("zh"/"ja";空=不可验证,维度记满分)
|
||||
# 块级 (type, text) 列表:供语言一致性维度排除 heading/table(照抄源/跟随模板)
|
||||
blocks: list[tuple[str, str]] = field(default_factory=list)
|
||||
# 块级 (type, text, caption) 列表:供语言一致性维度排除 heading / table.rows(照抄源/跟随模板)
|
||||
blocks: list[tuple[str, str, str]] = field(default_factory=list)
|
||||
|
||||
|
||||
# 逐章评估通过阈值(基于逐章总分)
|
||||
@@ -171,9 +171,9 @@ class ChapterScorer:
|
||||
continue
|
||||
from genesis.writer.models import ContentBlock
|
||||
if ch.blocks:
|
||||
# 优先用块级信息(可排除 heading/table)
|
||||
blocks = [ContentBlock(block_id=str(i), type=t, text=tx)
|
||||
for i, (t, tx) in enumerate(ch.blocks)]
|
||||
# 优先用块级信息(type, text, caption;可排除 heading/table.rows)
|
||||
blocks = [ContentBlock(block_id=str(i), type=t, text=tx, caption=cap or None)
|
||||
for i, (t, tx, cap) in enumerate(ch.blocks)]
|
||||
else:
|
||||
# 回退:整段正文作为单个 paragraph 块
|
||||
blocks = [ContentBlock(block_id="0", type="paragraph", text=ch.text or "")]
|
||||
|
||||
@@ -26,7 +26,7 @@ class QAValidator:
|
||||
source_uris=source_uris,
|
||||
template_sections_expected=[],
|
||||
expected_language=expected_language,
|
||||
blocks=[(b.type, b.text or "") for b in content.blocks],
|
||||
blocks=[(b.type, b.text or "", b.caption or "") for b in content.blocks],
|
||||
)
|
||||
|
||||
def validate_chapter(
|
||||
|
||||
@@ -5,8 +5,9 @@
|
||||
故 resolve_expected_language 采用两级推导:显式 > 标题假名 > 规则文档主导脚本。
|
||||
- 检测仅基于「是否含日文假名」:CJK 汉字零假名视为中文(日文不可能不含假名地
|
||||
使用汉字),反之中日混排含假名判日文。这是确定可机器验证的唯一稳健信号。
|
||||
- find_language_violations 仅检正文类块(paragraph/note/list);heading 跟随模板、
|
||||
table 照抄源 Excel 原文,二者均不检(design.md §7.2 内容准确性/可追溯性)。
|
||||
- find_language_violations 检正文类块(paragraph/note/list 的 text)与表格 caption
|
||||
(caption 为生成正文需跟随输出语言);heading 跟随模板、table 的 rows/headers 照抄源
|
||||
Excel 原文,不检(design.md §7.2 内容准确性/可追溯性)。
|
||||
- 短文本(<12 字)不误杀(如专有术语),阈值见 MIN_VIOLATION_LEN。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
@@ -81,24 +82,31 @@ def resolve_expected_language(
|
||||
def find_language_violations(blocks: list[ContentBlock], expected_language: str) -> list[str]:
|
||||
"""返回违规正文块文本片段(期望语言非空时才有意义)。
|
||||
|
||||
违规判定:
|
||||
- 期望 "ja":正文块含 CJK 汉字且零假名(即纯中文)且长度 ≥ 阈值
|
||||
- 期望 "zh":正文块含日文假名
|
||||
heading/table 块始终跳过(跟随模板 / 照抄源)。
|
||||
违规判定(对正文类块与表格 caption 一致):
|
||||
- 期望 "ja":含 CJK 汉字且零假名(即纯中文)且长度 ≥ 阈值
|
||||
- 期望 "zh":含日文假名
|
||||
受检范围:
|
||||
- paragraph/note/list 的 text(正文)
|
||||
- table 的 caption(生成正文,需跟随输出语言)
|
||||
不检:heading(跟随模板)、table 的 rows/headers(照抄源 Excel 原文,design §7.2)。
|
||||
"""
|
||||
if expected_language not in ("zh", "ja"):
|
||||
return []
|
||||
violations: list[str] = []
|
||||
for b in blocks:
|
||||
if b.type not in _CHECKED_BLOCK_TYPES:
|
||||
if b.type == "table":
|
||||
texts = [b.caption or ""] # 仅 caption;rows/headers 照抄源不检
|
||||
elif b.type in _CHECKED_BLOCK_TYPES:
|
||||
texts = [b.text or ""]
|
||||
else:
|
||||
continue
|
||||
text = b.text or ""
|
||||
if len(text) < MIN_VIOLATION_LEN:
|
||||
continue
|
||||
if expected_language == "ja":
|
||||
if has_cjk(text) and not has_kana(text):
|
||||
violations.append(text)
|
||||
else: # zh
|
||||
if has_kana(text):
|
||||
violations.append(text)
|
||||
for text in texts:
|
||||
if len(text) < MIN_VIOLATION_LEN:
|
||||
continue
|
||||
if expected_language == "ja":
|
||||
if has_cjk(text) and not has_kana(text):
|
||||
violations.append(text)
|
||||
else: # zh
|
||||
if has_kana(text):
|
||||
violations.append(text)
|
||||
return violations
|
||||
|
||||
@@ -16,6 +16,7 @@ from genesis.inference.factory import build_inference_engine
|
||||
from genesis.inference.prompt_registry import PromptRegistry
|
||||
from genesis.writer.context_builder import build_contexts
|
||||
from genesis.writer.docx_injector import Block, DocxInjector
|
||||
from genesis.writer.exceptions import WriterGenerationError
|
||||
from genesis.writer.models import ChapterContent
|
||||
from genesis.writer.renderer import render_chapter_blocks
|
||||
from genesis.writer.writer_agent import WriterAgent
|
||||
@@ -57,6 +58,7 @@ class WriteOrchestrator:
|
||||
impact_report=None,
|
||||
meta: dict | None = None,
|
||||
output_language: str = "auto",
|
||||
chapter_attempts: int = 3,
|
||||
) -> list[ChapterContent]:
|
||||
engine = engine or build_inference_engine()
|
||||
prompt_registry = prompt_registry or PromptRegistry()
|
||||
@@ -75,7 +77,21 @@ class WriteOrchestrator:
|
||||
contents: list[ChapterContent] = []
|
||||
sections: dict[str, list[Block]] = {}
|
||||
for ctx in ctxs:
|
||||
content = agent.generate_chapter(ctx)
|
||||
# 章级管道重试(#1/#2):真实 LLM 输出有随机方差,单章硬失败不连坐整次运行。
|
||||
# 每轮管道尝试内部已含 WriterAgent.max_retries 次 LLM 调用;chapter_attempts 为
|
||||
# 管道层兜底轮数(默认 3)。耗尽后仍抛错(不吞错)。
|
||||
content: ChapterContent | None = None
|
||||
last_err: Exception | None = None
|
||||
for attempt in range(max(1, chapter_attempts)):
|
||||
try:
|
||||
content = agent.generate_chapter(ctx)
|
||||
break
|
||||
except WriterGenerationError as e:
|
||||
last_err = e
|
||||
_LOGGER.warning("章节 %s 生成失败(第 %d/%d 轮管道重试): %s",
|
||||
ctx.chapter_id, attempt + 1, chapter_attempts, e)
|
||||
if content is None:
|
||||
raise WriterGenerationError(f"章节 {ctx.chapter_id} 管道重试耗尽: {last_err}")
|
||||
contents.append(content)
|
||||
blocks = render_chapter_blocks(content)
|
||||
sec_id = _section_id_of(ctx.template_marker.section_placeholder)
|
||||
|
||||
Reference in New Issue
Block a user