Files
2026Technology-Competition/tests/test_writer_language.py
T
lhl 3f4d17cf69 fix(writer): auto 模式语言 fallback 改用规则文档,修复日文文档被误判期望 zh
真实 ja 试运行暴露:auto 模式 fallback 误用 impact/data 源数据(含中文元素名
止损风控等 + 汉字标签无假名)→ 把日文文档误判为期望 zh → LLM 正常日文输出被
判违规 → 硬失败。修复:fallback 改用 write_rules/design_rules(RAG 自日文作成
说明书/记入规则检索,含假名可靠判 ja);加回归测试
test_generate_chapter_auto_with_ja_rules_but_chinese_source_data_enforces_ja。
修复后 ja 真实试运行成功(7章);zh 真实试运行成功(7章);425 passed / 99.15%
2026-08-26 00:36:54 +08:00

200 lines
8.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""步骤 A:确定性语言检测 + 输出语言强制测试。
- detect_script:含假名→jaCJK 零假名→zh,否则 None
- resolve_expected_language:显式 > 标题假名 > 规则文档主导脚本 > ""(不可验证)
- find_language_violations:仅检正文块(heading/table 不检),长度≥12 防误杀
- WriterAgent.generate_chapter:期望语言非空时违规即重试,耗尽→硬失败
"""
from __future__ import annotations
from genesis.writer.language import (
detect_script,
find_language_violations,
resolve_expected_language,
)
from genesis.writer.models import ChapterContent, ContentBlock, GenerationContext, ChapterSpec
from genesis.writer.writer_agent import WriterAgent
from genesis.writer.writer_state import WriterState
from genesis.writer.exceptions import WriterGenerationError
# ---------- detect_script ----------
def test_detect_script_kana_is_ja():
assert detect_script("機能一覧の説明を記述します。") == "ja"
def test_detect_script_cjk_no_kana_is_zh():
assert detect_script("功能一览的说明内容。") == "zh"
def test_detect_script_ascii_is_none():
assert detect_script("Hello world 123") is None
def test_detect_script_mixed_cjk_kana_is_ja():
# 含假名即判 ja,即使混有汉字(中文不可能含日文假名)
assert detect_script("DB設計の概要を説明する。") == "ja"
# ---------- resolve_expected_language ----------
def test_resolve_explicit_overrides():
assert resolve_expected_language("zh", "DB 設計", ["日本語の影響"]) == "zh"
assert resolve_expected_language("ja", "DB 设計", ["中文影响"]) == "ja"
def test_resolve_falls_back_to_title_kana():
# 标题含假名→ja(注意:纯汉字日文标题如「DB設計」无假名,会落入 fallback)
assert resolve_expected_language("auto", "機能一覧の説明", []) == "ja"
def test_resolve_falls_back_to_fallback_text():
# 标题纯汉字无假名(中文式)→ 看 fallback(日文影响书)→ ja
assert resolve_expected_language("auto", "DB 設計", ["影響調査の結果(日本語)"]) == "ja"
def test_resolve_unverifiable_when_no_hint():
assert resolve_expected_language("auto", "DB 設計", ["123 abc"]) == ""
# ---------- find_language_violations ----------
def _blocks(*specs):
out = []
for i, (t, text) in enumerate(specs):
out.append(ContentBlock(block_id=str(i), type=t, text=text))
return out
def test_violation_ja_expected_but_pure_chinese_paragraph():
blocks = _blocks(("paragraph", "这是一段纯中文的章节正文内容。"))
viol = find_language_violations(blocks, "ja")
assert viol # 期望日文却为中文 → 违规
def test_no_violation_ja_expected_japanese_paragraph():
blocks = _blocks(("paragraph", "本機能は注文処理を行う画面であり、詳細は以下の通り。"))
assert not find_language_violations(blocks, "ja")
def test_heading_and_table_blocks_excluded():
# heading 与 table(照抄原文)即使含中文也不算违规
blocks = _blocks(
("heading", "機能一覧表"),
("table", "機能ID 機能名"), # 表格不检
("paragraph", "本機能は注文処理を行う。"),
)
assert not find_language_violations(blocks, "ja")
def test_short_chinese_not_flagged_under_threshold():
# 长度<12 的短术语不误杀
blocks = _blocks(("paragraph", "中文术语"))
assert not find_language_violations(blocks, "ja")
def test_violation_zh_expected_but_kana_present():
blocks = _blocks(("paragraph", "本機能は注文処理を行う画面である。"))
assert find_language_violations(blocks, "zh")
# ---------- WriterAgent 强制 ----------
class MixedLanguageEngine:
"""返回中英混杂(期望日文时违规)的章节内容。"""
def __init__(self, texts):
self._texts = list(texts)
self._i = 0
def chat_structured(self, *, session_id, prompt, variables, schema, retry_count=2):
from types import SimpleNamespace
text = self._texts[min(self._i, len(self._texts) - 1)]
self._i += 1
return SimpleNamespace(
data={"title": variables["title"], "blocks": [{"type": "paragraph", "text": text}]},
status="ok",
)
class FakePromptRegistry:
@staticmethod
def get_or_create(name, template):
from types import SimpleNamespace
return SimpleNamespace(name=name, version="1", template=template)
def _ctx_with_lang(cid, title, lang):
return GenerationContext(
chapter_id=cid, title=title,
template_marker=ChapterSpec(chapter_id=cid, title=title),
structured_source=None, write_rules=[], design_rules=[],
template_styles=set(), output_language=lang,
)
def test_generate_chapter_retries_on_language_violation_then_passes():
# 第一次返回中文(ja 期望→违规),第二次返回日文(通过)
engine = MixedLanguageEngine([
"这是一段纯中文的章节正文内容,应当被判定为语言违规。",
"本機能は注文処理を行う画面であり、詳細は以下の通り記述する。",
])
agent = WriterAgent(session_id="s", engine=engine,
prompt_registry=FakePromptRegistry(),
state=WriterState(["db_design"]), max_retries=3)
content = agent.generate_chapter(_ctx_with_lang("db_design", "DB 設計", "ja"))
assert content is not None
assert "本機能" in content.blocks[0].text
def test_generate_chapter_hard_fails_when_violation_persists():
# 始终返回中文(ja 期望),重试耗尽→硬失败
engine = MixedLanguageEngine([
"这是一段纯中文的章节正文内容,应当被判定为语言违规。",
"还是一段纯中文的章节正文内容,依旧违规。",
])
agent = WriterAgent(session_id="s", engine=engine,
prompt_registry=FakePromptRegistry(),
state=WriterState(["db_design"]), max_retries=2)
try:
agent.generate_chapter(_ctx_with_lang("db_design", "DB 設計", "ja"))
assert False, "应抛出 WriterGenerationError"
except WriterGenerationError as e:
assert "语言" in str(e) or "language" in str(e).lower()
def test_generate_chapter_auto_unverifiable_not_enforced():
# auto 且无法推导期望语言(标题纯汉字无假名、无日文 fallback)→ 不强制,直接通过
engine = MixedLanguageEngine(["这是中文正文但 auto 不可验证所以放行。"])
agent = WriterAgent(session_id="s", engine=engine,
prompt_registry=FakePromptRegistry(),
state=WriterState(["db_design"]), max_retries=1)
content = agent.generate_chapter(_ctx_with_lang("db_design", "DB 設計", "auto"))
assert content is not None
class JaRuleEngine(MixedLanguageEngine):
"""返回日文正文(含假名)。"""
def test_generate_chapter_auto_with_ja_rules_but_chinese_source_data_enforces_ja():
"""回归:真实试运行暴露——auto 模式 fallback 只用规则文档(日文),
不得用 impact/data 源数据(含中文元素名)把期望误判为 zh。
构造:标题纯汉字「機能一覧」(无假名)+ 规则文档日文(含假名)+ 源数据中文,
期望应推导为 ja;日文正文应通过、中文正文应违规。
"""
ja_rules_ctx = GenerationContext(
chapter_id="function_list", title="2. 機能一覧",
template_marker=ChapterSpec(chapter_id="function_list", title="2. 機能一覧"),
structured_source="中文源数据:止损风控、下单流程",
write_rules=["書込規則に従って記述する。"], # 日文含假名
design_rules=["設計方針は本書の通りである。"], # 日文含假名
template_styles=set(),
output_language="auto",
)
engine = MixedLanguageEngine(["本機能は注文処理を行う画面であり、詳細は以下の通り。"])
agent = WriterAgent(session_id="s", engine=engine,
prompt_registry=FakePromptRegistry(),
state=WriterState(["function_list"]), max_retries=1)
content = agent.generate_chapter(ja_rules_ctx)
assert content is not None and "本機能" in content.blocks[0].text