60 lines
2.0 KiB
Python
60 lines
2.0 KiB
Python
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from genesis.parsers.excel_parser import ExcelParser
|
|
|
|
SAMPLES = Path(__file__).resolve().parents[1] / "samples"
|
|
|
|
|
|
def _x(name: str) -> Path:
|
|
return SAMPLES / name
|
|
|
|
|
|
def test_new_dev_sample_has_tables():
|
|
p = _x("要件定義_新規開発.xlsx")
|
|
if not p.exists():
|
|
pytest.skip("样本缺失")
|
|
result = ExcelParser().parse(p)
|
|
by_name = {t.name: t for t in result.tables}
|
|
assert "機能一覧" in by_name
|
|
assert by_name["機能一覧"].rows
|
|
assert any(t.name == "DB定義" for t in result.tables)
|
|
|
|
|
|
def test_additional_modification_has_tables():
|
|
p = _x("要件定義_追加改修.xlsx")
|
|
if not p.exists():
|
|
pytest.skip("样本缺失")
|
|
result = ExcelParser().parse(p)
|
|
assert result.tables
|
|
assert any(t.rows for t in result.tables)
|
|
|
|
|
|
def test_free_text_sample_detected():
|
|
p = _x("要件定義_自由記述.xlsx")
|
|
if not p.exists():
|
|
pytest.skip("样本缺失")
|
|
result = ExcelParser().parse(p)
|
|
assert any(t.extraction_method == "llm_from_free_text" for t in result.tables)
|
|
|
|
|
|
def test_mixed_sample_segments_detected():
|
|
p = _x("要件定義_混合型.xlsx")
|
|
if not p.exists():
|
|
pytest.skip("样本缺失")
|
|
result = ExcelParser().parse(p)
|
|
by_name = {t.name: t for t in result.tables}
|
|
assert "機能一覧" in by_name
|
|
assert result.mixed, "混合样本应产出段落"
|
|
mixed = result.mixed[0]
|
|
assert len(mixed.paragraphs) == 2 # 表格段 + 碎片段(・/■ 连续)
|
|
assert [p.kind for p in mixed.paragraphs] == ["table", "free_text"]
|
|
# 表格段无碎片污染
|
|
table = [p.table for p in mixed.paragraphs if p.kind == "table"][0]
|
|
assert table.rows[0]["機能ID"].value == "F101"
|
|
assert len(table.rows) == 3
|
|
# 碎片段含 ・ 与 ■ 两行文本
|
|
ft = [p for p in mixed.paragraphs if p.kind == "free_text"][0]
|
|
assert "改修ポイント" in (ft.text or "")
|
|
assert "対象期間" in (ft.text or "") |