176 lines
6.2 KiB
Python
176 lines
6.2 KiB
Python
from genesis.data_models import SheetType
|
||
from genesis.parsers.excel_parser import ExcelParseResult, ExcelParser
|
||
|
||
from tests.excel_helpers import new_workbook, save_workbook
|
||
|
||
|
||
def test_parse_skips_empty_sheet(tmp_path):
|
||
# 空 sheet → skipped 不进入(L skips 分支)
|
||
wb = new_workbook({"空白": [[]]})
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
assert "空白" in result.skipped
|
||
assert result.tables == []
|
||
|
||
|
||
def test_parse_mixed_formatting_outside_table_segment(tmp_path):
|
||
# 碎片段(free_text 段)存在取消线格式 → seg_fmt_map 构建时该行不在表格段 [s,e] 内,被跳过(66-65 分支)
|
||
from copy import copy
|
||
from openpyxl import Workbook
|
||
wb = Workbook()
|
||
ws = wb.active
|
||
ws.title = "混合"
|
||
data = [
|
||
["機能ID", "機能名"], ["F101", "社員登録"],
|
||
[],
|
||
["・改述ポイント"],
|
||
]
|
||
for r, row in enumerate(data, start=1):
|
||
for c, v in enumerate(row, start=1):
|
||
if v:
|
||
ws.cell(row=r, column=c, value=v)
|
||
font = copy(ws.cell(row=4, column=1).font) # 碎片段行(物理行 4)
|
||
font.strike = True
|
||
ws.cell(row=4, column=1).font = font
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
ms = result.mixed[0]
|
||
tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0]
|
||
# 表格段数据行不应带上碎片段的取消线格式
|
||
assert tbl.rows[0]["機能ID"].formatting is None
|
||
assert len(tbl.rows) == 1
|
||
|
||
|
||
def test_parse_table_sheets(tmp_path):
|
||
wb = new_workbook({
|
||
"機能一覧": [["機能ID", "機能名"], ["A001", "社員登録"]],
|
||
"バッチ一覧": [["バッチID", "処理名"], ["B1", "夜間集計"]],
|
||
})
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
assert isinstance(result, ExcelParseResult)
|
||
by_name = {t.name: t for t in result.tables}
|
||
assert by_name["機能一覧"].detected_type == SheetType.FUNCTION
|
||
assert len(by_name["機能一覧"].rows) == 1
|
||
assert by_name["バッチ一覧"].detected_type == SheetType.BATCH
|
||
|
||
|
||
def test_parse_free_text_sheet(tmp_path):
|
||
wb = new_workbook({"メモ": [["新入社員を登録"], [], [], [], []]})
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
assert len(result.tables) == 1
|
||
assert result.tables[0].extraction_method == "llm_from_free_text"
|
||
|
||
|
||
def test_parse_uses_detected_header_row(tmp_path):
|
||
# 标题行(单格)在首行,真实表头在第二行
|
||
wb = new_workbook({
|
||
"機能一覧": [["機能一覧"], ["機能ID", "機能名"], ["A001", "社員登録"]],
|
||
})
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
t = result.tables[0]
|
||
assert t.headers == ["機能ID", "機能名"]
|
||
assert len(t.rows) == 1
|
||
assert t.rows[0]["機能ID"].value == "A001"
|
||
|
||
|
||
def test_parse_attaches_strikethrough_formatting(tmp_path):
|
||
from copy import copy
|
||
from openpyxl import Workbook
|
||
wb = Workbook()
|
||
ws = wb.active
|
||
ws.title = "機能一覧"
|
||
ws["A1"] = "機能ID"
|
||
ws["B1"] = "機能名"
|
||
ws["A2"] = "A001"
|
||
ws["B2"] = "社員登録"
|
||
font = copy(ws["A2"].font)
|
||
font.strike = True
|
||
ws["A2"].font = font
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
cv = result.tables[0].rows[0]["機能ID"]
|
||
assert cv.formatting is not None
|
||
assert cv.formatting.strikethrough is True
|
||
|
||
|
||
from genesis.data_models import MixedParagraph, MixedSheet
|
||
from genesis.parsers.excel_parser import ExcelParseResult
|
||
|
||
|
||
def test_excel_parse_result_has_mixed_default():
|
||
r = ExcelParseResult(file_name="f.xlsx")
|
||
assert r.mixed == []
|
||
|
||
|
||
def test_mixed_paragraph_defaults():
|
||
p = MixedParagraph(kind="table")
|
||
assert p.table is None
|
||
assert p.text is None
|
||
assert p.source_range is None
|
||
|
||
|
||
def test_mixed_sheet_holds_paragraphs():
|
||
p1 = MixedParagraph(kind="table")
|
||
p2 = MixedParagraph(kind="free_text", text="备注")
|
||
ms = MixedSheet(name="混合", paragraphs=[p1, p2])
|
||
assert ms.name == "混合"
|
||
assert [p.kind for p in ms.paragraphs] == ["table", "free_text"]
|
||
|
||
|
||
def test_parse_mixed_sheet_segmented(tmp_path):
|
||
wb = new_workbook({
|
||
"混合": [
|
||
["機能ID", "機能名"],
|
||
["F101", "社員登録"],
|
||
["F102", "退職処理"],
|
||
[],
|
||
["・改修ポイント:F102 追加バリデーション"],
|
||
["■対象画面:SC001"],
|
||
],
|
||
})
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
assert result.mixed, "混合 sheet 应产产出 mixed 段落"
|
||
ms = result.mixed[0]
|
||
assert ms.name == "混合"
|
||
kinds = [p.kind for p in ms.paragraphs]
|
||
assert "table" in kinds and "free_text" in kinds
|
||
# 表格段无碎片污染:table 段应含 2 数据行,机能ID 首行为 F101
|
||
tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0]
|
||
assert tbl is not None and len(tbl.rows) == 2
|
||
assert tbl.rows[0]["機能ID"].value == "F101"
|
||
# 自由文本段捕获碎片
|
||
ft = [p for p in ms.paragraphs if p.kind == "free_text"][0]
|
||
assert "改修ポイント" in (ft.text or "")
|
||
|
||
|
||
def test_parse_mixed_sheet_formatting_in_mid_segment(tmp_path):
|
||
# 表格段不在物理行 0(自由文本段在前),数据行 F101 设取消线 —— 锁住 formatting_map 坐标错位
|
||
from copy import copy
|
||
from openpyxl import Workbook
|
||
wb = Workbook()
|
||
ws = wb.active
|
||
ws.title = "混合"
|
||
data = [
|
||
["■はじめに"], ["前提説明"], [],
|
||
["機能ID", "機能名"], ["F101", "社員登録"], ["F102", "退職処理"],
|
||
[], ["・改修ポイント"],
|
||
]
|
||
for r, row in enumerate(data, start=1):
|
||
for c, v in enumerate(row, start=1):
|
||
if v:
|
||
ws.cell(row=r, column=c, value=v)
|
||
font = copy(ws.cell(row=5, column=1).font) # F101 所在物理行(第 5 行)
|
||
font.strike = True
|
||
ws.cell(row=5, column=1).font = font
|
||
path = save_workbook(tmp_path, wb)
|
||
result = ExcelParser().parse(path)
|
||
ms = result.mixed[0]
|
||
tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0]
|
||
cv = tbl.rows[0]["機能ID"] # F101
|
||
assert cv.formatting is not None
|
||
assert cv.formatting.strikethrough is True
|