Files

176 lines
6.2 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from genesis.data_models import SheetType
from genesis.parsers.excel_parser import ExcelParseResult, ExcelParser
from tests.excel_helpers import new_workbook, save_workbook
def test_parse_skips_empty_sheet(tmp_path):
# 空 sheet → skipped 不进入(L skips 分支)
wb = new_workbook({"空白": [[]]})
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
assert "空白" in result.skipped
assert result.tables == []
def test_parse_mixed_formatting_outside_table_segment(tmp_path):
# 碎片段(free_text 段)存在取消线格式 → seg_fmt_map 构建时该行不在表格段 [s,e] 内,被跳过(66-65 分支)
from copy import copy
from openpyxl import Workbook
wb = Workbook()
ws = wb.active
ws.title = "混合"
data = [
["機能ID", "機能名"], ["F101", "社員登録"],
[],
["・改述ポイント"],
]
for r, row in enumerate(data, start=1):
for c, v in enumerate(row, start=1):
if v:
ws.cell(row=r, column=c, value=v)
font = copy(ws.cell(row=4, column=1).font) # 碎片段行(物理行 4
font.strike = True
ws.cell(row=4, column=1).font = font
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
ms = result.mixed[0]
tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0]
# 表格段数据行不应带上碎片段的取消线格式
assert tbl.rows[0]["機能ID"].formatting is None
assert len(tbl.rows) == 1
def test_parse_table_sheets(tmp_path):
wb = new_workbook({
"機能一覧": [["機能ID", "機能名"], ["A001", "社員登録"]],
"バッチ一覧": [["バッチID", "処理名"], ["B1", "夜間集計"]],
})
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
assert isinstance(result, ExcelParseResult)
by_name = {t.name: t for t in result.tables}
assert by_name["機能一覧"].detected_type == SheetType.FUNCTION
assert len(by_name["機能一覧"].rows) == 1
assert by_name["バッチ一覧"].detected_type == SheetType.BATCH
def test_parse_free_text_sheet(tmp_path):
wb = new_workbook({"メモ": [["新入社員を登録"], [], [], [], []]})
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
assert len(result.tables) == 1
assert result.tables[0].extraction_method == "llm_from_free_text"
def test_parse_uses_detected_header_row(tmp_path):
# 标题行(单格)在首行,真实表头在第二行
wb = new_workbook({
"機能一覧": [["機能一覧"], ["機能ID", "機能名"], ["A001", "社員登録"]],
})
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
t = result.tables[0]
assert t.headers == ["機能ID", "機能名"]
assert len(t.rows) == 1
assert t.rows[0]["機能ID"].value == "A001"
def test_parse_attaches_strikethrough_formatting(tmp_path):
from copy import copy
from openpyxl import Workbook
wb = Workbook()
ws = wb.active
ws.title = "機能一覧"
ws["A1"] = "機能ID"
ws["B1"] = "機能名"
ws["A2"] = "A001"
ws["B2"] = "社員登録"
font = copy(ws["A2"].font)
font.strike = True
ws["A2"].font = font
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
cv = result.tables[0].rows[0]["機能ID"]
assert cv.formatting is not None
assert cv.formatting.strikethrough is True
from genesis.data_models import MixedParagraph, MixedSheet
from genesis.parsers.excel_parser import ExcelParseResult
def test_excel_parse_result_has_mixed_default():
r = ExcelParseResult(file_name="f.xlsx")
assert r.mixed == []
def test_mixed_paragraph_defaults():
p = MixedParagraph(kind="table")
assert p.table is None
assert p.text is None
assert p.source_range is None
def test_mixed_sheet_holds_paragraphs():
p1 = MixedParagraph(kind="table")
p2 = MixedParagraph(kind="free_text", text="备注")
ms = MixedSheet(name="混合", paragraphs=[p1, p2])
assert ms.name == "混合"
assert [p.kind for p in ms.paragraphs] == ["table", "free_text"]
def test_parse_mixed_sheet_segmented(tmp_path):
wb = new_workbook({
"混合": [
["機能ID", "機能名"],
["F101", "社員登録"],
["F102", "退職処理"],
[],
["・改修ポイント:F102 追加バリデーション"],
["■対象画面:SC001"],
],
})
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
assert result.mixed, "混合 sheet 应产产出 mixed 段落"
ms = result.mixed[0]
assert ms.name == "混合"
kinds = [p.kind for p in ms.paragraphs]
assert "table" in kinds and "free_text" in kinds
# 表格段无碎片污染:table 段应含 2 数据行,机能ID 首行为 F101
tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0]
assert tbl is not None and len(tbl.rows) == 2
assert tbl.rows[0]["機能ID"].value == "F101"
# 自由文本段捕获碎片
ft = [p for p in ms.paragraphs if p.kind == "free_text"][0]
assert "改修ポイント" in (ft.text or "")
def test_parse_mixed_sheet_formatting_in_mid_segment(tmp_path):
# 表格段不在物理行 0(自由文本段在前),数据行 F101 设取消线 —— 锁住 formatting_map 坐标错位
from copy import copy
from openpyxl import Workbook
wb = Workbook()
ws = wb.active
ws.title = "混合"
data = [
["■はじめに"], ["前提説明"], [],
["機能ID", "機能名"], ["F101", "社員登録"], ["F102", "退職処理"],
[], ["・改修ポイント"],
]
for r, row in enumerate(data, start=1):
for c, v in enumerate(row, start=1):
if v:
ws.cell(row=r, column=c, value=v)
font = copy(ws.cell(row=5, column=1).font) # F101 所在物理行(第 5 行)
font.strike = True
ws.cell(row=5, column=1).font = font
path = save_workbook(tmp_path, wb)
result = ExcelParser().parse(path)
ms = result.mixed[0]
tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0]
cv = tbl.rows[0]["機能ID"] # F101
assert cv.formatting is not None
assert cv.formatting.strikethrough is True