from genesis.data_models import SheetType from genesis.parsers.excel_parser import ExcelParseResult, ExcelParser from tests.excel_helpers import new_workbook, save_workbook def test_parse_skips_empty_sheet(tmp_path): # 空 sheet → skipped 不进入(L skips 分支) wb = new_workbook({"空白": [[]]}) path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) assert "空白" in result.skipped assert result.tables == [] def test_parse_mixed_formatting_outside_table_segment(tmp_path): # 碎片段(free_text 段)存在取消线格式 → seg_fmt_map 构建时该行不在表格段 [s,e] 内,被跳过(66-65 分支) from copy import copy from openpyxl import Workbook wb = Workbook() ws = wb.active ws.title = "混合" data = [ ["機能ID", "機能名"], ["F101", "社員登録"], [], ["・改述ポイント"], ] for r, row in enumerate(data, start=1): for c, v in enumerate(row, start=1): if v: ws.cell(row=r, column=c, value=v) font = copy(ws.cell(row=4, column=1).font) # 碎片段行(物理行 4) font.strike = True ws.cell(row=4, column=1).font = font path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) ms = result.mixed[0] tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0] # 表格段数据行不应带上碎片段的取消线格式 assert tbl.rows[0]["機能ID"].formatting is None assert len(tbl.rows) == 1 def test_parse_table_sheets(tmp_path): wb = new_workbook({ "機能一覧": [["機能ID", "機能名"], ["A001", "社員登録"]], "バッチ一覧": [["バッチID", "処理名"], ["B1", "夜間集計"]], }) path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) assert isinstance(result, ExcelParseResult) by_name = {t.name: t for t in result.tables} assert by_name["機能一覧"].detected_type == SheetType.FUNCTION assert len(by_name["機能一覧"].rows) == 1 assert by_name["バッチ一覧"].detected_type == SheetType.BATCH def test_parse_free_text_sheet(tmp_path): wb = new_workbook({"メモ": [["新入社員を登録"], [], [], [], []]}) path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) assert len(result.tables) == 1 assert result.tables[0].extraction_method == "llm_from_free_text" def test_parse_uses_detected_header_row(tmp_path): # 标题行(单格)在首行,真实表头在第二行 wb = new_workbook({ "機能一覧": [["機能一覧"], ["機能ID", "機能名"], ["A001", "社員登録"]], }) path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) t = result.tables[0] assert t.headers == ["機能ID", "機能名"] assert len(t.rows) == 1 assert t.rows[0]["機能ID"].value == "A001" def test_parse_attaches_strikethrough_formatting(tmp_path): from copy import copy from openpyxl import Workbook wb = Workbook() ws = wb.active ws.title = "機能一覧" ws["A1"] = "機能ID" ws["B1"] = "機能名" ws["A2"] = "A001" ws["B2"] = "社員登録" font = copy(ws["A2"].font) font.strike = True ws["A2"].font = font path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) cv = result.tables[0].rows[0]["機能ID"] assert cv.formatting is not None assert cv.formatting.strikethrough is True from genesis.data_models import MixedParagraph, MixedSheet from genesis.parsers.excel_parser import ExcelParseResult def test_excel_parse_result_has_mixed_default(): r = ExcelParseResult(file_name="f.xlsx") assert r.mixed == [] def test_mixed_paragraph_defaults(): p = MixedParagraph(kind="table") assert p.table is None assert p.text is None assert p.source_range is None def test_mixed_sheet_holds_paragraphs(): p1 = MixedParagraph(kind="table") p2 = MixedParagraph(kind="free_text", text="备注") ms = MixedSheet(name="混合", paragraphs=[p1, p2]) assert ms.name == "混合" assert [p.kind for p in ms.paragraphs] == ["table", "free_text"] def test_parse_mixed_sheet_segmented(tmp_path): wb = new_workbook({ "混合": [ ["機能ID", "機能名"], ["F101", "社員登録"], ["F102", "退職処理"], [], ["・改修ポイント:F102 追加バリデーション"], ["■対象画面:SC001"], ], }) path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) assert result.mixed, "混合 sheet 应产产出 mixed 段落" ms = result.mixed[0] assert ms.name == "混合" kinds = [p.kind for p in ms.paragraphs] assert "table" in kinds and "free_text" in kinds # 表格段无碎片污染:table 段应含 2 数据行,机能ID 首行为 F101 tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0] assert tbl is not None and len(tbl.rows) == 2 assert tbl.rows[0]["機能ID"].value == "F101" # 自由文本段捕获碎片 ft = [p for p in ms.paragraphs if p.kind == "free_text"][0] assert "改修ポイント" in (ft.text or "") def test_parse_mixed_sheet_formatting_in_mid_segment(tmp_path): # 表格段不在物理行 0(自由文本段在前),数据行 F101 设取消线 —— 锁住 formatting_map 坐标错位 from copy import copy from openpyxl import Workbook wb = Workbook() ws = wb.active ws.title = "混合" data = [ ["■はじめに"], ["前提説明"], [], ["機能ID", "機能名"], ["F101", "社員登録"], ["F102", "退職処理"], [], ["・改修ポイント"], ] for r, row in enumerate(data, start=1): for c, v in enumerate(row, start=1): if v: ws.cell(row=r, column=c, value=v) font = copy(ws.cell(row=5, column=1).font) # F101 所在物理行(第 5 行) font.strike = True ws.cell(row=5, column=1).font = font path = save_workbook(tmp_path, wb) result = ExcelParser().parse(path) ms = result.mixed[0] tbl = [p.table for p in ms.paragraphs if p.kind == "table"][0] cv = tbl.rows[0]["機能ID"] # F101 assert cv.formatting is not None assert cv.formatting.strikethrough is True