feat: FreeTextExtractor(文本分段 + 占位结构化表)
This commit is contained in:
@@ -0,0 +1,51 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from genesis.data_models import CellValue, ExcelTable, Provenance, SheetType
|
||||
|
||||
|
||||
def extract_text_blocks(matrix: list[list[Any]]) -> list[str]:
|
||||
"""按全空行分段;行内非空单元格以「 」连接。"""
|
||||
blocks: list[str] = []
|
||||
current: list[str] = []
|
||||
for row in matrix:
|
||||
cells = [str(c).strip() for c in row if c is not None and str(c).strip() != ""]
|
||||
if not cells:
|
||||
if current:
|
||||
blocks.append(" ".join(current))
|
||||
current = []
|
||||
continue
|
||||
current.append(" ".join(cells))
|
||||
if current:
|
||||
blocks.append(" ".join(current))
|
||||
return blocks
|
||||
|
||||
|
||||
def build_free_text_table(
|
||||
sheet_name: str,
|
||||
blocks: list[str],
|
||||
file_name: str,
|
||||
detected_type: SheetType = SheetType.GENERIC,
|
||||
) -> ExcelTable:
|
||||
rows = []
|
||||
for i, text in enumerate(blocks, start=1):
|
||||
rows.append({
|
||||
"text": CellValue(
|
||||
value=text,
|
||||
provenance=Provenance(
|
||||
file_name=file_name,
|
||||
sheet_name=sheet_name,
|
||||
row=i,
|
||||
column="A",
|
||||
column_header="text",
|
||||
),
|
||||
),
|
||||
})
|
||||
return ExcelTable(
|
||||
name=sheet_name,
|
||||
detected_type=detected_type,
|
||||
extraction_method="llm_from_free_text",
|
||||
headers=["text"],
|
||||
rows=rows,
|
||||
)
|
||||
Reference in New Issue
Block a user