fix(data_process): 修复 Word 切分行号断档与预览标题缺失
- 正文抽取下钻 SDT 内容控件,目录等内容不再整段丢失 - 切片投影匹配剥离序列化插入的列表自动编号,新增行锚点兜底与游标防回退 - fixed/semantic 切分路径定位失败时保留切片,不再静默丢弃 - Word 预览按段落大纲级别识别标题,未套标题样式的小节正常渲染 - 本地嵌入模型抽为共享单例,供语义分块与质量评分共用
This commit is contained in:
@@ -9,6 +9,7 @@ from app.modules.data_process.document_chunking import (
|
||||
DocumentChunk,
|
||||
_compact_with_offsets,
|
||||
_document_converter,
|
||||
_nodes_to_chunks,
|
||||
_project_layout_span,
|
||||
chunk_fixed_text,
|
||||
chunk_semantic_text,
|
||||
@@ -103,6 +104,86 @@ def test_layout_projection_ignores_layout_whitespace_but_keeps_source_lines() ->
|
||||
assert cursor > 0
|
||||
|
||||
|
||||
def test_layout_projection_tolerates_list_numbers_inserted_by_serializer() -> None:
|
||||
# Word 自动编号存放在 numbering.xml,python-docx 抽取的正文没有编号,
|
||||
# 而 Docling 序列化切片时会补上 "1. " 前缀,投影不能因此失败。
|
||||
source = "接入方式说明\n结构化数据接入需要先配置连接地址。\n非结构化接入需要上传文档。"
|
||||
compact_source, offsets = _compact_with_offsets(source)
|
||||
start, end, cursor = _project_layout_span(
|
||||
source,
|
||||
"1. 结构化数据接入需要先配置连接地址。\n2. 非结构化接入需要上传文档。",
|
||||
compact_source=compact_source,
|
||||
source_offsets=offsets,
|
||||
compact_start=0,
|
||||
)
|
||||
|
||||
assert start is not None and end is not None
|
||||
assert source[start:end] == "结构化数据接入需要先配置连接地址。\n非结构化接入需要上传文档。"
|
||||
assert cursor > 0
|
||||
|
||||
|
||||
def test_layout_projection_never_moves_cursor_backwards() -> None:
|
||||
source = "重复段落内容。\n中间正文。\n重复段落内容。"
|
||||
compact_source, offsets = _compact_with_offsets(source)
|
||||
# 重复内容回退匹配命中已消费的更早位置时,游标必须保持不退。
|
||||
_, _, cursor = _project_layout_span(
|
||||
source,
|
||||
"重复段落内容。",
|
||||
compact_source=compact_source,
|
||||
source_offsets=offsets,
|
||||
compact_start=compact_source.index("中间正文"),
|
||||
)
|
||||
assert cursor >= compact_source.index("中间正文")
|
||||
|
||||
|
||||
def test_layout_projection_falls_back_to_line_anchors_for_inserted_content() -> None:
|
||||
# 表格跨切片时 Docling 会在续片中重复表头,正文不再是连续子串;
|
||||
# 按行锚点匹配仍应定位到表头所在行到末行数据之间的连续区间。
|
||||
source = "表头甲\t表头乙\n第一行数据\t说明一\n第二行数据\t说明二"
|
||||
compact_source, offsets = _compact_with_offsets(source)
|
||||
start, end, _ = _project_layout_span(
|
||||
source,
|
||||
"表头甲 表头乙\n第二行数据 说明二",
|
||||
compact_source=compact_source,
|
||||
source_offsets=offsets,
|
||||
compact_start=0,
|
||||
)
|
||||
|
||||
assert start is not None and end is not None
|
||||
assert source[start:end] == (
|
||||
"表头甲\t表头乙\n第一行数据\t说明一\n第二行数据\t说明二"
|
||||
)
|
||||
|
||||
|
||||
def test_layout_projection_refuses_low_coverage_anchor_match() -> None:
|
||||
source = "完全无关的正文内容甲。\n完全无关的正文内容乙。"
|
||||
compact_source, offsets = _compact_with_offsets(source)
|
||||
start, end, cursor = _project_layout_span(
|
||||
source,
|
||||
"找不到的数据行内容\n另一条找不到的数据行内容",
|
||||
compact_source=compact_source,
|
||||
source_offsets=offsets,
|
||||
compact_start=0,
|
||||
)
|
||||
|
||||
assert start is None
|
||||
assert end is None
|
||||
assert cursor == 0
|
||||
|
||||
|
||||
def test_text_splitter_keeps_chunks_that_cannot_be_located() -> None:
|
||||
class FakeNode:
|
||||
def get_content(self) -> str:
|
||||
return "这段文本在源文本中不存在。"
|
||||
|
||||
chunks = _nodes_to_chunks([FakeNode()], "完全不同的源文本。")
|
||||
|
||||
assert len(chunks) == 1
|
||||
assert chunks[0].original_content == "这段文本在源文本中不存在。"
|
||||
assert chunks[0].source_start is None
|
||||
assert chunks[0].source_start_line is None
|
||||
|
||||
|
||||
def test_short_layout_chunk_merges_with_neighbor_and_keeps_page_provenance() -> None:
|
||||
source = "短标题\n这是一段足够长的正文内容,用于测试相邻切片合并。"
|
||||
chunks = [
|
||||
|
||||
Reference in New Issue
Block a user