fix(data_process): 修复 Word 切分行号断档与预览标题缺失

- 正文抽取下钻 SDT 内容控件,目录等内容不再整段丢失
- 切片投影匹配剥离序列化插入的列表自动编号,新增行锚点兜底与游标防回退
- fixed/semantic 切分路径定位失败时保留切片,不再静默丢弃
- Word 预览按段落大纲级别识别标题,未套标题样式的小节正常渲染
- 本地嵌入模型抽为共享单例,供语义分块与质量评分共用
This commit is contained in:
caoxiaozhu
2026-08-19 14:21:41 +08:00
parent 93449b7f0e
commit f47611020a
6 changed files with 315 additions and 34 deletions

View File

@@ -0,0 +1,24 @@
"""数据处理算法 - 本地语义嵌入模型共享单例。"""
from __future__ import annotations
import os
from functools import lru_cache
from typing import Any
@lru_cache(maxsize=1)
def semantic_embedding_model() -> Any:
"""加载本地嵌入模型,供语义分块与语义质量评分共用。
模型可在部署环境覆盖;默认模型体积较小且适合中英文语义判断。
返回 LlamaIndex BaseEmbedding通过 ``get_text_embedding`` 使用。
"""
from llama_index.embeddings.huggingface import HuggingFaceEmbedding
return HuggingFaceEmbedding(
model_name=os.getenv("DATA_PROCESS_EMBEDDING_MODEL", "BAAI/bge-small-zh-v1.5"),
device=os.getenv("DATA_PROCESS_EMBEDDING_DEVICE", "cpu"),
trust_remote_code=False,
)

View File

@@ -7,12 +7,13 @@ import re
import unicodedata import unicodedata
import zipfile import zipfile
import xml.etree.ElementTree as ET import xml.etree.ElementTree as ET
from collections.abc import Mapping, Sequence from collections.abc import Iterator, Mapping, Sequence
from pathlib import PurePosixPath from pathlib import PurePosixPath
from typing import Any from typing import Any
from urllib.parse import unquote, urlsplit from urllib.parse import unquote, urlsplit
from docx import Document from docx import Document
from docx.oxml.ns import qn
from docx.oxml.table import CT_Tbl from docx.oxml.table import CT_Tbl
from docx.oxml.text.paragraph import CT_P from docx.oxml.text.paragraph import CT_P
from docx.table import Table from docx.table import Table
@@ -121,6 +122,21 @@ def _validate_office_archive(raw: bytes, file_format: TextFormat) -> None:
except zipfile.BadZipFile as exc: except zipfile.BadZipFile as exc:
raise ValueError(f"invalid {file_format.upper()} file: not an Office ZIP package") from exc raise ValueError(f"invalid {file_format.upper()} file: not an Office ZIP package") from exc
def iter_document_blocks(parent: Any) -> Iterator[Any]:
"""按文档顺序产出正文段落与表格,并下钻 SDT 内容控件。
Word 的目录、复选框等内容控件包在 ``w:sdt`` 元素里,只遍历 body
直接子级会把这些段落整段丢掉。
"""
for child in parent.iterchildren():
if isinstance(child, (CT_P, CT_Tbl)):
yield child
elif child.tag == qn("w:sdt"):
content = child.find(qn("w:sdtContent"))
if content is not None:
yield from iter_document_blocks(content)
def _extract_docx_text(raw: bytes) -> str: def _extract_docx_text(raw: bytes) -> str:
_validate_office_archive(raw, "docx") _validate_office_archive(raw, "docx")
try: try:
@@ -130,7 +146,7 @@ def _extract_docx_text(raw: bytes) -> str:
parts: list[str] = [] parts: list[str] = []
total = 0 total = 0
for child in document.element.body.iterchildren(): for child in iter_document_blocks(document.element.body):
if isinstance(child, CT_P): if isinstance(child, CT_P):
total = _append_bounded_text(parts, Paragraph(child, document).text, total) total = _append_bounded_text(parts, Paragraph(child, document).text, total)
continue continue

View File

@@ -18,12 +18,19 @@ from llama_index.core.base.embeddings.base import BaseEmbedding
from llama_index.core.node_parser import SemanticSplitterNodeParser, SentenceSplitter from llama_index.core.node_parser import SemanticSplitterNodeParser, SentenceSplitter
from app.modules.data_process.algorithms import normalize_text from app.modules.data_process.algorithms import normalize_text
from app.modules.data_process.algorithms.embedding import semantic_embedding_model
ChunkMethod = Literal["layout_hybrid", "semantic", "fixed"] ChunkMethod = Literal["layout_hybrid", "semantic", "fixed"]
_PAGE_FURNITURE = re.compile( _PAGE_FURNITURE = re.compile(
r"(?m)^\s*(?:第\s*\d+\s*页\s*共\s*\d+\s*页|[-—–]?\s*\d+\s*[/]\s*\d+\s*[-—–]?)\s*$" r"(?m)^\s*(?:第\s*\d+\s*页\s*共\s*\d+\s*页|[-—–]?\s*\d+\s*[/]\s*\d+\s*[-—–]?)\s*$"
) )
# Docling 的 markdown 序列化会给列表项补上自动编号,而 Word 的编号存放在
# numbering.xml 中python-docx 抽取的正文不含这些编号;紧凑匹配前剥掉
# 行首编号,否则带列表的切片会整体定位失败。
_LIST_MARKER_PREFIX = re.compile(
r"(?m)^[ \t>]*(?:(?:\d{1,3}[.)])+|\([a-zA-Z0-9]{1,3}\)|[a-zA-Z][.)]|[-*+•·])[ \t]+"
)
_COMPACT_CHARACTER = re.compile(r"[\w\u3400-\u4dbf\u4e00-\u9fff]", re.UNICODE) _COMPACT_CHARACTER = re.compile(r"[\w\u3400-\u4dbf\u4e00-\u9fff]", re.UNICODE)
_CONVERTER_LOCK = threading.Lock() _CONVERTER_LOCK = threading.Lock()
@@ -149,18 +156,6 @@ def chunk_fixed_text(
return _text_chunks(text, chunk_size=chunk_size, chunk_overlap=chunk_overlap) return _text_chunks(text, chunk_size=chunk_size, chunk_overlap=chunk_overlap)
@lru_cache(maxsize=1)
def _semantic_embedding_model() -> BaseEmbedding:
# 模型可在部署环境覆盖;默认模型体积较小且适合中英文语义边界判断。
from llama_index.embeddings.huggingface import HuggingFaceEmbedding
return HuggingFaceEmbedding(
model_name=os.getenv("DATA_PROCESS_EMBEDDING_MODEL", "BAAI/bge-small-zh-v1.5"),
device=os.getenv("DATA_PROCESS_EMBEDDING_DEVICE", "cpu"),
trust_remote_code=False,
)
def chunk_semantic_text( def chunk_semantic_text(
text: str, text: str,
*, *,
@@ -175,7 +170,7 @@ def chunk_semantic_text(
if not normalized: if not normalized:
return [] return []
splitter = SemanticSplitterNodeParser.from_defaults( splitter = SemanticSplitterNodeParser.from_defaults(
embed_model=embed_model or _semantic_embedding_model(), embed_model=embed_model or semantic_embedding_model(),
breakpoint_percentile_threshold=breakpoint_percentile_threshold, breakpoint_percentile_threshold=breakpoint_percentile_threshold,
buffer_size=1, buffer_size=1,
sentence_splitter=_sentence_chunks, sentence_splitter=_sentence_chunks,
@@ -193,6 +188,7 @@ def chunk_semantic_text(
if start is None: if start is None:
start = _locate_text(normalized, content, 0) start = _locate_text(normalized, content, 0)
if start is None: if start is None:
result.append(_unlocated_chunk(content))
continue continue
if len(_tokenizer().encode(content)) <= chunk_size: if len(_tokenizer().encode(content)) <= chunk_size:
result.append(_make_text_chunk(normalized, start, start + len(content))) result.append(_make_text_chunk(normalized, start, start + len(content)))
@@ -203,6 +199,7 @@ def chunk_semantic_text(
chunk_overlap=chunk_overlap, chunk_overlap=chunk_overlap,
): ):
if child.source_start is None or child.source_end is None: if child.source_start is None or child.source_end is None:
result.append(_unlocated_chunk(child.original_content))
continue continue
result.append( result.append(
_make_text_chunk( _make_text_chunk(
@@ -235,6 +232,7 @@ def _nodes_to_chunks(nodes: list[Any], source_text: str) -> list[DocumentChunk]:
if start is None: if start is None:
start = _locate_text(source_text, content, 0) start = _locate_text(source_text, content, 0)
if start is None: if start is None:
chunks.append(_unlocated_chunk(content))
continue continue
end = start + len(content) end = start + len(content)
chunks.append(_make_text_chunk(source_text, start, end)) chunks.append(_make_text_chunk(source_text, start, end))
@@ -247,6 +245,20 @@ def _locate_text(source: str, content: str, start: int) -> int | None:
return position if position >= 0 else None return position if position >= 0 else None
def _unlocated_chunk(content: str) -> DocumentChunk:
"""正文在源文本中定位失败时保底保留切片,只放弃行号信息。"""
return DocumentChunk(
original_content=content,
contextualized_content=content,
source_start=None,
source_end=None,
source_start_line=None,
source_end_line=None,
token_count=len(_tokenizer().encode(content)),
)
def _make_text_chunk(source: str, start: int, end: int) -> DocumentChunk: def _make_text_chunk(source: str, start: int, end: int) -> DocumentChunk:
content = source[start:end] content = source[start:end]
return DocumentChunk( return DocumentChunk(
@@ -316,6 +328,14 @@ def _compact_with_offsets(value: str) -> tuple[str, list[int]]:
return "".join(compact), offsets return "".join(compact), offsets
def _expand_to_line_boundaries(source_text: str, start: int, end: int) -> tuple[int, int]:
while start > 0 and source_text[start - 1] not in "\r\n":
start -= 1
while end < len(source_text) and source_text[end] not in "\r\n":
end += 1
return start, end
def _project_layout_span( def _project_layout_span(
source_text: str, source_text: str,
content: str, content: str,
@@ -324,21 +344,79 @@ def _project_layout_span(
source_offsets: list[int], source_offsets: list[int],
compact_start: int, compact_start: int,
) -> tuple[int | None, int | None, int]: ) -> tuple[int | None, int | None, int]:
compact_content, _ = _compact_with_offsets(content) for candidate in (content, _LIST_MARKER_PREFIX.sub("", content)):
compact_content, _ = _compact_with_offsets(candidate)
if len(compact_content) < 4: if len(compact_content) < 4:
return None, None, compact_start continue
position = compact_source.find(compact_content, compact_start) position = compact_source.find(compact_content, compact_start)
if position < 0: if position < 0:
position = compact_source.find(compact_content) position = compact_source.find(compact_content)
if position < 0: if position < 0:
continue
start, end = _expand_to_line_boundaries(
source_text,
source_offsets[position],
source_offsets[position + len(compact_content) - 1] + 1,
)
# 重复内容回退匹配可能命中已消费的更早位置,游标只进不退,
# 避免后续切片跟着错位。
return start, end, max(compact_start, position + len(compact_content))
return _project_layout_span_by_anchors(
source_text,
content,
compact_source=compact_source,
source_offsets=source_offsets,
compact_start=compact_start,
)
def _project_layout_span_by_anchors(
source_text: str,
content: str,
*,
compact_source: str,
source_offsets: list[int],
compact_start: int,
) -> tuple[int | None, int | None, int]:
"""按行锚点顺序匹配,容忍切片里插入的重复表头等非连续内容。"""
segments = [
compact
for compact in (
_compact_with_offsets(line)[0]
for line in _LIST_MARKER_PREFIX.sub("", content).split("\n")
)
if len(compact) >= 6
]
if not segments:
return None, None, compact_start return None, None, compact_start
start = source_offsets[position] total = sum(len(segment) for segment in segments)
end = source_offsets[position + len(compact_content) - 1] + 1
while start > 0 and source_text[start - 1] not in "\r\n": def match_from(cursor: int) -> tuple[list[tuple[int, int]], int]:
start -= 1 matched: list[tuple[int, int]] = []
while end < len(source_text) and source_text[end] not in "\r\n": position = cursor
end += 1 for segment in segments:
return start, end, position + len(compact_content) found = compact_source.find(segment, position)
if found < 0:
continue
matched.append((found, found + len(segment)))
position = found + len(segment)
return matched, sum(end - start for start, end in matched)
matched, covered = match_from(compact_start)
if covered * 2 < total:
retried, retry_covered = match_from(0)
if retry_covered > covered:
matched, covered = retried, retry_covered
# 覆盖不足一半时宁可不定位,也不能给出错误的行号。
if not matched or covered * 2 < total:
return None, None, compact_start
start, end = _expand_to_line_boundaries(
source_text,
source_offsets[matched[0][0]],
source_offsets[matched[-1][1] - 1] + 1,
)
return start, end, max(compact_start, matched[-1][1])
def chunk_layout_document( def chunk_layout_document(

View File

@@ -11,6 +11,7 @@ import re
from typing import Any from typing import Any
from docx import Document from docx import Document
from docx.oxml.ns import qn
from docx.oxml.table import CT_Tbl from docx.oxml.table import CT_Tbl
from docx.oxml.text.paragraph import CT_P from docx.oxml.text.paragraph import CT_P
from docx.table import Table from docx.table import Table
@@ -27,6 +28,7 @@ from app.modules.data_process.algorithms import (
_xlsx_sheet_merge_ranges, _xlsx_sheet_merge_ranges,
normalize_text, normalize_text,
) )
from app.modules.data_process.algorithms.parsers.office import iter_document_blocks
MAX_DOCX_PREVIEW_BLOCKS = 2_000 MAX_DOCX_PREVIEW_BLOCKS = 2_000
MAX_XLSX_PREVIEW_ROWS = 200 MAX_XLSX_PREVIEW_ROWS = 200
@@ -49,12 +51,21 @@ def _docx_alignment(paragraph: Paragraph) -> str:
def _docx_heading_level(paragraph: Paragraph) -> int | None: def _docx_heading_level(paragraph: Paragraph) -> int | None:
style = paragraph.style style = paragraph.style
if style is None: style_name = str(style.name or "") if style is not None else ""
return None style_id = str(style.style_id or "") if style is not None else ""
style_name = str(style.name or "")
style_id = str(style.style_id or "")
match = re.search(r"(?:heading|标题)\s*([1-6])", f"{style_name} {style_id}", re.IGNORECASE) match = re.search(r"(?:heading|标题)\s*([1-6])", f"{style_name} {style_id}", re.IGNORECASE)
return int(match.group(1)) if match else None if match:
return int(match.group(1))
# Word 的目录和导航窗格依据大纲级别识别标题;未套标题样式但带
# outlineLvl 的段落(如手工排版的编号小节)同样是标题。
outline = paragraph._p.find(f"{qn('w:pPr')}/{qn('w:outlineLvl')}")
if outline is not None:
value = outline.get(qn("w:val"))
if value is not None and value.isdigit():
level = int(value)
if 0 <= level <= 5:
return level + 1
return None
def build_docx_preview(raw: bytes) -> dict[str, Any]: def build_docx_preview(raw: bytes) -> dict[str, Any]:
@@ -84,7 +95,7 @@ def build_docx_preview(raw: bytes) -> dict[str, Any]:
has_source_content = True has_source_content = True
return text, start, source_cursor return text, start, source_cursor
for child in document.element.body.iterchildren(): for child in iter_document_blocks(document.element.body):
if rendered_blocks >= MAX_DOCX_PREVIEW_BLOCKS: if rendered_blocks >= MAX_DOCX_PREVIEW_BLOCKS:
truncated = True truncated = True
break break

View File

@@ -9,6 +9,8 @@ from decimal import Decimal
import pytest import pytest
from docx import Document from docx import Document
from docx.oxml import parse_xml
from docx.oxml.ns import nsdecls, qn
from openpyxl import Workbook from openpyxl import Workbook
from pptx import Presentation from pptx import Presentation
from pptx.util import Inches from pptx.util import Inches
@@ -38,6 +40,7 @@ from app.modules.data_process.algorithms import (
stable_split_assignments, stable_split_assignments,
structured_json_dumps, structured_json_dumps,
) )
from app.modules.data_process.office_preview import build_docx_preview
def _pdf_page_texts(*texts: str) -> tuple[PdfPageText, ...]: def _pdf_page_texts(*texts: str) -> tuple[PdfPageText, ...]:
@@ -318,6 +321,74 @@ def test_parse_pdf_docx_xlsx_and_pptx() -> None:
assert parsed_pptx.records == () assert parsed_pptx.records == ()
def _docx_with_sdt_bytes() -> bytes:
"""构造带 SDT 目录内容控件的 docx段落顺序为正文、SDT、正文。"""
document = Document()
document.add_paragraph("正文开头。")
sdt = parse_xml(
"<w:sdt %s><w:sdtPr><w:id w:val='1'/></w:sdtPr>"
"<w:sdtContent><w:p><w:r><w:t>目录条目 第一章 概述</w:t></w:r></w:p>"
"</w:sdtContent></w:sdt>" % nsdecls("w")
)
body = document.element.body
sect_pr = body.find(qn("w:sectPr"))
if sect_pr is not None:
sect_pr.addprevious(sdt)
else:
body.append(sdt)
document.add_paragraph("正文结尾。")
output = io.BytesIO()
document.save(output)
return output.getvalue()
def test_docx_extraction_and_preview_include_sdt_content() -> None:
raw = _docx_with_sdt_bytes()
parsed = parse_text_content(raw, filename="toc.docx")
assert "目录条目 第一章 概述" in parsed.text
assert (
parsed.text.index("正文开头。")
< parsed.text.index("目录条目 第一章 概述")
< parsed.text.index("正文结尾。")
)
preview = build_docx_preview(raw)
paragraph_texts = [
block["text"] for block in preview["blocks"] if block["type"] == "paragraph"
]
assert "目录条目 第一章 概述" in paragraph_texts
# 预览偏移必须与正文抽取规则一致,否则前端定位会错位。
sdt_block = next(
block
for block in preview["blocks"]
if block.get("text") == "目录条目 第一章 概述"
)
assert parsed.text[sdt_block["source_start"] : sdt_block["source_end"]] == (
"目录条目 第一章 概述"
)
def test_docx_preview_detects_outline_level_headings() -> None:
"""未套标题样式但设了大纲级别的段落Word 目录按此收录)也按标题渲染。"""
document = Document()
document.add_heading("一级标题", level=1)
plain = document.add_paragraph("4.2.1 数据管理")
p_pr = plain._p.get_or_add_pPr()
p_pr.append(parse_xml("<w:outlineLvl %s w:val='2'/>" % nsdecls("w")))
document.add_paragraph("普通正文段落。")
output = io.BytesIO()
document.save(output)
preview = build_docx_preview(output.getvalue())
blocks = {b["text"]: b for b in preview["blocks"] if b["type"] == "paragraph"}
assert blocks["一级标题"]["heading_level"] == 1
assert blocks["4.2.1 数据管理"]["heading_level"] == 3
assert blocks["普通正文段落。"]["heading_level"] is None
def test_xlsx_record_locators_distinguish_sheets_rows_and_duplicate_records() -> None: def test_xlsx_record_locators_distinguish_sheets_rows_and_duplicate_records() -> None:
workbook = Workbook() workbook = Workbook()
first = workbook.active first = workbook.active

View File

@@ -9,6 +9,7 @@ from app.modules.data_process.document_chunking import (
DocumentChunk, DocumentChunk,
_compact_with_offsets, _compact_with_offsets,
_document_converter, _document_converter,
_nodes_to_chunks,
_project_layout_span, _project_layout_span,
chunk_fixed_text, chunk_fixed_text,
chunk_semantic_text, chunk_semantic_text,
@@ -103,6 +104,86 @@ def test_layout_projection_ignores_layout_whitespace_but_keeps_source_lines() ->
assert cursor > 0 assert cursor > 0
def test_layout_projection_tolerates_list_numbers_inserted_by_serializer() -> None:
# Word 自动编号存放在 numbering.xmlpython-docx 抽取的正文没有编号,
# 而 Docling 序列化切片时会补上 "1. " 前缀,投影不能因此失败。
source = "接入方式说明\n结构化数据接入需要先配置连接地址。\n非结构化接入需要上传文档。"
compact_source, offsets = _compact_with_offsets(source)
start, end, cursor = _project_layout_span(
source,
"1. 结构化数据接入需要先配置连接地址。\n2. 非结构化接入需要上传文档。",
compact_source=compact_source,
source_offsets=offsets,
compact_start=0,
)
assert start is not None and end is not None
assert source[start:end] == "结构化数据接入需要先配置连接地址。\n非结构化接入需要上传文档。"
assert cursor > 0
def test_layout_projection_never_moves_cursor_backwards() -> None:
source = "重复段落内容。\n中间正文。\n重复段落内容。"
compact_source, offsets = _compact_with_offsets(source)
# 重复内容回退匹配命中已消费的更早位置时,游标必须保持不退。
_, _, cursor = _project_layout_span(
source,
"重复段落内容。",
compact_source=compact_source,
source_offsets=offsets,
compact_start=compact_source.index("中间正文"),
)
assert cursor >= compact_source.index("中间正文")
def test_layout_projection_falls_back_to_line_anchors_for_inserted_content() -> None:
# 表格跨切片时 Docling 会在续片中重复表头,正文不再是连续子串;
# 按行锚点匹配仍应定位到表头所在行到末行数据之间的连续区间。
source = "表头甲\t表头乙\n第一行数据\t说明一\n第二行数据\t说明二"
compact_source, offsets = _compact_with_offsets(source)
start, end, _ = _project_layout_span(
source,
"表头甲 表头乙\n第二行数据 说明二",
compact_source=compact_source,
source_offsets=offsets,
compact_start=0,
)
assert start is not None and end is not None
assert source[start:end] == (
"表头甲\t表头乙\n第一行数据\t说明一\n第二行数据\t说明二"
)
def test_layout_projection_refuses_low_coverage_anchor_match() -> None:
source = "完全无关的正文内容甲。\n完全无关的正文内容乙。"
compact_source, offsets = _compact_with_offsets(source)
start, end, cursor = _project_layout_span(
source,
"找不到的数据行内容\n另一条找不到的数据行内容",
compact_source=compact_source,
source_offsets=offsets,
compact_start=0,
)
assert start is None
assert end is None
assert cursor == 0
def test_text_splitter_keeps_chunks_that_cannot_be_located() -> None:
class FakeNode:
def get_content(self) -> str:
return "这段文本在源文本中不存在。"
chunks = _nodes_to_chunks([FakeNode()], "完全不同的源文本。")
assert len(chunks) == 1
assert chunks[0].original_content == "这段文本在源文本中不存在。"
assert chunks[0].source_start is None
assert chunks[0].source_start_line is None
def test_short_layout_chunk_merges_with_neighbor_and_keeps_page_provenance() -> None: def test_short_layout_chunk_merges_with_neighbor_and_keeps_page_provenance() -> None:
source = "短标题\n这是一段足够长的正文内容,用于测试相邻切片合并。" source = "短标题\n这是一段足够长的正文内容,用于测试相邻切片合并。"
chunks = [ chunks = [