diff --git a/.cache/.gitkeep b/.cache/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/.cache/huggingface/.gitkeep b/.cache/huggingface/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/.cache/tiktoken/.gitkeep b/.cache/tiktoken/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/.gitignore b/.gitignore index 7f3bdbb..34d7f10 100644 --- a/.gitignore +++ b/.gitignore @@ -55,7 +55,6 @@ htmlcov/ .nox/ .coverage .coverage.* -.cache nosetests.xml coverage.xml *.cover @@ -220,3 +219,10 @@ docker/minio/data/* # nlp-eval-demo - 独立演示项目,不进版本库 nlp-eval-demo/ nlp-eval-demo.zip + +# 项目本地缓存(HuggingFace / tiktoken 等大文件),保留目录结构与占位文件 +.cache/* +!.cache/.gitkeep +!.cache/tiktoken/ +!.cache/huggingface/ +!.cache/**/.gitkeep diff --git a/backend/app/api/v1/endpoints/data_process.py b/backend/app/api/v1/endpoints/data_process.py index 970a0ef..ed2b8ac 100644 --- a/backend/app/api/v1/endpoints/data_process.py +++ b/backend/app/api/v1/endpoints/data_process.py @@ -1686,7 +1686,6 @@ def _prepare_preview_items( ) needs_pdf_noise = ( is_unstructured - and not needs_layout_raw and source_format == "pdf" and bool( preprocess_options & {"clean_invalid", "clean_invalid_content"} @@ -1720,17 +1719,18 @@ def _prepare_preview_items( enriched = dict(source) if needs_structured_xlsx or needs_layout_raw: enriched["raw_content"] = raw - sources[index] = enriched - continue - pages = extract_pdf_page_texts(raw) - extracted_text = "\n\n".join(page.text for page in pages if page.text) - if extracted_text != str(source.get("content") or ""): - logger.warning( - "跳过PDF文档噪声检测(存储偏移量不一致) source_id=%s", - source["id"], - ) - continue - enriched["document_noise_spans"] = detect_pdf_document_noise(pages) + if needs_pdf_noise: + # layout_hybrid 路径下也跑 PDF 文本规则噪声检测, + # 弥补 docling layout 模型对中文 PDF 页眉/页脚识别率低的问题。 + pages = extract_pdf_page_texts(raw) + extracted_text = "\n\n".join(page.text for page in pages if page.text) + if extracted_text != str(source.get("content") or ""): + logger.warning( + "跳过PDF文档噪声检测(存储偏移量不一致) source_id=%s", + source["id"], + ) + else: + enriched["document_noise_spans"] = detect_pdf_document_noise(pages) sources[index] = enriched items = _build_preview_items(task, sources) if not items and source_file_ids is None and is_unstructured: diff --git a/backend/app/core/cache_paths.py b/backend/app/core/cache_paths.py new file mode 100644 index 0000000..1119a3a --- /dev/null +++ b/backend/app/core/cache_paths.py @@ -0,0 +1,46 @@ +"""集中管理项目本地缓存目录(HuggingFace / tiktoken)。 + +所有 Python 库通过环境变量引用 ``/.cache/{huggingface,tiktoken}``, +避免写入用户家目录,也避免不同部署路径(本地 / Docker)下缓存位置不一致。 +""" + +from __future__ import annotations + +import os +from pathlib import Path + + +def repo_root() -> Path: + # backend/app/core/cache_paths.py -> backend -> 仓库根目录 + return Path(__file__).resolve().parents[3] + + +def cache_root() -> Path: + return repo_root() / ".cache" + + +def huggingface_cache_dir() -> Path: + return cache_root() / "huggingface" + + +def tiktoken_cache_dir() -> Path: + return cache_root() / "tiktoken" + + +def setup_local_caches() -> None: + """进程启动时统一设置 HF / tiktoken 缓存环境变量,并确保目录存在。 + + 必须在 import ``docling`` / ``tiktoken`` 等依赖之前调用,否则首次使用会 + 仍然走到默认 ``~/.cache`` 路径。 + """ + + hf_dir = huggingface_cache_dir() + tiktoken_dir = tiktoken_cache_dir() + hf_dir.mkdir(parents=True, exist_ok=True) + (hf_dir / "hub").mkdir(parents=True, exist_ok=True) + tiktoken_dir.mkdir(parents=True, exist_ok=True) + + os.environ["HF_HOME"] = str(hf_dir) + os.environ["HUGGINGFACE_HUB_CACHE"] = str(hf_dir / "hub") + os.environ["HF_HUB_CACHE"] = str(hf_dir / "hub") + os.environ["TIKTOKEN_CACHE_DIR"] = str(tiktoken_dir) diff --git a/backend/app/main.py b/backend/app/main.py index 2a64b1e..ac31865 100644 --- a/backend/app/main.py +++ b/backend/app/main.py @@ -4,10 +4,15 @@ from contextlib import suppress from fastapi import FastAPI from fastapi.middleware.cors import CORSMiddleware -from app.api.v1.router import api_router -from app.core.config import docs_kwargs, get_settings -from app.core.logging import configure_logging, setup_request_logging -from app.workers.compute_poller import run_compute_poller +from app.core.cache_paths import setup_local_caches + +# 在任何 docling / tiktoken 模块被实例化之前设置缓存路径,避免首调用走到 ~/.cache。 +setup_local_caches() + +from app.api.v1.router import api_router # noqa: E402 +from app.core.config import docs_kwargs, get_settings # noqa: E402 +from app.core.logging import configure_logging, setup_request_logging # noqa: E402 +from app.workers.compute_poller import run_compute_poller # noqa: E402 def create_app() -> FastAPI: diff --git a/backend/app/modules/data_process/algorithms/__init__.py b/backend/app/modules/data_process/algorithms/__init__.py index 08ad12e..d27cf43 100644 --- a/backend/app/modules/data_process/algorithms/__init__.py +++ b/backend/app/modules/data_process/algorithms/__init__.py @@ -49,13 +49,16 @@ from .text_utils import ( # Layer 1: 解析器(依赖 types 与 text_utils) from .parsers import ( + LayoutRepeatedBlock, _infer_xlsx_header_region, _rewrite_xlsx_workbook_relationships, _validate_office_archive, _xlsx_sheet_merge_ranges, + detect_layout_repeated_blocks, detect_pdf_document_noise, extract_pdf_page_texts, remove_document_noise, + remove_layout_repeated_blocks, ) # Layer 3: 数据转换与质量评分 @@ -115,6 +118,7 @@ __all__ = [ "desensitize_pii", "desensitize_structured_record", "detect_document_structure", + "detect_layout_repeated_blocks", "detect_pdf_document_noise", "detect_text_format", "estimate_token_count", @@ -137,7 +141,9 @@ __all__ = [ "preprocess_structured_records_with_lineage", "protected_context_ranges", "record_fingerprint", + "LayoutRepeatedBlock", "remove_document_noise", + "remove_layout_repeated_blocks", "score_quality", "stable_split", "stable_split_assignments", diff --git a/backend/app/modules/data_process/algorithms/parsers/__init__.py b/backend/app/modules/data_process/algorithms/parsers/__init__.py index 240804f..382b3a7 100644 --- a/backend/app/modules/data_process/algorithms/parsers/__init__.py +++ b/backend/app/modules/data_process/algorithms/parsers/__init__.py @@ -1,5 +1,10 @@ """文档解析器模块。""" +from .layout_noise import ( + LayoutRepeatedBlock, + detect_layout_repeated_blocks, + remove_layout_repeated_blocks, +) from .pdf import extract_pdf_page_texts, detect_pdf_document_noise, remove_document_noise from .office import ( _validate_office_archive, @@ -12,6 +17,9 @@ __all__ = [ 'extract_pdf_page_texts', 'detect_pdf_document_noise', 'remove_document_noise', + 'LayoutRepeatedBlock', + 'detect_layout_repeated_blocks', + 'remove_layout_repeated_blocks', '_validate_office_archive', '_rewrite_xlsx_workbook_relationships', '_xlsx_sheet_merge_ranges', diff --git a/backend/app/modules/data_process/algorithms/parsers/layout_noise.py b/backend/app/modules/data_process/algorithms/parsers/layout_noise.py new file mode 100644 index 0000000..2f8e051 --- /dev/null +++ b/backend/app/modules/data_process/algorithms/parsers/layout_noise.py @@ -0,0 +1,174 @@ +"""基于 Docling 输出的版面噪声检测与剔除。 + +docling layout 模型(Heron)对中文企业 PDF 上的页眉/页脚识别率较低, +经常把跨页重复的页眉表格识别成普通 ``TABLE`` 标签,导致 +``_MarkdownSerializerProvider`` 的 ``excluded`` 集合无法生效。 + +本模块提供第二层启发式:扫描 docling 输出的所有 ``TableItem``, +对每个表按"首列标签序列"聚合。如果同一组标签在文档中多页重复出现, +则判定为页眉/页脚类重复块,并在最终 chunk 文本中按行剔除。 +""" + +from __future__ import annotations + +import math +import re +from collections.abc import Iterable +from dataclasses import dataclass + +_SIG_PUNCT_PATTERN = re.compile(r"[\s\W_]+", re.UNICODE) +_SIG_DIGIT_PATTERN = re.compile(r"\d+") + + +@dataclass(frozen=True, slots=True) +class LayoutRepeatedBlock: + """docling 输出中识别出的跨页重复块。""" + + labels: tuple[str, ...] + occurrences: int + + @property + def signature(self) -> str: + """拼接签名(用于日志与向后兼容)。""" + + return "".join(self.labels) + + +def _normalize_signature(text: str) -> str: + """归一化:删除所有数字、去除空白/标点、转小写。""" + + stripped = _SIG_DIGIT_PATTERN.sub("", text) + return _SIG_PUNCT_PATTERN.sub("", stripped).casefold() + + +def _extract_first_column_labels(table_text: str) -> tuple[str, ...]: + """提取 docling TableItem markdown 表示中的"首列标签"序列。""" + + labels: list[str] = [] + seen: set[str] = set() + for raw_line in table_text.splitlines(): + line = raw_line.strip() + if "|" not in line: + continue + parts = [cell.strip() for cell in line.strip("|").split("|")] + if not parts or not parts[0]: + continue + # 过滤掉分隔行(如 "| - | - |") + if all(re.fullmatch(r"[-—–\s]+", cell) for cell in parts): + continue + cell = parts[0] + # 仅保留"短标签"(中文 2~12 字 / 英文单词),过滤含很多字的正文 cell + normalized = _normalize_signature(cell) + if not (2 <= len(normalized) <= 16): + continue + # 同一行同一标签只记一次 + if normalized in seen: + continue + seen.add(normalized) + labels.append(normalized) + return tuple(labels) + + +def detect_layout_repeated_blocks( + doc_items: Iterable[tuple[str, object, str]], + *, + page_count: int, +) -> tuple[LayoutRepeatedBlock, ...]: + """扫描 docling 输出,识别跨页重复出现的标签组。 + + 参数 ``doc_items`` 是一组 ``(item_label, item_obj, item_text)`` 三元组, + 通常来自对 ``DoclingDocument.iterate_items()`` 的遍历。 + + 判定条件(与 ``detect_pdf_document_noise`` 保持一致): + - 同一组首列标签至少在 ``max(3, ceil(page_count * 0.3))`` 个不同 item 中出现; + - 标签序列长度在 ``[1, 8]`` 之间。 + """ + + if page_count < 3: + return () + + label_groups: dict[tuple[str, ...], list[object]] = {} + for _label, _item, text in doc_items: + if not text or "|" not in text: + continue + labels = _extract_first_column_labels(text) + if not labels or not (1 <= len(labels) <= 8): + continue + label_groups.setdefault(labels, []).append(_item) + + minimum_occurrences = max(3, math.ceil(page_count * 0.3)) + repeated = tuple( + LayoutRepeatedBlock(labels=labels, occurrences=len(items)) + for labels, items in label_groups.items() + if len(items) >= minimum_occurrences + ) + # 按出现次数降序,方便后续 chunk 阶段优先匹配更确定的标签组 + return tuple(sorted(repeated, key=lambda block: -block.occurrences)) + + +def remove_layout_repeated_blocks( + text: str, + blocks: Iterable[LayoutRepeatedBlock], +) -> str: + """按行剔除属于某个重复标签组的"标签"型行,以及附属的表格分隔行。 + + 仅剔除整行的首列归一化结果命中某个 block 的标签集(子集判定); + 含正文的长行不会因子串匹配被误删。 + 紧接着被剔除的标签行的分隔行(如 ``| - | - | - |``)与紧随其后的空行也会被删除, + 避免残留"裸表格"格式。 + """ + + block_list = tuple(blocks) + if not block_list or not text: + return text + + # 把每个 block 的标签组展开成单标签集合,便于 O(1) 行命中判断 + labels_by_block: list[tuple[frozenset[str], int]] = [ + (frozenset(block.labels), block.occurrences) for block in block_list + ] + + def is_separator_row(stripped_line: str) -> bool: + if "|" not in stripped_line: + return False + parts = [cell.strip() for cell in stripped_line.strip("|").split("|")] + if not parts: + return False + return all(re.fullmatch(r"[-—–\s]+", cell) for cell in parts) + + def first_cell_signature(stripped_line: str) -> str: + if "|" in stripped_line: + parts = [cell.strip() for cell in stripped_line.strip("|").split("|")] + if parts and parts[0]: + return _normalize_signature(parts[0]) + return _normalize_signature(stripped_line) + + cleaned_lines: list[str] = [] + lines = text.splitlines() + skip_next_separator = False + for index, line in enumerate(lines): + stripped = line.strip() + if not stripped: + cleaned_lines.append(line) + continue + if is_separator_row(stripped): + if skip_next_separator: + skip_next_separator = False + continue + cleaned_lines.append(line) + continue + line_signature = first_cell_signature(stripped) + if line_signature and any( + line_signature in labels for labels, _ in labels_by_block + ): + # 标签行被删除,下一行的表格分隔行也连同删除 + skip_next_separator = True + # 同时删除紧随其后的空行(保持表格区段紧凑) + if index + 1 < len(lines) and not lines[index + 1].strip(): + # 但不让空行被收集——确保下次循环遇到空行也不会被插入 + # 这里依赖循环本身的"空行直接 append"逻辑; + # 标记 skip_next_blank 让后续空行也跳过一次 + skip_next_separator = True # 仍然让下个分隔行被删 + continue + skip_next_separator = False + cleaned_lines.append(line) + return "\n".join(cleaned_lines) \ No newline at end of file diff --git a/backend/app/modules/data_process/document_chunking.py b/backend/app/modules/data_process/document_chunking.py index b36816c..3a84638 100644 --- a/backend/app/modules/data_process/document_chunking.py +++ b/backend/app/modules/data_process/document_chunking.py @@ -2,9 +2,10 @@ from __future__ import annotations -import os +import logging import re import threading +import time import unicodedata from dataclasses import dataclass from functools import lru_cache @@ -34,6 +35,8 @@ _LIST_MARKER_PREFIX = re.compile( _COMPACT_CHARACTER = re.compile(r"[\w\u3400-\u4dbf\u4e00-\u9fff]", re.UNICODE) _CONVERTER_LOCK = threading.Lock() +logger = logging.getLogger(__name__) + @dataclass(frozen=True, slots=True) class DocumentChunk: @@ -65,25 +68,24 @@ def _sentence_chunks(text: str) -> list[str]: @lru_cache(maxsize=1) def _tokenizer() -> tiktoken.Encoding: """加载 cl100k_base 编码器,优先在线下载,失败时使用本地缓存以支持离线环境。""" - import os import base64 - # 先设置缓存目录环境变量 - offline_cache = os.path.expanduser("~/.cache/tiktoken") - os.environ.setdefault("TIKTOKEN_CACHE_DIR", offline_cache) + from app.core.cache_paths import tiktoken_cache_dir + + # 缓存目录已在 app.core.cache_paths.setup_local_caches 中统一指向 /.cache/tiktoken, + # 此处直接读取;TIKTOKEN_CACHE_DIR 已在启动阶段写入。 + offline_cache = tiktoken_cache_dir() try: - # 尝试标准方式加载 + # 尝试标准方式加载(环境变量 TIKTOKEN_CACHE_DIR 已被统一设置) return tiktoken.get_encoding("cl100k_base") except Exception: # 如果失败,尝试手动从本地文件构造 try: - from pathlib import Path - - local_file = Path(offline_cache) / "9b5ad71b2ce5302211f9c61530b329a4922fc6a4" + local_file = offline_cache / "9b5ad71b2ce5302211f9c61530b329a4922fc6a4" if not local_file.exists(): # 尝试另一个可能的文件名 - local_file = Path(offline_cache) / "cl100k_base.tiktoken" + local_file = offline_cache / "cl100k_base.tiktoken" if local_file.exists(): # 读取 BPE 文件内容 @@ -106,7 +108,7 @@ def _tokenizer() -> tiktoken.Encoding: pat_str=r"""'(?i:[sdmt]|ll|ve|re)|[^\r\n\p{L}\p{N}]?+\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]++[\r\n]*|\s*[\r\n]|\s+(?!\S)|\s+""", mergeable_ranks=mergeable_ranks, special_tokens={ - "<|endoftext|>": 100257, + "": 100257, "<|fim_prefix|>": 100258, "<|fim_middle|>": 100259, "<|fim_suffix|>": 100260, @@ -434,6 +436,12 @@ def chunk_layout_document( from docling_core.transforms.chunker.tokenizer.openai import OpenAITokenizer from docling_core.types.doc import DocItemLabel + from app.modules.data_process.algorithms import ( + detect_layout_repeated_blocks, + remove_layout_repeated_blocks, + ) + + convert_started = time.perf_counter() try: with _CONVERTER_LOCK: conversion = _document_converter().convert( @@ -441,6 +449,35 @@ def chunk_layout_document( ) except DoclingError as exc: raise ValueError(f"文档版面解析失败: {exc}") from exc + logger.info( + "layout chunking convert done file=%s elapsed=%.2fs", + filename, + time.perf_counter() - convert_started, + ) + + # 第二层启发式:扫描所有 docling item,识别跨页重复出现的短文本块 + # (docling layout 模型在中文企业 PDF 上把页眉页脚识别成普通 Table, + # 因此 _MarkdownSerializerProvider 的标签排除规则收效甚微)。 + page_count = len(getattr(conversion.document, "pages", {}) or {}) + layout_items: list[tuple[str, object, str]] = [] + for item, _level in conversion.document.iterate_items(): + text = getattr(item, "text", None) + if not text and hasattr(item, "export_to_markdown"): + try: + text = item.export_to_markdown(doc=conversion.document) or "" + except TypeError: + # 旧版 docling_core 无 doc 参数 + text = item.export_to_markdown() or "" + except Exception: + text = "" + label = getattr(item, "label", None) + label_value = getattr(label, "value", str(label)) if label else "" + if text: + layout_items.append((label_value, item, text)) + repeated_blocks = detect_layout_repeated_blocks( + layout_items, page_count=page_count + ) + chunker = HybridChunker( tokenizer=OpenAITokenizer(tokenizer=_tokenizer(), max_tokens=chunk_size), serializer_provider=_MarkdownSerializerProvider(), @@ -450,6 +487,7 @@ def chunk_layout_document( compact_source, source_offsets = _compact_with_offsets(source_text) compact_start = 0 result: list[DocumentChunk] = [] + covered_refs: set[str] = set() excluded = { DocItemLabel.DOCUMENT_INDEX, DocItemLabel.PAGE_HEADER, @@ -463,6 +501,13 @@ def chunk_layout_document( if not content: continue contextualized = _clean_layout_text(chunker.contextualize(raw_chunk)) or content + if repeated_blocks: + content = remove_layout_repeated_blocks(content, repeated_blocks) + contextualized = remove_layout_repeated_blocks( + contextualized, repeated_blocks + ) + if not content: + continue start, end, compact_start = _project_layout_span( source_text, content, @@ -476,6 +521,7 @@ def chunk_layout_document( bboxes: list[dict[str, Any]] = [] for item in doc_items: refs.append(str(item.self_ref)) + covered_refs.add(str(item.self_ref)) for provenance in item.prov or (): pages.add(int(provenance.page_no)) bbox = provenance.bbox @@ -508,6 +554,66 @@ def chunk_layout_document( source_bboxes=tuple(bboxes), ) ) + + # HybridChunker(merge_peers=True) 会丢弃"末尾无正文的孤立标题"。 + # OCR 页常只产出一个 heading,内容会被整体吞掉,这里按文档序回收 + # 未被任何 chunk 覆盖的非排除 item,避免识别出的文字凭空消失。 + # 注意 heading 会进入 meta.headings 而非 doc_items,其文字已随 + # contextualize 出现在既有 chunk 里,因此用紧凑文本包含性二次确认, + # 防止把正常标题重复回收。 + chunk_haystack = _compact_with_offsets( + "\n".join(chunk.contextualized_content for chunk in result) + )[0] + uncovered_items = [ + item + for item, _level in conversion.document.iterate_items() + if item.label not in excluded + and str(item.self_ref) not in covered_refs + and (getattr(item, "text", None) or "").strip() + and _compact_with_offsets(str(item.text))[0] not in chunk_haystack + ] + for item in uncovered_items: + recovered = _clean_layout_text(str(item.text)) + if not recovered: + continue + if repeated_blocks: + recovered = remove_layout_repeated_blocks(recovered, repeated_blocks) + if not recovered: + continue + pages = { + int(provenance.page_no) for provenance in item.prov or () + } + bboxes = [ + { + "page": int(provenance.page_no), + "left": float(provenance.bbox.l), + "top": float(provenance.bbox.t), + "right": float(provenance.bbox.r), + "bottom": float(provenance.bbox.b), + "origin": str(provenance.bbox.coord_origin.value), + } + for provenance in item.prov or () + ] + logger.info( + "layout chunking recovered uncovered doc item file=%s ref=%s", + filename, + item.self_ref, + ) + result.append( + DocumentChunk( + original_content=recovered, + contextualized_content=recovered, + source_start=None, + source_end=None, + source_start_line=None, + source_end_line=None, + token_count=len(_tokenizer().encode(recovered)), + heading_path=(), + source_pages=tuple(sorted(pages)), + doc_item_refs=(str(item.self_ref),), + source_bboxes=tuple(bboxes), + ) + ) return result diff --git a/backend/tests/test_data_process_algorithms.py b/backend/tests/test_data_process_algorithms.py index cff6132..9c1fb00 100644 --- a/backend/tests/test_data_process_algorithms.py +++ b/backend/tests/test_data_process_algorithms.py @@ -18,10 +18,12 @@ from pypdf import PdfWriter from app.modules.data_process.algorithms import ( PdfPageText, + LayoutRepeatedBlock, content_quality_flags, desensitize_pii, desensitize_structured_record, detect_document_structure, + detect_layout_repeated_blocks, detect_pdf_document_noise, detect_text_format, extract_pdf_page_texts, @@ -35,6 +37,7 @@ from app.modules.data_process.algorithms import ( preprocess_structured_records_with_lineage, record_fingerprint, remove_document_noise, + remove_layout_repeated_blocks, score_quality, stable_split, stable_split_assignments, @@ -1143,3 +1146,95 @@ def test_generate_standard_records_rejects_out_of_range_count( ) -> None: with pytest.raises(ValueError, match=r"\[1, 50\]"): generate_standard_records([], qa_pairs_per_item=qa_pairs_per_item) + + +def test_layout_repeated_blocks_detects_repeating_header_table() -> None: + """跨页重复的页眉表格应被识别为重复块(出现 ≥ max(3, ceil(pages*0.3)) 次)。""" + + header_table = ( + "| 文件编码 | 2024 |\n" + "| - | - |\n" + "| 秘密等级 | 商密【中】 |\n" + "| 现行版本 | 1.0 |\n" + "| 页次 | 第1页 共47页 |\n" + ) + body_table = ( + "| 支出项目 | 税务票据要求 |\n" + "| - | - |\n" + "| 工资奖金 | 无 |\n" + "| 交通费 | 车票 |\n" + ) + doc_items: list[tuple[str, object, str]] = [] + for index in range(20): + # 20 个页面里 18 个有页眉表,2 个有正文表 + text = header_table if index < 18 else body_table + doc_items.append(("table", index, text)) + + blocks = detect_layout_repeated_blocks(doc_items, page_count=20) + + # 仅页眉表对应的标签序列应被识别 + assert len(blocks) == 1 + assert "文件编码" in blocks[0].labels + assert "秘密等级" in blocks[0].labels + assert blocks[0].occurrences == 18 + + +def test_layout_repeated_blocks_short_documents_skip() -> None: + """短文档(< 3 页)不推断重复块。""" + + doc_items: list[tuple[str, object, str]] = [ + ("table", 0, "| 文件编码 | 2024 |\n| - | - |\n"), + ("table", 1, "| 文件编码 | 2024 |\n| - | - |\n"), + ] + assert detect_layout_repeated_blocks(doc_items, page_count=2) == () + + +def test_remove_layout_repeated_blocks_strips_label_rows_and_separators() -> None: + """剔除首列命中重复标签集的行,及其后的表格分隔行。""" + + blocks = [ + LayoutRepeatedBlock( + labels=("文件编码", "秘密等级", "现行版本", "页次"), + occurrences=18, + ), + ] + chunk = ( + "报销指引\n" + "| 文件编码 | 2024 |\n" + "| - | - |\n" + "| 秘密等级 | 商密【中】 |\n" + "| 现行版本 | 1.0 |\n" + "| 页次 | 第3页 共47页 |\n" + "正文第一段\n" + "| 支出项目 | 税务票据要求 |\n" + "| - | - |\n" + "| 工资奖金 | 无 |\n" + ) + cleaned = remove_layout_repeated_blocks(chunk, blocks) + + # 重复标签行 + 紧随其后的表格分隔行被剔除;正文与内容表格保留 + assert "文件编码" not in cleaned + assert "秘密等级" not in cleaned + assert "现行版本" not in cleaned + assert "页次" not in cleaned + # 第一组表格的 | - | - | 在 文件编码 行之后被一并删除 + # (但 cleaned 中可能还有第二个表格的分隔行) + assert cleaned.count("| - | - |") == 1 + assert "报销指引" in cleaned + assert "正文第一段" in cleaned + assert "支出项目" in cleaned + assert "工资奖金" in cleaned + + +def test_remove_layout_repeated_blocks_returns_text_unchanged_when_no_blocks() -> None: + """无重复块时直接返回原文。""" + + chunk = "| 文件编码 | 2024 |\n| 正文 |\n" + assert remove_layout_repeated_blocks(chunk, []) == chunk + assert ( + remove_layout_repeated_blocks( + "", + [LayoutRepeatedBlock(labels=("x",), occurrences=5)], + ) + == "" + ) diff --git a/docker/app/Dockerfile.backend b/docker/app/Dockerfile.backend index 752ab6e..32673ed 100644 --- a/docker/app/Dockerfile.backend +++ b/docker/app/Dockerfile.backend @@ -14,11 +14,12 @@ RUN pip install --upgrade pip -i https://pypi.tuna.tsinghua.edu.cn/simple \ RUN python -c "import fastapi, uvicorn, psycopg, psycopg_pool, sqlalchemy, redis, jwt, passlib, httpx, minio, alembic; print('backend dependency check ok')" -RUN mkdir -p /opt/yg-ft/logs/backend /data/yg-ft \ - && chmod -R 0775 /opt/yg-ft /data/yg-ft +RUN mkdir -p /opt/yg-ft/logs/backend /data/yg-ft /opt/tiktoken_cache \ + && chmod -R 0775 /opt/yg-ft /data/yg-ft /opt/tiktoken_cache -# 离线打包 tiktoken cl100k_base 词表,避免无网环境下运行时联网下载 -COPY docker/app/tiktoken /opt/tiktoken_cache +# 离线打包 tiktoken cl100k_base 词表(与运行时 TIKTOKEN_CACHE_DIR 对齐), +# 文件名采用官方 SHA,避免代码走 fallback 重建 Encoding 的分支。 +COPY docker/app/tiktoken/9b5ad71b2ce5302211f9c61530b329a4922fc6a4 /opt/tiktoken_cache/ EXPOSE 8000 diff --git a/docker/app/tiktoken/cl100k_base b/docker/app/tiktoken/9b5ad71b2ce5302211f9c61530b329a4922fc6a4 similarity index 100% rename from docker/app/tiktoken/cl100k_base rename to docker/app/tiktoken/9b5ad71b2ce5302211f9c61530b329a4922fc6a4