- 添加 txt 到允许上传格式,跳过 MinerU OCR - 直接读取 UTF-8 文本,按 3000 字符/页分页后送入 LangExtract - 适合有中文文本的规范文件,避免扫描件 OCR 乱码问题
This commit is contained in:
@@ -110,6 +110,7 @@ async def list_formats():
|
||||
{"ext": "jpg", "description": "JPEG 图片(单页)", "max_size_mb": 200, "max_pages": 1, "requires_ocr": True},
|
||||
{"ext": "jpeg", "description": "JPEG 图片(单页)", "max_size_mb": 200, "max_pages": 1, "requires_ocr": True},
|
||||
{"ext": "html", "description": "HTML 文件", "max_size_mb": 200, "max_pages": 600, "requires_ocr": False},
|
||||
{"ext": "txt", "description": "纯文本文件(UTF-8 编码)", "max_size_mb": 200, "max_pages": 0, "requires_ocr": False},
|
||||
],
|
||||
"ocr_languages": [
|
||||
{"code": "ch", "name": "中文(默认)"},
|
||||
|
||||
@@ -7,7 +7,7 @@ from pathlib import Path
|
||||
|
||||
from storage import file_store as fs
|
||||
|
||||
ALLOWED_EXTENSIONS = {"pdf", "docx", "doc", "pptx", "ppt", "png", "jpg", "jpeg", "html"}
|
||||
ALLOWED_EXTENSIONS = {"pdf", "docx", "doc", "pptx", "ppt", "png", "jpg", "jpeg", "html", "txt"}
|
||||
MAX_FILE_SIZE_MB = 200
|
||||
|
||||
|
||||
|
||||
@@ -80,6 +80,50 @@ def _run_pipeline(job_id: str) -> None:
|
||||
_update_meta(job_id, status="cancelled", stage="Cancelled")
|
||||
return
|
||||
|
||||
from pipeline.text_assembler import PageText, BlockSpan
|
||||
from pipeline.entity_extractor import create_model, extract_entities
|
||||
from pipeline.kg_builder import build_kg, extractions_to_records
|
||||
|
||||
ext = pdf_path.suffix.lower().lstrip(".")
|
||||
|
||||
if ext in ("txt",):
|
||||
# Plain text — skip MinerU, split by paragraphs
|
||||
_update_meta(job_id, status="parsing", stage="Parsing plain text...")
|
||||
raw = pdf_path.read_text(encoding="utf-8")
|
||||
|
||||
# Split into ~3000-char pages at double-newline boundaries
|
||||
PAGE_SIZE = 3000
|
||||
raw_pages = []
|
||||
pos = 0
|
||||
while pos < len(raw):
|
||||
end = min(pos + PAGE_SIZE, len(raw))
|
||||
if end < len(raw):
|
||||
# Try to break at a double newline
|
||||
br = raw.rfind("\n\n", pos, end)
|
||||
if br > pos + PAGE_SIZE // 2:
|
||||
end = br + 2
|
||||
raw_pages.append(raw[pos:end])
|
||||
pos = end
|
||||
|
||||
pages = []
|
||||
for i, text in enumerate(raw_pages):
|
||||
pages.append(PageText(
|
||||
page_idx=i,
|
||||
text=text.strip(),
|
||||
block_spans=[BlockSpan(
|
||||
block_index=0, block_type="text",
|
||||
page_idx=i, char_start=0,
|
||||
char_end=len(text.strip()), bbox=[0, 0, 0, 0],
|
||||
)],
|
||||
))
|
||||
total_pages = len(pages)
|
||||
block_types = {"text": total_pages}
|
||||
|
||||
# Save full.txt in mineru_output for reference
|
||||
mineru_out_dir = job_dir / "mineru_output"
|
||||
mineru_out_dir.mkdir(parents=True, exist_ok=True)
|
||||
(mineru_out_dir / "full.md").write_text(raw, encoding="utf-8")
|
||||
else:
|
||||
_update_meta(job_id, status="parsing", stage="MinerU document parsing...")
|
||||
mineru_out_dir = job_dir / "mineru_output"
|
||||
mineru_out_dir.mkdir(parents=True, exist_ok=True)
|
||||
@@ -96,7 +140,8 @@ def _run_pipeline(job_id: str) -> None:
|
||||
raise RuntimeError(f"MinerU failed: {result.stderr[:500]}")
|
||||
|
||||
# Find content_list.json in MinerU output
|
||||
# MinerU writes output to mineru_mvp/output/{stem}/
|
||||
from pipeline.text_assembler import load_content_list, assemble_pages, count_blocks_by_type
|
||||
|
||||
stem = pdf_path.stem
|
||||
mineru_default_out = MINERU_PIPELINE.parent / "output" / stem
|
||||
content_list_path = None
|
||||
@@ -105,12 +150,10 @@ def _run_pipeline(job_id: str) -> None:
|
||||
matches = list(mineru_default_out.glob("*_content_list.json"))
|
||||
if matches:
|
||||
content_list_path = matches[0]
|
||||
# Copy to our job dir
|
||||
import shutil
|
||||
shutil.copytree(str(mineru_default_out), str(mineru_out_dir), dirs_exist_ok=True)
|
||||
|
||||
if not content_list_path:
|
||||
# Fallback: search job mineru_output dir
|
||||
matches = list(mineru_out_dir.glob("*_content_list.json"))
|
||||
if matches:
|
||||
content_list_path = matches[0]
|
||||
@@ -118,15 +161,6 @@ def _run_pipeline(job_id: str) -> None:
|
||||
if not content_list_path or not content_list_path.exists():
|
||||
raise RuntimeError(f"MinerU output content_list.json not found. stdout: {result.stdout[:300]}")
|
||||
|
||||
# ── Stage 2: extracting ───────────────────────────────────────────
|
||||
if _cancel_flags.get(job_id):
|
||||
_update_meta(job_id, status="cancelled", stage="Cancelled")
|
||||
return
|
||||
|
||||
from pipeline.text_assembler import load_content_list, assemble_pages, count_blocks_by_type
|
||||
from pipeline.entity_extractor import create_model, extract_entities
|
||||
from pipeline.kg_builder import build_kg, extractions_to_records
|
||||
|
||||
content_list = load_content_list(content_list_path)
|
||||
pages = assemble_pages(content_list)
|
||||
total_pages = len(pages)
|
||||
|
||||
Reference in New Issue
Block a user