feat: 新增多格式文件上传入库与认证体系

- 新增 JWT 认证模块,支持登录/注册/用户管理
- 新增文件上传接口,支持 .txt/.md/.html/.pdf/.docx 等格式解析入库
- 新增检索结果 AI 总结功能
- 新增文本去重缓存机制
- 新增全局认证夹具简化测试
- 新增配置项与环境变量支持
- 完善文档与测试覆盖
This commit is contained in:
2026-07-30 10:30:15 +08:00
parent 51dc8dc4f6
commit dce9e31bde
31 changed files with 3021 additions and 45 deletions
+312
View File
@@ -0,0 +1,312 @@
"""file_parser 单元测试:覆盖 txt/md/html/pdf/docx + 损坏文件 + 不支持扩展名 + PDF OCR 降级"""
import io
import sys
from typing import Any
import pytest
from app.config import settings
from app.core import file_parser as fp_module
from app.core.file_parser import parse_file, supported_extensions
def _make_minimal_pdf(text: str = "Hello PDF World") -> bytes:
"""构造一个含一页文本的最小 PDF(pypdf 可读出文本)"""
content_stream = f"BT /F1 24 Tf 100 700 Td ({text}) Tj ET".encode("latin-1")
content_obj = b"<< /Length " + str(len(content_stream)).encode() + b" >>\nstream\n" + content_stream + b"\nendstream"
return (
b"%PDF-1.0\n"
b"1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n"
b"2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n"
b"3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] "
b"/Resources << /Font << /F1 4 0 R >> >> /Contents 5 0 R >>\nendobj\n"
b"4 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>\nendobj\n"
b"5 0 obj\n" + content_obj + b"\nendobj\n"
b"xref\n0 6\n"
b"0000000000 65535 f\n"
b"0000000010 00000 n\n"
b"0000000059 00000 n\n"
b"0000000115 00000 n\n"
b"0000000241 00000 n\n"
b"0000000316 00000 n\n"
b"trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n414\n%%EOF\n"
)
def _make_docx(text_lines: list[str]) -> bytes:
from docx import Document # type: ignore[import-untyped]
document = Document()
for line in text_lines:
document.add_paragraph(line)
buf = io.BytesIO()
document.save(buf)
return buf.getvalue()
def test_parse_txt_returns_decoded_text() -> None:
"""txt:UTF-8 解码(含中文),无效字节 errors=replace 不抛错"""
text = "你好世界 hello"
assert parse_file("note.txt", text.encode("utf-8")) == text
# 无效 UTF-8 字节不抛错(errors=replace 兜底)
result = parse_file("bad.txt", b"\xff\xfe\x00invalid")
assert isinstance(result, str)
def test_parse_md_returns_decoded_text() -> None:
"""md:与 txt 走同一解析器"""
text = "# 标题\n\n正文内容"
assert parse_file("note.md", text.encode("utf-8")) == text
def test_parse_html_strips_tags() -> None:
"""html:剥离 script/style 与所有标签,仅保留可见文本"""
html = b"<html><body><h1>Hello</h1><p>World</p><script>x=1</script><style>p{}</style></body></html>"
result = parse_file("page.html", html)
assert "Hello" in result
assert "World" in result
assert "<" not in result
assert ">" not in result
assert "x=1" not in result
assert "p{}" not in result
def test_parse_htm_same_as_html() -> None:
"""htm:与 html 走同一解析器"""
html = b"<html><body><p>same content</p></body></html>"
assert parse_file("page.htm", html) == parse_file("page.html", html)
def test_parse_pdf_extracts_text() -> None:
"""pdf:从最小 PDF 中提取文本"""
pdf_bytes = _make_minimal_pdf("Hello PDF World")
result = parse_file("doc.pdf", pdf_bytes)
assert "Hello PDF World" in result
def test_parse_docx_extracts_paragraphs() -> None:
"""docx:提取段落文本"""
docx_bytes = _make_docx(["第一段落", "第二段落"])
result = parse_file("doc.docx", docx_bytes)
assert "第一段落" in result
assert "第二段落" in result
def test_parse_file_unsupported_extension_raises() -> None:
"""不支持的扩展名抛 ValueErrormessage 含扩展名"""
with pytest.raises(ValueError, match=r"不支持的文件类型: \.xlsx"):
parse_file("data.xlsx", b"binary")
def test_parse_file_no_extension_raises() -> None:
"""无扩展名抛 ValueError"""
with pytest.raises(ValueError, match=r"不支持的文件类型"):
parse_file("noext", b"text")
def test_parse_file_corrupted_pdf_raises() -> None:
"""损坏的 PDF 抛 ValueErrormessage 含'文件解析失败'"""
with pytest.raises(ValueError, match=r"文件解析失败"):
parse_file("bad.pdf", b"not a real pdf")
def test_parse_file_empty_html_returns_empty_string() -> None:
"""空 HTML 返回空字符串"""
assert parse_file("empty.html", b"<html></html>") == ""
def test_parse_file_html_with_script_style_excluded() -> None:
"""HTML 中 script/style 内容被排除"""
html = b"<html><body><p>visible</p><script>alert(1)</script><style>body{}</style></body></html>"
result = parse_file("page.html", html)
assert "visible" in result
assert "alert" not in result
assert "body{}" not in result
def test_parse_file_html_with_entities_decoded() -> None:
"""HTML 实体被解码"""
html = b"<html><body><p>Tom &amp; Jerry</p></body></html>"
result = parse_file("page.html", html)
assert "Tom & Jerry" in result
assert "&amp;" not in result
def test_supported_extensions_contains_expected_set() -> None:
"""supported_extensions 返回包含全部六种扩展名的集合"""
exts = supported_extensions()
assert {".txt", ".md", ".html", ".htm", ".pdf", ".docx"} <= exts
# --------------------------------------------------------------------------- #
# PDF OCR 降级路径测试(mock pypdf / pypdfium2 / rapidocr_onnxruntime,不真实下载模型)
# --------------------------------------------------------------------------- #
@pytest.fixture(autouse=True)
def _reset_ocr_state() -> Any:
"""每个 OCR 测试前后重置模块级 OCR 引擎状态,避免相互污染"""
saved_engine = fp_module._ocr_engine
saved_unavailable = fp_module._ocr_unavailable
yield
fp_module._ocr_engine = saved_engine
fp_module._ocr_unavailable = saved_unavailable
class _FakeTextPage:
"""pypdf PageObject 替身:返回固定文本"""
def __init__(self, text: str) -> None:
self._text = text
def extract_text(self) -> str:
return self._text
class _FakePdfReader:
"""pypdf.PdfReader 替身:构造时不解析,按预设页文本返回"""
def __init__(self, stream: Any) -> None:
self.pages = [_FakeTextPage(""), _FakeTextPage("")]
class _FakePilImage:
pass
class _FakeRenderResult:
def to_pil(self) -> _FakePilImage:
return _FakePilImage()
class _FakePdfiumPage:
def render(self, scale: float) -> _FakeRenderResult:
return _FakeRenderResult()
class _FakePdfDocument:
"""pypdfium2.PdfDocument 替身"""
def __init__(self, stream: Any) -> None:
self._n_pages = 2
def __len__(self) -> int:
return self._n_pages
def __getitem__(self, i: int) -> _FakePdfiumPage:
return _FakePdfiumPage()
def close(self) -> None:
pass
class _FakeOcrEngine:
"""rapidocr RapidOCR 替身:每次返回固定识别结果"""
def __init__(self) -> None:
self.call_count = 0
def __call__(self, image: Any) -> tuple[list[list[Any]], float]:
self.call_count += 1
# 返回 [[box, text, score], ...] 结构
return [[[0, 0], f"OCR文本第{self.call_count}", 0.95]], 0.1
def _patch_pdf_ocr_deps(monkeypatch: pytest.MonkeyPatch) -> None:
"""把 _parse_pdf/_ocr_pdf 内部用到的 pypdf / pypdfium2 / rapidocr_onnxruntime 全部替换"""
monkeypatch.setattr("pypdf.PdfReader", _FakePdfReader)
monkeypatch.setattr("pypdfium2.PdfDocument", _FakePdfDocument)
fake_module = type(sys)("rapidocr_onnxruntime")
fake_module.RapidOCR = _FakeOcrEngine
monkeypatch.setitem(sys.modules, "rapidocr_onnxruntime", fake_module)
def test_parse_pdf_ocr_fallback_when_text_layer_empty(monkeypatch: pytest.MonkeyPatch) -> None:
"""扫描件 PDF(文本层全空)触发 OCR 降级,返回识别文本"""
_patch_pdf_ocr_deps(monkeypatch)
monkeypatch.setattr(settings, "pdf_ocr_enabled", True)
monkeypatch.setattr(settings, "pdf_ocr_max_pages", 30)
monkeypatch.setattr(settings, "pdf_ocr_dpi", 200)
result = parse_file("scan.pdf", b"fake pdf bytes")
# 两页都跑了 OCR,每页返回一段文本
assert "OCR文本第1页" in result
assert "OCR文本第2页" in result
def test_parse_pdf_ocr_skipped_when_disabled(monkeypatch: pytest.MonkeyPatch) -> None:
"""pdf_ocr_enabled=False:文本层为空时直接返回空,不调 OCR"""
_patch_pdf_ocr_deps(monkeypatch)
monkeypatch.setattr(settings, "pdf_ocr_enabled", False)
result = parse_file("scan.pdf", b"fake pdf bytes")
assert result == ""
def test_parse_pdf_ocr_respects_max_pages(monkeypatch: pytest.MonkeyPatch) -> None:
"""max_pages=1:只 OCR 第一页,第二页跳过"""
_patch_pdf_ocr_deps(monkeypatch)
monkeypatch.setattr(settings, "pdf_ocr_enabled", True)
monkeypatch.setattr(settings, "pdf_ocr_max_pages", 1)
monkeypatch.setattr(settings, "pdf_ocr_dpi", 200)
result = parse_file("scan.pdf", b"fake pdf bytes")
assert "OCR文本第1页" in result
assert "OCR文本第2页" not in result
def test_parse_pdf_ocr_returns_empty_when_dependency_unavailable(monkeypatch: pytest.MonkeyPatch) -> None:
"""rapidocr 导入失败:降级返回空文本,且把 _ocr_unavailable 置 True 避免重试"""
monkeypatch.setattr("pypdf.PdfReader", _FakePdfReader)
# 故意让 rapidocr_onnxruntime 提供一个非类的 RapidOCR,构造时抛错
fake_module = type(sys)("rapidocr_onnxruntime")
def _boom(*args: Any, **kwargs: Any) -> None:
raise RuntimeError("model missing")
fake_module.RapidOCR = _boom # type: ignore[attr-defined]
monkeypatch.setitem(sys.modules, "rapidocr_onnxruntime", fake_module)
monkeypatch.setattr(settings, "pdf_ocr_enabled", True)
assert parse_file("scan.pdf", b"fake pdf bytes") == ""
assert fp_module._ocr_unavailable is True
def test_parse_pdf_ocr_runtime_exception_falls_back_to_empty(monkeypatch: pytest.MonkeyPatch) -> None:
"""OCR 运行时抛错:仅告警,降级返回空文本(不抛出 ValueError)"""
monkeypatch.setattr("pypdf.PdfReader", _FakePdfReader)
class _ExplodingPdfDocument:
def __init__(self, stream: Any) -> None:
raise RuntimeError("pdfium render failed")
monkeypatch.setattr("pypdfium2.PdfDocument", _ExplodingPdfDocument)
fake_module = type(sys)("rapidocr_onnxruntime")
fake_module.RapidOCR = _FakeOcrEngine
monkeypatch.setitem(sys.modules, "rapidocr_onnxruntime", fake_module)
monkeypatch.setattr(settings, "pdf_ocr_enabled", True)
assert parse_file("scan.pdf", b"fake pdf bytes") == ""
def test_parse_pdf_text_layer_present_skips_ocr(monkeypatch: pytest.MonkeyPatch) -> None:
"""文本层非空:直接返回文本,OCR 引擎不会被实例化"""
call_count = 0
class _CountingReader:
def __init__(self, stream: Any) -> None:
self.pages = [_FakeTextPage("这是文本层的内容")]
monkeypatch.setattr("pypdf.PdfReader", _CountingReader)
# 即便 OCR 依赖故意坏掉,也不应被调用
bad_module = type(sys)("rapidocr_onnxruntime")
bad_module.RapidOCR = lambda *a, **k: (_ for _ in ()).throw(RuntimeError("不应被调用")) # type: ignore[attr-defined]
monkeypatch.setitem(sys.modules, "rapidocr_onnxruntime", bad_module)
monkeypatch.setattr(settings, "pdf_ocr_enabled", True)
result = parse_file("text.pdf", b"fake pdf bytes")
assert result == "这是文本层的内容"
assert fp_module._ocr_engine is None