Initial commit: QMDSearch 分层信息检索服务

- FastAPI + Qdrant + Redis + Ollama 技术栈
- L1→L2→L3→chunk 四层分层检索(dense + sparse RRF 融合)
- 文档三级总结与 2.5 级回退
- query 解析路由与分类
- /admin 管理页面
This commit is contained in:
2026-07-29 21:24:40 +08:00
commit 51dc8dc4f6
83 changed files with 10794 additions and 0 deletions
+306
View File
@@ -0,0 +1,306 @@
"""文档入库模块
入库流程:文档输入 → 三级总结(Ollama) → 分类判定(L1总结) → 切分 chunk
→ 构建 L2/L3 大纲节点 → 批量向量化(dense + sparse)→ 写入 Qdrant 四层集合
L2/L3 大纲节点的构建策略见 _build_l2_nodes / _build_l3_nodes。
"""
import uuid
from collections.abc import Callable
from typing import Any
import structlog
from app.config import settings
from app.core.chunker import Chunker
from app.core.classifier import Classifier
from app.core.embeddings import EmbeddingService, create_embedding_service
from app.core.headings import Heading, parse_headings
from app.core.sparse import SparseEncoder
from app.core.summarizer import Summarizer
from app.models.document import ChunkModel, DocumentInput, DocumentSummary, IngestionResult, SummaryLevel
from app.models.knowledge import CategoryResult
from app.services.qdrant import COLLECTION_L2, COLLECTION_L3, QdrantService, SparseVectorTuple
logger = structlog.get_logger()
# IngestionError 阶段标识
STAGE_SUMMARIZE = "summarize"
STAGE_CLASSIFY = "classify"
STAGE_EMBED = "embed"
STAGE_QDRANT = "qdrant"
class IngestionError(Exception):
"""入库失败异常
携带失败阶段(stage)与已产出的总结(summary,如有),
Qdrant 写入失败时总结不丢,上层可按阶段重试。
"""
def __init__(self, stage: str, message: str, summary: DocumentSummary | None = None) -> None:
super().__init__(message)
self.stage = stage
self.summary = summary
def _heading_paths(headings: list[Heading]) -> list[tuple[str, str]]:
"""按文档顺序计算每个标题的 (标题文本, 祖先标题链含自身),链用 " / " 连接"""
paths: list[tuple[str, str]] = []
stack: list[Heading] = []
for heading in headings:
# 遇到同级或更高级标题时弹栈,维护当前祖先链
while stack and stack[-1].level >= heading.level:
stack.pop()
stack.append(heading)
paths.append((heading.title, " / ".join(h.title for h in stack)))
return paths
def _split_l3_blocks(outline: str) -> list[tuple[str, str]]:
"""将内容大纲按 "## " 行分块,返回 (块标题, 块文本) 列表
"## " 行时整块作为一个节点(块标题为空);
首个 "## " 之前的引导内容直接忽略。
"""
stripped = outline.strip()
if not stripped:
return []
blocks: list[tuple[str, list[str]]] = []
for line in stripped.splitlines():
if line.startswith("## "):
blocks.append((line[3:].strip(), [line]))
elif blocks:
blocks[-1][1].append(line)
if not blocks:
return [("", stripped)]
return [(title, "\n".join(lines).strip()) for title, lines in blocks]
class Ingester:
"""文档入库器:编排总结、分类、切分、向量化与 Qdrant 写入全链路"""
def __init__(
self,
summarizer: Summarizer | None = None,
classifier: Classifier | None = None,
chunker: Chunker | None = None,
embedding: EmbeddingService | None = None,
sparse: SparseEncoder | None = None,
qdrant: QdrantService | None = None,
) -> None:
self.summarizer = summarizer or Summarizer()
self.classifier = classifier or Classifier()
self.chunker = chunker or Chunker()
self.embedding = embedding or create_embedding_service()
self.sparse = sparse or SparseEncoder()
self.qdrant = qdrant or QdrantService()
async def ingest(self, doc: DocumentInput, progress_cb: Callable[[str], None] | None = None) -> IngestionResult:
"""执行文档入库
Args:
doc: 文档输入(文本内容 + 元数据)
progress_cb: 可选的阶段进度回调(同步函数),在各阶段边界以
"summarizing" / "classifying" / "embedding" / "writing" 调用
Returns:
IngestionResult: 入库结果
Raises:
IngestionError: 任一阶段失败时抛出,携带 stage 与已产出总结
"""
def _report(stage: str) -> None:
if progress_cb is not None:
progress_cb(stage)
logger.info("开始文档入库", title=doc.title, text_length=len(doc.text))
doc_id = uuid.uuid4().hex
# 1. 三级总结
_report("summarizing")
try:
summary = await self.summarizer.summarize(doc.text, title=doc.title)
except Exception as exc:
logger.error("入库失败:三级总结", stage=STAGE_SUMMARIZE, error=str(exc))
raise IngestionError(STAGE_SUMMARIZE, f"三级总结失败: {exc}") from exc
logger.info("三级总结完成", doc_id=doc_id, level=summary.level.value)
# 2. 分类判定(基于 L1 总结)
_report("classifying")
try:
category = await self.classifier.classify(summary.l1_summary, title=doc.title)
except Exception as exc:
logger.error("入库失败:分类判定", stage=STAGE_CLASSIFY, error=str(exc))
raise IngestionError(STAGE_CLASSIFY, f"分类判定失败: {exc}", summary=summary) from exc
logger.info("分类判定完成", doc_id=doc_id, category=category.main_category, confidence=category.confidence)
# 3. 切分 chunk 并构建 L2/L3 大纲节点((text, section_path) 列表)
chunks = self.chunker.chunk(doc.text, doc_id)
l2_nodes = self._build_l2_nodes(doc.text, summary)
l3_nodes = self._build_l3_nodes(doc.text, summary)
# 4. 批量 embeddingL1 + L2 + L3 + chunks 一次调用,按序切片取向量
texts = [
summary.l1_summary,
*(node_text for node_text, _ in l2_nodes),
*(node_text for node_text, _ in l3_nodes),
*(c.text for c in chunks),
]
try:
_report("embedding")
vectors = await self.embedding.embed(texts)
except Exception as exc:
logger.error("入库失败:向量化", stage=STAGE_EMBED, error=str(exc))
raise IngestionError(STAGE_EMBED, f"向量化失败: {exc}", summary=summary) from exc
l1_vector = vectors[0]
l2_vectors = vectors[1 : 1 + len(l2_nodes)]
l3_vectors = vectors[1 + len(l2_nodes) : 1 + len(l2_nodes) + len(l3_nodes)]
chunk_vectors = vectors[1 + len(l2_nodes) + len(l3_nodes) :]
# 5. sparse 向量(仅 L1 与 chunks 需要)
l1_sparse: SparseVectorTuple | None = None
chunk_sparses: list[SparseVectorTuple | None] = [None] * len(chunks)
if settings.sparse_enabled:
l1_sparse = self.sparse.encode(summary.l1_summary)
chunk_sparses = [self.sparse.encode(c.text) for c in chunks]
# 6. 写入 Qdrant 四层集合
_report("writing")
try:
await self._write_qdrant(
doc_id,
doc,
summary,
category,
chunks,
l2_nodes,
l3_nodes,
l1_vector,
l2_vectors,
l3_vectors,
chunk_vectors,
l1_sparse,
chunk_sparses,
)
except Exception as exc:
logger.error("入库失败:Qdrant 写入", stage=STAGE_QDRANT, doc_id=doc_id, error=str(exc))
raise IngestionError(STAGE_QDRANT, f"Qdrant 写入失败: {exc}", summary=summary) from exc
logger.info(
"文档入库完成",
doc_id=doc_id,
chunks_count=len(chunks),
l2_nodes=len(l2_nodes),
l3_nodes=len(l3_nodes),
category=category.main_category,
)
return IngestionResult(
document_id=doc_id,
summary=summary,
category=category.main_category,
tags=category.tags,
category_confidence=category.confidence,
collection="四层集合",
chunks_count=len(chunks),
)
def _build_l2_nodes(self, text: str, summary: DocumentSummary) -> list[tuple[str, str]]:
"""构建 L2 大纲节点,返回 (text, section_path) 列表
- 有标题结构(L3 级且标题数 >= 2):每个标题一个节点,
text 与 section_path 均为该节点的祖先标题链
- 否则若 l2_outline 非空(LLM 生成的大纲):按非空行拆节点,section_path 为空
- 2.5 级文档(l2_outline 为 None):无 L2 节点
"""
headings = parse_headings(text)
if summary.level == SummaryLevel.L3 and len(headings) >= 2:
return [(path, path) for _, path in _heading_paths(headings)]
if summary.l2_outline:
return [(line.strip(), "") for line in summary.l2_outline.splitlines() if line.strip()]
return []
def _build_l3_nodes(self, text: str, summary: DocumentSummary) -> list[tuple[str, str]]:
"""构建 L3 内容大纲节点,返回 (text, section_path) 列表
"## " 分块(无 "## " 则整块一个节点);
section_path 尽力匹配文档标题链(块标题与文档标题文本精确匹配),匹配不到用 ""
"""
path_by_title: dict[str, str] = {}
for title, path in _heading_paths(parse_headings(text)):
path_by_title.setdefault(title, path)
return [
(block_text, path_by_title.get(block_title, ""))
for block_title, block_text in _split_l3_blocks(summary.l3_content_outline)
]
async def _write_qdrant(
self,
doc_id: str,
doc: DocumentInput,
summary: DocumentSummary,
category: CategoryResult,
chunks: list[ChunkModel],
l2_nodes: list[tuple[str, str]],
l3_nodes: list[tuple[str, str]],
l1_vector: list[float],
l2_vectors: list[list[float]],
l3_vectors: list[list[float]],
chunk_vectors: list[list[float]],
l1_sparse: SparseVectorTuple | None,
chunk_sparses: list[SparseVectorTuple | None],
) -> None:
"""将 L1/L2/L3/chunks 四层数据写入 Qdrant(任一失败向上抛出)"""
await self.qdrant.upsert_l1(
doc_id=doc_id,
title=doc.title,
summary=summary.l1_summary,
category=category.main_category,
tags=category.tags,
dense_vector=l1_vector,
sparse_vector=l1_sparse,
)
# L2/L3 大纲节点(为空时跳过对应集合的 upsert)
for collection, nodes, vectors in (
(COLLECTION_L2, l2_nodes, l2_vectors),
(COLLECTION_L3, l3_nodes, l3_vectors),
):
if not nodes:
continue
await self.qdrant.upsert_nodes(
collection,
[
{
"doc_id": doc_id,
"section_path": section_path,
"text": node_text,
"category": category.main_category,
"tags": category.tags,
"dense_vector": vector,
}
for (node_text, section_path), vector in zip(nodes, vectors, strict=True)
],
)
if chunks:
# chunk dict 额外携带 doc_summary(= L1 总结),检索侧直接取用,不用回查 L1
chunk_dicts: list[dict[str, Any]] = [
{
"doc_id": doc_id,
"chunk_index": chunk.chunk_index,
"text": chunk.text,
"section_path": chunk.section_path,
"title": doc.title,
"category": category.main_category,
"tags": category.tags,
"dense_vector": vector,
"sparse_vector": sparse,
"doc_summary": summary.l1_summary,
}
for chunk, vector, sparse in zip(chunks, chunk_vectors, chunk_sparses, strict=True)
]
await self.qdrant.upsert_chunks(chunk_dicts)