Files
group_xinghuo_jinrong/app/tool/document_parser.py
T
zhanghongyu_0626 2945108f66 feat(kb): Enhance knowledge base with new collections and search functionality
- Added new configuration for knowledge base root directory in `.env.example` and `settings.py`.
- Implemented `find_products` method in `CoreReadOnlyRepository` for fuzzy product search based on user queries.
- Introduced `search_cs_knowledge` function in `rag_service.py` to facilitate semantic search across new `fin_*` collections.
- Updated document parsing to support Markdown and YAML front-matter for knowledge base entries.
- Created multiple new FAQ and policy documents in the `data/kb_collections` directory to enrich the knowledge base.

This update significantly improves the knowledge retrieval capabilities for customer service interactions, ensuring more relevant and accurate responses.
2026-09-09 20:00:06 +08:00

66 lines
2.2 KiB
Python

"""文档解析:Markdown + YAML front-matter → Chunk(客服 fin_* 知识库目录)。"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
FRONT_MATTER_RE = re.compile(r"^---\s*\n(?P<body>.*?)\n---\s*\n", re.DOTALL)
@dataclass
class Chunk:
chunk_id: str
text: str
metadata: dict[str, Any] = field(default_factory=dict)
def _parse_front_matter(text: str) -> tuple[dict[str, str], str]:
match = FRONT_MATTER_RE.match(text)
if not match:
return {}, text.strip()
meta: dict[str, str] = {}
for line in match.group("body").splitlines():
line = line.strip()
if not line or line.startswith("#") or ":" not in line:
continue
key, _, value = line.partition(":")
meta[key.strip()] = value.strip()
return meta, text[match.end() :].strip()
def _slug(value: str, fallback: str) -> str:
cleaned = re.sub(r"[^A-Za-z0-9._-]+", "-", (value or fallback).strip())[:48]
return cleaned or fallback
def _parse_markdown_file(path: Path, chunk_no: int) -> Chunk:
raw = path.read_text(encoding="utf-8")
meta, body = _parse_front_matter(raw)
source_doc = meta.get("source_doc") or path.stem
chunk_text = body or meta.get("answer") or meta.get("question") or path.stem
text = chunk_text
if meta.get("question") and meta.get("answer"):
text = f"{meta['question']}\n{meta['answer']}"
elif meta.get("question"):
text = f"{meta['question']}\n{chunk_text}"
record_meta = dict(meta)
record_meta.setdefault("source_doc", source_doc)
record_meta["chunk_no"] = str(chunk_no)
record_meta["chunk_text"] = chunk_text
chunk_id = f"{_slug(source_doc, path.stem)}_{chunk_no}"
return Chunk(chunk_id=chunk_id, text=text, metadata=record_meta)
def parse_collection_dir(dir_path: Path) -> tuple[list[Chunk], dict[str, Any]]:
"""解析目录下全部 ``*.md``,每文件 1 块(chunk_no 从 1 递增)。"""
if not dir_path.is_dir():
return [], {"file_count": 0}
files = sorted(dir_path.glob("*.md"))
chunks = [_parse_markdown_file(path, i + 1) for i, path in enumerate(files)]
return chunks, {"file_count": len(files)}