baodan/api/insurance/poster/manual_parser.py

73 lines
2.2 KiB
Python
Raw Normal View History

2026-07-23 15:04:16 +08:00
"""保司小册子解析 — 调用 LLM 提取产品规则。"""
import json
import logging
from insurance.ppt.llm_client import llm_client
logger = logging.getLogger(__name__)
MANUAL_PARSE_PROMPT = """请从以下保险产品手册中提取产品规则和卖点信息。
要求输出 JSON 格式不要包含 markdown 代码块标记
{
"product_name": "产品名称",
"features": [
{"code": "唯一编码", "title": "卖点标题", "summary": "一句话描述"}
],
"currency_options": ["USD", "HKD"],
"coverage_highlights": ["保障亮点1", "保障亮点2"]
}
注意
- features 列表提取 3-8 个核心卖点
- currency_options 提取支持的货币选项
- coverage_highlights 提取 3-5 个保障亮点
- 所有内容必须基于原文不得虚构
"""
async def parse_manual_pdf(filepath: str) -> dict:
"""解析小册子 PDF提取产品规则。
参数:
filepath: PDF 文件路径
返回:
解析后的结构化 JSON
"""
# 提取 PDF 文本
from insurance.ppt.extraction import _extract_pdf_text
text = _extract_pdf_text(filepath)
if not text:
raise ValueError("无法提取 PDF 文本")
# 截取前 8000 字符避免 token 超限
text = text[:8000]
user_prompt = f"以下是保险产品手册内容:\n\n{text}"
result, _response = await llm_client.structured_output(
user_prompt,
MANUAL_PARSE_PROMPT,
schema={
"type": "object",
"properties": {
"product_name": {"type": "string"},
"features": {
"type": "array",
"items": {
"type": "object",
"properties": {
"code": {"type": "string"},
"title": {"type": "string"},
"summary": {"type": "string"},
},
},
},
"currency_options": {"type": "array", "items": {"type": "string"}},
"coverage_highlights": {"type": "array", "items": {"type": "string"}},
},
"required": ["product_name", "features"],
},
)
return result