275 lines
12 KiB
Python
275 lines
12 KiB
Python
"""PPT 生成质量检查服务。
|
||
|
||
自动检查 PPTX 文件质量,并生成人工确认清单。
|
||
"""
|
||
import logging
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
|
||
class QualityChecker:
|
||
"""PPT 质量检查器。"""
|
||
|
||
def check(self, pptx_path: str, slides_data: dict | None,
|
||
extractions: list, expected_slide_count: int) -> dict:
|
||
"""执行质量检查,返回 qualityReport。
|
||
|
||
Args:
|
||
pptx_path: 生成的 PPTX 文件路径
|
||
slides_data: parse_slides 输出的结构化数据(可为 None)
|
||
extractions: 提取数据列表
|
||
expected_slide_count: 预期幻灯片数
|
||
|
||
Returns:
|
||
{"auto": [...], "manual": [...], "summary": {...}}
|
||
"""
|
||
auto = []
|
||
manual = []
|
||
|
||
# 1. 文件可打开
|
||
auto.append(self._check_file_openable(pptx_path))
|
||
|
||
# 2. 页数一致
|
||
actual_count = 0
|
||
if slides_data and slides_data.get("slides"):
|
||
actual_count = len(slides_data["slides"])
|
||
auto.append(self._check_slide_count(actual_count, expected_slide_count))
|
||
|
||
# 3. 无空白页
|
||
if slides_data and slides_data.get("slides"):
|
||
auto.append(self._check_blank_pages(slides_data["slides"]))
|
||
|
||
# 4. 必需页面存在
|
||
if slides_data and slides_data.get("slides"):
|
||
auto.append(self._check_required_pages(slides_data["slides"]))
|
||
|
||
# 5. 关键数据出现
|
||
auto.append(self._check_key_data(slides_data, extractions))
|
||
|
||
# 6. 文本溢出估算
|
||
if slides_data and slides_data.get("slides"):
|
||
overflow = self._check_text_overflow(slides_data["slides"])
|
||
auto.append(overflow)
|
||
|
||
# 人工确认项
|
||
manual = [
|
||
{"key": "confirm_customer_info", "confirmed": False,
|
||
"message": "请确认客户姓名和产品名称与原计划书一致"},
|
||
{"key": "confirm_premium_data", "confirmed": False,
|
||
"message": "请确认保费金额、缴费年期和回本年份正确"},
|
||
{"key": "confirm_benefit_table", "confirmed": False,
|
||
"message": "请确认利益演示表数据与原计划书一致"},
|
||
{"key": "confirm_compliance", "confirmed": False,
|
||
"message": "请确认营销表述和合规内容符合监管要求"},
|
||
]
|
||
|
||
pass_count = sum(1 for a in auto if a["status"] == "pass")
|
||
fail_count = sum(1 for a in auto if a["status"] == "fail")
|
||
warn_count = sum(1 for a in auto if a["status"] == "warn")
|
||
|
||
return {
|
||
"auto": auto,
|
||
"manual": manual,
|
||
"summary": {
|
||
"pass": pass_count,
|
||
"fail": fail_count,
|
||
"warn": warn_count,
|
||
"manualPending": len(manual),
|
||
},
|
||
}
|
||
|
||
@staticmethod
|
||
def _check_file_openable(pptx_path: str) -> dict:
|
||
"""检查 PPTX 可否正常打开。"""
|
||
try:
|
||
from pptx import Presentation
|
||
prs = Presentation(pptx_path)
|
||
count = len(prs.slides)
|
||
if count == 0:
|
||
return {"key": "file_openable", "status": "fail",
|
||
"message": "PPTX 文件可以打开但不包含幻灯片", "page": None}
|
||
return {"key": "file_openable", "status": "pass",
|
||
"message": f"PPTX 文件可正常打开({count} 页)", "page": None}
|
||
except Exception as e:
|
||
return {"key": "file_openable", "status": "fail",
|
||
"message": f"PPTX 文件无法打开: {e}", "page": None}
|
||
|
||
@staticmethod
|
||
def _check_slide_count(actual: int, expected: int) -> dict:
|
||
"""检查页数是否一致。"""
|
||
if expected <= 0:
|
||
return {"key": "slide_count_match", "status": "warn",
|
||
"message": "预期页数未知,无法比较", "page": None}
|
||
if actual == expected:
|
||
return {"key": "slide_count_match", "status": "pass",
|
||
"message": f"页数一致({actual} 页)", "page": None}
|
||
if abs(actual - expected) <= 1:
|
||
return {"key": "slide_count_match", "status": "warn",
|
||
"message": f"页数略有偏差: 实际 {actual} 页,预期 {expected} 页",
|
||
"page": None}
|
||
return {"key": "slide_count_match", "status": "fail",
|
||
"message": f"页数不一致: 实际 {actual} 页,预期 {expected} 页",
|
||
"page": None}
|
||
|
||
@staticmethod
|
||
def _check_blank_pages(slides: list) -> dict:
|
||
"""检查是否有空白或近似空白页。"""
|
||
blank_pages = []
|
||
for i, slide in enumerate(slides):
|
||
shapes = slide.get("shapes", [])
|
||
has_content = False
|
||
for shape in shapes:
|
||
stype = shape.get("type", "")
|
||
# 文本框有内容
|
||
if stype == "textbox":
|
||
for para in shape.get("paragraphs", []):
|
||
if para.get("text", "").strip():
|
||
has_content = True
|
||
break
|
||
# 图片、表格也算内容
|
||
elif stype in ("image", "table"):
|
||
has_content = True
|
||
# 有填充色的矩形也算内容(封面背景等)
|
||
elif stype == "rect" and shape.get("fill"):
|
||
has_content = True
|
||
# 分组形状中的子元素
|
||
elif stype == "group" and shape.get("children"):
|
||
has_content = True
|
||
if has_content:
|
||
break
|
||
if not has_content:
|
||
blank_pages.append(i + 1)
|
||
|
||
if not blank_pages:
|
||
return {"key": "no_blank_pages", "status": "pass",
|
||
"message": "所有页面均包含内容(文本/图片/表格/形状)", "page": None}
|
||
if len(blank_pages) <= 2:
|
||
return {"key": "no_blank_pages", "status": "warn",
|
||
"message": f"第 {', '.join(str(p) for p in blank_pages)} 页可能为空白页",
|
||
"page": blank_pages[0]}
|
||
return {"key": "no_blank_pages", "status": "fail",
|
||
"message": f"{len(blank_pages)} 页为空白: 第 {', '.join(str(p) for p in blank_pages)} 页",
|
||
"page": blank_pages[0]}
|
||
|
||
@staticmethod
|
||
def _check_required_pages(slides: list) -> dict:
|
||
"""检查必需页面类型是否存在(封面、数据页、总结页)。"""
|
||
if len(slides) < 2:
|
||
return {"key": "required_pages_present", "status": "warn",
|
||
"message": "幻灯片数量较少,无法判断必需页面", "page": None}
|
||
|
||
# 简单启发式:第 1 页应含封面关键词,最后 1 页应含总结关键词
|
||
first_texts = []
|
||
last_texts = []
|
||
for shape in slides[0].get("shapes", []):
|
||
for para in shape.get("paragraphs", []):
|
||
first_texts.append(para.get("text", ""))
|
||
for shape in slides[-1].get("shapes", []):
|
||
for para in shape.get("paragraphs", []):
|
||
last_texts.append(para.get("text", ""))
|
||
|
||
first_text = " ".join(first_texts).lower()
|
||
last_text = " ".join(last_texts).lower()
|
||
|
||
cover_keywords = ["方案", "计划", "客户", "保险", "财富"]
|
||
summary_keywords = ["总结", "总结", "summary", "感谢", "联系"]
|
||
|
||
has_cover = any(kw in first_text for kw in cover_keywords)
|
||
has_summary = any(kw in last_text for kw in summary_keywords)
|
||
|
||
if has_cover and has_summary:
|
||
return {"key": "required_pages_present", "status": "pass",
|
||
"message": "封面页和总结页均存在", "page": None}
|
||
missing = []
|
||
if not has_cover:
|
||
missing.append("封面")
|
||
if not has_summary:
|
||
missing.append("总结")
|
||
return {"key": "required_pages_present", "status": "warn",
|
||
"message": f"未检测到 {'、'.join(missing)} 页", "page": 1 if not has_cover else len(slides)}
|
||
|
||
@staticmethod
|
||
def _check_key_data(slides_data: dict | None, extractions: list) -> dict:
|
||
"""检查关键数据是否出现在幻灯片中。"""
|
||
if not slides_data or not slides_data.get("slides"):
|
||
return {"key": "key_data_present", "status": "warn",
|
||
"message": "无法获取幻灯片内容进行数据核对", "page": None}
|
||
|
||
# 收集所有幻灯片文本
|
||
all_text_parts = []
|
||
for slide in slides_data["slides"]:
|
||
for shape in slide.get("shapes", []):
|
||
for para in shape.get("paragraphs", []):
|
||
text = para.get("text", "").strip()
|
||
if text:
|
||
all_text_parts.append(text)
|
||
# 表格
|
||
for row in shape.get("rows", []):
|
||
for cell in row:
|
||
text = cell.get("text", "").strip()
|
||
if text:
|
||
all_text_parts.append(text)
|
||
all_text = " ".join(all_text_parts)
|
||
|
||
# 从提取数据中获取关键字段
|
||
missing = []
|
||
for ext in extractions:
|
||
if ext.get("status") not in ("success", "partial") or not ext.get("data"):
|
||
continue
|
||
data = ext["data"]
|
||
# 客户姓名
|
||
name = (data.get("insured", {}) or {}).get("name", "")
|
||
if name and name not in all_text:
|
||
missing.append(f"客户姓名({name})")
|
||
# 产品名称
|
||
product = data.get("product_name", "")
|
||
if product and product not in all_text:
|
||
missing.append(f"产品名称({product})")
|
||
# 年缴保费
|
||
policy = data.get("policy", {}) or {}
|
||
premium = policy.get("annual_premium") or policy.get("annualPremium", 0)
|
||
if premium and str(premium) not in all_text:
|
||
missing.append(f"年缴保费({premium})")
|
||
|
||
if not missing:
|
||
return {"key": "key_data_present", "status": "pass",
|
||
"message": "关键数据(客户名、产品名、保费)均已出现", "page": None}
|
||
return {"key": "key_data_present", "status": "warn",
|
||
"message": f"未找到: {', '.join(missing)}", "page": None}
|
||
|
||
@staticmethod
|
||
def _check_text_overflow(slides: list) -> dict:
|
||
"""估算文本是否可能溢出文本框。"""
|
||
overflow_pages = []
|
||
for i, slide in enumerate(slides):
|
||
for shape in slide.get("shapes", []):
|
||
if shape.get("type") != "textbox":
|
||
continue
|
||
box_w = shape.get("w", 0)
|
||
box_h = shape.get("h", 0)
|
||
if box_w <= 0 or box_h <= 0:
|
||
continue
|
||
total_text_len = 0
|
||
max_font = 14
|
||
for para in shape.get("paragraphs", []):
|
||
text = para.get("text", "")
|
||
total_text_len += len(text)
|
||
max_font = max(max_font, para.get("fontSize", 14))
|
||
# 粗略估算:每个中文字符占 fontSize 宽度,每行能放 box_w / fontSize 个字符
|
||
if max_font <= 0:
|
||
continue
|
||
chars_per_line = max(1, box_w / max_font)
|
||
est_lines = total_text_len / chars_per_line
|
||
line_height = max_font * 1.4
|
||
est_height = est_lines * line_height
|
||
if est_height > box_h * 1.3: # 超出 30% 视为溢出
|
||
overflow_pages.append(i + 1)
|
||
break
|
||
|
||
if not overflow_pages:
|
||
return {"key": "text_overflow", "status": "pass",
|
||
"message": "未检测到明显文本溢出", "page": None}
|
||
return {"key": "text_overflow", "status": "warn",
|
||
"message": f"第 {', '.join(str(p) for p in overflow_pages)} 页可能存在文本溢出",
|
||
"page": overflow_pages[0]}
|