阶段 当前状态 说明 Phase 0~3 基本完成 Document IR、证据链、人工确认、不可变 PlanData Snapshot、海报/PPT 投影已实现 Phase 4 代码完成 场景、策略、模板版本,校验/发布门禁,确定性 resolver 和严格槽位合并已实现 Phase 5 代码完成 输入冻结、PPTX/海报对账、失败关闭、幂等、心跳、重试和冻结输入重放已实现 Phase 6 基础完成 质量看板、结构化告警、Golden 审批、留存清理、孤儿检查和灰度开关已实现 正式上线 未完成 缺真实样本、生产模板、业务规则签字和灰度观察 目前验证基线: PPT/海报专项测试:107 passed, 1 skipped Vue TypeScript 检查:通过 前端生产构建:通过 代码变更仍在工作区,尚未提交 仓库全量测试仍有既有失败/挂起项,暂时不能宣称全仓测试完全绿色 仍未完成的代码任务主要有: 影子解析差异流水线 目前有灰度开关,但还没有完整的“新旧解析同时运行、字段差异入库、按保司/profile 聚合”的影子比较任务。 自动视觉回归 目前实现的是 DOM 模块、溢出、尺寸、文本和数值检查;还缺基于真实模板和标准图片的像素差异、字体缺失、遮挡和裁切回归。 告警通道接入 后台已经能产生结构化质量告警,但尚未自动推送到邮件、企微或其他通知通道。 留存任务生产化 dry-run、实删服务和失败审计已经具备,但尚未接入周期性 Celery/定时任务,也没有自动重试失败清理批次。 旧链路最终下线 旧 PPT 解析器和海报紧凑解析仍保留为回滚路径。需要全量灰度稳定后才能删除或彻底关闭写入口。 全仓测试收口 需要处理现有无关失败和挂起测试,建立真正全绿的 CI 基线。 仍需外部输入和生产环境完成的事项: 至少 30 份脱敏 Golden PDF,并完成双人标注和精确率验收。 业务专家确认派生公式、缺失值、可比较性和结论策略。 上传并标注真实生产 PPTX 的语义 shape、页面类型和容量。 安装生产字体并建立视觉基准图片。 实际执行数据库迁移 038~040。 完成留存 dry-run、实删演练以及 10% → 30% → 100% 灰度。 观察期通过后开启 SCENARIO_ENGINE_V2,目前它仍默认关闭;真实清理开关也默认关闭。 完整状态记录在 [PPT与海报Phase4至6补充实施记录](D:/work/code/python/coding/baodanagent/docs/PPT与海报Phase4至6补充实施记录_20260802.md)。
94 lines
4.6 KiB
Python
94 lines
4.6 KiB
Python
"""海报计划书上传模型。"""
|
|
from sqlalchemy import Column, String, Text, BigInteger, Integer, TIMESTAMP, func
|
|
from insurance.db.compat import db
|
|
|
|
|
|
class PosterCaseUpload(db.Model):
|
|
"""海报计划书上传记录表。"""
|
|
__tablename__ = "poster_case_uploads"
|
|
|
|
id = Column(BigInteger, primary_key=True, autoincrement=True)
|
|
user_id = Column(String(50), nullable=False, comment="用户 ID")
|
|
product_id = Column(String(50), nullable=False, comment="产品 ID")
|
|
product_source_type = Column(String(20), nullable=True, comment="library_product/user_material")
|
|
product_source_id = Column(String(64), nullable=True, comment="产品来源 ID")
|
|
product_snapshot_json = Column(Text, nullable=True, comment="产品信息快照 JSON")
|
|
source_file_url = Column(String(500), nullable=False, comment="源文件地址")
|
|
parse_status = Column(
|
|
String(20),
|
|
default="pending",
|
|
comment="解析状态: pending/queued/parsing/parsed/partial/failed",
|
|
)
|
|
parse_progress = Column(Integer, nullable=False, default=0, comment="解析进度 0-100")
|
|
parse_message = Column(String(500), nullable=False, default="", comment="解析进度说明")
|
|
parse_error = Column(Text, nullable=True, comment="解析失败原因")
|
|
parse_task_id = Column(String(200), nullable=True, comment="Celery 任务 ID")
|
|
parse_started_at = Column(TIMESTAMP, nullable=True)
|
|
parse_heartbeat_at = Column(TIMESTAMP, nullable=True)
|
|
parse_finished_at = Column(TIMESTAMP, nullable=True)
|
|
file_hash = Column(String(64), nullable=True, index=True, comment="源文件 SHA-256")
|
|
parsed_data = Column(Text, nullable=True, comment="系统解析结果 JSON")
|
|
confirmed_data = Column(Text, nullable=True, comment="人工核对后最终结果 JSON")
|
|
confirmed_by = Column(String(50), nullable=True, comment="核对人")
|
|
confirmed_at = Column(TIMESTAMP, nullable=True, comment="核对时间")
|
|
document_id = Column(BigInteger, nullable=True, index=True, comment="Document IR 文档 ID")
|
|
snapshot_id = Column(BigInteger, nullable=True, index=True, comment="最新 confirmed PlanData 快照 ID")
|
|
created_at = Column(TIMESTAMP, server_default=func.now())
|
|
|
|
def to_dict(self):
|
|
import json
|
|
|
|
def _safe_json(text):
|
|
if not text:
|
|
return None
|
|
try:
|
|
return json.loads(text)
|
|
except (json.JSONDecodeError, TypeError):
|
|
return None
|
|
|
|
def _safe_error(value):
|
|
if not value:
|
|
return ""
|
|
message = str(value).strip()
|
|
unsafe_markers = (
|
|
"traceback", "file \"", "\\", "/app/", "/api/",
|
|
"http://", "https://", "api_key", "token=",
|
|
)
|
|
if "\n" in message or any(marker in message.lower() for marker in unsafe_markers):
|
|
return "解析失败,请重试;如多次失败请联系管理员"
|
|
return message[:300]
|
|
|
|
return {
|
|
"id": self.id,
|
|
"userId": self.user_id,
|
|
"productId": self.product_id,
|
|
"productSource": {
|
|
"type": self.product_source_type or "library_product",
|
|
"id": self.product_source_id or self.product_id,
|
|
},
|
|
"productSnapshot": _safe_json(self.product_snapshot_json),
|
|
"sourceFileAvailable": bool(self.source_file_url),
|
|
"fileHash": self.file_hash,
|
|
"parseSnapshotHash": self._parse_snapshot_hash(),
|
|
"parseStatus": self.parse_status,
|
|
"parseProgress": self.parse_progress or 0,
|
|
"parseMessage": self.parse_message or "",
|
|
"parseError": _safe_error(self.parse_error),
|
|
"parseTaskId": self.parse_task_id,
|
|
"parseStartedAt": self.parse_started_at.isoformat() if self.parse_started_at else None,
|
|
"parseHeartbeatAt": self.parse_heartbeat_at.isoformat() if self.parse_heartbeat_at else None,
|
|
"parseFinishedAt": self.parse_finished_at.isoformat() if self.parse_finished_at else None,
|
|
"parsedData": _safe_json(self.parsed_data),
|
|
"confirmedData": _safe_json(self.confirmed_data),
|
|
"confirmedBy": self.confirmed_by,
|
|
"confirmedAt": self.confirmed_at.isoformat() if self.confirmed_at else None,
|
|
"documentId": self.document_id,
|
|
"snapshotId": self.snapshot_id,
|
|
"createdAt": self.created_at.isoformat() if self.created_at else None,
|
|
}
|
|
|
|
def _parse_snapshot_hash(self):
|
|
from insurance.plan_data.validators import parse_snapshot_hash
|
|
|
|
return parse_snapshot_hash(self)
|