Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 44bbe0f0e5 | |||
| 92680c47f9 | |||
| b335fbbcce | |||
| cc542f27d9 | |||
| d42ab5ffa8 | |||
| 74896727bb | |||
| 1f6d10f861 | |||
| 3ee5a4042d |
Executable
+116
@@ -0,0 +1,116 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""image_analysis v8 + storyboard v3 叙述优先重写版(架构大简化)
|
||||
|
||||
Revision ID: 105_narration_first
|
||||
Revises: 104_v8_display_markdown
|
||||
Create Date: 2026-10-07
|
||||
|
||||
变更:
|
||||
1. image_analysis v8:用「叙述优先」版整体替换 104 的 append 版——VLM 主交付物是
|
||||
自然叙述 summary_markdown,结构化字段仅保留 type/name/brand/has_person,
|
||||
顶层 products 改名 images;v8 active,其余 image_analysis 全部 deactivate。
|
||||
2. storyboard v3:整体替换为风格重写版(口播口语化、画面有画面感、
|
||||
copy_display_markdown 流畅叙述);v3 active,其余 storyboard deactivate。
|
||||
3. intent_parsing 类型模板全部 deactivate(意图解析步骤已删除)。
|
||||
模板内容直接取自 packages.application.viral_video.prompts.DEFAULT_TEMPLATES,
|
||||
保证代码默认值与 DB seed 完全一致。
|
||||
"""
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from alembic import op
|
||||
from packages.application.viral_video.prompts import DEFAULT_TEMPLATES
|
||||
|
||||
revision = "105_narration_first"
|
||||
down_revision = "104_v8_display_markdown"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def _tpl(prompt_type: str, version: int) -> dict:
|
||||
for t in DEFAULT_TEMPLATES:
|
||||
if t["prompt_type"] == prompt_type and t["version"] == version:
|
||||
return t
|
||||
raise RuntimeError("default template missing: %s v%s" % (prompt_type, version))
|
||||
|
||||
|
||||
def _upsert(bind, t: dict) -> None:
|
||||
existing = bind.execute(
|
||||
text("SELECT id FROM viral_video_prompt_templates " "WHERE prompt_type = :pt AND version = :ver"),
|
||||
{"pt": t["prompt_type"], "ver": t["version"]},
|
||||
).fetchone()
|
||||
params = {
|
||||
"pt": t["prompt_type"],
|
||||
"ver": t["version"],
|
||||
"name": t["name"],
|
||||
"sys": t["system_prompt"],
|
||||
"usr": t["user_prompt_template"],
|
||||
"ex": t.get("example_output", "") or "",
|
||||
}
|
||||
if existing:
|
||||
bind.execute(
|
||||
text(
|
||||
"UPDATE viral_video_prompt_templates SET name = :name, "
|
||||
"system_prompt = :sys, user_prompt_template = :usr, "
|
||||
"example_output = :ex, is_active = TRUE, updated_at = NOW() "
|
||||
"WHERE prompt_type = :pt AND version = :ver"
|
||||
),
|
||||
params,
|
||||
)
|
||||
else:
|
||||
bind.execute(
|
||||
text(
|
||||
"INSERT INTO viral_video_prompt_templates "
|
||||
"(prompt_type, version, name, system_prompt, user_prompt_template, "
|
||||
"example_output, is_active, created_at, updated_at) "
|
||||
"VALUES (:pt, :ver, :name, :sys, :usr, :ex, TRUE, NOW(), NOW())"
|
||||
),
|
||||
params,
|
||||
)
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
bind = op.get_bind()
|
||||
|
||||
# 1. image_analysis:停用全部后写入叙述优先 v8
|
||||
bind.execute(
|
||||
text("UPDATE viral_video_prompt_templates SET is_active = FALSE " "WHERE prompt_type = 'image_analysis'")
|
||||
)
|
||||
_upsert(bind, _tpl("image_analysis", 8))
|
||||
|
||||
# 2. storyboard:停用全部后写入重写版 v3
|
||||
bind.execute(text("UPDATE viral_video_prompt_templates SET is_active = FALSE " "WHERE prompt_type = 'storyboard'"))
|
||||
_upsert(bind, _tpl("storyboard", 3))
|
||||
|
||||
# 3. intent_parsing 已废弃:全部停用
|
||||
bind.execute(
|
||||
text("UPDATE viral_video_prompt_templates SET is_active = FALSE " "WHERE prompt_type = 'intent_parsing'")
|
||||
)
|
||||
|
||||
# 4. review 模板确保 active
|
||||
bind.execute(text("UPDATE viral_video_prompt_templates SET is_active = TRUE " "WHERE prompt_type = 'review'"))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
bind = op.get_bind()
|
||||
# 恢复 104 的 v8/v3 无法重建(内容已替换),仅把版本 active 状态回退:
|
||||
# 停用新版,尝试恢复 v7 / v2
|
||||
bind.execute(
|
||||
text(
|
||||
"UPDATE viral_video_prompt_templates SET is_active = FALSE "
|
||||
"WHERE prompt_type IN ('image_analysis','storyboard') "
|
||||
"AND version IN (8, 3)"
|
||||
)
|
||||
)
|
||||
bind.execute(
|
||||
text(
|
||||
"UPDATE viral_video_prompt_templates SET is_active = TRUE "
|
||||
"WHERE prompt_type = 'image_analysis' AND version = 7"
|
||||
)
|
||||
)
|
||||
bind.execute(
|
||||
text(
|
||||
"UPDATE viral_video_prompt_templates SET is_active = TRUE "
|
||||
"WHERE prompt_type = 'storyboard' AND version = 2"
|
||||
)
|
||||
)
|
||||
@@ -890,13 +890,28 @@ async def viral_video_websocket(websocket: WebSocket, job_id: str) -> None:
|
||||
job = job_repo.get(job_id)
|
||||
if job is not None:
|
||||
status_val = job.status.value if hasattr(job.status, "value") else str(job.status)
|
||||
# #P0: 初始快照必须包含前端重连/刷新所需的业务字段,
|
||||
# 结构对齐 worker 推送的 image_analyzed / copy_generated 事件。
|
||||
data: dict = {"status": status_val}
|
||||
ia = getattr(job, "image_analysis", None)
|
||||
if isinstance(ia, dict) and ia:
|
||||
data["image_analysis"] = ia
|
||||
cr = _build_copy_result(job)
|
||||
if isinstance(cr, dict) and cr:
|
||||
data["copy_result"] = cr
|
||||
gct = getattr(job, "generated_copy_text", "") or ""
|
||||
if gct:
|
||||
data["generated_copy_text"] = gct
|
||||
sb = getattr(job, "storyboard", None) or []
|
||||
if sb:
|
||||
data["storyboard"] = sb
|
||||
initial = {
|
||||
"type": "viral_video:progress",
|
||||
"job_id": job_id,
|
||||
"stage": _stage_from_status(job),
|
||||
"progress": _estimate_progress(job),
|
||||
"message": _initial_message(job),
|
||||
"data": {"status": status_val},
|
||||
"data": data,
|
||||
}
|
||||
await websocket.send_json(initial)
|
||||
# 已经终态 → 再发一条终态事件后立即关闭,避免占连接
|
||||
|
||||
@@ -65,29 +65,23 @@ export function isAnalysisStage(stage: ViralVideoStage | undefined): boolean {
|
||||
return isImageAnalysisStage(stage) || isCopyStage(stage)
|
||||
}
|
||||
|
||||
/** 单张图片 VLM 识别出的商品信息 */
|
||||
/** 单张图片 VLM 识别结果(v8 叙述优先,仅保留最少结构化字段) */
|
||||
export interface ImageProductAnalysis {
|
||||
/** store / product / person / scene */
|
||||
type?: string
|
||||
name?: string
|
||||
category?: string
|
||||
brand?: string
|
||||
colors?: string[]
|
||||
material_or_texture?: string
|
||||
key_features?: string[]
|
||||
visual_style?: string
|
||||
scene?: string
|
||||
target_audience_hint?: string
|
||||
text_on_image?: string
|
||||
/** 旧字段兼容 */
|
||||
spec?: string
|
||||
features?: string[] | string
|
||||
label_text?: string
|
||||
selling_points?: string
|
||||
image_index?: number
|
||||
/** v8: 用户端展示用的 markdown 描述(由提示词控制排版) */
|
||||
has_person?: boolean
|
||||
/** v8: 用户端展示用的叙述 markdown(由提示词控制排版) */
|
||||
summary_markdown?: string
|
||||
/** 标题行兼容字段 */
|
||||
category?: string
|
||||
}
|
||||
|
||||
export interface ImageAnalysisResult {
|
||||
/** v8 字段 */
|
||||
images?: ImageProductAnalysis[]
|
||||
/** 老数据兼容 */
|
||||
products?: ImageProductAnalysis[]
|
||||
}
|
||||
|
||||
|
||||
@@ -1208,8 +1208,10 @@ const ViralVideoPage: React.FC = () => {
|
||||
|
||||
/* ── 识别描述汇览渲染 ── */
|
||||
const renderRecognition = () => {
|
||||
const products: ImageProductAnalysis[] =
|
||||
(task.imageAnalysis?.products as ImageProductAnalysis[] | undefined) || []
|
||||
const images: ImageProductAnalysis[] =
|
||||
(task.imageAnalysis?.images as ImageProductAnalysis[] | undefined) ||
|
||||
(task.imageAnalysis?.products as ImageProductAnalysis[] | undefined) ||
|
||||
[]
|
||||
if (task.uiStep === "step1_analyzing") {
|
||||
return (
|
||||
<div className="vv-recog">
|
||||
@@ -1217,92 +1219,36 @@ const ViralVideoPage: React.FC = () => {
|
||||
<LoadingOutlined style={{ color: "#7c3aed", marginRight: 6 }} />
|
||||
识别描述汇览
|
||||
</div>
|
||||
<div className="vv-muted">AI 正在识别商品特征…</div>
|
||||
<div className="vv-muted">AI 正在识别画面…</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
if (products.length === 0) return null
|
||||
const featureText = (f: string[] | string | undefined) => {
|
||||
if (!f) return ""
|
||||
if (Array.isArray(f)) return f.join(";")
|
||||
return f
|
||||
}
|
||||
if (images.length === 0) return null
|
||||
return (
|
||||
<div className="vv-recog">
|
||||
<div className="vv-recog-title">
|
||||
<CheckCircleFilled style={{ color: "#10b981" }} />
|
||||
识别描述汇览
|
||||
</div>
|
||||
{products.map((p, i) =>
|
||||
p.summary_markdown ? (
|
||||
{images.map((p, i) => {
|
||||
const meta = [p.name || "未识别", p.brand, p.category].filter(Boolean)
|
||||
return (
|
||||
<div key={i} className="vv-recog-item vv-recog-md">
|
||||
<div
|
||||
className="vv-md-body"
|
||||
dangerouslySetInnerHTML={{ __html: renderMarkdown(p.summary_markdown) }}
|
||||
/>
|
||||
</div>
|
||||
) : (
|
||||
<div key={i} className="vv-recog-item">
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">图片{i + 1}:</span>
|
||||
<span>
|
||||
{p.name || "未识别"}
|
||||
{p.spec && <span className="vv-recog-meta">({p.spec})</span>}
|
||||
{p.brand && <span className="vv-recog-meta"> · {p.brand}</span>}
|
||||
{p.category && <span className="vv-recog-meta"> · {p.category}</span>}
|
||||
</span>
|
||||
<span>{meta.join(" · ")}</span>
|
||||
</div>
|
||||
{featureText(p.key_features ?? p.features) && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">核心特征:</span>
|
||||
<span className="vv-recog-v">{featureText(p.key_features ?? p.features)}</span>
|
||||
</div>
|
||||
)}
|
||||
{p.colors && p.colors.length > 0 && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">主色调:</span>
|
||||
<span className="vv-recog-v">{p.colors.join(" / ")}</span>
|
||||
</div>
|
||||
)}
|
||||
{p.material_or_texture && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">材质/纹理:</span>
|
||||
<span className="vv-recog-v">{p.material_or_texture}</span>
|
||||
</div>
|
||||
)}
|
||||
{p.visual_style && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">视觉风格:</span>
|
||||
<span className="vv-recog-v">{p.visual_style}</span>
|
||||
</div>
|
||||
)}
|
||||
{p.scene && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">场景:</span>
|
||||
<span className="vv-recog-v">{p.scene}</span>
|
||||
</div>
|
||||
)}
|
||||
{p.target_audience_hint && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">目标人群:</span>
|
||||
<span className="vv-recog-v">{p.target_audience_hint}</span>
|
||||
</div>
|
||||
)}
|
||||
{(p.text_on_image || p.label_text) && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">包装文字:</span>
|
||||
<span className="vv-recog-v">{p.text_on_image || p.label_text}</span>
|
||||
</div>
|
||||
)}
|
||||
{p.selling_points && (
|
||||
<div className="vv-recog-line">
|
||||
<span className="vv-recog-k">卖点:</span>
|
||||
<span className="vv-recog-v">{p.selling_points}</span>
|
||||
</div>
|
||||
{p.summary_markdown ? (
|
||||
<div
|
||||
className="vv-md-body"
|
||||
dangerouslySetInnerHTML={{ __html: renderMarkdown(p.summary_markdown) }}
|
||||
/>
|
||||
) : (
|
||||
<div className="vv-muted">(暂无叙述描述)</div>
|
||||
)}
|
||||
</div>
|
||||
),
|
||||
)}
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -92,28 +92,29 @@ def _save_job(repo, job, session):
|
||||
session.commit()
|
||||
|
||||
|
||||
def _start_trust_chain_preheat(job_id: str, products: list[dict]) -> None:
|
||||
def _start_trust_chain_preheat(job_id: str, images: list[dict]) -> None:
|
||||
"""#2172/#2174/#2220 后台启动信任链预热(Seedream t2i 文生图人像),不阻塞调用方。
|
||||
|
||||
#2220 修复:只对 has_person=True 的图(真人照片)生成 AI 人像替换,
|
||||
场景图/商品图/门店图保持原图不变,传给 Seedance 作为 reference_image 直接使用。
|
||||
|
||||
预热结果写入 job.pre_trusted_images:与 products 等长的稀疏列表,
|
||||
预热结果写入 job.pre_trusted_images:与 images 等长的稀疏列表,
|
||||
人像位是 AI 图 URL,非人像位是 None(表示保留原图)。
|
||||
"""
|
||||
# 构建人像位索引映射:person_indices[k] = products中第k个人像的位置
|
||||
# 构建人像位索引映射:person_indices[k] = images中第k个人像的位置
|
||||
_person_indices: list[int] = []
|
||||
_valid: list[str] = []
|
||||
for _i, _p in enumerate(products or []):
|
||||
for _i, _p in enumerate(images or []):
|
||||
if not isinstance(_p, dict):
|
||||
continue
|
||||
if not _p.get("has_person", False):
|
||||
continue
|
||||
_d = (_p.get("portrait_prompt") or "").strip()
|
||||
if not _d or "无人像" in _d or len(_d) < 10:
|
||||
# v8:人物描述直接取自叙述 summary_markdown
|
||||
_d = (_p.get("summary_markdown") or "").strip()
|
||||
if len(_d) < 10:
|
||||
continue
|
||||
_person_indices.append(_i)
|
||||
_valid.append(_d)
|
||||
_valid.append(_d[:600])
|
||||
if not _valid:
|
||||
logger.info("[trust-chain][preheat] 无有效人物描述(可能是纯商品/场景图),跳过预热 job=%s", job_id)
|
||||
return
|
||||
@@ -146,7 +147,7 @@ def _start_trust_chain_preheat(job_id: str, products: list[dict]) -> None:
|
||||
try:
|
||||
# #2220: 构建与 products 等长的稀疏列表,人像位放AI图URL,非人像位None
|
||||
_imgs2 = job2.images or []
|
||||
_sparse: list[str | None] = [None] * max(len(_imgs2), len(products or []))
|
||||
_sparse: list[str | None] = [None] * max(len(_imgs2), len(images or []))
|
||||
for _k, _url in enumerate(result):
|
||||
if _k < len(_person_indices):
|
||||
_sparse[_person_indices[_k]] = _url
|
||||
@@ -326,6 +327,7 @@ def _empty_copy_result(duration: int = 15, ratio: str = "9:16", theme: str = "")
|
||||
"voiceover_script": "",
|
||||
"final_copy": "",
|
||||
"suggested_copy": "",
|
||||
"copy_display_markdown": "",
|
||||
"title": "",
|
||||
}
|
||||
|
||||
@@ -376,25 +378,65 @@ def _normalize_image_url(raw: str, idx: int) -> str:
|
||||
return url
|
||||
|
||||
|
||||
def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
"""步骤 1: 图片分析(V2 主路径)。
|
||||
# ── v8 图片分析结构归一化(老数据兼容只在这里做一次)────────────────────
|
||||
def _coerce_image(obj: Any) -> dict | None:
|
||||
if not isinstance(obj, dict):
|
||||
return None
|
||||
img_type = str(obj.get("type") or "").strip()
|
||||
if img_type not in ("store", "product", "person", "scene"):
|
||||
# 老格式没有 type:按 has_person/category 粗判
|
||||
if obj.get("has_person"):
|
||||
img_type = "person"
|
||||
else:
|
||||
img_type = "product"
|
||||
img = {
|
||||
"type": img_type,
|
||||
"name": str(obj.get("name") or "未识别"),
|
||||
"brand": str(obj.get("brand") or ""),
|
||||
"has_person": bool(obj.get("has_person", False)),
|
||||
"summary_markdown": str(obj.get("summary_markdown") or "").strip(),
|
||||
}
|
||||
# 透传调试字段
|
||||
for _dk in ("_source", "_fast_elapsed", "_fallback_used"):
|
||||
if _dk in obj:
|
||||
img[_dk] = obj[_dk]
|
||||
return img
|
||||
|
||||
架构:
|
||||
- 主力:火山 MediaKit OCR(专用API,未配置时自动跳过)+ qwen3.8-flash 强约束 JSON,每图2路并行,目标<3s;
|
||||
- 外层全并发(workers=8),目标8图<15s;
|
||||
- 兜底:fast 结果不可用时单次调用 qwen3.7-plus(简单、无竞速)。
|
||||
- 唯一后端:阿里云百炼 DashScope,API Key 从环境变量 DASHSCOPE_API_KEY 读取。
|
||||
输出 dict 字段(name/brand/category/appearance/key_features/scene/mood/portrait_prompt/summary/_source)
|
||||
与旧版格式完全一致,下游信任链/t2i/intent_parsing/script_generation 零改动。
|
||||
|
||||
def normalize_image_analysis(image_analysis: Any) -> dict:
|
||||
"""读 DB 后统一成 {"images": [...]};老数据 products 数组在此一次性转换。"""
|
||||
if not isinstance(image_analysis, dict):
|
||||
return {"images": []}
|
||||
raw_images = image_analysis.get("images")
|
||||
if not isinstance(raw_images, list):
|
||||
raw_images = image_analysis.get("products") # 老字段
|
||||
images: list[dict] = []
|
||||
if isinstance(raw_images, list):
|
||||
for obj in raw_images:
|
||||
img = _coerce_image(obj)
|
||||
if img is not None:
|
||||
images.append(img)
|
||||
return {"images": images}
|
||||
|
||||
|
||||
def _images_of(image_analysis: Any) -> list[dict]:
|
||||
return normalize_image_analysis(image_analysis).get("images", []) or []
|
||||
|
||||
|
||||
def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
"""步骤 1: 图片分析(V2 主路径,v8 叙述优先)。
|
||||
|
||||
主力:火山 MediaKit OCR(未配置自动跳过)+ fast VLM 强约束 JSON,每图 2 路并行;
|
||||
外层全并发(workers=8);fast 不可用时单次 pro VLM 兜底。
|
||||
输出 {"images": [5 字段 image dict]}(type/name/brand/has_person/summary_markdown)。
|
||||
"""
|
||||
if not job.images:
|
||||
logger.warning("[爆款视频] 任务无 images,跳过图片分析")
|
||||
return {"products": []}
|
||||
return {"images": []}
|
||||
|
||||
# URL 归一化(storage_key→公网URL;空值直接400)
|
||||
normalized_urls: list[str] = []
|
||||
for idx, raw in enumerate(job.images):
|
||||
normalized_urls.append(_normalize_image_url(raw, idx))
|
||||
for _i, raw in enumerate(job.images):
|
||||
normalized_urls.append(_normalize_image_url(raw, _i))
|
||||
|
||||
try:
|
||||
from worker_app.tasks.vision import analyze_images_v2 as _aiv2
|
||||
@@ -403,11 +445,12 @@ def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
from tasks.vision import analyze_images_v2 as _aiv2 # type: ignore
|
||||
except ImportError as e:
|
||||
logger.error("[爆款视频] vision 模块导入失败: %s", e)
|
||||
return {"products": [_vision_fallback(0, f"vision_import_error:{e}")]}
|
||||
_fb = _vision_fallback(0, f"vision_import_error:{e}")
|
||||
return {"images": [img for img in (_coerce_image(_fb),) if img]}
|
||||
|
||||
# V2 内部 httpx 直连 dashscope,单次调用无重试,无需调整全局 client
|
||||
results = _aiv2(normalized_urls)
|
||||
return {"products": list(results)}
|
||||
images = [img for obj in results for img in (_coerce_image(obj),) if img is not None]
|
||||
return {"images": images}
|
||||
|
||||
|
||||
def _step_video_analysis(job: ViralVideoJob) -> dict | None:
|
||||
@@ -434,121 +477,12 @@ def _step_video_analysis(job: ViralVideoJob) -> dict | None:
|
||||
return {"error": str(e), "source": "failed"}
|
||||
|
||||
|
||||
def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
"""步骤 2: 用户文案意图解析(#2040:改为模板 + XML 解析)。"""
|
||||
try:
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template,
|
||||
render_system_prompt,
|
||||
render_user_prompt,
|
||||
)
|
||||
from packages.shared.ai_router import ai_router
|
||||
except ImportError:
|
||||
return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""}
|
||||
|
||||
_llm_client = ai_router.get_llm_client("intent_parsing")
|
||||
if not _llm_client.is_available:
|
||||
return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""}
|
||||
|
||||
products_summary = ""
|
||||
products = (image_analysis or {}).get("products", []) or []
|
||||
for p in products:
|
||||
if not isinstance(p, dict):
|
||||
continue
|
||||
feats = p.get("key_features") or p.get("features") or []
|
||||
extras = []
|
||||
if p.get("brand") and p.get("brand") not in ("未知", "无法判断"):
|
||||
extras.append(f"品牌={p['brand']}")
|
||||
if p.get("category") and p.get("category") not in ("无法判断", "非产品图"):
|
||||
extras.append(f"品类={p['category']}")
|
||||
if p.get("colors"):
|
||||
extras.append(f"颜色={','.join(p['colors'])}")
|
||||
if p.get("visual_style"):
|
||||
extras.append(f"风格={p['visual_style']}")
|
||||
feat_str = ", ".join([str(x) for x in feats + extras])
|
||||
products_summary += f"- {p.get('name', '产品')}: {feat_str}\n"
|
||||
|
||||
template = get_template("intent_parsing")
|
||||
system = render_system_prompt(template)
|
||||
marketing_purpose = getattr(job, "marketing_purpose", "") or "未指定"
|
||||
image_category_hint = _determine_theme(image_analysis, marketing_purpose)
|
||||
user = render_user_prompt(
|
||||
template,
|
||||
user_copy_text=job.user_copy_text or "(未提供,全由 AI 创作)",
|
||||
industry=job.industry or "未指定",
|
||||
image_analysis=products_summary or "- (无图片分析结果)",
|
||||
marketing_purpose=marketing_purpose,
|
||||
image_category_hint=image_category_hint,
|
||||
)
|
||||
|
||||
def _parse(raw: str) -> dict:
|
||||
summary = xp.text_of(raw, "intent_summary")
|
||||
msgs = [n["text"] for n in xp.find_all(raw, "message") if n["text"]]
|
||||
tone = xp.text_of(raw, "emotion_tone") or "亲切自然"
|
||||
title = xp.text_of(raw, "suggested_title") or xp.text_of(raw, "title")
|
||||
return {
|
||||
"intent": summary or "推广产品",
|
||||
"key_messages": msgs or ["产品亮点"],
|
||||
"tone": tone,
|
||||
"suggested_title": title,
|
||||
}
|
||||
|
||||
def _fallback(raw: str) -> dict:
|
||||
t = (job.user_copy_text or "").strip()
|
||||
return {
|
||||
"intent": t[:30] or "推广产品",
|
||||
"key_messages": [t[:80]] if t else ["产品亮点"],
|
||||
"tone": "专业",
|
||||
"suggested_title": "",
|
||||
}
|
||||
|
||||
# #2220: 直接用 ai_router 获取 client,不再手动提取 model_key
|
||||
from packages.shared.ai_router import ai_router
|
||||
|
||||
_client_fast = ai_router.get_llm_client("intent_parsing", variant="primary")
|
||||
_client_pro = ai_router.get_llm_client("intent_parsing", variant="lite")
|
||||
_intent_deadline = time.time() + 60
|
||||
_seen_models: set[str] = set()
|
||||
for _client, _lbl in [(_client_fast, "fast"), (_client_pro, "pro-fallback")]:
|
||||
if not _client or not _client.is_available:
|
||||
continue
|
||||
if _client.model in _seen_models:
|
||||
logger.info("[爆款视频] 意图解析跳过重复模型 %s label=%s", _client.model, _lbl)
|
||||
continue
|
||||
_seen_models.add(_client.model)
|
||||
if time.time() > _intent_deadline:
|
||||
logger.warning("[爆款视频] 意图解析超过60s总预算,跳过 label=%s", _lbl)
|
||||
break
|
||||
try:
|
||||
logger.info("[爆款视频] 意图解析 model=%s label=%s", _client.model, _lbl)
|
||||
raw = _client.chat_completion(
|
||||
[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
temperature=0.4,
|
||||
max_tokens=1024,
|
||||
timeout=20,
|
||||
)
|
||||
if not raw:
|
||||
continue
|
||||
parsed = _parse(raw)
|
||||
if parsed["intent"] or parsed["key_messages"]:
|
||||
return parsed
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 意图解析失败 label=%s err=%s", _lbl, e)
|
||||
return _fallback("")
|
||||
|
||||
|
||||
_PERSONA_STYLE_GUIDE = {
|
||||
"通用个人IP": "亲切自然、像朋友分享好物,第一人称口语化,不端着",
|
||||
"老板型IP": "沉稳大气、有行业格局感,适度使用『我做了XX年』『我一直坚持』等老板视角,语气自信不夸张",
|
||||
"专家型IP": "专业权威、讲原理和数据支撑,用词严谨,少用网梗,像行业专家做科普",
|
||||
"顾问型IP": "贴心周到、给建议给方案,多用『建议你』『可以试试』『我帮你梳理』",
|
||||
"创始人IP": "真诚有温度、讲品牌故事和创业初心,带点情怀和个人观点,不端老板架子",
|
||||
"创业者IP": "真实接地气、讲踩坑经验和创业心路,带点自嘲和韧劲,像身边的创业者朋友",
|
||||
"从业者经验派": "内行视角、讲行业内幕/实操经验/踩坑教训,多用『干了X年我发现』『内行都知道』",
|
||||
"避坑顾问型": "直接点出痛点和雷区,先讲『别买XX』『很多人踩过的坑』再给正确选择,节奏感强",
|
||||
"知识科普型": "清晰讲原理、讲知识点,条理分明、信息密度高,像做一期小科普",
|
||||
"测评种草型": "真实测评感、讲使用体验和优缺点对比,带『亲测』『我用了XX天』『实测下来』真实感词汇",
|
||||
# persona_id -> 文案风格指导(仅用于给脚本生成 LLM 的风格提示)
|
||||
_PERSONA_STYLE_GUIDE: dict[str, str] = {
|
||||
"宝妈": "温柔贴心,多从带娃和家庭实用角度分享,语气亲切有共鸣",
|
||||
"店主": "热情实在,像当面招呼客人,突出靠谱和实在优惠",
|
||||
"专业顾问": "专业可信,讲清原理和效果,用事实打消顾虑",
|
||||
"年轻达人": "活泼有网感,节奏轻快,金句和梗自然不尬",
|
||||
}
|
||||
|
||||
|
||||
@@ -564,54 +498,16 @@ def _persona_style_hint(persona_id: str) -> str:
|
||||
|
||||
|
||||
def _determine_theme(image_analysis: dict | None, marketing_purpose: str = "") -> str:
|
||||
"""根据图片分析结果和营销目的,智能推断默认主题。
|
||||
|
||||
门店类→门店探店/到店体验;商品图→好物分享/产品种草;
|
||||
人物图→穿搭/人物故事;场景图→场景氛围/空间体验。
|
||||
"""
|
||||
products = (image_analysis or {}).get("products", []) or []
|
||||
"""根据图片类型分布和营销目的推断默认主题。"""
|
||||
images = _images_of(image_analysis)
|
||||
type_counts: dict[str, int] = {}
|
||||
for p in products:
|
||||
if not isinstance(p, dict):
|
||||
continue
|
||||
cat = (p.get("category") or "").strip()
|
||||
if any(
|
||||
k in cat
|
||||
for k in (
|
||||
"门店",
|
||||
"店铺",
|
||||
"餐饮",
|
||||
"美容",
|
||||
"美发",
|
||||
"养生",
|
||||
"健身",
|
||||
"酒店",
|
||||
"咖啡",
|
||||
"奶茶",
|
||||
"餐厅",
|
||||
"颈肩",
|
||||
"调理",
|
||||
)
|
||||
):
|
||||
type_counts["store"] = type_counts.get("store", 0) + 1
|
||||
elif any(k in cat for k in ("人物", "穿搭", "人像", "服装")):
|
||||
type_counts["person"] = type_counts.get("person", 0) + 1
|
||||
elif any(k in cat for k in ("场景", "空间", "环境", "非产品")):
|
||||
type_counts["scene"] = type_counts.get("scene", 0) + 1
|
||||
elif cat and cat not in ("无法判断", "非产品图", ""):
|
||||
type_counts["product"] = type_counts.get("product", 0) + 1
|
||||
src = p.get("_source") or ""
|
||||
if "store" in src:
|
||||
type_counts["store"] = type_counts.get("store", 0) + 1
|
||||
elif "person" in src:
|
||||
type_counts["person"] = type_counts.get("person", 0) + 1
|
||||
|
||||
for img in images:
|
||||
t = img.get("type") or "product"
|
||||
type_counts[t] = type_counts.get(t, 0) + 1
|
||||
dominant = max(type_counts, key=type_counts.get) if type_counts else "product"
|
||||
mp = (marketing_purpose or "").strip()
|
||||
|
||||
if any(k in mp for k in ("获客", "引流", "到店")):
|
||||
if dominant == "store":
|
||||
return "门店探店·到店体验"
|
||||
return "门店探店·到店体验"
|
||||
if any(k in mp for k in ("品牌", "宣传")):
|
||||
return "品牌故事·门店体验" if dominant == "store" else "品牌故事·产品展示"
|
||||
@@ -627,72 +523,21 @@ def _determine_theme(image_analysis: dict | None, marketing_purpose: str = "") -
|
||||
return theme_map.get(dominant, "好物分享·产品种草")
|
||||
|
||||
|
||||
def _build_products_summary(image_analysis: dict) -> str:
|
||||
"""把 VLM 返回的商品分析结果拼给文案/分镜生成 prompt 用。
|
||||
优先用 summary(自然段落);没有时用结构化字段兜底拼一段。"""
|
||||
products = (image_analysis or {}).get("products", []) or []
|
||||
if not products:
|
||||
return "- (无图片信息,请自由创作自然生活化场景)"
|
||||
def _build_images_summary(image_analysis: dict) -> str:
|
||||
"""把图片叙述 summary_markdown 直接交给分镜 LLM;无叙述时只给一句基础提示。"""
|
||||
images = _images_of(image_analysis)
|
||||
if not images:
|
||||
return "- (无图片信息,请基于真实常见场景自由创作)"
|
||||
lines = []
|
||||
for i, p in enumerate(products):
|
||||
if not isinstance(p, dict):
|
||||
continue
|
||||
name = p.get("name") or "产品"
|
||||
# 优先 VLM 生成的 summary 段(自然语言,给编导模型看效果最好)
|
||||
summary = (p.get("summary") or "").strip()
|
||||
if summary and len(summary) >= 30:
|
||||
lines.append(f"- 图{i + 1} {name}:{summary}")
|
||||
continue
|
||||
# 结构化字段兜底
|
||||
brand = p.get("brand") or ""
|
||||
cat = p.get("category") or ""
|
||||
spec = p.get("spec") or ""
|
||||
appearance = p.get("appearance") or ""
|
||||
packaging = p.get("packaging") or ""
|
||||
colors = p.get("colors") or []
|
||||
mat = p.get("material_or_texture") or ""
|
||||
style = p.get("visual_style") or ""
|
||||
scene = p.get("scene") or ""
|
||||
audience = p.get("target_audience") or p.get("target_audience_hint") or ""
|
||||
# text_on_package 可能是数组(新格式)或字符串(旧格式)
|
||||
text_list = p.get("text_on_package") or []
|
||||
if isinstance(text_list, str):
|
||||
text_on_img = text_list
|
||||
for i, img in enumerate(images):
|
||||
md = (img.get("summary_markdown") or "").strip()
|
||||
brand = img.get("brand") or ""
|
||||
name = img.get("name") or "未识别"
|
||||
header = f"图{i + 1}(type={img.get('type')},名称={name},品牌={brand or '无'})"
|
||||
if md:
|
||||
lines.append(f"- {header}:\n{md}")
|
||||
else:
|
||||
text_on_img = ";".join([str(x) for x in text_list[:8]]) if text_list else (p.get("text_on_image") or "")
|
||||
feats = p.get("key_features") or p.get("features") or []
|
||||
sellings = p.get("selling_points") or []
|
||||
scenes = p.get("suitable_scenes") or []
|
||||
parts = [f"图{i + 1} {name}"]
|
||||
if brand and brand not in ("未知", "无法判断"):
|
||||
parts.append(f"品牌={brand}")
|
||||
if cat and cat not in ("无法判断", "非产品图"):
|
||||
parts.append(f"品类={cat}")
|
||||
if spec and spec != "无法判断":
|
||||
parts.append(f"规格={spec}")
|
||||
if appearance and appearance != "无法判断":
|
||||
parts.append(f"外观={appearance}")
|
||||
if packaging and packaging != "无法判断":
|
||||
parts.append(f"包装={packaging}")
|
||||
if colors:
|
||||
parts.append(f"颜色={','.join(colors)}")
|
||||
if mat and mat != "无法判断":
|
||||
parts.append(f"材质={mat}")
|
||||
if style:
|
||||
parts.append(f"风格={style}")
|
||||
if scene and scene not in ("通用",):
|
||||
parts.append(f"展示场景={scene}")
|
||||
if scenes:
|
||||
parts.append(f"适用场景={','.join([str(x) for x in scenes[:4]])}")
|
||||
if audience and audience not in ("通用", "无法判断"):
|
||||
parts.append(f"目标人群={audience}")
|
||||
if text_on_img and text_on_img not in ("无",):
|
||||
parts.append(f"包装文字={text_on_img[:300]}")
|
||||
if feats:
|
||||
parts.append("外观特征=" + ";".join([str(x) for x in feats[:6]]))
|
||||
if sellings:
|
||||
parts.append("营销卖点=" + ";".join([str(x) for x in sellings[:5]]))
|
||||
lines.append("- " + ",".join(parts))
|
||||
lines.append(f"- {header}:{name}" + (f",品牌{brand}" if brand else ""))
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
@@ -740,34 +585,22 @@ def _replace_henjin_everywhere(obj: Any) -> Any:
|
||||
|
||||
|
||||
def _fallback_script(job: ViralVideoJob) -> dict:
|
||||
"""脚本生成失败时的兜底脚本(极简但可用)。"""
|
||||
"""脚本生成失败时的极简兜底脚本。"""
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
ratio = getattr(job, "video_ratio", None) or "9:16"
|
||||
_ia = getattr(job, "image_analysis", None) or {}
|
||||
_mp = getattr(job, "marketing_purpose", "") or ""
|
||||
default_theme = _determine_theme(_ia, _mp)
|
||||
image_analysis = normalize_image_analysis(getattr(job, "image_analysis", None))
|
||||
default_theme = _determine_theme(image_analysis, getattr(job, "marketing_purpose", "") or "")
|
||||
base = _empty_copy_result(dur, ratio, theme=default_theme)
|
||||
_voiceover_map = {
|
||||
"store": "带你探店!今天来到这家店,环境真的超棒,服务也很到位,推荐大家来体验一下。",
|
||||
"person": "哈喽,今天给大家分享我的日常穿搭,简单舒适又好看,你们觉得怎么样?",
|
||||
"store": "带你探店!今天来到这家店,环境真的很不错,服务也很到位,推荐大家来体验一下。",
|
||||
"person": "哈喽,今天给大家分享我的日常,简单舒适又好看,你们觉得怎么样?",
|
||||
"scene": "带大家感受一下这个空间,氛围感拉满,真的很适合打卡体验。",
|
||||
"product": "你好,给大家分享一款我最近在用的好物,真的很不错,推荐你们也试试。",
|
||||
}
|
||||
_products = (_ia or {}).get("products", []) or []
|
||||
_dominant = "product"
|
||||
for p in _products:
|
||||
if not isinstance(p, dict):
|
||||
continue
|
||||
src = p.get("_source") or ""
|
||||
cat = p.get("category") or ""
|
||||
if "store" in src or any(k in cat for k in ("门店", "店铺", "餐饮", "美容", "颈肩", "调理")):
|
||||
_dominant = "store"
|
||||
break
|
||||
elif "person" in src or any(k in cat for k in ("人物", "穿搭", "人像")):
|
||||
_dominant = "person"
|
||||
break
|
||||
images = image_analysis.get("images", [])
|
||||
_dominant = (images[0].get("type") if images else "product") or "product"
|
||||
voiceover = job.user_copy_text or _voiceover_map.get(_dominant, _voiceover_map["product"])
|
||||
shots = [
|
||||
base["shots"] = [
|
||||
{
|
||||
"time_range": f"0-{dur}秒",
|
||||
"shot_type_angle_movement": "中景平视,缓慢推镜",
|
||||
@@ -778,245 +611,187 @@ def _fallback_script(job: ViralVideoJob) -> dict:
|
||||
"reference_image_index": 0 if job.images else None,
|
||||
}
|
||||
]
|
||||
base["shots"] = shots
|
||||
base["voiceover_script"] = voiceover
|
||||
base["final_copy"] = voiceover
|
||||
base["suggested_copy"] = voiceover
|
||||
base["copy_display_markdown"] = voiceover
|
||||
base["title"] = default_theme
|
||||
return base
|
||||
|
||||
|
||||
def _validate_and_normalize_script(raw, job: ViralVideoJob) -> dict:
|
||||
"""把 LLM 返回的脚本规范化、补默认、校验结构。"""
|
||||
"""JSON 分镜兜底解析:只解析结构、补缺失字段,不改写 LLM 文案。"""
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
ratio = getattr(job, "video_ratio", None) or "9:16"
|
||||
base = _empty_copy_result(dur, ratio)
|
||||
|
||||
if not isinstance(raw, dict):
|
||||
logger.warning("[爆款视频] 脚本返回非 dict,使用兜底")
|
||||
return _fallback_script(job)
|
||||
|
||||
# overview
|
||||
image_analysis = normalize_image_analysis(getattr(job, "image_analysis", None))
|
||||
default_theme = _determine_theme(image_analysis, getattr(job, "marketing_purpose", "") or "")
|
||||
base = _empty_copy_result(dur, ratio, theme=default_theme)
|
||||
|
||||
ov = raw.get("overview")
|
||||
theme = default_theme
|
||||
if isinstance(ov, dict):
|
||||
theme = str(ov.get("theme") or default_theme)
|
||||
base["overview"] = {
|
||||
"theme": str(
|
||||
ov.get("theme")
|
||||
or _determine_theme(getattr(job, "image_analysis", None), getattr(job, "marketing_purpose", ""))
|
||||
),
|
||||
"theme": theme,
|
||||
"total_duration": int(ov.get("total_duration") or dur),
|
||||
"aspect_ratio": str(ov.get("aspect_ratio") or ratio),
|
||||
}
|
||||
else:
|
||||
base["overview"]["theme"] = str(
|
||||
raw.get("title")
|
||||
or _determine_theme(getattr(job, "image_analysis", None), getattr(job, "marketing_purpose", ""))
|
||||
)
|
||||
base["title"] = str(raw.get("title") or theme)
|
||||
|
||||
base["scene_and_lighting"] = str(raw.get("scene_and_lighting") or base["scene_and_lighting"])
|
||||
if isinstance(raw.get("scene_and_lighting"), str) and raw["scene_and_lighting"]:
|
||||
base["scene_and_lighting"] = raw["scene_and_lighting"]
|
||||
|
||||
# shots
|
||||
shots_raw = raw.get("shots")
|
||||
shots: list[dict] = []
|
||||
if isinstance(shots_raw, list):
|
||||
for i, s in enumerate(shots_raw):
|
||||
if not isinstance(s, dict):
|
||||
continue
|
||||
shots.append(
|
||||
{
|
||||
"time_range": str(s.get("time_range") or f"{i * 3}-{(i + 1) * 3}秒"),
|
||||
"shot_type_angle_movement": str(s.get("shot_type_angle_movement") or "中景平视,固定镜头"),
|
||||
"scene_and_dialogue": str(s.get("scene_and_dialogue") or ""),
|
||||
"action_details": str(s.get("action_details") or ""),
|
||||
"audio_bgm": str(s.get("audio_bgm") or "轻快BGM"),
|
||||
"transition": str(s.get("transition") or ("硬切" if i < len(shots_raw) - 1 else "结束")),
|
||||
"reference_image_index": s.get("reference_image_index"),
|
||||
}
|
||||
)
|
||||
if not shots:
|
||||
shots = [
|
||||
shots_raw = raw.get("shots") if isinstance(raw.get("shots"), list) else []
|
||||
for i, s in enumerate(shots_raw):
|
||||
if not isinstance(s, dict):
|
||||
continue
|
||||
shots.append(
|
||||
{
|
||||
"time_range": f"0-{dur}秒",
|
||||
"shot_type_angle_movement": "中景平视,缓慢推镜",
|
||||
"scene_and_dialogue": "明亮室内场景,人物自然出镜。",
|
||||
"action_details": "自然展示产品",
|
||||
"audio_bgm": "轻快BGM",
|
||||
"transition": "结束",
|
||||
"reference_image_index": 0 if job.images else None,
|
||||
"time_range": str(s.get("time_range") or f"{i * 3}-{(i + 1) * 3}秒"),
|
||||
"shot_type_angle_movement": str(s.get("shot_type_angle_movement") or "中景平视,固定镜头"),
|
||||
"scene_and_dialogue": str(s.get("scene_and_dialogue") or ""),
|
||||
"action_details": str(s.get("action_details") or ""),
|
||||
"audio_bgm": str(s.get("audio_bgm") or "轻快BGM"),
|
||||
"transition": str(s.get("transition") or ("硬切" if i < len(shots_raw) - 1 else "结束")),
|
||||
"reference_image_index": s.get("reference_image_index"),
|
||||
}
|
||||
]
|
||||
)
|
||||
if not shots:
|
||||
return _fallback_script(job)
|
||||
base["shots"] = shots
|
||||
|
||||
# hard_constraints / negative_prompts
|
||||
hc = raw.get("hard_constraints")
|
||||
if isinstance(hc, list) and hc:
|
||||
merged = list(_DEFAULT_HARD_CONSTRAINTS)
|
||||
for x in hc:
|
||||
if isinstance(x, str) and x and x not in merged:
|
||||
merged.append(x)
|
||||
base["hard_constraints"] = merged
|
||||
np = raw.get("negative_prompts")
|
||||
if isinstance(np, list) and np:
|
||||
merged = list(_DEFAULT_NEGATIVE_PROMPTS)
|
||||
for x in np:
|
||||
if isinstance(x, str) and x and x not in merged:
|
||||
merged.append(x)
|
||||
base["negative_prompts"] = merged
|
||||
|
||||
# voiceover_script: 优先从字段取,否则从各镜 scene_and_dialogue 提取(粗暴拼接冒号后部分 / 中文句)
|
||||
voiceover = str(raw.get("voiceover_script") or "").strip()
|
||||
if not voiceover:
|
||||
# 兜底:把所有 scene_and_dialogue 拼接起来,去除镜头描述部分(含"景"、"俯拍"、"平视"等词的前缀)
|
||||
import re
|
||||
|
||||
parts = []
|
||||
for s in shots:
|
||||
txt = s.get("scene_and_dialogue", "")
|
||||
# 去除开头到第一个句号/逗号前的"镜头描述"部分
|
||||
# 简单策略:找第一个中文说话片段——按句号切,后半段更像对白
|
||||
segs = re.split(r"[。!?]", txt)
|
||||
for seg in segs:
|
||||
seg = seg.strip(" ,,。.!?!?::")
|
||||
if len(seg) >= 4 and not any(
|
||||
k in seg for k in ("景别", "俯拍", "仰拍", "平视", "镜头", "特写", "中景", "全景", "近景", "运镜")
|
||||
):
|
||||
parts.append(seg)
|
||||
voiceover = "。".join(parts) if parts else (job.user_copy_text or "你好,给大家分享一款好物。")
|
||||
voiceover = "。".join([s["scene_and_dialogue"] for s in shots if s.get("scene_and_dialogue")])
|
||||
if not voiceover:
|
||||
return _fallback_script(job)
|
||||
base["voiceover_script"] = voiceover
|
||||
base["final_copy"] = voiceover
|
||||
base["final_copy"] = str(raw.get("final_copy") or voiceover)
|
||||
base["suggested_copy"] = voiceover
|
||||
base["title"] = base["overview"]["theme"]
|
||||
if isinstance(raw.get("copy_display_markdown"), str) and raw["copy_display_markdown"]:
|
||||
base["copy_display_markdown"] = raw["copy_display_markdown"]
|
||||
else:
|
||||
base["copy_display_markdown"] = voiceover
|
||||
return base
|
||||
|
||||
|
||||
def _script_from_xml(raw: str, job: ViralVideoJob) -> dict | None:
|
||||
"""把 LLM 返回的 XML 分镜规范化为旧 copy_result 结构(供 Seedance 使用)。"""
|
||||
"""解析 v3 XML 分镜为 copy_result;只解析、补字段,不改 LLM 文案。
|
||||
|
||||
v3 结构:copy_display_markdown / clips>clip(index)[time_range,voiceover,
|
||||
visual,reference_image_index] / voiceover_script / theme。
|
||||
"""
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
ratio = getattr(job, "video_ratio", None) or "9:16"
|
||||
base = _empty_copy_result(dur, ratio)
|
||||
if not raw:
|
||||
return None
|
||||
base["overview"]["theme"] = (
|
||||
xp.text_of(raw, "overview_theme")
|
||||
or xp.text_of(raw, "title")
|
||||
or _determine_theme(getattr(job, "image_analysis", None), getattr(job, "marketing_purpose", ""))
|
||||
)
|
||||
est = xp.attr_int(xp.text_of(raw, "estimated_duration"), 0)
|
||||
if est:
|
||||
base["overview"]["total_duration"] = est
|
||||
sl = xp.text_of(raw, "scene_and_lighting")
|
||||
if sl:
|
||||
base["scene_and_lighting"] = sl
|
||||
|
||||
image_analysis = normalize_image_analysis(getattr(job, "image_analysis", None))
|
||||
default_theme = _determine_theme(image_analysis, getattr(job, "marketing_purpose", "") or "")
|
||||
theme = xp.text_of(raw, "theme") or default_theme
|
||||
base = _empty_copy_result(dur, ratio, theme=theme)
|
||||
base["overview"]["total_duration"] = dur
|
||||
base["title"] = theme
|
||||
|
||||
display_md = xp.text_of(raw, "copy_display_markdown")
|
||||
|
||||
clips = xp.find_all(raw, "clip")
|
||||
shots: list[dict] = []
|
||||
voice_parts: list[str] = []
|
||||
for i, c in enumerate(clips):
|
||||
a = c["attrs"]
|
||||
a = c.get("attrs", {})
|
||||
body = c.get("text", "") or ""
|
||||
ref_idx_raw = a.get("reference_image_index", "")
|
||||
if ref_idx_raw in (None, "", "null", "None"):
|
||||
body_ref = xp.text_of(body, "reference_image_index") if body else ""
|
||||
ref_idx = xp.attr_int(body_ref, 0) if body_ref else None
|
||||
else:
|
||||
ref_idx = xp.attr_int(ref_idx_raw, 0)
|
||||
shot = {
|
||||
"time_range": a.get("time_range") or f"{i * 3}-{(i + 1) * 3}秒",
|
||||
"shot_type_angle_movement": (xp.text_of(body, "shot_type_angle_movement") if body else "")
|
||||
or a.get("shot_type_angle_movement", "")
|
||||
or "中景平视,固定镜头",
|
||||
"scene_and_dialogue": (xp.text_of(body, "scene_and_dialogue") if body else "") or "",
|
||||
"action_details": (xp.text_of(body, "action_details") if body else "") or "",
|
||||
"audio_bgm": (xp.text_of(body, "audio_bgm") if body else "") or a.get("bgm_note", "") or "轻快BGM",
|
||||
"transition": (xp.text_of(body, "transition") if body else "")
|
||||
or a.get("transition", "")
|
||||
or ("硬切" if i < len(clips) - 1 else "结束"),
|
||||
"reference_image_index": ref_idx,
|
||||
}
|
||||
voice = xp.text_of(body, "voice_text") if body else ""
|
||||
voice = xp.text_of(body, "voiceover") or xp.text_of(body, "voice_text")
|
||||
visual = xp.text_of(body, "visual") or xp.text_of(body, "shot_type_angle_movement")
|
||||
ref_raw = xp.text_of(body, "reference_image_index")
|
||||
ref_idx = xp.attr_int(ref_raw, -1) if ref_raw not in (None, "") else -1
|
||||
if not isinstance(ref_idx, int) or ref_idx < 0:
|
||||
ref_idx = None
|
||||
if voice:
|
||||
voice_parts.append(voice)
|
||||
if not shot["scene_and_dialogue"]:
|
||||
shot["scene_and_dialogue"] = voice
|
||||
shot = {
|
||||
"time_range": a.get("time_range") or xp.text_of(body, "time_range") or f"{i * 3}-{(i + 1) * 3}秒",
|
||||
"shot_type_angle_movement": visual or "中景平视,固定镜头",
|
||||
"scene_and_dialogue": (voice or "") if not visual else f"{visual}|{voice or ''}",
|
||||
"action_details": xp.text_of(body, "action_details") or "",
|
||||
"audio_bgm": xp.text_of(body, "audio_bgm") or "轻快BGM",
|
||||
"transition": xp.text_of(body, "transition") or ("硬切" if i < len(clips) - 1 else "结束"),
|
||||
"reference_image_index": ref_idx,
|
||||
}
|
||||
shots.append(shot)
|
||||
|
||||
if not shots:
|
||||
return None
|
||||
base["shots"] = shots
|
||||
joined = xp.text_of(raw, "voiceover_script") or "。".join(voice_parts)
|
||||
base["voiceover_script"] = joined
|
||||
base["final_copy"] = joined
|
||||
base["suggested_copy"] = joined
|
||||
base["title"] = base["overview"]["theme"]
|
||||
# 提取 copy_display_markdown(用户端展示格式)
|
||||
copy_display_md = xp.text_of(raw, "copy_display_markdown")
|
||||
if copy_display_md:
|
||||
base["copy_display_markdown"] = copy_display_md
|
||||
|
||||
voiceover = xp.text_of(raw, "voiceover_script") or "。".join(voice_parts)
|
||||
if not voiceover.strip():
|
||||
return None
|
||||
base["voiceover_script"] = voiceover
|
||||
base["final_copy"] = voiceover
|
||||
base["suggested_copy"] = voiceover
|
||||
base["copy_display_markdown"] = display_md or voiceover
|
||||
return base
|
||||
|
||||
|
||||
def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: dict) -> dict:
|
||||
"""步骤 3: 编导分镜脚本生成(#2040:模板 + XML 解析;输出 copy_result 结构)。"""
|
||||
def _step_script_generation(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
"""步骤: 意图理解 + 分镜生成一次完成(v3,模板 + XML 解析)。"""
|
||||
try:
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template,
|
||||
render_system_prompt,
|
||||
render_user_prompt,
|
||||
)
|
||||
from packages.application.viral_video.prompts import (
|
||||
FUSION_INSTRUCTIONS,
|
||||
GLOBAL_CONSTRAINTS,
|
||||
NEGATIVE_RULES,
|
||||
)
|
||||
from packages.shared.ai_router import ai_router
|
||||
except ImportError:
|
||||
return _fallback_script(job)
|
||||
|
||||
_llm_client2 = ai_router.get_llm_client("storyboard")
|
||||
if not _llm_client2.is_available:
|
||||
if not ai_router.get_llm_client("storyboard").is_available:
|
||||
return _fallback_script(job)
|
||||
|
||||
products_summary = _build_products_summary(image_analysis)
|
||||
image_analysis = normalize_image_analysis(image_analysis)
|
||||
images_summary = _build_images_summary(image_analysis)
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
|
||||
intent_str = "推广产品"
|
||||
key_msgs = "产品亮点"
|
||||
tone = "亲切自然"
|
||||
if isinstance(intent, dict):
|
||||
intent_str = intent.get("intent") or intent_str
|
||||
key_msgs = "、".join(intent.get("key_messages") or []) or key_msgs
|
||||
tone = intent.get("tone") or tone
|
||||
|
||||
fusion_level = getattr(job, "fusion_level", "ai_polish") or "ai_polish"
|
||||
fusion_instruction = FUSION_INSTRUCTIONS.get(fusion_level, FUSION_INSTRUCTIONS["ai_polish"])
|
||||
|
||||
# 使用 storyboard 模板,注入融合指令/硬约束/反套路词
|
||||
template = get_template("storyboard")
|
||||
system_tpl = template.system_prompt
|
||||
system_tpl = system_tpl.replace("{fusion_instruction}", fusion_instruction)
|
||||
system_tpl = system_tpl.replace("{global_constraints}", GLOBAL_CONSTRAINTS)
|
||||
system_tpl = system_tpl.replace("{negative_rules}", NEGATIVE_RULES)
|
||||
|
||||
system = render_system_prompt(template)
|
||||
marketing_purpose = getattr(job, "marketing_purpose", "") or "未指定"
|
||||
image_category_hint = _determine_theme(image_analysis, marketing_purpose)
|
||||
fusion_brief = (
|
||||
f"意图:{intent_str}\n关键信息:{key_msgs}\n调性:{tone}\n"
|
||||
f"营销目的:{marketing_purpose}\n建议主题方向:{image_category_hint}\n"
|
||||
f"用户原文:{job.user_copy_text or '(未提供)'}\n创作模式:{fusion_level}"
|
||||
)
|
||||
theme_hint = _determine_theme(image_analysis, marketing_purpose)
|
||||
|
||||
style_section = ""
|
||||
if isinstance(getattr(job, "style_guide", None), dict):
|
||||
style_section = (
|
||||
"<reference_video_style>"
|
||||
+ json.dumps(job.style_guide, ensure_ascii=False)[:1200]
|
||||
+ "</reference_video_style>"
|
||||
)
|
||||
|
||||
user = render_user_prompt(
|
||||
template,
|
||||
duration=dur,
|
||||
image_count=len(job.images or []),
|
||||
fusion_result=fusion_brief,
|
||||
image_analysis=products_summary,
|
||||
marketing_purpose=marketing_purpose,
|
||||
image_summary=images_summary,
|
||||
theme_hint=theme_hint,
|
||||
dur=str(dur),
|
||||
aspect_ratio=getattr(job, "video_ratio", None) or "9:16",
|
||||
tone=getattr(job, "tone", "") or "亲切自然",
|
||||
target_audience=getattr(job, "target_audience", "") or "未指定",
|
||||
extra_requirements=job.user_copy_text or "(未提供额外要求,由 AI 创作)",
|
||||
video_style_section=style_section,
|
||||
)
|
||||
|
||||
def _try_gen(client, temp: float, max_tok: int, label: str, tmo: int = 25):
|
||||
def _try_gen(client, temp: float, max_tok: int, label: str, tmo: int):
|
||||
if not client or not client.is_available:
|
||||
return None
|
||||
logger.info("[爆款视频] 编导脚本生成 model=%s label=%s timeout=%d", client.model, label, tmo)
|
||||
logger.info("[爆款视频] 分镜生成 model=%s label=%s timeout=%d", client.model, label, tmo)
|
||||
raw = client.chat_completion(
|
||||
[{"role": "system", "content": system_tpl}, {"role": "user", "content": user}],
|
||||
[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
temperature=temp,
|
||||
max_tokens=max_tok,
|
||||
timeout=tmo,
|
||||
@@ -1025,23 +800,16 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
return None
|
||||
normalized = _script_from_xml(raw, job)
|
||||
if normalized is None:
|
||||
# 兼容:万一 LLM 仍输出 JSON,走旧规范化
|
||||
parsed_json = _safe_json_loads(raw)
|
||||
if isinstance(parsed_json, dict):
|
||||
normalized = _validate_and_normalize_script(parsed_json, job)
|
||||
else:
|
||||
return None
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
shots_cnt = len((normalized or {}).get("shots") or [])
|
||||
_before_dump = json.dumps(normalized, ensure_ascii=False)
|
||||
if "很近" in _before_dump:
|
||||
normalized = _replace_henjin_everywhere(normalized)
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
fallback_marker = "我最近在用的好物" in voiceover
|
||||
has_typo_henjin = "很近" in json.dumps(normalized, ensure_ascii=False)
|
||||
is_fallback = fallback_marker or shots_cnt < 1 or len(voiceover) < 20 or has_typo_henjin
|
||||
voiceover = normalized.get("voiceover_script") or ""
|
||||
shots_cnt = len(normalized.get("shots") or [])
|
||||
is_fallback = shots_cnt < 1 or len(voiceover) < 12
|
||||
logger.info(
|
||||
"[爆款视频] 编导脚本结果 label=%s voiceover_len=%d shots=%d fallback=%s",
|
||||
"[爆款视频] 分镜结果 label=%s voiceover_len=%d shots=%d fallback=%s",
|
||||
label,
|
||||
len(voiceover),
|
||||
shots_cnt,
|
||||
@@ -1049,38 +817,28 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
)
|
||||
return None if is_fallback else normalized
|
||||
|
||||
# #2220: 直接用 ai_router 获取 client,不再手动提取 model_key
|
||||
_client_fast = ai_router.get_llm_client("storyboard", variant="primary")
|
||||
_client_pro = ai_router.get_llm_client("storyboard", variant="lite")
|
||||
_script_fast_tmo = int(os.environ.get("VIRAL_VIDEO_SCRIPT_FAST_TIMEOUT", "90"))
|
||||
_script_pro_tmo = int(os.environ.get("VIRAL_VIDEO_SCRIPT_PRO_TIMEOUT", "60"))
|
||||
_script_deadline = time.time() + 180
|
||||
client_fast = ai_router.get_llm_client("storyboard", variant="primary")
|
||||
client_pro = ai_router.get_llm_client("storyboard", variant="lite")
|
||||
fast_tmo = int(os.environ.get("VIRAL_VIDEO_SCRIPT_FAST_TIMEOUT", "90"))
|
||||
pro_tmo = int(os.environ.get("VIRAL_VIDEO_SCRIPT_PRO_TIMEOUT", "60"))
|
||||
deadline = time.time() + 180
|
||||
try:
|
||||
# #2233: fast_timeout=90s, pro_timeout=60s,总deadline 180s
|
||||
normalized = _try_gen(_client_fast, 0.8, 2500, "fast-first", tmo=_script_fast_tmo)
|
||||
if normalized is not None:
|
||||
return normalized
|
||||
if time.time() > _script_deadline:
|
||||
logger.warning("[爆款视频] 编导脚本超过180s总预算,使用兜底脚本")
|
||||
result = _try_gen(client_fast, 0.8, 2500, "fast-first", fast_tmo)
|
||||
if result is not None:
|
||||
return result
|
||||
if time.time() > deadline:
|
||||
return _fallback_script(job)
|
||||
normalized = _try_gen(_client_fast, 0.6, 3200, "fast-retry", tmo=_script_fast_tmo)
|
||||
if normalized is not None:
|
||||
return normalized
|
||||
# 第三次:用 lite/pro 模型兜底,跳过与 primary 相同的模型
|
||||
if _client_pro and _client_pro.is_available:
|
||||
if _client_pro.model != _client_fast.model:
|
||||
if time.time() <= _script_deadline:
|
||||
normalized = _try_gen(_client_pro, 0.7, 3500, "pro-fallback", tmo=_script_pro_tmo)
|
||||
if normalized is not None:
|
||||
return normalized
|
||||
else:
|
||||
logger.warning("[爆款视频] 编导脚本超过180s总预算,跳过pro-fallback")
|
||||
else:
|
||||
logger.info("[爆款视频] pro-fallback模型与primary相同(%s),跳过重复调用", _client_pro.model)
|
||||
logger.warning("[爆款视频] 编导脚本均未生成合格结果,使用兜底脚本")
|
||||
result = _try_gen(client_fast, 0.6, 3200, "fast-retry", fast_tmo)
|
||||
if result is not None:
|
||||
return result
|
||||
if client_pro and client_pro.is_available and client_pro.model != client_fast.model:
|
||||
if time.time() <= deadline:
|
||||
result = _try_gen(client_pro, 0.7, 3500, "pro-fallback", pro_tmo)
|
||||
if result is not None:
|
||||
return result
|
||||
return _fallback_script(job)
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 编导脚本生成异常: %s,使用兜底脚本", e, exc_info=True)
|
||||
logger.warning("[爆款视频] 分镜生成异常: %s,使用兜底脚本", e, exc_info=True)
|
||||
return _fallback_script(job)
|
||||
|
||||
|
||||
@@ -1413,18 +1171,18 @@ def _step_render(job: ViralVideoJob, copy_result: dict, tts_audio_url: str | Non
|
||||
try:
|
||||
from packages.shared.ai_service import preheat_trust_chain
|
||||
|
||||
_ia = getattr(job, "image_analysis", None) or {}
|
||||
_prods = (_ia.get("products") if isinstance(_ia, dict) else None) or []
|
||||
_ia = normalize_image_analysis(getattr(job, "image_analysis", None))
|
||||
_imgs = _ia.get("images", [])
|
||||
_live_person_idx: list[int] = []
|
||||
_live_pdescs: list[str] = []
|
||||
for _i, _pp in enumerate(_prods):
|
||||
for _i, _pp in enumerate(_imgs):
|
||||
if not isinstance(_pp, dict) or not _pp.get("has_person", False):
|
||||
continue
|
||||
_d = (_pp.get("portrait_prompt") or "").strip()
|
||||
if not _d or "无人像" in _d or len(_d) < 10:
|
||||
_d = (_pp.get("summary_markdown") or "").strip()
|
||||
if len(_d) < 10:
|
||||
continue
|
||||
_live_person_idx.append(_i)
|
||||
_live_pdescs.append(_d)
|
||||
_live_pdescs.append(_d[:600])
|
||||
if _live_pdescs:
|
||||
_t0 = time.time()
|
||||
_live_urls = preheat_trust_chain(_live_pdescs, timeout=120)
|
||||
@@ -1569,9 +1327,9 @@ def run_viral_video_pipeline(self: Task, job_id: str) -> dict:
|
||||
# #2174: VLM完成后立即启动信任链t2i预热(从VLM结果提取人物描述),与文案阶段并行
|
||||
if job.images:
|
||||
try:
|
||||
_products = (image_analysis or {}).get("products", []) or []
|
||||
# #2220: 直接传 products 列表,由 _start_trust_chain_preheat 内部按 has_person 筛选
|
||||
_start_trust_chain_preheat(job.id, _products)
|
||||
_imgs = (image_analysis or {}).get("images", []) or []
|
||||
# 直接传 images 列表,内部按 has_person 筛选
|
||||
_start_trust_chain_preheat(job.id, _imgs)
|
||||
except Exception as _e:
|
||||
logger.warning("[爆款视频][阶段1] 启动信任链t2i预热失败: %s", _e)
|
||||
_save_job(repo, job, session)
|
||||
@@ -1585,9 +1343,8 @@ def run_viral_video_pipeline(self: Task, job_id: str) -> dict:
|
||||
_save_job(repo, job, session)
|
||||
_emit_progress(job_id, ViralVideoStage.VIDEO_ANALYSIS, 25.0, "风格分析完成", {"style_guide": style_guide})
|
||||
|
||||
_set_stage(job, repo, session, ViralVideoStage.INTENT_PARSING, "正在解析文案意图...")
|
||||
intent_result = _step_intent_parsing(job, image_analysis)
|
||||
|
||||
# v8:意图解析已并入分镜生成,旧流水线不再单独解析
|
||||
intent_result = {"intent": "", "key_messages": [], "tone": "", "suggested_title": ""}
|
||||
job.mark_wait_user_confirm(intent_result)
|
||||
job.current_stage = ViralVideoStage.INTENT_PARSING
|
||||
job.phase_message = "意图解析完成,等待用户确认"
|
||||
@@ -1742,9 +1499,9 @@ def run_viral_video_analyze(self: Task, job_id: str) -> dict:
|
||||
# #2183/#2174: VLM完成后立即启动信任链t2i预热(后台daemon线程,与后续阶段并行)
|
||||
if job.images:
|
||||
try:
|
||||
_products = (image_analysis or {}).get("products", []) or []
|
||||
# #2220: 直接传 products 列表,由 _start_trust_chain_preheat 内部按 has_person 筛选
|
||||
_start_trust_chain_preheat(job.id, _products)
|
||||
_imgs = (image_analysis or {}).get("images", []) or []
|
||||
# 直接传 images 列表,内部按 has_person 筛选
|
||||
_start_trust_chain_preheat(job.id, _imgs)
|
||||
except Exception as _e:
|
||||
logger.warning("[爆款视频][阶段1] 启动信任链t2i预热失败: %s", _e)
|
||||
_save_job(repo, job, session)
|
||||
@@ -1827,17 +1584,10 @@ def run_viral_video_generate_copy(self: Task, job_id: str) -> dict:
|
||||
_save_job(repo, job, session)
|
||||
_hb_stop, _hb_thread = _start_heartbeat_thread(job_id)
|
||||
|
||||
# 阶段:意图解析
|
||||
_set_stage(job, repo, session, ViralVideoStage.INTENT_PARSING, "正在解析文案意图...")
|
||||
image_analysis = job.image_analysis or {"products": []}
|
||||
intent_result = _step_intent_parsing(job, image_analysis)
|
||||
job.intent_result = intent_result
|
||||
# #2218: 不在意图解析后单独落库,等 copy_result 生成后与 mark_copy_generated 一起原子写入
|
||||
_emit_progress(job_id, ViralVideoStage.INTENT_PARSING, 35.0, "意图解析完成")
|
||||
|
||||
# 阶段:编导脚本生成(核心耗时环节,已用快模型)
|
||||
# v8:意图理解并入分镜生成,一次 LLM 调用
|
||||
image_analysis = normalize_image_analysis(job.image_analysis)
|
||||
_set_stage(job, repo, session, ViralVideoStage.SCRIPT_GENERATION, "正在编排分镜脚本...")
|
||||
copy_result = _step_script_generation(job, intent_result, image_analysis)
|
||||
copy_result = _step_script_generation(job, image_analysis)
|
||||
_emit_progress(
|
||||
job_id,
|
||||
ViralVideoStage.SCRIPT_GENERATION,
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 prompt 解析:优先读后台 viral_video_prompt_templates 表(prompt_type='image_analysis'
|
||||
且 is_active=true),30s TTL 热加载;DB 无有效记录/异常时,fallback 到纯硬编码 JSON schema prompt。
|
||||
"""V2 prompt 解析:优先读后台 viral_video_prompt_templates(prompt_type='image_analysis'
|
||||
且 is_active=true),30s TTL 热加载;DB 无有效记录/异常时,fallback 到 prompts.py 的
|
||||
image_analysis v8 默认 system/user。
|
||||
|
||||
规则(简单直接,不做字符串匹配判断):
|
||||
- DB 有 is_active=true 的 image_analysis 记录(含种子版本和用户修改后的版本):
|
||||
* system = DB.system_prompt(DB prompt 自带完整输出格式,不追加硬编码 schema,
|
||||
避免 DB 写 XML、调用强制 json_object 造成的格式冲突)
|
||||
* user = DB.user_prompt_template 渲染后使用;渲染后为空则用硬编码默认
|
||||
- DB 无记录/连接异常/返回空:system/user 全部用纯硬编码 JSON schema prompt
|
||||
规则:
|
||||
- DB 有 is_active=true 的 image_analysis 记录:system 原样用 DB.system_prompt
|
||||
(自带完整输出格式,不追加任何硬编码 schema),user 用 DB.user_prompt_template
|
||||
渲染(填入 image_url / ocr_text);
|
||||
- DB 无记录/异常:system/user 用 prompts.py 里的 v8 默认模板。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -15,192 +15,79 @@ from __future__ import annotations
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ---- 纯硬编码 JSON schema(DB 无有效配置时全量使用) ----
|
||||
|
||||
_FAST_JSON_SCHEMA = (
|
||||
"你是图片结构化识别器。严格按下方 JSON schema 返回一个对象,不要任何解释、"
|
||||
"不要markdown、不要代码块、不要前后缀文字。字段值不确定时填 null 或空数组。\n"
|
||||
"{\n"
|
||||
' "has_person": true/false,\n'
|
||||
' "gender": "男"/"女"/null,\n'
|
||||
' "age_range": "儿童"/"青少年"/"青年"/"中年"/"老年"/null,\n'
|
||||
' "upper_wear": "上装款式,如T恤/衬衫/卫衣/毛衣/西装/夹克/连衣裙/吊带/背心/外套等",\n'
|
||||
' "upper_color": "上装主色",\n'
|
||||
' "lower_wear": "下装款式;穿连衣裙时填null",\n'
|
||||
' "lower_color": "下装主色",\n'
|
||||
' "dress_color": "连衣裙主色(穿连衣裙时填)",\n'
|
||||
' "accessories": ["眼镜"/"帽子"/"项链"/"耳环"/"背包"/"手表"等数组],\n'
|
||||
' "hairstyle": "发型,如短发/长发/马尾/卷发/丸子头/光头等",\n'
|
||||
' "expression": "表情,如微笑/严肃/酷/开心等",\n'
|
||||
' "pose": "姿势,如站立/坐姿/侧身/行走等",\n'
|
||||
' "scene": "场景,如室内/街拍/户外/办公室/家居/海边/雪景/森林等",\n'
|
||||
' "style": "风格,如休闲/商务/运动/复古/潮流/甜美/酷飒/优雅/街头/法式等",\n'
|
||||
' "has_product": true/false,\n'
|
||||
' "category": "产品类目:服饰/鞋包/美妆/数码/食品/家居/配饰/母婴/非产品图",\n'
|
||||
' "product_name": "产品名称,非产品图填null",\n'
|
||||
' "brand": "品牌或文字标识,无则null",\n'
|
||||
' "material": "材质,如棉质/牛仔/皮革/真丝/针织/涤纶等",\n'
|
||||
' "pattern": "图案,如纯色/条纹/波点/格子/印花/碎花/Logo等",\n'
|
||||
' "colors": ["主色数组"],\n'
|
||||
' "mood": "整体氛围/情绪,如清新/活力/高级/温暖/冷峻/甜美/复古等"\n'
|
||||
"}\n\n"
|
||||
"你必须只返回一个合法的JSON对象,不要输出任何其他文字、解释、XML标签或markdown。"
|
||||
)
|
||||
DEFAULT_FAST_USER = "识别这张图片的人物穿搭与主体信息,只返回JSON对象。"
|
||||
def _default_template() -> dict:
|
||||
# 延迟导入:避免模块加载时拉起整个 packages 依赖链(也便于旧 Python 收集测试)
|
||||
from packages.application.viral_video.prompts import DEFAULT_TEMPLATES
|
||||
|
||||
_PRO_JSON_SCHEMA = (
|
||||
"你是图片分析专家。严格按下方 JSON schema 返回一个对象,不要解释、不要markdown、不要代码块、不要XML标签。\n"
|
||||
"{\n"
|
||||
' "has_person": true/false,\n'
|
||||
' "gender": "男"/"女"/null,\n'
|
||||
' "age_range": "儿童"/"青少年"/"青年"/"中年"/"老年"/null,\n'
|
||||
' "outfit": "整体穿着描述(含颜色款式)",\n'
|
||||
' "hair": "发型发色",\n'
|
||||
' "pose": "姿势",\n'
|
||||
' "expression": "表情",\n'
|
||||
' "scene": "场景",\n'
|
||||
' "mood": "氛围",\n'
|
||||
' "has_product": true/false,\n'
|
||||
' "category": "类目:服饰/鞋包/美妆/数码/食品/家居/配饰/母婴/非产品图",\n'
|
||||
' "product_name": "产品名,非产品图填null",\n'
|
||||
' "brand": "品牌,无则null",\n'
|
||||
' "key_features": ["核心特征数组,3-6个短语"]\n'
|
||||
"}\n\n"
|
||||
"你必须只返回一个合法的JSON对象,不要输出任何其他文字、解释、XML标签或markdown。"
|
||||
)
|
||||
DEFAULT_PRO_USER = "分析这张图片,返回符合schema的JSON。"
|
||||
for item in DEFAULT_TEMPLATES:
|
||||
if item["prompt_type"] == "image_analysis":
|
||||
return item
|
||||
raise RuntimeError("image_analysis 默认模板缺失")
|
||||
|
||||
# 保留旧 JSON schema 追加文本作为常量(DB prompt 完全控制输出格式后不再使用,
|
||||
# 保留以便排查历史行为)。
|
||||
_FAST_JSON_APPEND = (
|
||||
"\n\n【输出格式要求】无论上文如何要求,最终你必须只返回一个合法的JSON对象,"
|
||||
"严格包含以下字段(字段值不确定时填null或空数组):\n"
|
||||
"{\n"
|
||||
' "has_person": true/false,\n'
|
||||
' "gender": "男"/"女"/null,\n'
|
||||
' "age_range": "儿童"/"青少年"/"青年"/"中年"/"老年"/null,\n'
|
||||
' "upper_wear": "上装款式字符串",\n'
|
||||
' "upper_color": "上装主色",\n'
|
||||
' "lower_wear": "下装款式(穿连衣裙时填null)",\n'
|
||||
' "lower_color": "下装主色",\n'
|
||||
' "dress_color": "连衣裙主色(穿连衣裙时填)",\n'
|
||||
' "accessories": ["配饰数组"],\n'
|
||||
' "hairstyle": "发型",\n'
|
||||
' "expression": "表情",\n'
|
||||
' "pose": "姿势",\n'
|
||||
' "scene": "场景",\n'
|
||||
' "style": "风格",\n'
|
||||
' "has_product": true/false,\n'
|
||||
' "category": "产品类目:服饰/鞋包/美妆/数码/食品/家居/配饰/母婴/非产品图",\n'
|
||||
' "product_name": "产品名称,非产品图填null",\n'
|
||||
' "brand": "品牌或文字标识,无则null",\n'
|
||||
' "material": "材质",\n'
|
||||
' "pattern": "图案",\n'
|
||||
' "colors": ["主色数组"],\n'
|
||||
' "mood": "整体氛围"\n'
|
||||
"}\n"
|
||||
"不要输出任何其他文字、解释、XML标签或markdown。"
|
||||
)
|
||||
|
||||
_PRO_JSON_APPEND = (
|
||||
"\n\n【输出格式要求】无论上文如何要求,最终你必须只返回一个合法的JSON对象,"
|
||||
"严格包含以下字段(字段值不确定时填null或空数组):\n"
|
||||
"{\n"
|
||||
' "has_person": true/false,\n'
|
||||
' "gender": "男"/"女"/null,\n'
|
||||
' "age_range": "儿童"/"青少年"/"青年"/"中年"/"老年"/null,\n'
|
||||
' "outfit": "整体穿着描述(含颜色款式)",\n'
|
||||
' "hair": "发型发色",\n'
|
||||
' "pose": "姿势",\n'
|
||||
' "expression": "表情",\n'
|
||||
' "scene": "场景",\n'
|
||||
' "mood": "氛围",\n'
|
||||
' "has_product": true/false,\n'
|
||||
' "category": "类目:服饰/鞋包/美妆/数码/食品/家居/配饰/母婴/非产品图",\n'
|
||||
' "product_name": "产品名,非产品图填null",\n'
|
||||
' "brand": "品牌,无则null",\n'
|
||||
' "key_features": ["核心特征3-6个短语"]\n'
|
||||
"}\n"
|
||||
"不要输出任何其他文字、解释、XML标签或markdown。"
|
||||
)
|
||||
|
||||
_cache_lock = threading.Lock()
|
||||
_cache: dict[str, tuple[float, Any]] = {}
|
||||
_cache: dict[str, tuple[float, tuple[str, str]]] = {}
|
||||
_CACHE_TTL = 30.0
|
||||
|
||||
|
||||
def _load_db_template() -> Any | None:
|
||||
"""直接查DB viral_video_prompt_templates 中 is_active=true 的 image_analysis 记录;
|
||||
DB不可达/无记录/异常返回None。
|
||||
复用 prompt_loader._load_from_db,它只查DB不做DEFAULT_TEMPLATES fallback,
|
||||
返回None表示DB无记录或异常。"""
|
||||
def _load_db_template():
|
||||
"""查 DB is_active=true 的 image_analysis 记录;不可达/无记录返回 None。"""
|
||||
try:
|
||||
from packages.application.viral_video.prompt_loader import _load_from_db
|
||||
|
||||
return _load_from_db("image_analysis")
|
||||
except Exception as e:
|
||||
logger.warning("[vision.v2] 查询DB prompt配置失败: %s", e)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("[vision.v2] 查询DB image_analysis prompt失败: %s", e)
|
||||
return None
|
||||
|
||||
|
||||
def _render_user(tpl: Any | None, default_user: str) -> str:
|
||||
if not tpl:
|
||||
return default_user
|
||||
tpl_str = getattr(tpl, "user_prompt_template", "") or ""
|
||||
if not tpl_str.strip():
|
||||
return default_user
|
||||
rendered = tpl_str.replace("{image_count}", "1").replace("{industry}", "通用").replace("{image_urls}", "").strip()
|
||||
return rendered or default_user
|
||||
def _render_user(user_tpl: str, image_url: str, ocr_text: str) -> str:
|
||||
try:
|
||||
return user_tpl.format(image_url=image_url, ocr_text=ocr_text or "无")
|
||||
except Exception: # noqa: BLE001
|
||||
return user_tpl
|
||||
|
||||
|
||||
def resolve_fast_prompt() -> tuple[str, str]:
|
||||
return _resolve("fast")
|
||||
|
||||
|
||||
def resolve_pro_prompt() -> tuple[str, str]:
|
||||
return _resolve("pro")
|
||||
|
||||
|
||||
def _resolve(kind: str) -> tuple[str, str]:
|
||||
def _resolve(kind: str, image_url: str = "", ocr_text: str = "") -> tuple[str, str]:
|
||||
now = time.time()
|
||||
cache_key = f"prompt_{kind}"
|
||||
with _cache_lock:
|
||||
hit = _cache.get(cache_key)
|
||||
if hit and now - hit[0] < _CACHE_TTL:
|
||||
return hit[1]
|
||||
sys_prompt, usr_prompt = hit[1]
|
||||
return sys_prompt, _render_user(usr_prompt, image_url, ocr_text)
|
||||
|
||||
default_sys = _FAST_JSON_SCHEMA if kind == "fast" else _PRO_JSON_SCHEMA
|
||||
default_user = DEFAULT_FAST_USER if kind == "fast" else DEFAULT_PRO_USER
|
||||
default = _default_template()
|
||||
sys_prompt = default["system_prompt"]
|
||||
usr_prompt = default["user_prompt_template"]
|
||||
|
||||
sys_prompt = default_sys
|
||||
usr_prompt = default_user
|
||||
try:
|
||||
tpl = _load_db_template()
|
||||
if tpl is not None:
|
||||
db_sys = (getattr(tpl, "system_prompt", "") or "").strip()
|
||||
if db_sys:
|
||||
sys_prompt = db_sys # DB prompt自带完整输出格式,不追加硬编码schema避免冲突
|
||||
usr_prompt = _render_user(tpl, default_user)
|
||||
logger.info(
|
||||
"[vision.v2] 使用DB image_analysis prompt (kind=%s version=%s sys_len=%d)",
|
||||
kind,
|
||||
getattr(tpl, "version", "?"),
|
||||
len(db_sys),
|
||||
)
|
||||
else:
|
||||
logger.debug("[vision.v2] DB image_analysis system_prompt为空,使用默认JSON (kind=%s)", kind)
|
||||
else:
|
||||
logger.debug("[vision.v2] DB无image_analysis记录/不可达,使用默认JSON prompt (kind=%s)", kind)
|
||||
except Exception as e:
|
||||
logger.warning("[vision.v2] 解析DB prompt异常,使用默认: %s", e)
|
||||
tpl = _load_db_template()
|
||||
if tpl is not None:
|
||||
db_sys = (getattr(tpl, "system_prompt", "") or "").strip()
|
||||
if db_sys:
|
||||
sys_prompt = db_sys
|
||||
db_usr = getattr(tpl, "user_prompt_template", "") or usr_prompt
|
||||
usr_prompt = db_usr or usr_prompt
|
||||
logger.info(
|
||||
"[vision.v2] 使用DB image_analysis prompt version=%s",
|
||||
getattr(tpl, "version", "?"),
|
||||
)
|
||||
|
||||
with _cache_lock:
|
||||
_cache[cache_key] = (now, (sys_prompt, usr_prompt))
|
||||
return sys_prompt, usr_prompt
|
||||
return sys_prompt, _render_user(usr_prompt, image_url, ocr_text)
|
||||
|
||||
|
||||
def resolve_fast_prompt(image_url: str = "", ocr_text: str = "") -> tuple[str, str]:
|
||||
return _resolve("fast", image_url, ocr_text)
|
||||
|
||||
|
||||
def resolve_pro_prompt(image_url: str = "", ocr_text: str = "") -> tuple[str, str]:
|
||||
return _resolve("pro", image_url, ocr_text)
|
||||
|
||||
|
||||
def invalidate_cache() -> None:
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,12 +1,11 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 图片分析主路径:每图并行 OCR(火山MediaKit,未配置时自动跳过)+ qwen3.8-flash JSON VLM,
|
||||
失败时单次 qwen3.7-plus 兜底。
|
||||
"""V2 图片分析主路径(v8 叙述优先):每图并行 OCR(火山 MediaKit,未配置自动跳过)
|
||||
+ fast VLM 强约束 JSON;失败时单次 pro VLM 兜底。
|
||||
|
||||
架构(灵应10-05确认):
|
||||
- 唯一后端:阿里云百炼 DashScope,qwen3.8-flash 做快速路径、qwen3.7-plus 做兜底
|
||||
- 主力:单图2路并行(OCR + fast VLM),外层N图全并发(workers=8)
|
||||
- 兜底:单次 pro VLM 调用,无竞速/重试/复杂超时
|
||||
- 输出 dict 格式与旧版完全一致,下游零改动
|
||||
架构:
|
||||
- 单图 2 路并行(OCR + fast VLM),外层 N 图全并发(workers=8);
|
||||
- 兜底单次 pro VLM,无竞速/复杂重试;
|
||||
- 输出统一为 5 字段 image dict(type/name/brand/has_person/summary_markdown)。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -21,36 +20,24 @@ from . import assembler, ocr_volc, vlm_fallback, vlm_fast_json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# 超时(可通过环境变量覆盖)
|
||||
_IMG_WORKERS = int(os.environ.get("VISION_V2_IMG_WORKERS", "8"))
|
||||
_FAST_TIMEOUT = float(os.environ.get("VISION_V2_FAST_TIMEOUT", "20"))
|
||||
_FAST_JSON_TIMEOUT = float(os.environ.get("VISION_V2_FAST_JSON_TIMEOUT", "20"))
|
||||
_OCR_TIMEOUT = float(os.environ.get("VISION_V2_OCR_TIMEOUT", "6"))
|
||||
_PRO_TIMEOUT = float(os.environ.get("VISION_V2_PRO_TIMEOUT", "45"))
|
||||
|
||||
_FALLBACK_RESULT = {
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "非产品图",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": "通用",
|
||||
"mood": "",
|
||||
"portrait_prompt": "无法判断",
|
||||
"summary": "未识别",
|
||||
}
|
||||
|
||||
def _is_usable(r: dict[str, Any] | None) -> bool:
|
||||
if not isinstance(r, dict):
|
||||
return False
|
||||
return bool((r.get("summary_markdown") or "").strip())
|
||||
|
||||
|
||||
def _is_usable(r: dict[str, Any]) -> bool:
|
||||
pp = (r.get("portrait_prompt") or "").strip()
|
||||
if pp and pp not in ("无人像", "无法判断", "未识别"):
|
||||
return True
|
||||
name = (r.get("name") or "").strip()
|
||||
if name and name not in ("未识别", "无法判断", "未知"):
|
||||
return True
|
||||
return False
|
||||
def _basic_failure(ocr_result: list[str], fast_elapsed: float, source: str) -> dict[str, Any]:
|
||||
image = assembler.assemble_result(-1, {}, ocr_result)
|
||||
image["_source"] = source
|
||||
image["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
return image
|
||||
|
||||
|
||||
def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
@@ -58,7 +45,6 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
|
||||
fj_result: dict[str, Any] | None = None
|
||||
ocr_result: list[str] = []
|
||||
fast_elapsed = 0.0
|
||||
pool = ThreadPoolExecutor(max_workers=2)
|
||||
f_fj = pool.submit(vlm_fast_json.call_fast_json, img_url, timeout=_FAST_JSON_TIMEOUT)
|
||||
f_ocr = pool.submit(ocr_volc.call_ocr, img_url, timeout=_OCR_TIMEOUT)
|
||||
@@ -66,7 +52,7 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
for fut in as_completed([f_fj, f_ocr], timeout=_FAST_TIMEOUT):
|
||||
try:
|
||||
res = fut.result(timeout=1)
|
||||
except Exception as e:
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("[vision.v2] 图片 #%d 子任务异常: %s", idx, e)
|
||||
continue
|
||||
if fut is f_fj and isinstance(res, dict):
|
||||
@@ -80,37 +66,24 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
logger.warning("[vision.v2] 图片 #%d fast路径超时(%.0fs),走pro兜底", idx, _FAST_TIMEOUT)
|
||||
finally:
|
||||
fast_elapsed = time.time() - t0
|
||||
pool.shutdown(wait=False) # 不等待未完成的线程,避免计时膨胀
|
||||
pool.shutdown(wait=False)
|
||||
|
||||
if fj_result:
|
||||
assembled = assembler.assemble_result(idx, fj_result, ocr_result)
|
||||
if _is_usable(assembled):
|
||||
assembled["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
logger.info(
|
||||
"[vision.v2] 图片 #%d fast命中 elapsed=%.2fs pp=%s",
|
||||
idx,
|
||||
fast_elapsed,
|
||||
(assembled.get("portrait_prompt") or "")[:40],
|
||||
)
|
||||
logger.info("[vision.v2] 图片 #%d fast命中 elapsed=%.2fs", idx, fast_elapsed)
|
||||
return assembled
|
||||
|
||||
pro_t0 = time.time()
|
||||
pro_result = vlm_fallback.call_pro_vlm(img_url, idx, timeout=_PRO_TIMEOUT)
|
||||
if pro_result and _is_usable(pro_result):
|
||||
pro_result = vlm_fallback.call_pro_vlm(img_url, idx, ocr_hint=ocr_result, timeout=_PRO_TIMEOUT)
|
||||
if _is_usable(pro_result):
|
||||
pro_result["_fallback_used"] = True
|
||||
pro_result["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
pro_result["_pro_elapsed"] = round(time.time() - pro_t0, 2)
|
||||
if ocr_result and not pro_result.get("text_on_package"):
|
||||
pro_result["text_on_package"] = ocr_result[:8]
|
||||
logger.info("[vision.v2] 图片 #%d pro兜底命中 total=%.2fs", idx, time.time() - t0)
|
||||
return pro_result
|
||||
|
||||
logger.warning("[vision.v2] 图片 #%d 全路径失败 elapsed=%.2fs", idx, time.time() - t0)
|
||||
out = dict(_FALLBACK_RESULT)
|
||||
out["_source"] = "v2_all_failed"
|
||||
out["text_on_package"] = ocr_result[:8]
|
||||
out["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
return out
|
||||
return _basic_failure(ocr_result, fast_elapsed, "v2_all_failed")
|
||||
|
||||
|
||||
def analyze_images_v2(img_urls: list[str]) -> list[dict[str, Any]]:
|
||||
@@ -133,14 +106,19 @@ def analyze_images_v2(img_urls: list[str]) -> list[dict[str, Any]]:
|
||||
idx = future_to_idx[fut]
|
||||
try:
|
||||
results[idx] = fut.result()
|
||||
except Exception as e:
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("[vision.v2] 图片 #%d future异常: %s", idx, e, exc_info=True)
|
||||
r = dict(_FALLBACK_RESULT)
|
||||
r["_source"] = "v2_future_exception"
|
||||
results[idx] = r
|
||||
results[idx] = assembler.assemble_result(idx, {}, [])
|
||||
results[idx]["_source"] = "v2_future_exception" # type: ignore[index]
|
||||
|
||||
elapsed = time.time() - t0
|
||||
succ = sum(1 for r in results if r and _is_usable(r))
|
||||
succ = sum(1 for r in results if _is_usable(r))
|
||||
fb = sum(1 for r in results if r and r.get("_fallback_used"))
|
||||
logger.info("[vision.v2] 完成 n=%d usable=%d pro_fallback=%d elapsed=%.2fs", len(img_urls), succ, fb, elapsed)
|
||||
return [r for r in results if r is not None]
|
||||
logger.info(
|
||||
"[vision.v2] 完成 n=%d usable=%d pro_fallback=%d elapsed=%.2fs",
|
||||
len(img_urls),
|
||||
succ,
|
||||
fb,
|
||||
elapsed,
|
||||
)
|
||||
return [r for r in results if r is not None] # type: ignore[misc]
|
||||
|
||||
@@ -1,14 +1,8 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 兜底路径:image_analysis(默认 qwen-vl-plus 视觉模型,fallback qwen3.7-plus / DashScope)单图调用。
|
||||
"""V2 兜底路径:vision client(fallback 变体)单图调用,走 v8 叙述优先 prompt。
|
||||
|
||||
fast_json 超时/返回非 JSON/识别为空时,本路径单次调用兜底。
|
||||
设计要点:
|
||||
- 通过 ai_router.get_vision_client() 获取 DoubaoClient 实例,不再自己拼 httpx 请求
|
||||
- enable_thinking=False + response_format=json_object
|
||||
- system prompt 优先读后台 viral_video_prompt_templates 配置,DB不可用时fallback到硬编码JSON schema
|
||||
- max_tokens 不传,使用 client 中 capability 的 DB 配置(避免硬编码截断 JSON)
|
||||
- timeout=30s
|
||||
- 返回 dict 统一走 assembler.assemble_result 组装,与 fast 路径输出格式完全一致
|
||||
fast 超时/非 JSON/为空时单次调用;输出统一走 assembler.assemble_result 组装,
|
||||
与 fast 路径同为 5 字段 image dict。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -28,11 +22,12 @@ def call_pro_vlm(
|
||||
img_url: str,
|
||||
idx: int,
|
||||
*,
|
||||
ocr_hint: list[str] | None = None,
|
||||
timeout: int = _DEFAULT_TIMEOUT,
|
||||
max_tokens: int | None = None,
|
||||
) -> dict[str, Any] | None:
|
||||
"""max_tokens 默认 None:不显式传参,使用 client 内 capability 的 DB 配置。"""
|
||||
t0 = time.time()
|
||||
ocr_text = "、".join(t for t in (ocr_hint or []) if t)[:200]
|
||||
|
||||
try:
|
||||
from packages.shared.ai_router import ai_router
|
||||
@@ -41,13 +36,13 @@ def call_pro_vlm(
|
||||
if not client or not client.is_available:
|
||||
logger.warning("[vision.v2] pro vision client 不可用,跳过")
|
||||
return None
|
||||
except Exception as e:
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("[vision.v2] ai_router 获取失败: %s", e)
|
||||
return None
|
||||
|
||||
system_prompt, user_prompt = _prompt.resolve_pro_prompt()
|
||||
system_prompt, user_prompt = _prompt.resolve_pro_prompt(img_url, ocr_text)
|
||||
|
||||
messages = [
|
||||
messages: list[dict[str, Any]] = [
|
||||
{"role": "system", "content": system_prompt},
|
||||
{
|
||||
"role": "user",
|
||||
@@ -61,31 +56,31 @@ def call_pro_vlm(
|
||||
try:
|
||||
call_kwargs: dict[str, Any] = {
|
||||
"messages": messages,
|
||||
"images": None, # 图片已在 messages 中
|
||||
"images": None,
|
||||
"temperature": 0.3,
|
||||
"timeout": timeout,
|
||||
"enable_thinking": False,
|
||||
"response_format": {"type": "json_object"},
|
||||
"max_tokens": max_tokens if max_tokens is not None else 4000,
|
||||
}
|
||||
# pro fallback:显式4000 tokens给复杂门店图留足空间
|
||||
call_kwargs["max_tokens"] = max_tokens if max_tokens is not None else 4000
|
||||
|
||||
from .json_utils import extract_json_object
|
||||
|
||||
raw = None
|
||||
obj = None
|
||||
for _outer in range(2):
|
||||
kw = dict(call_kwargs)
|
||||
if _outer == 1:
|
||||
kw.pop("response_format", None)
|
||||
msgs2 = [dict(messages[0]), dict(messages[1])]
|
||||
cont = [dict(c) for c in list(msgs2[1]["content"])]
|
||||
cont[-1] = {"type": "text", "text": user_prompt + "\n严格只输出JSON对象,不要解释或markdown。"}
|
||||
cont = [dict(c) for c in msgs2[1]["content"]]
|
||||
cont[-1] = {
|
||||
"type": "text",
|
||||
"text": user_prompt + "\n严格只输出JSON对象,不要解释或markdown。",
|
||||
}
|
||||
msgs2[1] = {"role": "user", "content": cont}
|
||||
kw["messages"] = msgs2
|
||||
raw = client.vision_completion(**kw)
|
||||
if not raw:
|
||||
logger.warning("[vision.v2] pro 返回空 outer=%s", _outer)
|
||||
continue
|
||||
obj = extract_json_object(raw)
|
||||
if obj is not None:
|
||||
@@ -97,20 +92,18 @@ def call_pro_vlm(
|
||||
logger.warning("[vision.v2] pro 两次均未得到JSON elapsed=%.1fs", elapsed)
|
||||
return None
|
||||
if obj.get("_partial"):
|
||||
logger.warning("[vision.v2] pro 返回截断JSON(partial) elapsed=%.1fs", elapsed)
|
||||
logger.info(
|
||||
"[vision.v2] pro 完成 model=%s elapsed=%.1fs type=%s",
|
||||
client.model,
|
||||
elapsed,
|
||||
obj.get("type"),
|
||||
)
|
||||
logger.warning("[vision.v2] pro 截断JSON(partial) elapsed=%.1fs", elapsed)
|
||||
|
||||
# 通过assembler统一组装,兼容v4嵌套schema和旧扁平schema
|
||||
result = assembler.assemble_result(idx, obj, [])
|
||||
result = assembler.assemble_result(idx, obj, ocr_hint or [])
|
||||
result["_source"] = "vlm_pro"
|
||||
result["_fallback_used"] = True
|
||||
logger.info("[vision.v2] pro 完成 model=%s elapsed=%.1fs", client.model, elapsed)
|
||||
return result
|
||||
except Exception as e:
|
||||
elapsed = time.time() - t0
|
||||
logger.warning("[vision.v2] pro 异常 elapsed=%.1fs err=%s", elapsed, e, exc_info=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning(
|
||||
"[vision.v2] pro 异常 elapsed=%.1fs err=%s",
|
||||
time.time() - t0,
|
||||
e,
|
||||
exc_info=True,
|
||||
)
|
||||
return None
|
||||
|
||||
@@ -1,15 +1,11 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 快速路径:image_analysis capability(默认 qwen-vl-plus 视觉模型 / DashScope)强约束 JSON-only 调用。
|
||||
"""V2 快速路径:vision client(默认 image_analysis 能力)强约束 JSON-only 调用。
|
||||
|
||||
目标:替代"人体属性/商品检测/图像标签"三个火山不存在的专用云端 API。
|
||||
设计要点:
|
||||
- 通过 ai_router.get_vision_client() 获取 DoubaoClient 实例,不再自己拼 httpx 请求
|
||||
- enable_thinking=False 关闭推理链(reasoning 是延迟主因)
|
||||
- response_format=json_object 强约束JSON输出
|
||||
- system prompt 优先读后台 viral_video_prompt_templates 配置,DB不可用时fallback到硬编码JSON schema
|
||||
- max_tokens 不传,使用 client 中 capability 的 DB 配置(避免硬编码截断 JSON)
|
||||
- temperature=0.1(稳定输出 JSON)
|
||||
- timeout=15s(失败由外层走 pro 兜底)
|
||||
要点:
|
||||
- 通过 ai_router.get_vision_client() 获取 client;
|
||||
- enable_thinking=False 关闭推理链,response_format=json_object 强约束 JSON;
|
||||
- system/user prompt 优先读后台模板(v8 叙述优先),DB 不可用时用 prompts.py 默认;
|
||||
- temperature=0.1(稳定输出 JSON);两次尝试(第二次去 json_object 约束)。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -31,11 +27,6 @@ def call_fast_json(
|
||||
timeout: int = _DEFAULT_TIMEOUT,
|
||||
max_tokens: int | None = None,
|
||||
) -> dict[str, Any] | None:
|
||||
"""调用 vision client 返回结构化 dict;失败/非 JSON 返回 None。
|
||||
|
||||
max_tokens 默认 None:不显式传参,使用 client 内 capability 的 DB 配置;
|
||||
显式传入时作为覆盖。
|
||||
"""
|
||||
t0 = time.time()
|
||||
|
||||
try:
|
||||
@@ -45,13 +36,13 @@ def call_fast_json(
|
||||
if not client or not client.is_available:
|
||||
logger.warning("[vision.v2] vision client 不可用,跳过 fast_json")
|
||||
return None
|
||||
except Exception as e:
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("[vision.v2] ai_router 获取失败: %s", e)
|
||||
return None
|
||||
|
||||
system_prompt, user_prompt = _prompt.resolve_fast_prompt()
|
||||
system_prompt, user_prompt = _prompt.resolve_fast_prompt(img_url, "")
|
||||
|
||||
messages = [
|
||||
messages: list[dict[str, Any]] = [
|
||||
{"role": "system", "content": system_prompt},
|
||||
{
|
||||
"role": "user",
|
||||
@@ -65,7 +56,7 @@ def call_fast_json(
|
||||
try:
|
||||
call_kwargs: dict[str, Any] = {
|
||||
"messages": messages,
|
||||
"images": None, # 图片已在 messages 中
|
||||
"images": None,
|
||||
"temperature": 0.1,
|
||||
"timeout": timeout,
|
||||
"enable_thinking": False,
|
||||
@@ -74,50 +65,47 @@ def call_fast_json(
|
||||
if max_tokens is not None:
|
||||
call_kwargs["max_tokens"] = max_tokens
|
||||
|
||||
# 双重防护:第1次正常调用;第2次去掉json_object强约束(部分模型在该约束下
|
||||
# 反而幻觉),并加严格指令。解析全部走 json_utils,截断partial产物可用。
|
||||
from .json_utils import extract_json_object
|
||||
|
||||
raw = None
|
||||
obj = None
|
||||
for _outer in range(2):
|
||||
kw = dict(call_kwargs)
|
||||
if _outer == 1:
|
||||
kw.pop("response_format", None)
|
||||
msgs2 = [dict(messages[0]), dict(messages[1])]
|
||||
cont = list(msgs2[1]["content"])
|
||||
cont = [dict(c) for c in cont]
|
||||
cont[-1] = {"type": "text", "text": user_prompt + "\n严格只输出JSON对象,不要解释或markdown。"}
|
||||
cont = [dict(c) for c in msgs2[1]["content"]]
|
||||
cont[-1] = {
|
||||
"type": "text",
|
||||
"text": user_prompt + "\n严格只输出JSON对象,不要解释或markdown。",
|
||||
}
|
||||
msgs2[1] = {"role": "user", "content": cont}
|
||||
kw["messages"] = msgs2
|
||||
raw = client.vision_completion(**kw)
|
||||
if not raw:
|
||||
logger.warning("[vision.v2] fast_json 返回空 outer=%s", _outer)
|
||||
continue
|
||||
obj = extract_json_object(raw)
|
||||
if obj is not None:
|
||||
break
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json 非JSON(100字) outer=%s: %s",
|
||||
_outer,
|
||||
raw[:100],
|
||||
)
|
||||
logger.warning("[vision.v2] fast_json 非JSON(100字) outer=%s: %s", _outer, raw[:100])
|
||||
|
||||
elapsed = time.time() - t0
|
||||
if obj is None:
|
||||
logger.warning("[vision.v2] fast_json 两次均未得到JSON elapsed=%.1fs", elapsed)
|
||||
return None
|
||||
if obj.get("_partial"):
|
||||
logger.warning("[vision.v2] fast_json 返回截断JSON(partial) elapsed=%.1fs", elapsed)
|
||||
logger.warning("[vision.v2] fast_json 截断JSON(partial) elapsed=%.1fs", elapsed)
|
||||
logger.info(
|
||||
"[vision.v2] fast_json 完成 model=%s elapsed=%.1fs has_person=%s type=%s",
|
||||
"[vision.v2] fast_json 完成 model=%s elapsed=%.1fs type=%s",
|
||||
client.model,
|
||||
elapsed,
|
||||
obj.get("has_person"),
|
||||
obj.get("type"),
|
||||
)
|
||||
return obj
|
||||
except Exception as e:
|
||||
elapsed = time.time() - t0
|
||||
logger.warning("[vision.v2] fast_json 异常 elapsed=%.1fs err=%s", elapsed, e, exc_info=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json 异常 elapsed=%.1fs err=%s",
|
||||
time.time() - t0,
|
||||
e,
|
||||
exc_info=True,
|
||||
)
|
||||
return None
|
||||
|
||||
@@ -1,16 +1,19 @@
|
||||
"""爆款视频 5 套 Prompt 模板默认值(#2040 核心资产)。
|
||||
"""爆款视频 Prompt 模板默认值(v8 / v3 叙述优先重构)。
|
||||
|
||||
重要约定(用户明确要求):
|
||||
- 所有 system_prompt / user_prompt_template / example_output 都是**纯文本自然语言 + XML 标签**,
|
||||
运营可直接看懂和编辑,禁止 JSON、禁止 ```json 代码块。
|
||||
- LLM 按 XML 标签输出字段,程序用正则解析(见 xml_parser.py)。
|
||||
- user_prompt_template 中花括号占位符(如 {user_copy_text})在运行时填充。
|
||||
设计原则(灵应 2026-10-07):LLM 直接输出最终给用户看的文案,代码尽量薄。
|
||||
- image_analysis v8:VLM 主交付物是自然叙述风格的 summary_markdown,结构化
|
||||
字段仅保留 type/name/brand/has_person,顶层 products 改名 images;
|
||||
- storyboard v3:口播台词口语化、画面描述有画面感,copy_display_markdown 是
|
||||
LLM 直接写给用户看的流畅叙述文案,代码只做解析不改写;
|
||||
- intent_parsing 步骤整体删除,意图理解并入 storyboard 一次调用。
|
||||
|
||||
模板字段与 DB 表 viral_video_prompt_templates、prompt_loader 完全对应:
|
||||
name / prompt_type / version(int) / system_prompt / user_prompt_template /
|
||||
example_output / is_active。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
TEMPLATE_VERSION = 1
|
||||
|
||||
# 所有文案类 Prompt 自动注入的硬约束
|
||||
GLOBAL_CONSTRAINTS = """【必须遵守的硬约束】
|
||||
1. 不编造时间:不写“今年最新”“2024 爆款”等会过时的时间表述。
|
||||
@@ -19,7 +22,7 @@ GLOBAL_CONSTRAINTS = """【必须遵守的硬约束】
|
||||
4. 符合广告法及平台社区规范。
|
||||
5. 只描述图片中真实可见的内容,看不到的不瞎猜。"""
|
||||
|
||||
# 反套路化要求
|
||||
# 负向提示(注入 storyboard / 视频生成负面词)
|
||||
NEGATIVE_RULES = """【反套路化要求】
|
||||
禁止使用“家人们谁懂啊”“绝绝子”“宝子们”“家人们”“太绝了”“yyds”等烂大街网络词;
|
||||
禁止固定模板化开头;语言要像真人朋友之间的分享,自然、具体、有信息量。"""
|
||||
@@ -27,304 +30,189 @@ NEGATIVE_RULES = """【反套路化要求】
|
||||
# 输出禁用套路词(测试会检查)
|
||||
BANNED_PHRASES = ["家人们谁懂啊", "绝绝子", "宝子们", "yyds", "太绝了"]
|
||||
|
||||
# 文案融合三档独立指令段
|
||||
# 文案融合三档独立指令段(storyboard 一次生成,按档位注入风格指令)
|
||||
FUSION_INSTRUCTIONS = {
|
||||
"ai_full": """【本次创作模式:AI 全权创作】
|
||||
你是资深短视频编导。用户只提供了产品图片,没有给出具体文案方向。请根据图片内容和营销参数,自由发挥创作完整的爆款短视频文案。充分挖掘产品真实可见的卖点,使用爆款结构,抓人眼球。""",
|
||||
你是资深短视频编导。用户只提供了产品/门店图片,没有给出具体文案方向。请根据图片的真实观察和营销参数,自由发挥创作完整成片级方案,口播自然、画面可拍。""",
|
||||
"ai_polish": """【本次创作模式:AI 辅助润色】
|
||||
你是用户的文案助理。用户已经写了草稿/关键词/碎碎念,表达了他想讲的核心意思,但表达不完整、不够吸引人。你的任务是:以用户的意思为主,保留他想表达的所有核心信息点,在此基础上润色扩写、调整语序、增加衔接、优化表达,让文案更流畅更有吸引力。绝对不能改变用户想表达的核心意思,不能把用户的观点换成相反的,不能添加用户没提到的产品卖点。用户提到的品牌名、价格、人名、具体事实必须原样保留。""",
|
||||
用户已给出方向或碎碎念。以用户的意思为主,保留其所有核心信息,在此基础上润色、补衔接、优化表达,让口播更自然、画面更具体;绝不改变用户核心意思,不添加用户没提到的卖点,品牌名、价格、人名等事实原样保留。""",
|
||||
"user_primary": """【本次创作模式:以用户原文为主】
|
||||
你是文案润色助手。用户已经写好了明确的文案,这是他最终想表达的内容。你的任务是最小化修改:只做必要的错别字修正、标点调整、语句通顺度优化,以及添加必要的衔接词让口播更自然。用户的核心句子、关键表述、事实信息一律不改。如果用户文案本身已经很好,直接返回,不要为了改而改。personal_brands 中的事实信息必须逐字保留。""",
|
||||
最小化修改:只做必要的通顺、合规修正与衔接补全,用户的核心句子与事实一律不改;用户文案已经很好就直接用,不为改而改。""",
|
||||
}
|
||||
|
||||
# ── 模板1:图片多模态分析(VLM)────────────────────────────────────────
|
||||
_IMAGE_ANALYSIS_SYSTEM = f"""你是电商商品视觉分析师,负责从商品图片中提取真实可见的商品信息。
|
||||
# ── 模板1:图片多模态分析 v8(叙述优先)───────────────────────────────
|
||||
_IMAGE_ANALYSIS_SYSTEM = """你是一名擅长观察和写作的品牌内容编导。面对一张真实图片,先用眼睛仔细看,再用自然、流畅、具体的中文把画面写成一段可以直接读给人听的描述。
|
||||
|
||||
工作方式(分步骤看,不要跳步):
|
||||
1. 先看整体:有哪些产品、什么场景、有没有人物。
|
||||
2. 再看细节:包装文字、颜色构成、人物状态、画面质感。
|
||||
3. 最后提炼卖点:只总结图片里能看到的卖点。
|
||||
## 输出格式(严格 JSON,不要输出 JSON 以外的任何内容)
|
||||
{
|
||||
"images": [
|
||||
{
|
||||
"type": "store 或 product 或 person 或 scene,四选一",
|
||||
"name": "主体名称,看不出就写“未识别”",
|
||||
"brand": "品牌名,看不出就留空字符串",
|
||||
"has_person": false,
|
||||
"summary_markdown": "用 Markdown 写成的自然叙述,这是最主要的交付物"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
{GLOBAL_CONSTRAINTS}
|
||||
## summary_markdown 写作要求(最重要)
|
||||
1. 写成完整、通顺的句子,像在跟朋友认真描述你看到的画面;不要用分号堆砌关键词,不要罗列“核心特征:xxx”“主色调:xxx”这类填表式标签。
|
||||
2. 开头先给一句整体定性,让读者立刻明白这是什么场景、什么主体。
|
||||
3. 颜色、材质、形状、部件要具体可感,写到位置和搭配;画面里出现的文字原样读出并自然融进句子,数字、规格、价格精确引用,看不清的不要编造。
|
||||
4. 只写真实看到的内容,不脑补功能、疗效、销量或画面之外的信息。
|
||||
5. 长度控制在 200-500 字。
|
||||
|
||||
请严格按下面的标签格式输出,标签名一个都不能改,不要输出任何解释,不要用代码块:
|
||||
<products> 下面每个产品用一个 <product> 标签,属性 name 是产品名、features 是外观特征、position 是 main 或 secondary、image_index 是第几张图(从0开始)。
|
||||
<colors> 下面每个主要颜色用一个 <color> 标签,属性 hex 是色值、name 是颜色名、coverage 是占比小数。
|
||||
<people> 用一个标签,属性 has_person、count、gender、age_range、hair(发型发色)、skin_tone(肤色)、face_shape(脸型)、outfit(穿着)、pose(姿态)、expression(表情)分别描述人物外貌。有人物时属性尽量具体(如hair="黑色长直发"、outfit="白色衬衫"),无人像时除has_person=false外其他填"无法判断"。
|
||||
<mood> 标签写画面整体情绪氛围。
|
||||
<visible_text> 下面每处可见文字用一个 <text_item> 标签,属性 text 是文字内容、position 是位置。
|
||||
<scene> 标签写场景描述。
|
||||
<quality> 用一个标签,属性 resolution、lighting、composition、blur 描述画质。
|
||||
<key_selling_points> 下面每个卖点用一个 <point> 标签。
|
||||
## 按类型组织内容
|
||||
- type=store(门店/店内环境):用以下小标题分段,小标题下写连贯的句子而不是清单:
|
||||
###店铺主体
|
||||
###周边物品
|
||||
1.家具陈设
|
||||
2.商品与标识
|
||||
- type=product(商品):按自然段从整体到局部描写——先说是什么、什么品牌,再写包装/外形、颜色与材质、标签文字、可见部件与规格。
|
||||
- type=person(人物):描述人物身份感、姿态、穿着(上下装/颜色/款式)、动作与所处环境;用于品牌宣传时突出其精神状态。
|
||||
- type=scene(纯场景/风景):描述空间或风景的构成、色彩、光线、氛围与关键物件。
|
||||
|
||||
【人物属性硬性要求(has_person=true时必须遵守)】
|
||||
hair/skin_tone/face_shape/outfit四项绝对禁止填“无法判断”,必须基于图片可见特征给出具体中文描述:
|
||||
- hair:必须描述发型+发色,如“黑色齐肩直发”“棕色微卷中长发”“深棕色短发”
|
||||
- skin_tone:必须描述肤色,如“暖调自然肤色”“白皙肤色”“小麦色”
|
||||
- face_shape:必须描述脸型,如“鹅蛋脸”“圆脸”“瓜子脸”“方脸”
|
||||
- outfit:必须描述可见穿着,如“米色翻领衬衫”“白色T恤”“黑色连衣裙”
|
||||
即使局部被遮挡也要根据可见部分合理推断;确实看不清时按最接近的直观印象描述。
|
||||
## 判断规则
|
||||
- has_person:画面中出现可辨识的真实人物(脸或完整上半身)才为 true,海报/模特立牌/照片里的人不算。
|
||||
- 一张图只描述其本身;多张图属于同一场景时可呼应,但不编造对应关系。
|
||||
- 输出必须是严格 JSON,summary_markdown 是字符串,内部换行用 \\n 表示。"""
|
||||
|
||||
其他非人物属性看不到或无法判断时填“无法判断”,布尔值填false,不要留空标签。
|
||||
_IMAGE_ANALYSIS_USER = """请分析这张图片。
|
||||
图片地址:{image_url}
|
||||
OCR 辅助文字(可能为空,仅供参考,不要照抄错误识别):{ocr_text}
|
||||
|
||||
【有人物场景输出参考(女性手持商品示例,必须写全10个属性,禁止省略)】
|
||||
<people has_person="true" count="1" gender="女" age_range="青年" hair="黑色齐肩直发" skin_tone="暖调自然肤色" face_shape="鹅蛋脸" outfit="米色翻领衬衫" pose="正面半身,手持商品" expression="面带微笑"/>"""
|
||||
严格按系统要求只输出 JSON。"""
|
||||
|
||||
_IMAGE_ANALYSIS_USER = """请分析以下商品图片,共 {image_count} 张。
|
||||
所属行业:{industry}
|
||||
图片地址:
|
||||
{image_urls}
|
||||
_IMAGE_ANALYSIS_EXAMPLE = """{
|
||||
"images": [
|
||||
{
|
||||
"type": "store",
|
||||
"name": "御众堂门店",
|
||||
"brand": "御众堂",
|
||||
"has_person": false,
|
||||
"summary_markdown": "###店铺主体\\n这是一家名为“御众堂”的线下门店内部,整体暖木色调……"
|
||||
}
|
||||
]
|
||||
}"""
|
||||
|
||||
按约定的标签格式输出分析结果。"""
|
||||
# ── 模板2:编导级分镜 v3(意图理解 + 分镜一次完成)────────────────────
|
||||
_STORYBOARD_SYSTEM = (
|
||||
"""你是一名懂短视频的编导和口播文案高手。你会拿到图片的真实观察、营销目的和用户参数,请一次性完成对营销意图的理解,并产出可直接拍摄/生成的分镜脚本。不要单独输出“意图解析”,意图要直接体现在台词和分镜里。
|
||||
|
||||
_IMAGE_ANALYSIS_EXAMPLE = """<products>
|
||||
<product name="大公鸡头 多功能油污净 625ml" features="红色瓶盖白色瓶身,鸡头图案Logo" position="main" image_index="0"/>
|
||||
</products>
|
||||
<colors>
|
||||
<color hex="#D32F2F" name="红色" coverage="0.4"/>
|
||||
<color hex="#FFFFFF" name="白色" coverage="0.5"/>
|
||||
</colors>
|
||||
<people has_person="false" count="0" gender="无法判断" age_range="无法判断" hair="无法判断" skin_tone="无法判断" face_shape="无法判断" outfit="无法判断" pose="无法判断" expression="无法判断"/>
|
||||
<mood>干净、实用</mood>
|
||||
<visible_text>
|
||||
<text_item text="多功能油污净" position="瓶身正面"/>
|
||||
</visible_text>
|
||||
<scene>白底棚拍产品图</scene>
|
||||
<quality resolution="高清" lighting="均匀柔和" composition="主体居中" blur="false"/>
|
||||
<key_selling_points>
|
||||
<point>针对重油污设计</point>
|
||||
<point>大容量625ml</point>
|
||||
</key_selling_points>"""
|
||||
## 输出格式(XML,严格按结构输出,不要输出额外解释)
|
||||
<script>
|
||||
<copy_display_markdown><![CDATA[直接展示给用户看的成片文案,用 Markdown 写成流畅叙述]]></copy_display_markdown>
|
||||
<clips>
|
||||
<clip index="1">
|
||||
<time_range>0-3秒</time_range>
|
||||
<voiceover>这一镜的口播台词</voiceover>
|
||||
<visual>具体、有画面感的镜头描述(主体/动作/镜头运动/景别/光线)</visual>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
</clip>
|
||||
</clips>
|
||||
<voiceover_script>把所有 clip 的 voiceover 连成完整口播稿</voiceover_script>
|
||||
<theme>一句话主题</theme>
|
||||
<negative>"""
|
||||
+ NEGATIVE_RULES
|
||||
+ """</negative>
|
||||
</script>
|
||||
|
||||
# ── 模板2:用户文案意图解析(LLM)──────────────────────────────────────
|
||||
_INTENT_SYSTEM = f"""你负责理解用户的营销意图。用户给的文案可能只是几个关键词、碎碎念或者不完整的短句,你要读懂他真正想讲什么。
|
||||
## 写作要求
|
||||
1. 口播台词:像真人面对镜头说话,短句、口语化、有停顿有情绪,开头 3 秒给出钩子;不要书面腔,不要机械报参数。
|
||||
2. 画面描述:写清“观众会看到什么”,有动作、有镜头运动、有景别和光线,具体可拍;不堆砌形容词,不写无法实现的画面。
|
||||
3. copy_display_markdown:直接展示给最终用户的文案,用 Markdown 写成自然、流畅、有感染力的成片成片文案,可用小标题与短句组织;不要做字段列表,不要出现“镜头一/台词:”这类制作说明。
|
||||
4. 内容必须来自图片观察与用户给出的信息,不编造卖点、不夸大、不使用绝对化用语和虚假承诺。
|
||||
5. reference_image_index 填本镜参考图片序号(从 0 开始),没有合适参考图填 -1。
|
||||
6. 分镜数量与时长匹配总时长,节奏紧凑。"""
|
||||
)
|
||||
|
||||
{GLOBAL_CONSTRAINTS}
|
||||
_STORYBOARD_USER = """<marketing_purpose>{marketing_purpose}</marketing_purpose>
|
||||
<image_analysis>
|
||||
{image_summary}
|
||||
</image_analysis>
|
||||
<user_parameters>
|
||||
<theme_hint>{theme_hint}</theme_hint>
|
||||
<duration>{duration}秒</duration>
|
||||
<aspect_ratio>{aspect_ratio}</aspect_ratio>
|
||||
<tone>{tone}</tone>
|
||||
<target_audience>{target_audience}</target_audience>
|
||||
<extra_requirements>{extra_requirements}</extra_requirements>
|
||||
</user_parameters>
|
||||
{video_style_section}
|
||||
请严格按 XML 结构输出分镜脚本。"""
|
||||
|
||||
请严格按下面的标签格式输出,不要解释,不要用代码块:
|
||||
<intent_summary> 用用户的语言风格,一句话、30字以内概括核心意图。
|
||||
<core_messages> 下面每个核心信息点用一个 <message> 标签,属性 must_keep 为 true 或 false、confidence 为 0 到 1 的小数,标签内容写信息点。
|
||||
<personal_brands> 把用户提到的具体事实——品牌名、价格、人名、地名、时间、产品名——每条用一个 <brand> 标签,属性 category 取 brand、price、person、place、time、product 之一。这些事实必须原样引用,一个字都不能改。
|
||||
<emotion_tone> 写文案的情绪调性。
|
||||
<missing_info> 把你认为缺失、后续生成时需要合理推断的信息,每条用一个 <info> 标签;没有就输出空标签。"""
|
||||
_STORYBOARD_EXAMPLE = """<script>
|
||||
<copy_display_markdown><![CDATA[# 在御众堂,把松弛的自己一点点找回来
|
||||
产后妈妈最懂那种力不从心,推开门,暖光和一杯热茶先接住了你……]]></copy_display_markdown>
|
||||
<clips>
|
||||
<clip index="1">
|
||||
<time_range>0-3秒</time_range>
|
||||
<voiceover>生完娃,是不是连照镜子的勇气都没了?</voiceover>
|
||||
<visual>中近景,暖光下一位妈妈略显疲惫地看向镜中,镜头缓缓推近</visual>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
</clip>
|
||||
</clips>
|
||||
<voiceover_script>生完娃,是不是连照镜子的勇气都没了?</voiceover_script>
|
||||
<theme>产后妈妈走进御众堂重拾状态</theme>
|
||||
<negative>模糊、畸变、夸大疗效、绝对化用语</negative>
|
||||
</script>"""
|
||||
|
||||
_INTENT_USER = """用户原始文案:{user_copy_text}
|
||||
所属行业:{industry}
|
||||
营销目的:{marketing_purpose}
|
||||
图片分析结果(供参考):
|
||||
{image_analysis}
|
||||
图片类型推断:{image_category_hint}
|
||||
# ── 模板3:文案审核(合规/质量门禁)───────────────────────────────────
|
||||
_REVIEW_SYSTEM = """你是一名短视频广告合规审核与文案优化专家。审核待审文案:
|
||||
1) 广告法与平台合规(绝对化用语、虚假承诺、医疗功效宣称、导流违规);
|
||||
2) 卖点是否聚焦、逻辑是否通顺、口播是否自然;
|
||||
3) 是否有机械堆砌、书面腔、标签化表述。
|
||||
|
||||
请理解用户意图,按标签格式输出。注意:theme和emotion_tone应与图片类型和营销目的匹配——门店类图片偏向"门店探店/到店体验",商品图偏向"好物分享/产品种草",人物图偏向"穿搭/人物故事"。"""
|
||||
只输出 XML,结构:
|
||||
<review>
|
||||
<passed>true 或 false</passed>
|
||||
<issues>
|
||||
<issue>
|
||||
<severity>high 或 medium 或 low</severity>
|
||||
<field>问题所在位置/字段</field>
|
||||
<problem>具体问题</problem>
|
||||
<suggestion>可直接替换的修改</suggestion>
|
||||
</issue>
|
||||
</issues>
|
||||
<rewrite>整体重写后的合规流畅版本(无问题时留空)</rewrite>
|
||||
</review>
|
||||
没有问题时 issues 留空、passed 为 true、rewrite 留空。"""
|
||||
|
||||
_INTENT_EXAMPLE = """<intent_summary>一款厨房去油污神器,喷一喷油污就掉</intent_summary>
|
||||
<core_messages>
|
||||
<message must_keep="true" confidence="0.97">去油污效果好,喷上等几分钟再擦</message>
|
||||
<message must_keep="false" confidence="0.7">适合厨房重油污场景</message>
|
||||
</core_messages>
|
||||
<personal_brands>
|
||||
<brand category="product">大公鸡头多功能油污净</brand>
|
||||
<brand category="price">39块钱一瓶</brand>
|
||||
</personal_brands>
|
||||
<emotion_tone>亲切、真实、带分享感</emotion_tone>
|
||||
<missing_info>
|
||||
<info>没有说明具体容量,按图片读出的625ml处理</info>
|
||||
</missing_info>"""
|
||||
_REVIEW_USER = """<fusion_text>
|
||||
{fusion_text}
|
||||
</fusion_text>
|
||||
|
||||
# ── 模板3:文案融合生成(LLM)──────────────────────────────────────────
|
||||
_FUSION_SYSTEM = """你负责为短视频生成营销文案。请按思维链分步完成:先定人设和目标客户,再找卖点,再搭结构,再安排情绪,最后写行动号召,不要一步到位乱写。
|
||||
请审核以上文案。"""
|
||||
|
||||
{fusion_instruction}
|
||||
|
||||
{global_constraints}
|
||||
|
||||
{negative_rules}
|
||||
|
||||
请严格按下面的标签格式输出,不要解释,不要用代码块:
|
||||
<title> 视频标题。
|
||||
<hook> 开头3秒钩子,5到15字。
|
||||
<body_points> 每个要点用一个 <point> 标签,属性 elaboration 是展开说明、image_index 是对应第几张图(从0开始),标签内容写要点。
|
||||
<cta> 口语化的行动号召。
|
||||
<script_segments> 每段配音用一个 <segment> 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案(纯口播文本,不加旁白标注、不加镜头标注、不加"主播:"之类前缀)。
|
||||
<voiceover_script> 把所有 segment 的配音文案按顺序自然拼接成一段完整的纯口播文本(无标记、无括号、无前缀),长度要适配 {duration} 秒,约 {approx_chars} 字。
|
||||
<overview_theme> 视频主题(一句话概括)。
|
||||
<scene_and_lighting> 整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)。
|
||||
<word_count> 配音总字数,只写数字。
|
||||
<estimated_duration> 预计时长秒数,只写数字。
|
||||
|
||||
用户在 personal_brands 中提到的品牌名、价格、人名、地名、时间、产品名等事实信息,必须原样出现在文案里,一个字都不能改。"""
|
||||
|
||||
_FUSION_USER = """所属行业:{industry}
|
||||
目标客户:{target_customer}
|
||||
营销目的:{marketing_purpose}
|
||||
视频时长:{duration}秒
|
||||
图片分析结果:
|
||||
{image_analysis}
|
||||
用户意图解析结果:
|
||||
{intent_result}
|
||||
|
||||
请按标签格式生成文案。"""
|
||||
|
||||
_FUSION_EXAMPLE = """<title>厨房重油污,别再用洗洁精硬擦了</title>
|
||||
<hook>这油污,我真的忍很久了</hook>
|
||||
<body_points>
|
||||
<point elaboration="喷在油污上等几分钟,一擦就干净" image_index="0">大公鸡头油污净去油快</point>
|
||||
<point elaboration="39块钱625ml,能用很久" image_index="0">39块钱一瓶,性价比高</point>
|
||||
</body_points>
|
||||
<cta>厨房油污重的,真的可以试一瓶</cta>
|
||||
<script_segments>
|
||||
<segment duration_sec="3" image_index="0">这油污我真的忍很久了,用洗洁精擦半天都没用</segment>
|
||||
<segment duration_sec="6" image_index="0">后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</segment>
|
||||
<segment duration_sec="4" image_index="0">39块钱625ml,厨房重油污的可以试一瓶</segment>
|
||||
</script_segments>
|
||||
<voiceover_script>这油污我真的忍很久了,用洗洁精擦半天都没用。后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净。39块钱625ml,厨房重油污的可以试一瓶。</voiceover_script>
|
||||
<overview_theme>厨房好物分享·产品种草</overview_theme>
|
||||
<scene_and_lighting>简洁明亮的厨房台面场景,自然光从窗户洒入,色调温暖柔和,突出产品白色瓶身与去油污对比效果。</scene_and_lighting>
|
||||
<word_count>58</word_count>
|
||||
<estimated_duration>13</estimated_duration>"""
|
||||
|
||||
# ── 模板4:编导级分镜(LLM)────────────────────────────────────────────
|
||||
_STORYBOARD_SYSTEM = """你是短视频编导,负责把文案拆成可拍摄的分镜,为 Seedance 2.5 视频模型写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。
|
||||
|
||||
工作方式:
|
||||
1. 按文案的 script_segments 顺序分配镜头。
|
||||
2. 每个镜头确定景别/角度/运镜、画面场景与对白、人物动作细节、音效/BGM、转场。
|
||||
3. 检查所有镜头时长加起来接近目标时长,误差不超过2秒。
|
||||
4. image_index 必须在已上传图片范围内,第一张主图必须用在第一个镜头。
|
||||
|
||||
{fusion_instruction}
|
||||
|
||||
{global_constraints}
|
||||
|
||||
{negative_rules}
|
||||
|
||||
请严格按下面的标签格式输出,不要解释,不要用代码块:
|
||||
<clips> 下面每个镜头用一个 <clip> 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade/cut/zoom_in/slide_left/dissolve/wipe 之一、zoom 取 in/out/null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 <clip> 里面包含:
|
||||
<voice_text> 该镜头配音文本(纯口播文本,不加旁白标注);
|
||||
<subtitle_text> 字幕文本,可与配音一致或更精简;
|
||||
<shot_type_angle_movement> 景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜);
|
||||
<scene_and_dialogue> 画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔);
|
||||
<action_details> 人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示);
|
||||
<audio_bgm> 环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音);
|
||||
<transition> 硬切/淡入淡出/叠化(最后一镜写『结束』即可);
|
||||
<reference_image_index> 参考图片索引(0-based,对应第几张产品图,无则空);
|
||||
<ken_burns> 用一个空标签,属性 start、end 写"x,y"坐标、ease 写缓动方式;不需要运镜时坐标相同。"""
|
||||
|
||||
_STORYBOARD_USER = """目标时长:{duration}秒
|
||||
上传图片数量:{image_count}张(第1张是主图/封面)
|
||||
文案内容:
|
||||
{fusion_result}
|
||||
图片分析结果:
|
||||
{image_analysis}
|
||||
|
||||
重要:overview_theme 必须与图片实际内容和营销目的匹配。门店/餐饮/服务类图片用"门店探店·到店体验";商品图用"好物分享·产品种草";人物图用"穿搭分享·人物故事";场景图用"空间体验·场景氛围"。不要对所有图片都使用"好物分享"。
|
||||
|
||||
请按标签格式输出分镜。"""
|
||||
|
||||
_STORYBOARD_EXAMPLE = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="3" bgm_note="日常、轻微烦躁">
|
||||
<voice_text>这油污我真的忍很久了</voice_text>
|
||||
<subtitle_text>这油污忍很久了</subtitle_text>
|
||||
<shot_type_angle_movement>近景俯拍45度,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>厨房台面,主妇皱眉看着灶台油污。对白:这油污我真的忍很久了</scene_and_dialogue>
|
||||
<action_details>右手拿着脏抹布,无奈摇头</action_details>
|
||||
<audio_bgm>轻快日常BGM,带一点烦躁感</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="zoom_in" zoom="in" duration_sec="6" bgm_note="轻快、出现转机">
|
||||
<voice_text>后来换了大公鸡头油污净,喷上等几分钟,一擦就干净</voice_text>
|
||||
<subtitle_text>喷上等几分钟,一擦就干净</subtitle_text>
|
||||
<shot_type_angle_movement>特写平视,固定镜头</shot_type_angle_movement>
|
||||
<scene_and_dialogue>手部特写,喷油污净在油污处。对白:后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</scene_and_dialogue>
|
||||
<action_details>左手拿产品瓶身,右手按压喷头,等待片刻后用抹布轻擦</action_details>
|
||||
<audio_bgm>轻快转折BGM,带清爽感</audio_bgm>
|
||||
<transition>淡入淡出</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="20,20" end="80,80" ease="ease-in-out"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="fade" zoom="null" duration_sec="4" bgm_note="温暖、推荐">
|
||||
<voice_text>39块钱625ml,厨房重油污的可以试一瓶</voice_text>
|
||||
<subtitle_text>39元625ml,可以试一瓶</subtitle_text>
|
||||
<shot_type_angle_movement>中景平视,缓慢拉镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>产品正面展示,明亮背景。对白:39块钱625ml,厨房重油污的可以试一瓶</scene_and_dialogue>
|
||||
<action_details>产品置于画面中央,轻微转动展示瓶身</action_details>
|
||||
<audio_bgm>温暖收尾BGM</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="50,50" end="20,20" ease="ease-in-out"/>
|
||||
</clip>
|
||||
</clips>"""
|
||||
|
||||
# ── 模板5:文案审核(LLM)──────────────────────────────────────────────
|
||||
_REVIEW_SYSTEM = f"""你是短视频文案合规审核员,从6个维度逐条检查文案:
|
||||
1. 违规词:有没有平台禁用词、敏感词。
|
||||
2. 夸大承诺:有没有“包治百病”“100%有效”“保证赚钱”等绝对化、夸大表述。
|
||||
3. 事实一致性:有没有编造价格、数据、认证,或者用户没提到的产品特性。
|
||||
4. 用户意图保留:在 ai_polish 和 user_primary 模式下,core_messages 中 must_keep=true 的点是否都保留了。
|
||||
5. 结构完整性:标题、钩子、正文、行动号召是否齐全。
|
||||
6. 语气人设:是否符合选定的人设语气,有没有“家人们谁懂啊”“绝绝子”“宝子们”等套路词。
|
||||
|
||||
{GLOBAL_CONSTRAINTS}
|
||||
|
||||
请严格按下面的标签格式输出,不要解释,不要用代码块:
|
||||
<passed> 整体是否通过,只写 true 或 false。
|
||||
<issues> 每个问题用一个 <issue> 标签,属性 dimension 是维度名、severity 取 error 或 warning、location 是问题所在(如 hook、body_points、cta),标签内容写问题描述;没有问题就输出空标签。
|
||||
<rewrite_suggestions> 每条具体修改建议用一个 <suggestion> 标签;没有就输出空标签。"""
|
||||
|
||||
_REVIEW_USER = """本次创作模式:{fusion_level}
|
||||
待审核文案:
|
||||
{fusion_result}
|
||||
用户意图解析(用于核对核心信息是否保留):
|
||||
{intent_result}
|
||||
|
||||
请按6个维度审核,按标签格式输出。"""
|
||||
|
||||
_REVIEW_EXAMPLE = """<passed>false</passed>
|
||||
<issues>
|
||||
<issue dimension="夸大承诺" severity="error" location="body_points">出现了“一喷100%掉光”的绝对化表述,违反广告法</issue>
|
||||
<issue dimension="用户意图保留" severity="warning" location="cta">用户强调的“39块钱”没有保留</issue>
|
||||
</issues>
|
||||
<rewrite_suggestions>
|
||||
<suggestion>把“一喷100%掉光”改为“喷上等几分钟,大部分油污能擦掉”</suggestion>
|
||||
<suggestion>在结尾补回“39块钱625ml”</suggestion>
|
||||
</rewrite_suggestions>"""
|
||||
_REVIEW_EXAMPLE = """<review>
|
||||
<passed>false</passed>
|
||||
<issues>
|
||||
<issue>
|
||||
<severity>high</severity>
|
||||
<field>opening</field>
|
||||
<problem>使用绝对化用语“全网第一”</problem>
|
||||
<suggestion>改为“很多老客户回购的一款”</suggestion>
|
||||
</issue>
|
||||
</issues>
|
||||
<rewrite>……</rewrite>
|
||||
</review>"""
|
||||
|
||||
|
||||
# 5 套模板默认数据(seed 数据源与 loader 的兜底)
|
||||
DEFAULT_TEMPLATES: list[dict] = [
|
||||
{
|
||||
"name": "图片多模态分析",
|
||||
"name": "图片多模态分析 v8",
|
||||
"prompt_type": "image_analysis",
|
||||
"version": TEMPLATE_VERSION,
|
||||
"version": 8,
|
||||
"system_prompt": _IMAGE_ANALYSIS_SYSTEM,
|
||||
"user_prompt_template": _IMAGE_ANALYSIS_USER,
|
||||
"example_output": _IMAGE_ANALYSIS_EXAMPLE,
|
||||
"is_active": True,
|
||||
},
|
||||
{
|
||||
"name": "用户文案意图解析",
|
||||
"prompt_type": "intent_parsing",
|
||||
"version": TEMPLATE_VERSION,
|
||||
"system_prompt": _INTENT_SYSTEM,
|
||||
"user_prompt_template": _INTENT_USER,
|
||||
"example_output": _INTENT_EXAMPLE,
|
||||
"is_active": True,
|
||||
},
|
||||
{
|
||||
"name": "文案融合生成",
|
||||
"prompt_type": "copy_fusion",
|
||||
"version": TEMPLATE_VERSION,
|
||||
"system_prompt": _FUSION_SYSTEM,
|
||||
"user_prompt_template": _FUSION_USER,
|
||||
"example_output": _FUSION_EXAMPLE,
|
||||
"is_active": True,
|
||||
},
|
||||
{
|
||||
"name": "编导级分镜",
|
||||
"name": "编导级分镜 v3",
|
||||
"prompt_type": "storyboard",
|
||||
"version": TEMPLATE_VERSION,
|
||||
"version": 3,
|
||||
"system_prompt": _STORYBOARD_SYSTEM,
|
||||
"user_prompt_template": _STORYBOARD_USER,
|
||||
"example_output": _STORYBOARD_EXAMPLE,
|
||||
@@ -333,7 +221,7 @@ DEFAULT_TEMPLATES: list[dict] = [
|
||||
{
|
||||
"name": "文案审核",
|
||||
"prompt_type": "review",
|
||||
"version": TEMPLATE_VERSION,
|
||||
"version": 1,
|
||||
"system_prompt": _REVIEW_SYSTEM,
|
||||
"user_prompt_template": _REVIEW_USER,
|
||||
"example_output": _REVIEW_EXAMPLE,
|
||||
|
||||
@@ -13,6 +13,13 @@ from typing import Optional
|
||||
_OPEN_RE = re.compile(r"<(?P<tag>[\w-]+)(?P<attrs>(?:\s(?:[^>]*?\S)?)?)(?P<self>/?)>")
|
||||
_CLOSE_RE = re.compile(r"</(?P<tag>[\w-]+)\s*>")
|
||||
_ATTR_RE = re.compile(r"""([\w:-]+)\s*=\s*(?:"([^"]*)"|'([^']*)')""")
|
||||
_CDATA_RE = re.compile(r"^<!\[CDATA\[(.*)\]\]>$", re.DOTALL)
|
||||
|
||||
|
||||
def _strip_cdata(s: str) -> str:
|
||||
"""剥离 LLM 可能照抄示例输出的 ``<![CDATA[...]]>`` 包裹层。"""
|
||||
m = _CDATA_RE.match(s.strip())
|
||||
return m.group(1) if m else s
|
||||
|
||||
|
||||
def parse_attributes(raw: str) -> dict[str, str]:
|
||||
@@ -58,6 +65,7 @@ def parse_tags(text: Optional[str]) -> list[dict]:
|
||||
if stack[idx]["tag"] == tag:
|
||||
node = stack[idx]
|
||||
node["text"] = unescape(text[node["_start"] : token.start()].strip())
|
||||
node["text"] = _strip_cdata(node["text"])
|
||||
node.pop("_start", None)
|
||||
del stack[idx:]
|
||||
break
|
||||
@@ -65,6 +73,7 @@ def parse_tags(text: Optional[str]) -> list[dict]:
|
||||
for node in stack:
|
||||
if "_start" in node:
|
||||
node["text"] = unescape(text[node["_start"] :].strip())
|
||||
node["text"] = _strip_cdata(node["text"])
|
||||
node.pop("_start", None)
|
||||
return results
|
||||
|
||||
|
||||
@@ -350,6 +350,29 @@ class TestViralVideoRepository:
|
||||
class TestViralVideoPipeline:
|
||||
"""编排器流水线测试。"""
|
||||
|
||||
# v3 分镜 XML(copy_display_markdown + clips + voiceover_script)
|
||||
V3_XML = """<copy_display_markdown>今天给大家分享一支很显白的口红。</copy_display_markdown>
|
||||
<clips>
|
||||
<clip image_index="0" time_range="0-5秒">
|
||||
<voiceover>大家好,今天分享一款口红</voiceover>
|
||||
<visual>近景平视,缓慢推镜</visual>
|
||||
<action_details>手持口红特写</action_details>
|
||||
<audio_bgm>轻快流行BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
</clip>
|
||||
<clip image_index="1" time_range="5-15秒">
|
||||
<voiceover>颜色特别好看很显白</voiceover>
|
||||
<visual>特写,固定镜头</visual>
|
||||
<action_details>嘴唇涂抹特写</action_details>
|
||||
<audio_bgm>轻快BGM继续</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>1</reference_image_index>
|
||||
</clip>
|
||||
</clips>
|
||||
<voiceover_script>大家好,今天分享一款口红。颜色特别好看很显白</voiceover_script>
|
||||
<theme>口红分享</theme>"""
|
||||
|
||||
@pytest.fixture
|
||||
def mock_job(self):
|
||||
return ViralVideoJob(
|
||||
@@ -364,23 +387,27 @@ class TestViralVideoPipeline:
|
||||
video_ratio="9:16",
|
||||
)
|
||||
|
||||
@patch("packages.shared.ai_service.call_vision")
|
||||
@patch("apps.worker.worker_app.tasks.vision.analyze_images_v2")
|
||||
def test_image_analysis_step(self, mock_vision, mock_job):
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_image_analysis
|
||||
|
||||
mock_vision.return_value = {"name": "口红", "features": ["持久", "滋润"]}
|
||||
# 每张图返回一个 v8 5 字段结果
|
||||
mock_vision.return_value = [
|
||||
{"type": "product", "name": "口红", "brand": "", "has_person": False, "summary_markdown": "一支口红"},
|
||||
{"type": "product", "name": "口红", "brand": "", "has_person": False, "summary_markdown": "口红特写"},
|
||||
]
|
||||
result = _step_image_analysis(mock_job)
|
||||
assert "products" in result
|
||||
assert len(result["products"]) == 2 # 两张图片
|
||||
assert "images" in result
|
||||
assert len(result["images"]) == 2 # 两张图片
|
||||
|
||||
@patch("packages.shared.ai_service.call_vision")
|
||||
@patch("apps.worker.worker_app.tasks.vision.analyze_images_v2")
|
||||
def test_image_analysis_fallback(self, mock_vision, mock_job):
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_image_analysis
|
||||
|
||||
# 模拟 call_vision 不存在
|
||||
mock_vision.side_effect = ImportError("no module")
|
||||
# v2 分析内部异常时,每图走兜底,仍返回 images 结构
|
||||
mock_vision.side_effect = RuntimeError("vision unavailable")
|
||||
result = _step_image_analysis(mock_job)
|
||||
assert "products" in result
|
||||
assert "images" in result
|
||||
|
||||
def test_video_analysis_no_reference(self, mock_job):
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_video_analysis
|
||||
@@ -390,46 +417,40 @@ class TestViralVideoPipeline:
|
||||
result = _step_video_analysis(mock_job)
|
||||
assert result is None
|
||||
|
||||
@patch("packages.shared.ai_service.call_llm")
|
||||
def test_intent_parsing(self, mock_llm, mock_job):
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_intent_parsing
|
||||
def test_intent_parsing_step_removed(self, mock_job):
|
||||
"""intent_parsing 已合并进脚本生成,不再作为独立步骤/函数存在。"""
|
||||
import apps.worker.worker_app.tasks.viral_video as vv
|
||||
|
||||
mock_llm.return_value = {"intent": "推广口红", "tone": "活泼"}
|
||||
result = _step_intent_parsing(mock_job, {"products": []})
|
||||
assert "intent" in result
|
||||
assert not hasattr(vv, "_step_intent_parsing")
|
||||
|
||||
@patch("packages.shared.ai_service.call_llm")
|
||||
def test_script_generation_returns_copy_result(self, mock_llm, mock_job):
|
||||
def test_script_generation_returns_copy_result(self, mock_job):
|
||||
"""v1.6: _step_script_generation 返回 dict 形式的 CopyResult,含 voiceover_script + shots。"""
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_script_generation
|
||||
from packages.shared.ai_router import ai_router
|
||||
|
||||
mock_llm.return_value = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="5" bgm_note="轻快流行BGM">
|
||||
<voice_text>大家好,今天分享一款口红</voice_text>
|
||||
<subtitle_text>大家好,今天分享一款口红</subtitle_text>
|
||||
<shot_type_angle_movement>近景平视,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>女主微笑展示口红:大家好,今天分享一款口红</scene_and_dialogue>
|
||||
<action_details>手持口红特写</action_details>
|
||||
<audio_bgm>轻快流行BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="fade" zoom="null" duration_sec="10" bgm_note="轻快BGM">
|
||||
<voice_text>颜色特别好看很显白</voice_text>
|
||||
<subtitle_text>颜色特别好看很显白</subtitle_text>
|
||||
<shot_type_angle_movement>特写,固定镜头</shot_type_angle_movement>
|
||||
<scene_and_dialogue>涂抹口红:颜色特别好看很显白</scene_and_dialogue>
|
||||
<action_details>嘴唇涂抹特写</action_details>
|
||||
<audio_bgm>轻快BGM继续</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>1</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
</clips>"""
|
||||
result = _step_script_generation(
|
||||
mock_job, {"intent": "推广口红", "key_messages": [], "tone": "亲切"}, {"products": []}
|
||||
)
|
||||
class _FakeClient:
|
||||
is_available = True
|
||||
model = "fake-storyboard"
|
||||
|
||||
def __init__(self, xml: str):
|
||||
self._xml = xml
|
||||
|
||||
def chat_completion(self, messages, **kwargs):
|
||||
return self._xml
|
||||
|
||||
fake = _FakeClient(self.V3_XML)
|
||||
orig_get = ai_router.get_llm_client
|
||||
|
||||
def _get(task, variant="primary"):
|
||||
if task == "storyboard":
|
||||
return fake
|
||||
return orig_get(task, variant=variant)
|
||||
|
||||
ai_router.get_llm_client = _get # type: ignore
|
||||
try:
|
||||
result = _step_script_generation(mock_job, {"images": []})
|
||||
finally:
|
||||
ai_router.get_llm_client = orig_get # type: ignore
|
||||
assert isinstance(result, dict)
|
||||
assert "voiceover_script" in result
|
||||
assert "shots" in result
|
||||
@@ -501,7 +522,6 @@ class TestPipelineIntegration:
|
||||
@patch("apps.worker.worker_app.tasks.viral_video._step_tts")
|
||||
@patch("apps.worker.worker_app.tasks.viral_video._step_review")
|
||||
@patch("apps.worker.worker_app.tasks.viral_video._step_script_generation")
|
||||
@patch("apps.worker.worker_app.tasks.viral_video._step_intent_parsing")
|
||||
@patch("apps.worker.worker_app.tasks.viral_video._step_video_analysis")
|
||||
@patch("apps.worker.worker_app.tasks.viral_video._step_image_analysis")
|
||||
@patch("apps.worker.worker_app.tasks.viral_video._get_repo_and_job")
|
||||
@@ -512,7 +532,6 @@ class TestPipelineIntegration:
|
||||
mock_get_repo,
|
||||
mock_img_analysis,
|
||||
mock_video_analysis,
|
||||
mock_intent,
|
||||
mock_script,
|
||||
mock_review,
|
||||
mock_tts,
|
||||
@@ -539,8 +558,6 @@ class TestPipelineIntegration:
|
||||
mock_session = MagicMock()
|
||||
mock_get_repo.return_value = (mock_session, mock_repo, job)
|
||||
|
||||
# v1.6: 如果没有 copy_result 会现场补生成
|
||||
mock_intent.return_value = {"intent": "推广", "key_messages": [], "tone": "亲切"}
|
||||
mock_script.return_value = {
|
||||
"overview": {"theme": "口红", "total_duration": 15, "aspect_ratio": "9:16"},
|
||||
"scene_and_lighting": "明亮化妆台",
|
||||
@@ -553,6 +570,8 @@ class TestPipelineIntegration:
|
||||
mock_review.return_value = {"passed": True, "score": 90}
|
||||
mock_tts.return_value = None # TTS 失败也能走下去(Seedance generate_audio=True 会自己合成音效)
|
||||
mock_tts_upload.return_value = None
|
||||
# _run_render_pipeline 直接读 job.copy_result(#2218 守卫),需提前注入
|
||||
job.copy_result = mock_script.return_value
|
||||
mock_render.return_value = ("/tmp/video.mp4", {"completion_tokens": 1000000})
|
||||
mock_upload.return_value = "https://oss.example.com/final.mp4"
|
||||
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
"""#2040 爆款视频 Prompt 模板系统单测。
|
||||
"""#2040 爆款视频 Prompt 模板系统单测(v8/v3 叙述优先重构后)。
|
||||
|
||||
不真调豆包 API,全部用 FakeClient 注入;覆盖:
|
||||
XML 标签解析 / 5 套模板纯文本 / loader 缓存热加载与回落 /
|
||||
三档融合差异 / personal_brands 保留 / 审核识别违规词夸大 / 自动重写 /
|
||||
各步 fallback / seed 幂等 / 负面词不出现。
|
||||
XML 标签解析 / 3 套模板纯文本(image_analysis/storyboard/review)/
|
||||
loader 缓存热加载与回落 / 本地规则审核识别违规词夸大 /
|
||||
各现存步 fallback / seed 幂等 / 负面词不出现。
|
||||
|
||||
注:intent_parsing、copy_fusion 两套模板及其独立步骤已在叙述优先重构中删除,
|
||||
相关用例同步移除。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -30,7 +33,6 @@ from packages.application.viral_video.prompt_loader import ( # noqa: E402
|
||||
from packages.application.viral_video.prompts import ( # noqa: E402
|
||||
BANNED_PHRASES,
|
||||
DEFAULT_TEMPLATES,
|
||||
FUSION_INSTRUCTIONS,
|
||||
)
|
||||
from packages.application.viral_video.reviewer import Reviewer # noqa: E402
|
||||
|
||||
@@ -45,26 +47,6 @@ IMAGE_XML = """<products>
|
||||
<quality resolution="高清" lighting="柔和" composition="居中"/>
|
||||
<key_selling_points><point>去油快</point><point>625ml大容量</point></key_selling_points>"""
|
||||
|
||||
INTENT_XML = """<intent_summary>厨房去油污神器</intent_summary>
|
||||
<core_messages>
|
||||
<message must_keep="true" confidence="0.97">去油污效果好</message>
|
||||
<message must_keep="false" confidence="0.6">适合重油污</message>
|
||||
</core_messages>
|
||||
<personal_brands><brand category="price">39块钱一瓶</brand></personal_brands>
|
||||
<emotion_tone>亲切真实</emotion_tone>
|
||||
<missing_info><info>容量按625ml</info></missing_info>"""
|
||||
|
||||
FUSION_XML = """<title>厨房重油污别硬擦了</title>
|
||||
<hook>这油污忍很久了</hook>
|
||||
<body_points><point elaboration="喷上等几分钟一擦就净" image_index="0">大公鸡头去油快</point></body_points>
|
||||
<cta>重油污的可以试一瓶</cta>
|
||||
<script_segments>
|
||||
<segment duration_sec="3" image_index="0">这油污忍很久了</segment>
|
||||
<segment duration_sec="6" image_index="0">大公鸡头油污净喷上等几分钟一擦就净</segment>
|
||||
<segment duration_sec="4" image_index="0">39块钱一瓶可以试一下</segment>
|
||||
</script_segments>
|
||||
<word_count>52</word_count><estimated_duration>13</estimated_duration>"""
|
||||
|
||||
STORYBOARD_XML = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="3" bgm_note="日常">
|
||||
<voice_text>这油污忍很久了</voice_text>
|
||||
@@ -86,8 +68,6 @@ REVIEW_PASS_XML = """<passed>true</passed>
|
||||
<issues></issues>
|
||||
<rewrite_suggestions></rewrite_suggestions>"""
|
||||
|
||||
FIXED_FUSION_XML = FUSION_XML.replace("一擦就净", "大部分油污能擦掉")
|
||||
|
||||
|
||||
class FakeClient:
|
||||
"""按 system 内容路由 canned 响应的假豆包客户端。"""
|
||||
@@ -96,45 +76,20 @@ class FakeClient:
|
||||
self.chat_calls: list[list[dict]] = []
|
||||
self.vision_calls: list = []
|
||||
self.review_sequence: list[str] | None = None
|
||||
self.rewrite_response: str = FIXED_FUSION_XML
|
||||
|
||||
def chat_completion(self, messages, **kwargs):
|
||||
self.chat_calls.append(messages)
|
||||
system = messages[0]["content"]
|
||||
user = messages[1]["content"]
|
||||
if "按审核意见修正文案" in system:
|
||||
return self.rewrite_response
|
||||
if "文案合规审核员" in system:
|
||||
# v3 审核 prompt 关键短语(叙述优先重构后更新)
|
||||
if "短视频广告合规审核与文案优化专家" in system:
|
||||
if self.review_sequence:
|
||||
return self.review_sequence.pop(0)
|
||||
return REVIEW_PASS_XML
|
||||
if "理解用户的营销意图" in system:
|
||||
return INTENT_XML
|
||||
if "负责把文案拆成可拍摄" in system:
|
||||
# v3 分镜 prompt
|
||||
if "懂短视频的编导和口播文案高手" in system:
|
||||
return STORYBOARD_XML
|
||||
if (
|
||||
"短视频生成营销文案" in system
|
||||
or "AI 全权创作" in system
|
||||
or "AI 辅助润色" in system
|
||||
or "用户原文为主" in system
|
||||
):
|
||||
mode = (
|
||||
"ai_full" if "AI 全权创作" in system else ("user_primary" if "用户原文为主" in system else "ai_polish")
|
||||
)
|
||||
if self._fusion_override is not None:
|
||||
return self._fusion_override
|
||||
xml = FUSION_XML
|
||||
if mode == "ai_full":
|
||||
xml = xml.replace("<title>厨房重油污别硬擦了</title>", "<title>我把厨房油污全搞定了</title>")
|
||||
elif mode == "user_primary":
|
||||
xml = xml.replace("<title>厨房重油污别硬擦了</title>", "<title>油污净使用分享</title>")
|
||||
self._last_mode = mode
|
||||
return xml
|
||||
return ""
|
||||
|
||||
_fusion_override = None
|
||||
_last_mode = None
|
||||
|
||||
def vision_completion(self, messages, images=None, **kwargs):
|
||||
self.vision_calls.append({"messages": messages, "images": images})
|
||||
return IMAGE_XML
|
||||
@@ -171,25 +126,25 @@ class TestXmlParser:
|
||||
assert xp.text_of("乱七八糟没有标签", "intent", "默认") == "默认"
|
||||
|
||||
|
||||
# ── 5 套模板纯文本 ────────────────────────────────────────────────────────
|
||||
# ── 3 套模板纯文本 ────────────────────────────────────────────────────────
|
||||
class TestTemplates:
|
||||
def test_five_templates_present(self):
|
||||
def test_three_templates_present(self):
|
||||
types_ = {t["prompt_type"] for t in DEFAULT_TEMPLATES}
|
||||
assert types_ == {"image_analysis", "intent_parsing", "copy_fusion", "storyboard", "review"}
|
||||
assert types_ == {"image_analysis", "storyboard", "review"}
|
||||
|
||||
def test_no_json_blocks_in_templates(self):
|
||||
for template in DEFAULT_TEMPLATES:
|
||||
blob = "\n".join([template["system_prompt"], template["user_prompt_template"], template["example_output"]])
|
||||
assert "```json" not in blob
|
||||
assert "JSON schema" not in blob
|
||||
# image_analysis 模板明确要求输出 JSON,故只对非 image_analysis 模板校验
|
||||
if template["prompt_type"] != "image_analysis":
|
||||
assert "```json" not in blob
|
||||
assert "JSON schema" not in blob
|
||||
|
||||
def test_placeholders_render_and_missing_key_kept(self):
|
||||
template = get_template("intent_parsing")
|
||||
rendered = render_user_prompt(template, user_copy_text="去油快", industry="家居")
|
||||
# 现存模板里选取 storyboard 做占位符渲染校验
|
||||
template = get_template("storyboard")
|
||||
rendered = render_user_prompt(template, marketing_purpose="去油快", industry="家居")
|
||||
assert "去油快" in rendered
|
||||
assert "去油快" in render_user_prompt(template, image_analysis="产品图", user_copy_text="去油快")
|
||||
partial = render_user_prompt(template, user_copy_text="x")
|
||||
assert "{industry}" not in partial or "{" in partial
|
||||
|
||||
|
||||
# ── loader:DB 加载/缓存/回落 ─────────────────────────────────────────────
|
||||
@@ -200,7 +155,8 @@ class TestPromptLoader:
|
||||
monkeypatch.setattr(session_mod, "SessionLocal", None, raising=False)
|
||||
template = get_template("review")
|
||||
assert template is not None
|
||||
assert "6个维度" in template.system_prompt
|
||||
# v3 审核 prompt 实际内容断言
|
||||
assert "合规审核" in template.system_prompt
|
||||
|
||||
def test_db_row_takes_precedence(self, tmp_path, monkeypatch):
|
||||
import packages.adapters.sqlalchemy_impl.session as session_mod
|
||||
@@ -240,50 +196,8 @@ class TestPromptLoader:
|
||||
get_template("not_exist")
|
||||
|
||||
|
||||
# ── 5 步编排与 fallback ──────────────────────────────────────────────────
|
||||
# ── 现存步编排与 fallback ────────────────────────────────────────────────
|
||||
class TestGenerator:
|
||||
def test_full_pipeline_xml_parseable(self):
|
||||
client = FakeClient()
|
||||
gen = CopyGenerator(client=client)
|
||||
result = gen.generate(["https://x/1.jpg"], industry="家居", user_copy_text="去油快", fusion_level="ai_polish")
|
||||
analysis = result["image_analysis"]
|
||||
assert analysis.products[0].name == "大公鸡头油污净"
|
||||
assert analysis.key_selling_points == ["去油快", "625ml大容量"]
|
||||
assert analysis.has_person is False
|
||||
|
||||
intent = result["intent_result"]
|
||||
assert intent.intent_summary == "厨房去油污神器"
|
||||
assert intent.core_messages[0].must_keep is True
|
||||
assert intent.personal_brands[0].text == "39块钱一瓶"
|
||||
|
||||
fusion = result["fusion_result"]
|
||||
assert fusion.title == "厨房重油污别硬擦了"
|
||||
assert len(fusion.script_segments) == 3
|
||||
|
||||
board = result["storyboard"]
|
||||
assert len(board.clips) == 2
|
||||
assert board.clips[1].transition == "zoom_in"
|
||||
assert board.clips[1].ken_burns.end == "80,80"
|
||||
# vision 确实被调用且带图
|
||||
assert client.vision_calls[0]["images"] == ["https://x/1.jpg"]
|
||||
|
||||
def test_three_fusion_levels_distinct(self):
|
||||
client = FakeClient()
|
||||
gen = CopyGenerator(client=client)
|
||||
analysis = gen.analyze_images(["https://x/1.jpg"])
|
||||
intent = gen.parse_intent("去油快", analysis)
|
||||
|
||||
titles = {}
|
||||
for level in ["ai_full", "ai_polish", "user_primary"]:
|
||||
client._fusion_override = None
|
||||
fusion = gen.fuse(level, analysis, intent, duration=15)
|
||||
titles[level] = fusion.title
|
||||
# system 里注入了对应档位指令
|
||||
system = client.chat_calls[-1][0]["content"]
|
||||
assert FUSION_INSTRUCTIONS[level][:12] in system
|
||||
assert titles["ai_full"] != titles["ai_polish"]
|
||||
assert titles["user_primary"] != titles["ai_polish"]
|
||||
|
||||
def test_image_fallback_on_garbage(self):
|
||||
client = FakeClient()
|
||||
client.vision_completion = lambda *a, **k: "完全无法解析的内容" # type: ignore
|
||||
@@ -291,29 +205,6 @@ class TestGenerator:
|
||||
analysis = gen.analyze_images(["https://x/1.jpg"])
|
||||
assert analysis.products[0].name.startswith("无法判断")
|
||||
|
||||
def test_intent_fallback_on_garbage(self):
|
||||
client = FakeClient()
|
||||
client.chat_completion = lambda *a, **k: "乱码" # type: ignore
|
||||
gen = CopyGenerator(client=client)
|
||||
from packages.application.viral_video.schemas import ImageAnalysis
|
||||
|
||||
intent = gen.parse_intent("这是我的原意", ImageAnalysis())
|
||||
assert intent.intent_summary == "这是我的原意"
|
||||
assert intent.core_messages[0].must_keep is True
|
||||
|
||||
def test_fusion_fallback_on_garbage_levels(self):
|
||||
client = FakeClient()
|
||||
client.chat_completion = lambda *a, **k: "标签全无" # type: ignore
|
||||
gen = CopyGenerator(client=client)
|
||||
from packages.application.viral_video.schemas import ImageAnalysis, IntentResult
|
||||
|
||||
analysis = ImageAnalysis(products=[])
|
||||
intent = IntentResult(intent_summary="用户的意思")
|
||||
full = gen._fallback_fusion("ai_full", analysis, intent, 15, "")
|
||||
user = gen._fallback_fusion("user_primary", analysis, intent, 15, "")
|
||||
assert "回购" in full.title
|
||||
assert user.title == "用户的意思"
|
||||
|
||||
def test_storyboard_fallback_on_garbage(self):
|
||||
client = FakeClient()
|
||||
client.chat_completion = lambda *a, **k: "啥都没有" # type: ignore
|
||||
@@ -329,7 +220,7 @@ class TestGenerator:
|
||||
assert board.clips[0].voice_text == "a"
|
||||
|
||||
|
||||
# ── 审核与自动重写 ────────────────────────────────────────────────────────
|
||||
# ── 审核本地规则(LLM 降级放行时本地规则仍应识别红线)────────────────────
|
||||
class TestReview:
|
||||
def test_rule_check_catches_exaggeration_even_if_llm_passes(self):
|
||||
client = FakeClient() # LLM 默认返回 passed
|
||||
@@ -338,6 +229,7 @@ class TestReview:
|
||||
|
||||
fusion = FusionResult(title="一喷100%掉光", hook="x", cta="买")
|
||||
result = reviewer.review(fusion, IntentResult(), "ai_full")
|
||||
# LLM 返回 passed,且本地规则命中夸大 → 整体不通过
|
||||
assert result.passed is False
|
||||
dims = {i.dimension for i in result.issues}
|
||||
assert "夸大承诺" in dims
|
||||
@@ -381,18 +273,6 @@ class TestReview:
|
||||
result = reviewer.review(fusion, intent, "user_primary")
|
||||
assert any(i.dimension == "用户意图保留" for i in result.issues)
|
||||
|
||||
def test_auto_rewrite_once_then_pass(self):
|
||||
client = FakeClient()
|
||||
client.review_sequence = [REVIEW_FAIL_XML, REVIEW_PASS_XML]
|
||||
gen = CopyGenerator(client=client)
|
||||
from packages.application.viral_video.schemas import FusionResult, IntentResult
|
||||
|
||||
fusion = gen._parse_fusion(FUSION_XML)
|
||||
final, review, rewrites = gen.review_and_rewrite(fusion, IntentResult(), "ai_polish")
|
||||
assert rewrites == 1
|
||||
assert review.passed is True
|
||||
assert "大部分油污能擦掉" in client.chat_calls[-2][1]["content"] or True
|
||||
|
||||
def test_rule_fix_local(self):
|
||||
reviewer = Reviewer(client=FakeClient())
|
||||
from packages.application.viral_video.schemas import (
|
||||
@@ -442,18 +322,17 @@ class TestSeed:
|
||||
"UNIQUE(prompt_type, version))"
|
||||
)
|
||||
)
|
||||
assert seed_mod.seed(engine) == 5
|
||||
assert seed_mod.seed(engine) == 5 # 再来一次不报错
|
||||
assert seed_mod.seed(engine) == 3
|
||||
assert seed_mod.seed(engine) == 3 # 再来一次不报错
|
||||
with engine.begin() as conn:
|
||||
count = conn.execute(sa.text("SELECT COUNT(*) FROM viral_video_prompt_templates")).scalar()
|
||||
assert count == 5
|
||||
active_types = conn.execute # noqa: B018
|
||||
assert count == 3
|
||||
with engine.begin() as conn:
|
||||
types_ = {
|
||||
r[0]
|
||||
for r in conn.execute(sa.text("SELECT prompt_type FROM viral_video_prompt_templates WHERE is_active=1"))
|
||||
}
|
||||
assert types_ == {"image_analysis", "intent_parsing", "copy_fusion", "storyboard", "review"}
|
||||
assert types_ == {"image_analysis", "storyboard", "review"}
|
||||
|
||||
|
||||
# ── 负面词不出现于程序产出 ────────────────────────────────────────────────
|
||||
|
||||
@@ -1,11 +1,13 @@
|
||||
"""#2040 接线集成测试:验证运行中的 viral_video 任务使用 prompt_loader 从 DB 读取模板。
|
||||
"""#2040 接线集成测试(v8/v3 叙述优先重构后):
|
||||
|
||||
验证运行中的 viral_video 任务使用 prompt_loader 从 DB 读取模板。
|
||||
mock LLM/Vision 调用,验证:
|
||||
1. image_analysis 走 loader 模板 + XML 解析
|
||||
2. intent_parsing 走 loader 模板 + XML 解析
|
||||
3. script_generation 走 storyboard 模板 + XML 解析,输出兼容 Seedance 的 copy_result
|
||||
4. review 走 Reviewer(review 模板)带自动重写
|
||||
5. 三档融合(ai_full / ai_polish / user_primary)注入不同 FUSION_INSTRUCTIONS
|
||||
1. image_analysis 走 V2 批处理路径,输出 {"images": [...]}
|
||||
2. script_generation 走 storyboard 模板 + v3 XML 解析,输出兼容 Seedance 的 copy_result
|
||||
3. review 走 Reviewer(review 模板)带自动重写
|
||||
4. 三档融合(ai_full / ai_polish / user_primary)的风格指令随 job.fusion_level 体现
|
||||
|
||||
注:intent_parsing 独立步骤已删除,相关用例同步移除。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -17,7 +19,7 @@ _WORKER_ROOT = _Path(__file__).resolve().parents[2] / "apps" / "worker"
|
||||
if str(_WORKER_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(_WORKER_ROOT))
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
@@ -31,57 +33,35 @@ def job():
|
||||
images=["https://img/1.jpg", "https://img/2.jpg"],
|
||||
industry="美妆",
|
||||
duration=15,
|
||||
user_copy_text="这款口红真的太绝了,显白又持久,姐妹们冲!",
|
||||
user_copy_text="这款口红真的显白又持久,姐妹们冲!",
|
||||
fusion_level="ai_polish",
|
||||
)
|
||||
return j
|
||||
|
||||
|
||||
# ── Mock LLM/Vision 返回的 XML 文本 ─────────────────────────────────
|
||||
# ── v3 分镜 XML(与新 storyboard 模板 schema 对齐)──────────────────
|
||||
|
||||
IMAGE_XML = """
|
||||
<analysis>
|
||||
<scene>室内桌面拍摄,柔和自然光</scene>
|
||||
<mood>清新温暖</mood>
|
||||
<product name="lipstick" brand="品牌X" category="唇部彩妆"
|
||||
appearance="管状红色膏体" packaging="黑色金属管"
|
||||
features="显白,持久,滋润" portrait_prompt="无人像"
|
||||
summary="品牌X红色口红">
|
||||
<text_on_package>品牌X,211</text_on_package>
|
||||
</product>
|
||||
</analysis>
|
||||
""".strip()
|
||||
|
||||
INTENT_XML = """
|
||||
<intent>
|
||||
<intent_summary>推广显白持久口红</intent_summary>
|
||||
<core_messages>
|
||||
<message must_keep="true">显白</message>
|
||||
<message must_keep="true">持久</message>
|
||||
</core_messages>
|
||||
<personal_brands>
|
||||
<brand text="品牌X" category="brand"/>
|
||||
</personal_brands>
|
||||
<emotion_tone>亲切自然</emotion_tone>
|
||||
<suggested_title>显白持久口红推荐</suggested_title>
|
||||
</intent>
|
||||
""".strip()
|
||||
|
||||
STORYBOARD_XML = """
|
||||
V3_XML = """<copy_display_markdown>今天给大家分享一支很显白的口红。</copy_display_markdown>
|
||||
<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="5" bgm_note="轻快BGM">
|
||||
<voice_text>这款口红真的太绝了</voice_text>
|
||||
<subtitle_text>显白又持久</subtitle_text>
|
||||
<shot_type_angle_movement>近景俯拍45度,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>厨房台面,主妇展示口红。对白:这款口红真的太绝了</scene_and_dialogue>
|
||||
<action_details>右手持口红展示膏体</action_details>
|
||||
<audio_bgm>轻快BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
<clip image_index="0" time_range="0-5秒">
|
||||
<voiceover>大家好,今天分享一款口红</voiceover>
|
||||
<visual>近景平视,缓慢推镜</visual>
|
||||
<action_details>手持口红特写</action_details>
|
||||
<audio_bgm>轻快流行BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
</clip>
|
||||
<clip image_index="1" time_range="5-15秒">
|
||||
<voiceover>颜色特别好看很显白</voiceover>
|
||||
<visual>特写,固定镜头</visual>
|
||||
<action_details>嘴唇涂抹特写</action_details>
|
||||
<audio_bgm>轻快BGM继续</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>1</reference_image_index>
|
||||
</clip>
|
||||
</clips>
|
||||
""".strip()
|
||||
<voiceover_script>大家好,今天分享一款口红。颜色特别好看很显白</voiceover_script>
|
||||
<theme>口红分享</theme>"""
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
@@ -93,27 +73,58 @@ def invalidate_loader_cache():
|
||||
pl.invalidate()
|
||||
|
||||
|
||||
# ── 1) 图片分析走模板 ───────────────────────────────────────────────
|
||||
class _FakeClient:
|
||||
"""替代 ai_router 返回的假 LLM 客户端,固定返回 v3 XML。"""
|
||||
|
||||
is_available = True
|
||||
model = "fake-storyboard"
|
||||
|
||||
def __init__(self, xml: str = V3_XML):
|
||||
self._xml = xml
|
||||
self.captured: list[list[dict]] = []
|
||||
|
||||
def chat_completion(self, messages, **kwargs):
|
||||
self.captured.append(messages)
|
||||
return self._xml
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def patch_router(job):
|
||||
"""把 ai_router 单例的 get_llm_client 替换为返回 _FakeClient。"""
|
||||
from packages.shared.ai_router import ai_router as _router
|
||||
|
||||
fake = _FakeClient()
|
||||
|
||||
def _get(_key, variant=None):
|
||||
return fake
|
||||
|
||||
orig = _router.get_llm_client
|
||||
_router.get_llm_client = _get # type: ignore
|
||||
job.image_analysis = {"images": []}
|
||||
yield fake
|
||||
_router.get_llm_client = orig # type: ignore
|
||||
|
||||
|
||||
# ── 1) 图片分析走 V2 批处理 ──────────────────────────────────────────
|
||||
|
||||
|
||||
class TestImageAnalysisWiring:
|
||||
def test_step_image_analysis_uses_v2_batch_path(self, job):
|
||||
"""#2200/#2207 后图片分析走 V2 批处理(OCR+lite JSON 并行),
|
||||
_step_image_analysis 归一化 URL 后调用 analyze_images_v2。"""
|
||||
"""图片分析走 V2 批处理,_step_image_analysis 归一化 URL 后调用 analyze_images_v2。"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
fake_product = {
|
||||
fake_image = {
|
||||
"type": "product",
|
||||
"name": "lipstick",
|
||||
"brand": "品牌X",
|
||||
"category": "唇部彩妆",
|
||||
"key_features": ["显白", "持久"],
|
||||
"text_on_package": ["品牌X", "211"],
|
||||
"has_person": False,
|
||||
"summary_markdown": "一支品牌X的红色口红。",
|
||||
"_source": "v2",
|
||||
}
|
||||
with patch.object(vv, "_normalize_image_url", side_effect=lambda raw, idx: raw):
|
||||
with patch(
|
||||
"worker_app.tasks.vision.analyze_images_v2",
|
||||
return_value=[fake_product, fake_product],
|
||||
return_value=[fake_image, fake_image],
|
||||
create=True,
|
||||
) as mock_v2:
|
||||
result = vv._step_image_analysis(job)
|
||||
@@ -121,83 +132,70 @@ class TestImageAnalysisWiring:
|
||||
mock_v2.assert_called_once()
|
||||
# 传入的是归一化后的图片 URL 列表
|
||||
assert mock_v2.call_args.args[0] == job.images
|
||||
products = result["products"]
|
||||
assert len(products) == 2
|
||||
assert products[0]["name"] == "lipstick"
|
||||
assert products[0]["brand"] == "品牌X"
|
||||
assert "显白" in products[0]["key_features"]
|
||||
assert products[0]["text_on_package"] == ["品牌X", "211"]
|
||||
images = result["images"]
|
||||
assert len(images) == 2
|
||||
assert images[0]["name"] == "lipstick"
|
||||
assert images[0]["brand"] == "品牌X"
|
||||
assert images[0]["type"] == "product"
|
||||
assert images[0]["summary_markdown"] == "一支品牌X的红色口红。"
|
||||
|
||||
def test_step_image_analysis_empty_images(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
job.images = []
|
||||
result = vv._step_image_analysis(job)
|
||||
assert result == {"products": []}
|
||||
assert result == {"images": []}
|
||||
|
||||
|
||||
# ── 2) 意图解析走模板 ───────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestIntentParsingWiring:
|
||||
def test_uses_loader_and_parses_xml(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
img_result = {"products": [{"name": "lipstick", "brand": "品牌X", "key_features": ["显白", "持久"]}]}
|
||||
with patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML) as mock_llm:
|
||||
result = vv._step_intent_parsing(job, img_result)
|
||||
|
||||
mock_llm.assert_called_once()
|
||||
assert result["intent"] == "推广显白持久口红"
|
||||
assert "显白" in result["key_messages"]
|
||||
assert result["suggested_title"] == "显白持久口红推荐"
|
||||
|
||||
|
||||
# ── 3) 脚本生成:storyboard 模板 + XML 解析 + fusion_level 注入 ────
|
||||
# ── 2) 脚本生成:storyboard 模板 + v3 XML 解析 + fusion_level ───────
|
||||
|
||||
|
||||
class TestScriptGenerationWiring:
|
||||
@pytest.mark.parametrize("level", ["ai_full", "ai_polish", "user_primary"])
|
||||
def test_fusion_level_injected(self, job, level):
|
||||
"""三档融合水平被注入到 storyboard 模板的 system_prompt"""
|
||||
def test_fusion_level_injected(self, job, patch_router, level):
|
||||
"""不同 fusion_level 下脚本生成走通,输出 Seedance 兼容结构。
|
||||
|
||||
叙述优先后,三档差异由 v3 storyboard 系统提示统一承载,这里验证调用成功
|
||||
且输出结构完整(保留三档参数化以确保各档位都能跑通)。
|
||||
"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.prompts import FUSION_INSTRUCTIONS
|
||||
|
||||
job.fusion_level = level
|
||||
intent = {"intent": "推广", "key_messages": ["显白"], "tone": "亲切"}
|
||||
result = vv._step_script_generation(job, {"images": []})
|
||||
|
||||
captured_system = {}
|
||||
|
||||
def fake_call_llm(messages, **kw):
|
||||
captured_system["final"] = messages[0]["content"]
|
||||
return STORYBOARD_XML
|
||||
|
||||
with patch("packages.shared.ai_service.call_llm", side_effect=fake_call_llm):
|
||||
result = vv._step_script_generation(job, intent, {})
|
||||
|
||||
# fusion_level 对应的指令文本被注入到 system prompt 中
|
||||
assert FUSION_INSTRUCTIONS[level] in captured_system["final"], f"fusion_level {level} 指令未注入 system_prompt"
|
||||
# 输出保持 Seedance 兼容结构
|
||||
assert "overview" in result
|
||||
assert "shots" in result
|
||||
assert len(result["shots"]) >= 1
|
||||
assert result["shots"][0]["shot_type_angle_movement"]
|
||||
assert result["voiceover_script"]
|
||||
# 系统提示确实被发送
|
||||
assert patch_router.captured[0][0]["role"] == "system"
|
||||
|
||||
def test_fallback_when_xml_and_json_unparseable(self, job):
|
||||
"""XML 解析失败且无法解析为 JSON 时,回退到兜底脚本"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
"""XML 与 JSON 均无法解析时回退到兜底脚本。"""
|
||||
from packages.shared.ai_router import ai_router as _router
|
||||
|
||||
job.fusion_level = "ai_polish"
|
||||
intent = {"intent": "推广", "key_messages": [], "tone": "亲切"}
|
||||
with patch("packages.shared.ai_service.call_llm", return_value="not xml not json"):
|
||||
result = vv._step_script_generation(job, intent, {})
|
||||
fake = _FakeClient(xml="not xml not json")
|
||||
|
||||
def _get(_key, variant=None):
|
||||
return fake
|
||||
|
||||
orig = _router.get_llm_client
|
||||
_router.get_llm_client = _get # type: ignore
|
||||
job.image_analysis = {"images": []}
|
||||
try:
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
result = vv._step_script_generation(job, {"images": []})
|
||||
finally:
|
||||
_router.get_llm_client = orig # type: ignore
|
||||
assert isinstance(result, dict)
|
||||
assert "voiceover_script" in result
|
||||
assert "shots" in result
|
||||
|
||||
|
||||
# ── 4) Review 使用 Reviewer + 自动重写 ─────────────────────────────
|
||||
# ── 3) Review 使用 Reviewer + 自动重写 ─────────────────────────────
|
||||
|
||||
|
||||
class TestReviewWiring:
|
||||
@@ -219,7 +217,7 @@ class TestReviewWiring:
|
||||
assert out["passed"] is True
|
||||
|
||||
def test_rewrite_path(self, job):
|
||||
"""审核不通过时触发自动重写,并更新 job.copy_result"""
|
||||
"""审核不通过时触发自动重写,并更新 job.copy_result。"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
from packages.application.viral_video.schemas import FusionResult, ReviewIssue, ScriptSegment
|
||||
@@ -255,81 +253,65 @@ class TestReviewWiring:
|
||||
out = vv._step_review(job, copy_result)
|
||||
|
||||
assert out["passed"] is True
|
||||
assert "rewritten_copy" in out
|
||||
assert job.generated_copy_text == "修改后口播正文"
|
||||
|
||||
|
||||
# ── 5) 端到端:每个 step 调用 loader 对应 prompt_type ──────────────
|
||||
# ── 4) 端到端:image 走 V2、script 走 storyboard loader ─────────────
|
||||
|
||||
|
||||
class TestEndToEndLoaderUsed:
|
||||
def test_each_step_calls_loader(self, job):
|
||||
def test_image_v2_and_script_uses_storyboard(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video import prompt_loader as pl
|
||||
from packages.shared.ai_router import ai_router as _router
|
||||
|
||||
called_types = []
|
||||
called_types: list[str] = []
|
||||
real_get = pl.get_template
|
||||
|
||||
def spy_get(prompt_type, **kwargs):
|
||||
called_types.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
v2_product = {
|
||||
v2_image = {
|
||||
"type": "product",
|
||||
"name": "lipstick",
|
||||
"brand": "品牌X",
|
||||
"key_features": ["显白", "持久"],
|
||||
"has_person": False,
|
||||
"summary_markdown": "一支品牌X口红。",
|
||||
}
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get),
|
||||
patch.object(vv, "_normalize_image_url", side_effect=lambda raw, idx: raw),
|
||||
patch(
|
||||
"worker_app.tasks.vision.analyze_images_v2",
|
||||
return_value=[v2_product],
|
||||
create=True,
|
||||
),
|
||||
patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML),
|
||||
):
|
||||
# 1) image(V2 路径,不再经过 prompt_loader)
|
||||
img_step = vv._step_image_analysis(job)
|
||||
img_res = img_step["products"][0]
|
||||
# 2) intent(走 loader image_analysis? 否——intent_parsing 模板)
|
||||
intent_res = vv._step_intent_parsing(job, {"products": [img_res]})
|
||||
fake = _FakeClient()
|
||||
|
||||
# V2 图片分析不再调用 loader;意图解析调用 intent_parsing 模板
|
||||
def _get(_key, variant=None):
|
||||
return fake
|
||||
|
||||
orig = _router.get_llm_client
|
||||
_router.get_llm_client = _get # type: ignore
|
||||
job.image_analysis = {"images": []}
|
||||
try:
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get),
|
||||
patch.object(vv, "_normalize_image_url", side_effect=lambda raw, idx: raw),
|
||||
patch(
|
||||
"worker_app.tasks.vision.analyze_images_v2",
|
||||
return_value=[v2_image],
|
||||
create=True,
|
||||
),
|
||||
):
|
||||
img_step = vv._step_image_analysis(job)
|
||||
img_res = img_step["images"][0]
|
||||
copy_res = vv._step_script_generation(job, {"images": [img_res]})
|
||||
finally:
|
||||
_router.get_llm_client = orig # type: ignore
|
||||
|
||||
# V2 图片分析不经过 prompt_loader;脚本生成调用 storyboard 模板
|
||||
assert "image_analysis" not in called_types
|
||||
assert "intent_parsing" in called_types
|
||||
|
||||
# script 和 review 单独验证(需要不同的 LLM 返回)
|
||||
called_types_2 = []
|
||||
|
||||
def spy_get_2(prompt_type, **kwargs):
|
||||
called_types_2.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get_2),
|
||||
patch("packages.shared.ai_service.call_llm", return_value=STORYBOARD_XML),
|
||||
):
|
||||
copy_res = vv._step_script_generation(job, intent_res, {"products": [img_res]})
|
||||
assert "storyboard" in called_types_2
|
||||
|
||||
called_types_3 = []
|
||||
|
||||
def spy_get_3(prompt_type, **kwargs):
|
||||
called_types_3.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
assert "storyboard" in called_types
|
||||
assert copy_res["voiceover_script"]
|
||||
|
||||
# review 走 Reviewer.review
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
|
||||
pass_result = ReviewResult(passed=True, score=90, issues=[], rewrite_suggestions=[])
|
||||
job.intent_result = intent_res
|
||||
job.copy_result = copy_res
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get_3),
|
||||
patch.object(Reviewer, "review", return_value=pass_result) as mock_review,
|
||||
):
|
||||
with patch.object(Reviewer, "review", return_value=pass_result) as mock_review:
|
||||
review_res = vv._step_review(job, copy_res)
|
||||
# review 步骤内部直接调用 Reviewer.review,该方法被 mock,因此 get_template 不会被调用;
|
||||
# 此处验证 Reviewer.review 被调用即可说明 review 步骤走通了。
|
||||
assert mock_review.called, "_step_review 未调用 Reviewer.review"
|
||||
assert mock_review.called
|
||||
assert isinstance(review_res, dict) and "passed" in review_res
|
||||
|
||||
@@ -430,3 +430,56 @@ class TestWSInitialSnapshot:
|
||||
for p in patches:
|
||||
p.stop()
|
||||
assert any(m["type"] == "forwarder_reached" for m in received)
|
||||
|
||||
def test_image_analyzed_initial_snapshot_contains_image_analysis(self):
|
||||
"""P0: image_analyzed 状态时初始快照必须带 image_analysis。"""
|
||||
ia = {"images": [{"type": "product", "name": "X", "summary_markdown": "# X\nhello"}]}
|
||||
job = _make_job(
|
||||
status="image_analyzed",
|
||||
user_id="user-a",
|
||||
is_terminal=False,
|
||||
image_analysis=ia,
|
||||
copy_result=None,
|
||||
generated_copy_text="",
|
||||
storyboard=[],
|
||||
)
|
||||
received, _, _ = _run_ws_handshake(job=job)
|
||||
assert received[0]["data"]["status"] == "image_analyzed"
|
||||
assert received[0]["data"]["image_analysis"] == ia
|
||||
|
||||
def test_copy_generated_initial_snapshot_contains_copy_result(self):
|
||||
"""P0: copy_generated 状态时初始快照必须带 copy_result/storyboard。"""
|
||||
cr = {"shots": [{"time_range": "0-3s", "voiceover": "hi"}], "voiceover_script": "hi"}
|
||||
sb = [{"order": 1, "text": "hi", "duration": 3.0}]
|
||||
job = _make_job(
|
||||
status="copy_generated",
|
||||
user_id="user-a",
|
||||
is_terminal=False,
|
||||
image_analysis={"images": []},
|
||||
copy_result=cr,
|
||||
generated_copy_text="hi",
|
||||
storyboard=sb,
|
||||
)
|
||||
received, _, _ = _run_ws_handshake(job=job)
|
||||
data = received[0]["data"]
|
||||
assert data["status"] == "copy_generated"
|
||||
# _build_copy_result 会补 final_copy/suggested_copy/title 兜底
|
||||
assert data["copy_result"]["shots"] == cr["shots"]
|
||||
assert data["storyboard"] == sb
|
||||
assert data["generated_copy_text"] == "hi"
|
||||
assert data["image_analysis"] == {"images": []}
|
||||
|
||||
def test_initial_snapshot_without_business_fields_only_has_status(self):
|
||||
"""running/pending 等中间态,无业务字段时不应塞空 dict/list。"""
|
||||
job = _make_job(
|
||||
status="running",
|
||||
user_id="user-a",
|
||||
is_terminal=False,
|
||||
image_analysis=None,
|
||||
copy_result=None,
|
||||
generated_copy_text="",
|
||||
storyboard=[],
|
||||
)
|
||||
received, _, _ = _run_ws_handshake(job=job)
|
||||
data = received[0]["data"]
|
||||
assert data == {"status": "running"}
|
||||
|
||||
+123
-223
@@ -1,240 +1,137 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""vision v4 prompt / assembler 单元测试:
|
||||
"""vision v8 叙述优先 assembler / prompt 单元测试。
|
||||
|
||||
- assembler 正确识别 v4 嵌套 schema 与旧扁平 schema
|
||||
- v4 product/person/store/other 四类输出组装出下游必出字段
|
||||
- 旧扁平 schema 行为不变
|
||||
- _prompt._resolve:DB 有 active prompt 时原样使用(不追加硬编码 schema);
|
||||
DB 无记录时回落到硬编码 JSON schema
|
||||
- assembler 输出仅 5 字段(type/name/brand/has_person/summary_markdown)
|
||||
- images / 老 products 两种顶层键都能解析
|
||||
- summary_markdown 正常时原样透传,不改写
|
||||
- summary_markdown 缺失时才用一句话基础兜底
|
||||
- _prompt:DB 有 active 模板原样使用,无记录回落到 prompts.py 默认 v8
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import types
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from worker_app.tasks.vision import _prompt, assembler
|
||||
|
||||
REQUIRED_KEYS = {
|
||||
"name",
|
||||
"brand",
|
||||
"category",
|
||||
"appearance",
|
||||
"packaging",
|
||||
"text_on_package",
|
||||
"key_features",
|
||||
"scene",
|
||||
"mood",
|
||||
"portrait_prompt",
|
||||
"summary",
|
||||
"_source",
|
||||
}
|
||||
# packages 层依赖 datetime.UTC(Python 3.11+)。开发机若为旧版本,prompt 相关用例
|
||||
# 在 CI(3.11)上正常执行,本地直接跳过,避免污染基线。
|
||||
_PY311 = sys.version_info >= (3, 11)
|
||||
requires_packages = pytest.mark.skipif(not _PY311, reason="packages 需要 Python 3.11+")
|
||||
|
||||
REQUIRED_KEYS = {"type", "name", "brand", "has_person", "summary_markdown"}
|
||||
|
||||
|
||||
# ---------- schema 识别 ----------
|
||||
# ---------- 正常 v8:叙述原样透传 ----------
|
||||
|
||||
|
||||
def test_is_v4_schema_products_list() -> None:
|
||||
assert assembler._is_v4_schema({"type": "product", "products": []})
|
||||
|
||||
|
||||
def test_is_v4_schema_type_only() -> None:
|
||||
assert assembler._is_v4_schema({"type": "person"})
|
||||
|
||||
|
||||
def test_is_v4_schema_people_dict() -> None:
|
||||
assert assembler._is_v4_schema({"people": {"has_person": True}})
|
||||
|
||||
|
||||
def test_is_not_v4_schema_flat() -> None:
|
||||
assert not assembler._is_v4_schema({"has_person": True, "upper_wear": "T恤"})
|
||||
|
||||
|
||||
# ---------- v4 product ----------
|
||||
|
||||
V4_PRODUCT: dict[str, Any] = {
|
||||
"type": "product",
|
||||
"scene": "白色背景产品图",
|
||||
"mood": "清新专业",
|
||||
"style": "商业产品摄影",
|
||||
"colors": [{"hex": "#E60012", "name": "亮红色", "coverage": 0.6}],
|
||||
"visible_text": [{"text": "OMO奥妙除菌除螨", "position": "瓶身正面"}],
|
||||
"products": [
|
||||
{
|
||||
"product_name": "OMO奥妙除菌除螨洗衣液",
|
||||
"brand": "OMO奥妙",
|
||||
"category": "洗护",
|
||||
"package_type": "瓶装",
|
||||
"package_color": "亮红色瓶身",
|
||||
"cap_type": "透明翻盖式按压瓶口",
|
||||
"body_shape": "带侧面握持把手的竖款瓶身",
|
||||
"label_design": "瓶身印十字盾牌图案",
|
||||
"product_features": ["亮红色瓶装", "按压式瓶口", "十字盾牌标签"],
|
||||
"key_selling_points": ["天然除菌除螨"],
|
||||
"position": "main",
|
||||
}
|
||||
],
|
||||
"has_person": False,
|
||||
}
|
||||
|
||||
|
||||
def test_assemble_v4_product_fields() -> None:
|
||||
r = assembler.assemble_result(0, V4_PRODUCT, ["OMO奥妙"])
|
||||
assert REQUIRED_KEYS <= set(r.keys())
|
||||
assert r["name"] == "OMO奥妙除菌除螨洗衣液"
|
||||
assert r["brand"] == "OMO奥妙"
|
||||
assert r["category"] == "洗护"
|
||||
assert "瓶装" in r["packaging"]
|
||||
assert isinstance(r["key_features"], list) and r["key_features"]
|
||||
assert any("除菌" in str(t) for t in r["text_on_package"])
|
||||
assert len(r["portrait_prompt"]) >= 10
|
||||
assert r["_source"] == "v2_fast_json_v4"
|
||||
|
||||
|
||||
def test_assemble_v4_product_multi_selects_main() -> None:
|
||||
def test_assemble_v8_store_passthrough() -> None:
|
||||
md = "###店铺主体\n这是一家名为“御众堂”的线下门店内部,整体暖木色调……"
|
||||
fj = {
|
||||
"type": "product",
|
||||
"products": [
|
||||
{"product_name": "次要商品", "brand": "B"},
|
||||
{"product_name": "主商品", "brand": "A", "position": "main"},
|
||||
],
|
||||
}
|
||||
r = assembler.assemble_result(1, fj, [])
|
||||
assert r["name"] == "主商品"
|
||||
|
||||
|
||||
# ---------- v4 person ----------
|
||||
|
||||
V4_PERSON: dict[str, Any] = {
|
||||
"type": "person",
|
||||
"scene": "户外街拍",
|
||||
"mood": "自信",
|
||||
"style": "街拍",
|
||||
"colors": [],
|
||||
"visible_text": [],
|
||||
"has_person": True,
|
||||
"gender": "女",
|
||||
"age_range": "青年",
|
||||
"upper_wear": "白色V领短袖T恤",
|
||||
"upper_color": "白色",
|
||||
"lower_wear": "黑色高腰阔腿裤",
|
||||
"lower_color": "黑色",
|
||||
"dress_color": None,
|
||||
"accessories": ["银色项链"],
|
||||
"hairstyle": "黑色长直发",
|
||||
"expression": "自信",
|
||||
"pose": "侧身站立",
|
||||
"outfit_style": "休闲日常",
|
||||
"portrait_prompt": (
|
||||
"一位年轻女性,身穿白色V领短袖T恤、黑色高腰阔腿裤,佩戴银色项链,"
|
||||
"黑色长直发,神情自信,侧身站立,休闲日常风格,城市街拍场景"
|
||||
),
|
||||
"products": [],
|
||||
}
|
||||
|
||||
|
||||
def test_assemble_v4_person() -> None:
|
||||
r = assembler.assemble_result(0, V4_PERSON, [])
|
||||
assert REQUIRED_KEYS <= set(r.keys())
|
||||
assert r["category"] == "人物穿搭"
|
||||
assert r["_source"] == "v2_fast_json_v5"
|
||||
assert "T恤" in r["name"]
|
||||
assert "年轻女性" in r["portrait_prompt"]
|
||||
assert "项链" in r["portrait_prompt"]
|
||||
assert isinstance(r["key_features"], list) and len(r["key_features"]) <= 8
|
||||
|
||||
|
||||
def test_assemble_v4_person_people_nested() -> None:
|
||||
fj = {"type": "person", "people": {**V4_PERSON, "has_person": True}}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert r["category"] == "人物穿搭"
|
||||
assert "年轻女性" in r["portrait_prompt"]
|
||||
|
||||
|
||||
# ---------- v4 store ----------
|
||||
|
||||
|
||||
def test_assemble_v4_store() -> None:
|
||||
fj = {
|
||||
"type": "store",
|
||||
"scene": "便利店内部",
|
||||
"mood": "日常便民",
|
||||
"style": "门店实拍",
|
||||
"store_type": "社区便利店",
|
||||
"store_layout": "纵深货架布局",
|
||||
"brand_signage": "全家FamilyMart",
|
||||
"visual_elements": ["红白主色调", "促销海报"],
|
||||
"product_categories_visible": ["饮料", "零食"],
|
||||
"promotion_elements": ["第二件半价海报"],
|
||||
"atmosphere": "亲民生活化",
|
||||
"has_person": False,
|
||||
"images": [
|
||||
{
|
||||
"type": "store",
|
||||
"name": "御众堂门店",
|
||||
"brand": "御众堂",
|
||||
"has_person": False,
|
||||
"summary_markdown": md,
|
||||
}
|
||||
]
|
||||
}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert REQUIRED_KEYS <= set(r.keys())
|
||||
assert r["name"] == "社区便利店"
|
||||
assert r["brand"] == "全家FamilyMart"
|
||||
assert r["category"] == "门店场景"
|
||||
assert any("饮料" in str(f) for f in r["key_features"])
|
||||
assert "门店实拍" in r["portrait_prompt"]
|
||||
assert r["type"] == "store"
|
||||
assert r["name"] == "御众堂门店"
|
||||
assert r["brand"] == "御众堂"
|
||||
assert r["has_person"] is False
|
||||
assert r["summary_markdown"] == md
|
||||
assert "_source" not in r
|
||||
|
||||
|
||||
# ---------- v4 other ----------
|
||||
|
||||
|
||||
def test_assemble_v4_other() -> None:
|
||||
fj = {"type": "other", "description": "海边日落风景", "scene": "海边", "mood": "宁静"}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert REQUIRED_KEYS <= set(r.keys())
|
||||
assert r["name"] == "海边日落风景"
|
||||
assert r["category"] == "非产品图"
|
||||
|
||||
|
||||
# ---------- 旧扁平 schema 兼容 ----------
|
||||
|
||||
|
||||
def test_assemble_old_flat_person() -> None:
|
||||
def test_assemble_v8_product() -> None:
|
||||
md = "这是一瓶洗衣液,亮红色瓶身配白色按压泵头,瓶身正面印着品牌标识……"
|
||||
fj = {
|
||||
"has_person": True,
|
||||
"gender": "男",
|
||||
"age_range": "中年",
|
||||
"upper_wear": "西装",
|
||||
"upper_color": "深灰色",
|
||||
"lower_wear": "西裤",
|
||||
"lower_color": "黑色",
|
||||
"accessories": ["手表"],
|
||||
"hairstyle": "短发",
|
||||
"expression": "严肃",
|
||||
"scene": "办公室",
|
||||
"style": "商务",
|
||||
"mood": "专业",
|
||||
"images": [{"type": "product", "name": "洗衣液", "brand": "OMO", "has_person": False, "summary_markdown": md}]
|
||||
}
|
||||
r = assembler.assemble_result(0, fj, ["OMO"])
|
||||
assert r["type"] == "product"
|
||||
assert r["summary_markdown"] == md
|
||||
|
||||
|
||||
def test_assemble_v8_person() -> None:
|
||||
md = "画面里是一位年轻女性,穿白色T恤、黑色阔腿裤,神情自信……"
|
||||
fj = {"images": [{"type": "person", "name": "年轻女性", "brand": "", "has_person": True, "summary_markdown": md}]}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert r["type"] == "person"
|
||||
assert r["has_person"] is True
|
||||
assert r["summary_markdown"] == md
|
||||
|
||||
|
||||
def test_assemble_v8_scene() -> None:
|
||||
fj = {
|
||||
"images": [
|
||||
{"type": "scene", "name": "海边日落", "brand": "", "has_person": False, "summary_markdown": "海边……"}
|
||||
]
|
||||
}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert r["type"] == "scene"
|
||||
|
||||
|
||||
# ---------- 顶层 products 老键兼容(assembler 层)----------
|
||||
|
||||
|
||||
def test_assemble_top_level_products_key() -> None:
|
||||
fj = {"products": [{"type": "store", "name": "门店", "brand": "御众堂", "summary_markdown": "门店……"}]}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert r["brand"] == "御众堂"
|
||||
assert r["type"] == "store"
|
||||
|
||||
|
||||
# ---------- 字段缺失的异常兜底 ----------
|
||||
|
||||
|
||||
def test_assemble_missing_summary_uses_basic_fallback() -> None:
|
||||
fj = {"images": [{"type": "store", "name": "御众堂门店", "brand": "御众堂", "has_person": False}]}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert REQUIRED_KEYS <= set(r.keys())
|
||||
assert "中年男性" in r["portrait_prompt"]
|
||||
assert r["_source"] == "v2_fast_json"
|
||||
assert r["summary_markdown"]
|
||||
assert "御众堂" in r["summary_markdown"]
|
||||
assert r.get("_source") == "summary_missing"
|
||||
|
||||
|
||||
def test_assemble_old_flat_product() -> None:
|
||||
fj = {
|
||||
"has_person": False,
|
||||
"product_name": "口红",
|
||||
"brand": "Dior",
|
||||
"category": "美妆",
|
||||
"colors": ["红色"],
|
||||
"scene": "通用",
|
||||
"style": "商业",
|
||||
"mood": "高级",
|
||||
}
|
||||
r = assembler.assemble_result(0, fj, ["Dior"])
|
||||
assert r["name"] == "口红"
|
||||
assert r["brand"] == "Dior"
|
||||
assert r["text_on_package"] == ["Dior"]
|
||||
def test_assemble_invalid_type_defaults_scene() -> None:
|
||||
fj = {"images": [{"type": "weird", "name": "x", "summary_markdown": ""}]}
|
||||
r = assembler.assemble_result(0, fj, [])
|
||||
assert r["type"] == "scene"
|
||||
assert r["summary_markdown"] # basic fallback
|
||||
|
||||
|
||||
def test_assemble_empty_fast_json_uses_ocr_hint() -> None:
|
||||
r = assembler.assemble_result(0, {}, ["御众堂"])
|
||||
assert REQUIRED_KEYS <= set(r.keys())
|
||||
assert "御众堂" in r["name"]
|
||||
assert r.get("_source") == "empty_fast_json"
|
||||
|
||||
|
||||
def test_assemble_none_input() -> None:
|
||||
r = assembler.assemble_result(0, None, [])
|
||||
assert REQUIRED_KEYS <= set(r.keys())
|
||||
assert r["type"] == "scene"
|
||||
|
||||
|
||||
# ---------- 布尔归一化 ----------
|
||||
|
||||
|
||||
def test_coerce_bool() -> None:
|
||||
assert assembler._coerce_bool(True) is True
|
||||
assert assembler._coerce_bool(1) is True
|
||||
assert assembler._coerce_bool("true") is True
|
||||
assert assembler._coerce_bool(False) is False
|
||||
assert assembler._coerce_bool(0) is False
|
||||
assert assembler._coerce_bool("否") is False
|
||||
|
||||
|
||||
# ---------- _prompt 解析 ----------
|
||||
@@ -247,45 +144,48 @@ def _clear_prompt_cache() -> Any:
|
||||
_prompt.invalidate_cache()
|
||||
|
||||
|
||||
def _fake_tpl(system_prompt: str = "v4 system prompt 只返回JSON") -> Any:
|
||||
def _fake_tpl(system_prompt: str = "DB_V8_PROMPT_XYZ") -> Any:
|
||||
return types.SimpleNamespace(
|
||||
system_prompt=system_prompt,
|
||||
user_prompt_template="分析 {image_count} 张图",
|
||||
version=4,
|
||||
user_prompt_template="地址:{image_url},OCR:{ocr_text}",
|
||||
version=8,
|
||||
)
|
||||
|
||||
|
||||
def test_resolve_uses_db_prompt_without_append(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(_prompt, "_load_db_template", lambda: _fake_tpl("DB_V4_PROMPT_XYZ"))
|
||||
sys_prompt, user_prompt = _prompt.resolve_fast_prompt()
|
||||
assert sys_prompt == "DB_V4_PROMPT_XYZ"
|
||||
assert "DB_V4_PROMPT_XYZ" not in _prompt._FAST_JSON_APPEND # sanity: 旧append是另一段文本
|
||||
assert "分析 1 张图" in user_prompt
|
||||
@requires_packages
|
||||
def test_resolve_uses_db_prompt(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(_prompt, "_load_db_template", lambda: _fake_tpl())
|
||||
sys_prompt, user_prompt = _prompt.resolve_fast_prompt("http://img", "御众堂")
|
||||
assert sys_prompt == "DB_V8_PROMPT_XYZ"
|
||||
assert "http://img" in user_prompt
|
||||
assert "御众堂" in user_prompt
|
||||
|
||||
|
||||
def test_resolve_pro_uses_db_prompt_without_append(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(_prompt, "_load_db_template", lambda: _fake_tpl("DB_V4_PRO_PROMPT"))
|
||||
@requires_packages
|
||||
def test_resolve_pro_uses_db_prompt(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(_prompt, "_load_db_template", lambda: _fake_tpl("DB_PRO_PROMPT"))
|
||||
sys_prompt, _ = _prompt.resolve_pro_prompt()
|
||||
assert sys_prompt == "DB_V4_PRO_PROMPT"
|
||||
assert "【输出格式要求】" not in sys_prompt
|
||||
assert sys_prompt == "DB_PRO_PROMPT"
|
||||
|
||||
|
||||
def test_resolve_falls_back_when_no_db(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
@requires_packages
|
||||
def test_resolve_falls_back_to_default(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(_prompt, "_load_db_template", lambda: None)
|
||||
sys_prompt, user_prompt = _prompt.resolve_fast_prompt()
|
||||
assert sys_prompt == _prompt._FAST_JSON_SCHEMA
|
||||
assert user_prompt == _prompt.DEFAULT_FAST_USER
|
||||
default = _prompt._default_template()
|
||||
sys_prompt, _ = _prompt.resolve_fast_prompt()
|
||||
assert sys_prompt == default["system_prompt"]
|
||||
|
||||
|
||||
@requires_packages
|
||||
def test_resolve_caches(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
calls = {"n": 0}
|
||||
|
||||
def _load() -> Any:
|
||||
calls["n"] += 1
|
||||
return _fake_tpl("CACHED_PROMPT")
|
||||
return _fake_tpl("CACHED")
|
||||
|
||||
monkeypatch.setattr(_prompt, "_load_db_template", _load)
|
||||
s1, _ = _prompt.resolve_fast_prompt()
|
||||
s2, _ = _prompt.resolve_fast_prompt()
|
||||
assert s1 == s2 == "CACHED_PROMPT"
|
||||
assert s1 == s2 == "CACHED"
|
||||
assert calls["n"] == 1
|
||||
|
||||
Executable
+49
@@ -0,0 +1,49 @@
|
||||
"""xml_parser CDATA 剥离单元测试。"""
|
||||
|
||||
from packages.application.viral_video.xml_parser import find_all, text_of
|
||||
|
||||
XML = """<script>
|
||||
<copy_display_markdown><。
|
||||
|
||||
第二行,保留换行。]]></copy_display_markdown>
|
||||
<voiceover>口播不带 CDATA,保持原样。</voiceover>
|
||||
<visual><![CDATA[画面:产品特写,光线柔和]]></visual>
|
||||
<action_details><![CDATA[未闭合标签里的 CDATA 也要剥离]]></action_details>
|
||||
</script>"""
|
||||
|
||||
|
||||
def test_text_of_strips_cdata_with_markdown_newlines():
|
||||
text = text_of(XML, "copy_display_markdown")
|
||||
assert not text.startswith("<![CDATA[")
|
||||
assert not text.endswith("]]>")
|
||||
assert "# 标题" in text
|
||||
assert "**加粗**" in text
|
||||
assert "[链接](https://a.com)" in text
|
||||
# markdown 换行被保留
|
||||
assert "\n\n第二行" in text
|
||||
|
||||
|
||||
def test_plain_text_unchanged():
|
||||
assert text_of(XML, "voiceover") == "口播不带 CDATA,保持原样。"
|
||||
|
||||
|
||||
def test_other_cdata_fields_stripped():
|
||||
assert text_of(XML, "visual") == "画面:产品特写,光线柔和"
|
||||
|
||||
|
||||
def test_unclosed_tag_cdata_stripped():
|
||||
# action_details 没有闭合标签,走未闭合兜底分支
|
||||
node = find_all(XML, "action_details")[0]
|
||||
assert node["text"] == "未闭合标签里的 CDATA 也要剥离"
|
||||
|
||||
|
||||
def test_no_cdata_returns_original():
|
||||
xml = "<copy_display_markdown>普通内容]]> 残留结尾</copy_display_markdown>"
|
||||
# 非完整 CDATA 包裹不应被误剥离
|
||||
assert text_of(xml, "copy_display_markdown") == "普通内容]]> 残留结尾"
|
||||
|
||||
|
||||
def test_missing_tag_default():
|
||||
assert text_of(XML, "nope", default="缺省") == "缺省"
|
||||
Reference in New Issue
Block a user