Compare commits
23 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 6a7709ba43 | |||
| 2c70dd4c29 | |||
| 5548e78eee | |||
| 44bd96b148 | |||
| dda67cd10f | |||
| 233d0272a9 | |||
| 631dd643c0 | |||
| 50489b05d6 | |||
| 252ea71d50 | |||
| d5f6e9499d | |||
| afd6b9c0b8 | |||
| 8a6ebb49be | |||
| 9e024caff5 | |||
| c7e9878f0d | |||
| b6243c8ab8 | |||
| 67e4b3fc6f | |||
| df654bce19 | |||
| bd99c161dd | |||
| 7fee203693 | |||
| 562ffc53cc | |||
| 9a67f727b3 | |||
| 5bb714f25b | |||
| 0f592add29 |
@@ -1187,6 +1187,14 @@ jobs:
|
||||
DOUBAO_MODEL: "${{ secrets.DOUBAO_MODEL }}"
|
||||
DOUBAO_BASE_URL: "${{ secrets.DOUBAO_BASE_URL }}"
|
||||
DOUBAO_VISION_MODEL: "${{ secrets.DOUBAO_VISION_MODEL }}"
|
||||
DOUBAO_VISION_LITE_MODEL: "${{ secrets.DOUBAO_VISION_LITE_MODEL }}"
|
||||
DOUBAO_VISION_USE_LITE: "${{ secrets.DOUBAO_VISION_USE_LITE }}"
|
||||
DOUBAO_IMAGE_MODEL: "${{ secrets.DOUBAO_IMAGE_MODEL }}"
|
||||
DOUBAO_IMAGE_SIZE: "${{ secrets.DOUBAO_IMAGE_SIZE }}"
|
||||
DOUBAO_IMAGE_TIMEOUT: "${{ secrets.DOUBAO_IMAGE_TIMEOUT }}"
|
||||
DOUBAO_FAST_MODEL: "${{ secrets.DOUBAO_FAST_MODEL }}"
|
||||
DOUBAO_TIMEOUT: "${{ secrets.DOUBAO_TIMEOUT }}"
|
||||
DOUBAO_MAX_RETRIES: "${{ secrets.DOUBAO_MAX_RETRIES }}"
|
||||
WECHAT_APP_ID: "${{ secrets.WECHAT_APP_ID }}"
|
||||
WECHAT_APP_SECRET: "${{ secrets.WECHAT_APP_SECRET }}"
|
||||
TIKHUB_API_KEY: "${{ secrets.TIKHUB_API_KEY }}"
|
||||
|
||||
@@ -22,6 +22,7 @@ from __future__ import annotations
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
@@ -47,7 +48,6 @@ from packages.shared import get_shared_settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# ── WS 进度推送 ──────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -312,34 +312,6 @@ def _empty_copy_result(duration: int = 15, ratio: str = "9:16") -> dict:
|
||||
# ── 流水线各步骤 ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
_IMAGE_ANALYSIS_SYSTEM_PROMPT = """你是电商商品视觉分析师,从商品图片中提取关键商品信息。严格规则:
|
||||
1. 只说图片里真实可见的内容,看不清/没有的填「无法判断」,不要瞎猜。
|
||||
2. 输出必须是严格 JSON(不要 Markdown 代码块,不要额外解释文字)。
|
||||
3. 字段说明:
|
||||
{
|
||||
"name": "商品全名(品牌+产品名+规格,如『大公鸡头管家 多功能油污净 625ml』,从包装 OCR 读出)",
|
||||
"brand": "品牌名(从 Logo/包装文字读出,看不清填『无法判断』)",
|
||||
"category": "商品品类(如『家用清洁/油污清洁剂』『日化/洗衣液』;非产品图填『非产品图』)",
|
||||
"appearance": "外观特征(50-100字:瓶身形状、颜色、瓶盖、标签颜色、尺寸感)",
|
||||
"packaging": "包装细节(50-100字:标签分区、图案元素、瓶盖/泵头样式、塑封状态)",
|
||||
"text_on_package": ["包装上清晰可见的文字列表(品牌、产品名、卖点、规格等,看不清的不列)"],
|
||||
"key_features": [
|
||||
"3-5 条图片中能看到的外观/视觉特征(如『红色瓶盖白色瓶身』『鸡头图案 Logo』等)"
|
||||
],
|
||||
"scene": "图片场景(如白底棚拍/浴室实拍/桌面静物/手持实拍等)",
|
||||
"portrait_prompt": "如果图片中有清晰人物面部,用60-100字中文描述该人物外貌(性别、年龄段、发型/发色、肤色、脸型、五官特征、当前穿着、表情姿态),用于AI生图参考;没有人物或看不清面部填「无人像」",
|
||||
"summary": "100-180字中文导购描述,连贯自然段落,像电商详情页介绍,前端直接展示,必须提到品牌/品名/核心外观特征,不能写『无法判断』"
|
||||
}"""
|
||||
|
||||
_IMAGE_ANALYSIS_USER_PROMPT = """请分析这张商品图片,输出严格 JSON。重点:
|
||||
1. name/brand/text_on_package 从图片包装 OCR 读取,不编造;
|
||||
2. appearance/packaging 各写 50-100 字,要具体;
|
||||
3. portrait_prompt:有人物时详细描述外貌(性别/年龄/发型/肤色/穿着/表情)用于AI人像生成参考,无人像填「无人像」;
|
||||
4. summary 必须是 100-180 字连贯中文段落,说清商品是什么、长什么样、适合谁用,不要写「无法判断」;
|
||||
5. 非产品图时 category 填「非产品图」,name 填实际看到的内容;
|
||||
6. 看不清的字段填「无法判断」。"""
|
||||
|
||||
|
||||
def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict:
|
||||
d = {
|
||||
"name": "未识别",
|
||||
@@ -385,85 +357,111 @@ def _analyze_single_image(
|
||||
*,
|
||||
pro_fallback_model: str | None = None,
|
||||
) -> dict:
|
||||
"""单张图片 VLM 分析(线程池并行调用)。
|
||||
- lite 失败/结果不可用 时自动用 pro 模型降级重试 1 次。
|
||||
- 失败/None/不可用最终返回含默认字段的 dict(不会让用户看到「未识别·无法判断」裸结果)。
|
||||
"""单张图片 VLM 分析(#2040:改为从 prompt_loader 读模板 + XML 解析)。
|
||||
|
||||
lite 失败/不可用时用 pro 降级重试 1 次。失败/None 最终返回含默认字段的 dict。
|
||||
"""
|
||||
from packages.shared.ai_service import call_vision
|
||||
try:
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template,
|
||||
render_system_prompt,
|
||||
render_user_prompt,
|
||||
)
|
||||
from packages.shared.ai_service import call_vision
|
||||
except ImportError as e:
|
||||
logger.warning("[爆款视频] prompt 模板/解析模块不可用: %s", e)
|
||||
return _vision_fallback(idx, f"fallback_import_error:{e}")
|
||||
|
||||
if not img_url or not isinstance(img_url, str):
|
||||
return _vision_fallback(idx, "invalid_url")
|
||||
|
||||
template = get_template("image_analysis")
|
||||
system = render_system_prompt(template)
|
||||
user = render_user_prompt(
|
||||
template,
|
||||
image_count=1,
|
||||
industry="通用",
|
||||
image_urls=f"第1张:{img_url}",
|
||||
)
|
||||
|
||||
def _call(model: str, tmo: int):
|
||||
try:
|
||||
return call_vision(
|
||||
image_url=img_url,
|
||||
prompt=_IMAGE_ANALYSIS_USER_PROMPT,
|
||||
prompt=user,
|
||||
model=model,
|
||||
max_tokens=800,
|
||||
temperature=0.1,
|
||||
max_tokens=2048,
|
||||
temperature=0.3,
|
||||
timeout=tmo,
|
||||
system_prompt=_IMAGE_ANALYSIS_SYSTEM_PROMPT,
|
||||
system_prompt=system,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片 #%d call_vision(%s) 异常 err=%s", idx, model, e)
|
||||
return None
|
||||
|
||||
def _xml_to_product(nodes: list, raw_text: str) -> dict:
|
||||
product_nodes = [n for n in nodes if n["tag"] == "product"]
|
||||
scene = xp.text_of(raw_text, "scene") or "通用"
|
||||
mood = xp.text_of(raw_text, "mood") or ""
|
||||
for p in product_nodes:
|
||||
a = p["attrs"]
|
||||
text_on_pkg = a.get("text_on_package", "")
|
||||
p_body = p.get("text", "") or ""
|
||||
if not text_on_pkg and p_body:
|
||||
text_on_pkg = xp.text_of(p_body, "text_on_package") or ""
|
||||
text_list = [x.strip() for x in re.split(r"[,,;;]", text_on_pkg) if x.strip()] if text_on_pkg else []
|
||||
features = a.get("features", "")
|
||||
feat_list = [x.strip() for x in re.split(r"[,,;;]", features) if x.strip()] if features else []
|
||||
name = a.get("name", "") or "未识别"
|
||||
brand = a.get("brand", "") or "无法判断"
|
||||
category = a.get("category", "") or "无法判断"
|
||||
appearance = a.get("appearance", "") or "无法判断"
|
||||
packaging = a.get("packaging", "") or "无法判断"
|
||||
summary = a.get("summary", "") or f"{brand} {name}"
|
||||
return {
|
||||
"name": name,
|
||||
"brand": brand,
|
||||
"category": category,
|
||||
"appearance": appearance,
|
||||
"packaging": packaging,
|
||||
"text_on_package": text_list,
|
||||
"key_features": feat_list or [features] if features else ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": a.get("portrait_prompt", "无人像"),
|
||||
"summary": summary,
|
||||
"_source": "xml",
|
||||
}
|
||||
return _vision_fallback(idx, "no_product_tag")
|
||||
|
||||
def _normalize(raw, source: str) -> dict:
|
||||
if raw is None:
|
||||
return _vision_fallback(idx, f"{source}_none")
|
||||
if isinstance(raw, str):
|
||||
logger.warning("[爆款视频] 图片 #%d VLM(%s) 返回非 JSON: %s", idx, source, raw[:200])
|
||||
return _vision_fallback(idx, f"{source}_text", {"_raw": raw[:500]})
|
||||
if not isinstance(raw, dict):
|
||||
if not isinstance(raw, str):
|
||||
return _vision_fallback(idx, f"{source}_badtype")
|
||||
raw.setdefault("_source", source)
|
||||
raw.setdefault("name", "未识别")
|
||||
raw.setdefault("brand", "无法判断")
|
||||
raw.setdefault("category", "无法判断")
|
||||
raw.setdefault("appearance", "无法判断")
|
||||
raw.setdefault("packaging", "无法判断")
|
||||
raw.setdefault("text_on_package", [])
|
||||
raw.setdefault("key_features", [])
|
||||
raw.setdefault("scene", "通用")
|
||||
raw.setdefault("portrait_prompt", "无人像")
|
||||
raw.setdefault("summary", "")
|
||||
if not isinstance(raw.get("text_on_package"), list):
|
||||
raw["text_on_package"] = []
|
||||
if not isinstance(raw.get("key_features"), list):
|
||||
raw["key_features"] = []
|
||||
return raw
|
||||
nodes = xp.parse_tags(raw)
|
||||
if not nodes:
|
||||
logger.warning("[爆款视频] 图片 #%d XML 解析失败 source=%s", idx, source)
|
||||
return _vision_fallback(idx, f"{source}_xml_fail", {"_raw": raw[:500]})
|
||||
product = _xml_to_product(nodes, raw)
|
||||
product.setdefault("_source", source)
|
||||
product["raw"] = raw[:500]
|
||||
return product
|
||||
|
||||
# 第一次:传入模型(通常是 lite)
|
||||
first_raw = _call(vision_model, timeout)
|
||||
tag1 = vision_model.split("/")[-1] if "/" in vision_model else vision_model
|
||||
first_result = _normalize(first_raw, tag1)
|
||||
if _is_vision_result_usable(first_result):
|
||||
return first_result
|
||||
|
||||
logger.warning(
|
||||
"[爆款视频] 图片 #%d VLM(%s) 结果不可用 name=%r summary_len=%d,尝试 pro 降级",
|
||||
idx,
|
||||
vision_model,
|
||||
first_result.get("name"),
|
||||
len(first_result.get("summary") or ""),
|
||||
)
|
||||
|
||||
# 第二次:pro 降级重试
|
||||
if pro_fallback_model and pro_fallback_model != vision_model:
|
||||
pro_raw = _call(pro_fallback_model, 60) # #2180: pro VLM 实测也需25-38s,原25s太短,提到60s
|
||||
pro_result = _normalize(pro_raw, "pro_fallback")
|
||||
if _is_vision_result_usable(pro_result):
|
||||
pro_result["_fallback_used"] = True
|
||||
return pro_result
|
||||
logger.warning(
|
||||
"[爆款视频] 图片 #%d pro 降级仍不可用 name=%r summary_len=%d",
|
||||
idx,
|
||||
pro_result.get("name"),
|
||||
len(pro_result.get("summary") or ""),
|
||||
)
|
||||
return pro_result
|
||||
|
||||
return first_result
|
||||
|
||||
|
||||
@@ -548,8 +546,14 @@ def _step_video_analysis(job: ViralVideoJob) -> dict | None:
|
||||
|
||||
|
||||
def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
"""步骤 2: 用户文案意图解析。"""
|
||||
"""步骤 2: 用户文案意图解析(#2040:改为模板 + XML 解析)。"""
|
||||
try:
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template,
|
||||
render_system_prompt,
|
||||
render_user_prompt,
|
||||
)
|
||||
from packages.shared.ai_service import call_llm
|
||||
except ImportError:
|
||||
return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""}
|
||||
@@ -572,25 +576,35 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
feat_str = ", ".join([str(x) for x in feats + extras])
|
||||
products_summary += f"- {p.get('name', '产品')}: {feat_str}\n"
|
||||
|
||||
prompt = f"""你是一个营销编导。请分析以下信息,理解用户的营销意图并给出短视频主题建议:
|
||||
template = get_template("intent_parsing")
|
||||
system = render_system_prompt(template)
|
||||
user = render_user_prompt(
|
||||
template,
|
||||
user_copy_text=job.user_copy_text or "(未提供,全由 AI 创作)",
|
||||
industry=job.industry or "未指定",
|
||||
image_analysis=products_summary or "- (无图片分析结果)",
|
||||
)
|
||||
|
||||
用户原始文案:{job.user_copy_text or "(未提供,全由 AI 创作)"}
|
||||
行业:{job.industry or "未指定"}
|
||||
目标客户:{job.target_customer or "未指定"}
|
||||
营销目的:{job.marketing_purpose or "未指定"}
|
||||
视频时长:{job.duration}秒
|
||||
产品信息:
|
||||
{products_summary or "- (无图片分析结果)"}
|
||||
def _parse(raw: str) -> dict:
|
||||
summary = xp.text_of(raw, "intent_summary")
|
||||
msgs = [n["text"] for n in xp.find_all(raw, "message") if n["text"]]
|
||||
tone = xp.text_of(raw, "emotion_tone") or "亲切自然"
|
||||
title = xp.text_of(raw, "suggested_title") or xp.text_of(raw, "title")
|
||||
return {
|
||||
"intent": summary or "推广产品",
|
||||
"key_messages": msgs or ["产品亮点"],
|
||||
"tone": tone,
|
||||
"suggested_title": title,
|
||||
}
|
||||
|
||||
请返回严格 JSON(不要 Markdown,不要解释):
|
||||
{{
|
||||
"intent": "核心营销意图(一句话)",
|
||||
"key_messages": ["要传达的3-5个关键信息"],
|
||||
"tone": "文案调性(如亲切/专业/高端/活力/治愈/搞笑)",
|
||||
"target_emotion": "希望触发的用户情感",
|
||||
"call_to_action": "行动号召短句(口语化,5-10字)",
|
||||
"suggested_title": "视频主题标题(5-15字)"
|
||||
}}"""
|
||||
def _fallback(raw: str) -> dict:
|
||||
t = (job.user_copy_text or "").strip()
|
||||
return {
|
||||
"intent": t[:30] or "推广产品",
|
||||
"key_messages": [t[:80]] if t else ["产品亮点"],
|
||||
"tone": "专业",
|
||||
"suggested_title": "",
|
||||
}
|
||||
|
||||
_s = get_shared_settings()
|
||||
_fast = _s.doubao_fast_model
|
||||
@@ -598,22 +612,23 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
for _m, _lbl in [(_fast, "fast"), (_pro, "pro-fallback")]:
|
||||
try:
|
||||
logger.info("[爆款视频] 意图解析 model=%s label=%s", _m, _lbl)
|
||||
result = call_llm(prompt, temperature=0.4, max_tokens=800, model=_m, timeout=45)
|
||||
return (
|
||||
result
|
||||
if isinstance(result, dict)
|
||||
else {"intent": str(result)[:200], "key_messages": [], "tone": "专业", "suggested_title": ""}
|
||||
)
|
||||
raw = call_llm(
|
||||
[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
temperature=0.4,
|
||||
max_tokens=1024,
|
||||
model=_m,
|
||||
timeout=60,
|
||||
) # #2180: 意图解析 LLM 实测需更长响应,原25s太紧
|
||||
if not raw:
|
||||
continue
|
||||
parsed = _parse(raw)
|
||||
if parsed["intent"] or parsed["key_messages"]:
|
||||
return parsed
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 意图解析失败 label=%s err=%s", _lbl, e)
|
||||
continue
|
||||
return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""}
|
||||
return _fallback("")
|
||||
|
||||
|
||||
# ── 编导分镜脚本生成(核心,v1.6 新 prompt) ──────────────────────────────
|
||||
|
||||
|
||||
# 人设 IP 类型 → 文案/出镜风格指导(前端下拉 10 个 IP 类型)
|
||||
_PERSONA_STYLE_GUIDE = {
|
||||
"通用个人IP": "亲切自然、像朋友分享好物,第一人称口语化,不端着",
|
||||
"老板型IP": "沉稳大气、有行业格局感,适度使用『我做了XX年』『我一直坚持』等老板视角,语气自信不夸张",
|
||||
@@ -639,73 +654,6 @@ def _persona_style_hint(persona_id: str) -> str:
|
||||
return "【人设风格:未指定】亲切自然、像朋友分享好物"
|
||||
|
||||
|
||||
_SCRIPT_GENERATION_PROMPT = """你是资深短视频导演,为 Seedance 2.5(单次生成最多{duration}秒)写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。
|
||||
|
||||
## 产品
|
||||
{products_summary}
|
||||
|
||||
## 营销参数
|
||||
- 主题/意图:{intent}
|
||||
- 关键信息:{key_messages}
|
||||
- 调性:{tone}
|
||||
- 目标客户:{target_customer}
|
||||
- 用户原始卖点(必须融入口播):{user_copy}
|
||||
- 时长:{duration}秒 / 画幅:{ratio} / 产品图:{n_images}张(第1张通常是主图/首帧)
|
||||
- 风格参考:{style_hint}
|
||||
- 爆款结构(必须严格遵循节奏/段落顺序):{viral_structure_block}
|
||||
- 人设/出镜口吻(必须贯穿全部对白和动作描写):{persona_hint}
|
||||
|
||||
## 输出格式(必须输出严格 JSON,不要 Markdown,不要解释,字段一个都不能少)
|
||||
|
||||
```json
|
||||
{{
|
||||
"overview": {{
|
||||
"theme": "视频主题(一句话概括)",
|
||||
"total_duration": {duration},
|
||||
"aspect_ratio": "{ratio}"
|
||||
}},
|
||||
"scene_and_lighting": "整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)",
|
||||
"shots": [
|
||||
{{
|
||||
"time_range": "0-3秒",
|
||||
"shot_type_angle_movement": "景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜)",
|
||||
"scene_and_dialogue": "画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔)",
|
||||
"action_details": "人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示)",
|
||||
"audio_bgm": "环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音)",
|
||||
"transition": "硬切/淡入淡出/叠化(最后一镜写『结束』即可)",
|
||||
"reference_image_index": 0
|
||||
}}
|
||||
// ... 按时间顺序列出所有镜头,总时长累计 = {duration} 秒
|
||||
],
|
||||
"hard_constraints": [
|
||||
"无字幕、无水印、无任何自动生成文字、无logo",
|
||||
"同一人物全程五官、发型、服装、身材保持一致,不得换脸变形",
|
||||
"口播语音在总时长内自然念完,语速自然,口型与语音严格同步",
|
||||
"画面流畅无闪烁、无多余肢体、无扭曲变形、无穿模",
|
||||
"色彩自然、曝光正确、电影级质感、高清细节"
|
||||
],
|
||||
"negative_prompts": [
|
||||
"字幕","自动字幕","水印","logo","图标","错误文字","乱码文字",
|
||||
"男女声错配","中途换声","五官崩坏","脸部变形","多余手指",
|
||||
"肢体扭曲","闪烁","画面抖动","模糊","低分辨率"
|
||||
],
|
||||
"voiceover_script": "完整口播稿(把 shots 里所有对白自然拼接成一段,口语化,不加旁白标注、不加镜头标注、不加'主播:'之类前缀,就是纯念出来的文本,长度适配{duration}秒,约{approx_chars}字)"
|
||||
}}
|
||||
```
|
||||
|
||||
## 关键要求
|
||||
1. 每镜写清景别/角度/运镜(特写/近景/中景+平视/俯拍+推/拉/固定)。
|
||||
2. 画面具体:主体(性别/年龄/穿着)、场景、动作、光线、镜头运动要可落地。
|
||||
3. 对白自然口语化,像朋友分享好物;拒绝"家人们""宝子们""太好用了"等浮夸/硬广腔。
|
||||
4. reference_image_index 填 0-based 索引(产品特写用索引0主图),人像/场景可 null。
|
||||
5. shots time_range 累计={duration}秒,单镜2-8秒。
|
||||
6. hard_constraints/negative_prompts 保留默认项可追加,不要删减。
|
||||
7. voiceover_script 为纯口播文本(无标记/括号/前缀),{duration}秒约{approx_chars}字。
|
||||
8. 严格按上方「爆款结构」的节奏/段落顺序编排(钩子/痛点/反转/案例/行动号召与结构对齐)。
|
||||
9. 输出前自检:口播对白禁止错别字和语病,**严禁使用"很近",正确用词是"最近"**(指"最近一段时间/最近在用",绝不能写成"很近");其他同音字、形近字错误一律修正。
|
||||
10. 必须使用产品信息中真实的品牌、品名和外观特征,不要编造与产品无关的内容。"""
|
||||
|
||||
|
||||
def _build_products_summary(image_analysis: dict) -> str:
|
||||
"""把 VLM 返回的商品分析结果拼给文案/分镜生成 prompt 用。
|
||||
优先用 summary(自然段落);没有时用结构化字段兜底拼一段。"""
|
||||
@@ -940,109 +888,165 @@ def _validate_and_normalize_script(raw, job: ViralVideoJob) -> dict:
|
||||
return base
|
||||
|
||||
|
||||
def _script_from_xml(raw: str, job: ViralVideoJob) -> dict | None:
|
||||
"""把 LLM 返回的 XML 分镜规范化为旧 copy_result 结构(供 Seedance 使用)。"""
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
ratio = getattr(job, "video_ratio", None) or "9:16"
|
||||
base = _empty_copy_result(dur, ratio)
|
||||
if not raw:
|
||||
return None
|
||||
base["overview"]["theme"] = xp.text_of(raw, "overview_theme") or xp.text_of(raw, "title") or "好物分享"
|
||||
est = xp.attr_int(xp.text_of(raw, "estimated_duration"), 0)
|
||||
if est:
|
||||
base["overview"]["total_duration"] = est
|
||||
sl = xp.text_of(raw, "scene_and_lighting")
|
||||
if sl:
|
||||
base["scene_and_lighting"] = sl
|
||||
clips = xp.find_all(raw, "clip")
|
||||
shots: list[dict] = []
|
||||
voice_parts: list[str] = []
|
||||
for i, c in enumerate(clips):
|
||||
a = c["attrs"]
|
||||
body = c.get("text", "") or ""
|
||||
ref_idx_raw = a.get("reference_image_index", "")
|
||||
if ref_idx_raw in (None, "", "null", "None"):
|
||||
body_ref = xp.text_of(body, "reference_image_index") if body else ""
|
||||
ref_idx = xp.attr_int(body_ref, 0) if body_ref else None
|
||||
else:
|
||||
ref_idx = xp.attr_int(ref_idx_raw, 0)
|
||||
shot = {
|
||||
"time_range": a.get("time_range") or f"{i * 3}-{(i + 1) * 3}秒",
|
||||
"shot_type_angle_movement": (xp.text_of(body, "shot_type_angle_movement") if body else "")
|
||||
or a.get("shot_type_angle_movement", "")
|
||||
or "中景平视,固定镜头",
|
||||
"scene_and_dialogue": (xp.text_of(body, "scene_and_dialogue") if body else "") or "",
|
||||
"action_details": (xp.text_of(body, "action_details") if body else "") or "",
|
||||
"audio_bgm": (xp.text_of(body, "audio_bgm") if body else "") or a.get("bgm_note", "") or "轻快BGM",
|
||||
"transition": (xp.text_of(body, "transition") if body else "")
|
||||
or a.get("transition", "")
|
||||
or ("硬切" if i < len(clips) - 1 else "结束"),
|
||||
"reference_image_index": ref_idx,
|
||||
}
|
||||
voice = xp.text_of(body, "voice_text") if body else ""
|
||||
if voice:
|
||||
voice_parts.append(voice)
|
||||
if not shot["scene_and_dialogue"]:
|
||||
shot["scene_and_dialogue"] = voice
|
||||
shots.append(shot)
|
||||
if not shots:
|
||||
return None
|
||||
base["shots"] = shots
|
||||
joined = xp.text_of(raw, "voiceover_script") or "。".join(voice_parts)
|
||||
base["voiceover_script"] = joined
|
||||
base["final_copy"] = joined
|
||||
base["suggested_copy"] = joined
|
||||
base["title"] = base["overview"]["theme"]
|
||||
return base
|
||||
|
||||
|
||||
def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: dict) -> dict:
|
||||
"""步骤 3: 编导分镜脚本生成(v1.6 核心,输出 copy_result 结构)。"""
|
||||
"""步骤 3: 编导分镜脚本生成(#2040:模板 + XML 解析;输出 copy_result 结构)。"""
|
||||
try:
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template,
|
||||
render_user_prompt,
|
||||
)
|
||||
from packages.application.viral_video.prompts import (
|
||||
FUSION_INSTRUCTIONS,
|
||||
GLOBAL_CONSTRAINTS,
|
||||
NEGATIVE_RULES,
|
||||
)
|
||||
from packages.shared.ai_service import call_llm
|
||||
except ImportError:
|
||||
return _fallback_script(job)
|
||||
|
||||
products_summary = _build_products_summary(image_analysis)
|
||||
style_hint = "无"
|
||||
if isinstance(job.style_guide, dict):
|
||||
style_hint = (
|
||||
f"节奏{job.style_guide.get('cut_speed', '')}、转场{job.style_guide.get('transition', '')}、"
|
||||
f"色调{job.style_guide.get('color_grade', '')}、能量{job.style_guide.get('energy', '')}"
|
||||
)
|
||||
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
ratio = getattr(job, "video_ratio", None) or "9:16"
|
||||
approx_chars = max(20, dur * 4)
|
||||
|
||||
intent_str = ""
|
||||
key_msgs = ""
|
||||
tone = ""
|
||||
intent_str = "推广产品"
|
||||
key_msgs = "产品亮点"
|
||||
tone = "亲切自然"
|
||||
if isinstance(intent, dict):
|
||||
intent_str = intent.get("intent") or "推广产品"
|
||||
key_msgs = "、".join(intent.get("key_messages") or [])
|
||||
tone = intent.get("tone") or "亲切自然"
|
||||
else:
|
||||
intent_str = "推广产品"
|
||||
tone = "亲切自然"
|
||||
intent_str = intent.get("intent") or intent_str
|
||||
key_msgs = "、".join(intent.get("key_messages") or []) or key_msgs
|
||||
tone = intent.get("tone") or tone
|
||||
|
||||
# 爆款结构:用户在 STEP1/STEP2 选的中文结构名,必须严格注入 prompt 指导 AI 编排
|
||||
_vs = (job.viral_structure or "").strip()
|
||||
if _vs:
|
||||
viral_structure_block = f"【{_vs}】—— 请严格按照这个爆款结构的节奏/段落顺序编排镜头、台词和情绪节点(开场钩子、痛点、反转、案例、行动号召等按结构走),不要打乱顺序"
|
||||
else:
|
||||
viral_structure_block = "未指定(自由编排,但仍需有钩子开头+产品展示+行动号召的基本节奏)"
|
||||
fusion_level = getattr(job, "fusion_level", "ai_polish") or "ai_polish"
|
||||
fusion_instruction = FUSION_INSTRUCTIONS.get(fusion_level, FUSION_INSTRUCTIONS["ai_polish"])
|
||||
|
||||
persona_hint = _persona_style_hint(getattr(job, "persona_id", ""))
|
||||
prompt = _SCRIPT_GENERATION_PROMPT.format(
|
||||
products_summary=products_summary,
|
||||
intent=intent_str,
|
||||
key_messages=key_msgs or "产品亮点",
|
||||
tone=tone,
|
||||
target_customer=job.target_customer or "通用人群",
|
||||
user_copy=job.user_copy_text or "(未提供,自由创作)",
|
||||
duration=dur,
|
||||
ratio=ratio,
|
||||
n_images=len(job.images or []),
|
||||
style_hint=style_hint,
|
||||
approx_chars=approx_chars,
|
||||
viral_structure_block=viral_structure_block,
|
||||
persona_hint=persona_hint,
|
||||
# 使用 storyboard 模板,注入融合指令/硬约束/反套路词
|
||||
template = get_template("storyboard")
|
||||
system_tpl = template.system_prompt
|
||||
system_tpl = system_tpl.replace("{fusion_instruction}", fusion_instruction)
|
||||
system_tpl = system_tpl.replace("{global_constraints}", GLOBAL_CONSTRAINTS)
|
||||
system_tpl = system_tpl.replace("{negative_rules}", NEGATIVE_RULES)
|
||||
|
||||
fusion_brief = (
|
||||
f"意图:{intent_str}\n关键信息:{key_msgs}\n调性:{tone}\n"
|
||||
f"用户原文:{job.user_copy_text or '(未提供)'}\n创作模式:{fusion_level}"
|
||||
)
|
||||
user = render_user_prompt(
|
||||
template,
|
||||
duration=dur,
|
||||
image_count=len(job.images or []),
|
||||
fusion_result=fusion_brief,
|
||||
image_analysis=products_summary,
|
||||
)
|
||||
|
||||
def _try_gen(model: str, temp: float, max_tok: int, label: str, tmo: int = 25):
|
||||
logger.info("[爆款视频] 编导脚本生成 model=%s label=%s timeout=%d", model, label, tmo)
|
||||
raw = call_llm(
|
||||
[{"role": "system", "content": system_tpl}, {"role": "user", "content": user}],
|
||||
temperature=temp,
|
||||
max_tokens=max_tok,
|
||||
model=model,
|
||||
timeout=tmo,
|
||||
)
|
||||
if not raw:
|
||||
return None
|
||||
normalized = _script_from_xml(raw, job)
|
||||
if normalized is None:
|
||||
# 兼容:万一 LLM 仍输出 JSON,走旧规范化
|
||||
parsed_json = _safe_json_loads(raw)
|
||||
if isinstance(parsed_json, dict):
|
||||
normalized = _validate_and_normalize_script(parsed_json, job)
|
||||
else:
|
||||
return None
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
shots_cnt = len((normalized or {}).get("shots") or [])
|
||||
_before_dump = json.dumps(normalized, ensure_ascii=False)
|
||||
if "很近" in _before_dump:
|
||||
normalized = _replace_henjin_everywhere(normalized)
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
fallback_marker = "我最近在用的好物" in voiceover
|
||||
has_typo_henjin = "很近" in json.dumps(normalized, ensure_ascii=False)
|
||||
is_fallback = fallback_marker or shots_cnt < 1 or len(voiceover) < 20 or has_typo_henjin
|
||||
logger.info(
|
||||
"[爆款视频] 编导脚本结果 label=%s voiceover_len=%d shots=%d fallback=%s",
|
||||
label,
|
||||
len(voiceover),
|
||||
shots_cnt,
|
||||
is_fallback,
|
||||
)
|
||||
return None if is_fallback else normalized
|
||||
|
||||
_s = get_shared_settings()
|
||||
_fast = _s.doubao_fast_model
|
||||
_pro = getattr(_s, "doubao_model", None) or _fast
|
||||
|
||||
def _try_gen(model: str, temp: float, max_tok: int, label: str, tmo: int = 25):
|
||||
logger.info("[爆款视频] 编导脚本生成 model=%s label=%s timeout=%d", model, label, tmo)
|
||||
r = call_llm(prompt, temperature=temp, max_tokens=max_tok, model=model, timeout=tmo)
|
||||
if r is None:
|
||||
logger.warning("[爆款视频] 编导脚本返回None label=%s", label)
|
||||
return None
|
||||
parsed = _safe_json_loads(r)
|
||||
normalized = _validate_and_normalize_script(parsed, job)
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
voiceover_len = len(voiceover)
|
||||
shots_cnt = len((normalized or {}).get("shots") or [])
|
||||
# 判定是否"退化到兜底质量":口播过短(<20字)或镜头数<1;正常的短口播(如15s视频~40字)不视为兜底
|
||||
# v1.6.1 P1修复:递归替换 copy_result 里所有字符串字段的"很近"→"最近"(覆盖 overview/scene_and_lighting/voiceover/shots.* 全部字段)
|
||||
_before_dump = json.dumps(normalized, ensure_ascii=False)
|
||||
if "很近" in _before_dump:
|
||||
logger.warning("[爆款视频] 编导脚本含错别字'很近',递归替换为'最近' label=%s", label)
|
||||
normalized = _replace_henjin_everywhere(normalized)
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
fallback_marker = "我最近在用的好物" in voiceover # _fallback_script 的特征串
|
||||
has_typo_henjin = "很近" in json.dumps(normalized, ensure_ascii=False) # 递归检查仍有"很近"视为不合格
|
||||
is_fallback = fallback_marker or shots_cnt < 1 or voiceover_len < 20 or has_typo_henjin
|
||||
logger.info(
|
||||
"[爆款视频] 编导脚本结果 label=%s voiceover_len=%d shots=%d fallback=%s raw_type=%s",
|
||||
label,
|
||||
voiceover_len,
|
||||
shots_cnt,
|
||||
is_fallback,
|
||||
type(r).__name__,
|
||||
)
|
||||
if is_fallback:
|
||||
return None # 触发重试
|
||||
return normalized
|
||||
|
||||
try:
|
||||
# 第一次:快模型 25s
|
||||
normalized = _try_gen(_fast, 0.8, 2500, "fast-first", tmo=45)
|
||||
normalized = _try_gen(_fast, 0.8, 2500, "fast-first", tmo=90)
|
||||
if normalized is not None:
|
||||
return normalized
|
||||
# 第二次:快模型降温度+加大 max_tokens,25s
|
||||
normalized = _try_gen(_fast, 0.6, 3200, "fast-retry", tmo=45)
|
||||
# #2183: 实测pro 1500tok输出需75.8s,单次timeout提到90s
|
||||
normalized = _try_gen(_fast, 0.6, 3200, "fast-retry", tmo=90)
|
||||
if normalized is not None:
|
||||
return normalized
|
||||
# 第三次:用主力模型兜底,给 40s
|
||||
# 第三次:用主力模型兜底,给 120s
|
||||
if _pro and _pro != _fast:
|
||||
normalized = _try_gen(_pro, 0.7, 3500, "pro-fallback", tmo=60)
|
||||
normalized = _try_gen(_pro, 0.7, 3500, "pro-fallback", tmo=120)
|
||||
if normalized is not None:
|
||||
return normalized
|
||||
logger.warning("[爆款视频] 编导脚本三次都未生成合格结果,使用兜底脚本")
|
||||
@@ -1053,38 +1057,94 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
|
||||
|
||||
def _step_review(job: ViralVideoJob, copy_result: dict) -> dict:
|
||||
"""步骤 4: 合规审核(简化版:基于脚本的 voiceover_script+shots 文本)。"""
|
||||
dimensions = ["广告法合规", "平台规范", "内容真实性", "版权安全", "价值观", "风格一致性"]
|
||||
"""步骤 4: 合规审核(#2040:使用 Reviewer + review 模板,6 维度 + 自动重写 1 次)。
|
||||
|
||||
返回结构与旧版兼容:{passed, score, details, issues, rewritten_copy?}
|
||||
"""
|
||||
try:
|
||||
from packages.shared.ai_service import call_llm
|
||||
except ImportError:
|
||||
return {"passed": True, "score": 90, "details": {d: "通过" for d in dimensions}}
|
||||
from packages.application.viral_video.reviewer import Reviewer
|
||||
from packages.application.viral_video.schemas import (
|
||||
CoreMessage,
|
||||
FusionResult,
|
||||
IntentResult,
|
||||
PersonalBrand,
|
||||
ScriptSegment,
|
||||
)
|
||||
except ImportError as e:
|
||||
logger.warning("[爆款视频] reviewer 模块不可用,跳过审核: %s", e)
|
||||
return {"passed": True, "score": 80, "details": {}, "issues": []}
|
||||
|
||||
voiceover = (copy_result or {}).get("voiceover_script", "")
|
||||
shots_preview = json.dumps((copy_result or {}).get("shots", [])[:3], ensure_ascii=False)
|
||||
prompt = f"""请对以下短视频编导脚本进行合规审核,检查6个维度:{", ".join(dimensions)}
|
||||
voiceover = (copy_result or {}).get("voiceover_script", "") or ""
|
||||
title = (copy_result or {}).get("title") or (copy_result or {}).get("overview", {}).get("theme", "")
|
||||
|
||||
口播文案:{voiceover}
|
||||
前3个镜头:{shots_preview}
|
||||
行业:{job.industry}
|
||||
intent_data = job.intent_result or {}
|
||||
core_msgs = [
|
||||
CoreMessage(text=str(m), must_keep=True, confidence=0.9) for m in (intent_data.get("key_messages") or [])
|
||||
]
|
||||
brands: list[PersonalBrand] = []
|
||||
brand_text = intent_data.get("brand_text") or intent_data.get("suggested_title") or ""
|
||||
if brand_text:
|
||||
brands.append(PersonalBrand(text=str(brand_text), category="brand"))
|
||||
intent_obj = IntentResult(
|
||||
intent_summary=intent_data.get("intent", "") or "推广产品",
|
||||
core_messages=core_msgs,
|
||||
personal_brands=brands,
|
||||
)
|
||||
fusion_obj = FusionResult(
|
||||
title=title or "",
|
||||
hook=(voiceover[:30] if voiceover else ""),
|
||||
cta="",
|
||||
script_segments=[],
|
||||
word_count=len(voiceover),
|
||||
estimated_duration=int(getattr(job, "duration", 15) or 15),
|
||||
)
|
||||
for shot in (copy_result or {}).get("shots", []) or []:
|
||||
if isinstance(shot, dict) and shot.get("scene_and_dialogue"):
|
||||
fusion_obj.script_segments.append(ScriptSegment(text=shot["scene_and_dialogue"]))
|
||||
|
||||
请以JSON格式返回:
|
||||
- passed: bool(是否全部通过)
|
||||
- score: int(0-100分)
|
||||
- details: 各维度评分和说明
|
||||
- issues: 需要修改的问题列表(如有)"""
|
||||
_s = get_shared_settings()
|
||||
_fast = _s.doubao_fast_model
|
||||
_pro = _s.doubao_model
|
||||
for _m, _lbl in [(_fast, "fast"), (_pro, "pro-fallback")]:
|
||||
try:
|
||||
logger.info("[爆款视频] 合规审核 model=%s label=%s", _m, _lbl)
|
||||
result = call_llm(prompt, temperature=0.1, max_tokens=500, model=_m, timeout=30)
|
||||
return result if isinstance(result, dict) else {"passed": True, "score": 80, "details": {}}
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 合规审核失败 label=%s err=%s", _lbl, e)
|
||||
continue
|
||||
return {"passed": True, "score": 75, "details": {d: "默认通过" for d in dimensions}}
|
||||
fusion_level = getattr(job, "fusion_level", "ai_polish") or "ai_polish"
|
||||
try:
|
||||
reviewer = Reviewer()
|
||||
review_res = reviewer.review(fusion_obj, intent_obj, fusion_level)
|
||||
new_copy = copy_result
|
||||
rewritten_voice = None
|
||||
if not review_res.passed and review_res.rewrite_suggestions:
|
||||
try:
|
||||
rewritten = reviewer.rewrite(fusion_obj, review_res, intent_obj, fusion_level)
|
||||
if rewritten and (rewritten.title or rewritten.script_segments):
|
||||
new_voice = (
|
||||
rewritten.script_segments[0].text
|
||||
if rewritten.script_segments
|
||||
else (rewritten.hook or voiceover)
|
||||
)
|
||||
new_copy = dict(copy_result)
|
||||
new_copy["voiceover_script"] = new_voice
|
||||
new_copy["final_copy"] = new_voice
|
||||
new_copy["suggested_copy"] = new_voice
|
||||
if rewritten.title:
|
||||
new_copy.setdefault("overview", {})["theme"] = rewritten.title
|
||||
new_copy["title"] = rewritten.title
|
||||
rewritten_voice = new_voice
|
||||
review_res = reviewer.review(rewritten, intent_obj, fusion_level)
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 自动重写失败: %s", e)
|
||||
result = {
|
||||
"passed": review_res.passed,
|
||||
"score": 90 if review_res.passed else 60,
|
||||
"details": {i.dimension: i.text for i in review_res.issues},
|
||||
"issues": [
|
||||
{"dimension": i.dimension, "severity": i.severity, "location": i.location, "text": i.text}
|
||||
for i in review_res.issues
|
||||
],
|
||||
}
|
||||
if rewritten_voice is not None:
|
||||
result["rewritten_copy"] = new_copy
|
||||
job.copy_result = new_copy
|
||||
job.generated_copy_text = rewritten_voice
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 审核异常,跳过: %s", e, exc_info=True)
|
||||
return {"passed": True, "score": 75, "details": {}, "issues": []}
|
||||
|
||||
|
||||
def _step_tts(job: ViralVideoJob, voiceover_script: str):
|
||||
@@ -1251,13 +1311,47 @@ def _step_render(job: ViralVideoJob, copy_result: dict, tts_audio_url: str | Non
|
||||
pre_trusted = list(pti)
|
||||
logger.info("[爆款视频] 使用信任链预热结果 n=%d,跳过现场 Seedream AI 化", len(pre_trusted))
|
||||
elif all_portrait_urls and _mcfg.get("provider", "doubao") == "doubao":
|
||||
# #2183: 真·现场跑信任链——同步调用 Seedream t2i,拿到 AI 人像 URL 后再传 Seedance
|
||||
logger.info(
|
||||
"[爆款视频] 预热结果不可用(%s/%d张),将现场跑信任链",
|
||||
"[爆款视频] 预热结果不可用(%s/%d张),现场同步跑信任链Seedream t2i",
|
||||
"缺失" if not pti else f"{len(pti)}/{len(all_portrait_urls)}",
|
||||
len(all_portrait_urls),
|
||||
)
|
||||
try:
|
||||
from packages.shared.ai_service import preheat_trust_chain
|
||||
|
||||
# 第一次调用:带参考图/首帧/音频/参考视频
|
||||
_ia = getattr(job, "image_analysis", None) or {}
|
||||
_prods = (_ia.get("products") if isinstance(_ia, dict) else None) or []
|
||||
_pdescs = []
|
||||
if _prods:
|
||||
_pdescs = [(pp.get("portrait_prompt") or "无人像") for pp in _prods]
|
||||
elif isinstance(_ia, dict):
|
||||
_pp0 = _ia.get("portrait_prompt") or "无人像"
|
||||
if _pp0 and _pp0 != "无人像":
|
||||
_pdescs = [_pp0]
|
||||
_valid = [d for d in _pdescs if d and isinstance(d, str) and "无人像" not in d and len(d) >= 10]
|
||||
if _valid:
|
||||
_t0 = time.time()
|
||||
_live_urls = preheat_trust_chain(_valid, timeout=120)
|
||||
if _live_urls and len(_live_urls) == len(all_portrait_urls):
|
||||
pre_trusted = list(_live_urls)
|
||||
logger.info(
|
||||
"[爆款视频] 现场信任链t2i完成 %d张 耗时%.1fs,将用AI人像传Seedance",
|
||||
len(pre_trusted),
|
||||
time.time() - _t0,
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
"[爆款视频] 现场信任链t2i返回不匹配 urls=%s n_portraits=%d,回退原图+400降级纯t2v",
|
||||
_live_urls,
|
||||
len(all_portrait_urls),
|
||||
)
|
||||
else:
|
||||
logger.info("[爆款视频] 无有效人物描述(可能是商品图),无需现场跑信任链")
|
||||
except Exception as _te:
|
||||
logger.warning("[爆款视频] 现场跑信任链异常: %s,回退原图+400降级纯t2v", _te, exc_info=True)
|
||||
|
||||
# 第一次调用:带参考图/首帧/音频/参考视频(pre_trusted有值→走信任链;无值→原图;若400 ai_client内部自动降级纯t2v)
|
||||
result = call_video_generation(
|
||||
prompt=prompt,
|
||||
image_url=first_image,
|
||||
@@ -1550,8 +1644,26 @@ def run_viral_video_analyze(self: Task, job_id: str) -> dict:
|
||||
|
||||
image_analysis = _step_image_analysis(job)
|
||||
job.image_analysis = image_analysis
|
||||
# #2183/#2174: VLM完成后立即启动信任链t2i预热(后台daemon线程,与后续阶段并行)
|
||||
if job.images:
|
||||
try:
|
||||
_products = (image_analysis or {}).get("products", []) or []
|
||||
_portrait_descs = [(p.get("portrait_prompt") or "无人像") for p in _products] if _products else []
|
||||
if not _portrait_descs and isinstance(image_analysis, dict):
|
||||
_pp = image_analysis.get("portrait_prompt") or "无人像"
|
||||
if _pp and _pp != "无人像":
|
||||
_portrait_descs = [_pp]
|
||||
_start_trust_chain_preheat(job.id, _portrait_descs)
|
||||
except Exception as _e:
|
||||
logger.warning("[爆款视频][阶段1] 启动信任链t2i预热失败: %s", _e)
|
||||
_save_job(repo, job, session)
|
||||
_emit_progress(job_id, ViralVideoStage.IMAGE_ANALYSIS, 60.0, "图片分析完成", {"result": image_analysis})
|
||||
_emit_progress(
|
||||
job_id,
|
||||
ViralVideoStage.IMAGE_ANALYSIS,
|
||||
60.0,
|
||||
"图片分析完成,AI人像预热中...",
|
||||
{"result": image_analysis, "trust_chain_preheating": True},
|
||||
)
|
||||
|
||||
style_guide = None
|
||||
if job.reference_video_url or job.style_template_id:
|
||||
@@ -1598,8 +1710,8 @@ def run_viral_video_analyze(self: Task, job_id: str) -> dict:
|
||||
bind=True,
|
||||
max_retries=1,
|
||||
name="worker.run_viral_video_generate_copy",
|
||||
soft_time_limit=360, # #2173: 6min(编导脚本含意图+三级重试+审核,fast超时转pro)
|
||||
time_limit=420, # #2173: 7min hard limit
|
||||
soft_time_limit=600, # #2183: pro长脚本~120s+三级重试最坏300s+intent/review ~105s,给到10min
|
||||
time_limit=660, # #2183: 11min hard limit
|
||||
)
|
||||
def run_viral_video_generate_copy(self: Task, job_id: str) -> dict:
|
||||
"""v1.6 阶段2(v1.6.1 提速版):意图解析 → 编导分镜脚本生成 → 直接返回,合规审核后置到出片前。
|
||||
@@ -1863,9 +1975,14 @@ def _run_render_pipeline(job_id: str, session, repo, job) -> dict:
|
||||
review_result = _step_review(job, copy_result)
|
||||
if not review_result.get("passed", True):
|
||||
_emit_progress(job_id, ViralVideoStage.REVIEW, 67.0, "审核未通过,正在自动重写...")
|
||||
intent = job.intent_result or _step_intent_parsing(job, image_analysis)
|
||||
copy_result = _step_script_generation(job, intent, image_analysis)
|
||||
_step_review(job, copy_result) # 二次审核,不通过也继续出片(避免反复循环)
|
||||
# #2040: Reviewer 已在 _step_review 内完成 1 次自动重写
|
||||
rewritten = review_result.get("rewritten_copy")
|
||||
if isinstance(rewritten, dict) and rewritten:
|
||||
copy_result = rewritten
|
||||
else:
|
||||
intent = job.intent_result or _step_intent_parsing(job, image_analysis)
|
||||
copy_result = _step_script_generation(job, intent, image_analysis)
|
||||
_step_review(job, copy_result)
|
||||
job.copy_result = copy_result
|
||||
job.generated_copy_text = copy_result.get("voiceover_script", "") or ""
|
||||
_save_job(repo, job, session)
|
||||
|
||||
@@ -241,19 +241,35 @@ DOUBAO_API_KEY=${DOUBAO_API_KEY}
|
||||
|
||||
# 模型 Endpoint ID(在 ARK 控制台创建推理接入点后获得)
|
||||
DOUBAO_MODEL=${DOUBAO_MODEL}
|
||||
DOUBAO_FAST_MODEL=${DOUBAO_FAST_MODEL}
|
||||
|
||||
# API Base URL
|
||||
DOUBAO_BASE_URL=${DOUBAO_BASE_URL}
|
||||
|
||||
# 请求超时(秒)
|
||||
DOUBAO_TIMEOUT=60
|
||||
DOUBAO_TIMEOUT=${DOUBAO_TIMEOUT}
|
||||
|
||||
# 最大重试次数
|
||||
DOUBAO_MAX_RETRIES=2
|
||||
DOUBAO_MAX_RETRIES=${DOUBAO_MAX_RETRIES}
|
||||
|
||||
# 视觉模型 Endpoint ID(支持图片/视频理解的模型)
|
||||
# 视觉模型(支持图片/视频理解的模型,model name 格式)
|
||||
DOUBAO_VISION_MODEL=${DOUBAO_VISION_MODEL}
|
||||
|
||||
# 快速视觉模型(viral-video 图片分析 lite 路径)
|
||||
DOUBAO_VISION_LITE_MODEL=${DOUBAO_VISION_LITE_MODEL}
|
||||
|
||||
# 是否启用 lite 视觉路径(true/false)
|
||||
DOUBAO_VISION_USE_LITE=${DOUBAO_VISION_USE_LITE}
|
||||
|
||||
# 信任链文生图模型(Seedream)
|
||||
DOUBAO_IMAGE_MODEL=${DOUBAO_IMAGE_MODEL}
|
||||
|
||||
# 文生图尺寸
|
||||
DOUBAO_IMAGE_SIZE=${DOUBAO_IMAGE_SIZE}
|
||||
|
||||
# 文生图超时(秒)
|
||||
DOUBAO_IMAGE_TIMEOUT=${DOUBAO_IMAGE_TIMEOUT}
|
||||
|
||||
|
||||
# ==================== 微信开放平台 OAuth(网页扫码登录)====================
|
||||
# 回调域名:xiaoxiajianji.com(微信开放平台已配置)
|
||||
|
||||
@@ -132,7 +132,10 @@ _FUSION_SYSTEM = """你负责为短视频生成营销文案。请按思维链分
|
||||
<hook> 开头3秒钩子,5到15字。
|
||||
<body_points> 每个要点用一个 <point> 标签,属性 elaboration 是展开说明、image_index 是对应第几张图(从0开始),标签内容写要点。
|
||||
<cta> 口语化的行动号召。
|
||||
<script_segments> 每段配音用一个 <segment> 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案。
|
||||
<script_segments> 每段配音用一个 <segment> 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案(纯口播文本,不加旁白标注、不加镜头标注、不加"主播:"之类前缀)。
|
||||
<voiceover_script> 把所有 segment 的配音文案按顺序自然拼接成一段完整的纯口播文本(无标记、无括号、无前缀),长度要适配 {duration} 秒,约 {approx_chars} 字。
|
||||
<overview_theme> 视频主题(一句话概括)。
|
||||
<scene_and_lighting> 整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)。
|
||||
<word_count> 配音总字数,只写数字。
|
||||
<estimated_duration> 预计时长秒数,只写数字。
|
||||
|
||||
@@ -161,25 +164,38 @@ _FUSION_EXAMPLE = """<title>厨房重油污,别再用洗洁精硬擦了</title
|
||||
<segment duration_sec="6" image_index="0">后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</segment>
|
||||
<segment duration_sec="4" image_index="0">39块钱625ml,厨房重油污的可以试一瓶</segment>
|
||||
</script_segments>
|
||||
<voiceover_script>这油污我真的忍很久了,用洗洁精擦半天都没用。后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净。39块钱625ml,厨房重油污的可以试一瓶。</voiceover_script>
|
||||
<overview_theme>厨房油污清洁好物分享</overview_theme>
|
||||
<scene_and_lighting>简洁明亮的厨房台面场景,自然光从窗户洒入,色调温暖柔和,突出产品白色瓶身与去油污对比效果。</scene_and_lighting>
|
||||
<word_count>58</word_count>
|
||||
<estimated_duration>13</estimated_duration>"""
|
||||
|
||||
# ── 模板4:编导级分镜(LLM)────────────────────────────────────────────
|
||||
_STORYBOARD_SYSTEM = f"""你是短视频编导,负责把文案拆成可拍摄的分镜。
|
||||
_STORYBOARD_SYSTEM = """你是短视频编导,负责把文案拆成可拍摄的分镜,为 Seedance 2.5 视频模型写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。
|
||||
|
||||
工作方式:
|
||||
1. 按文案的 script_segments 顺序分配镜头。
|
||||
2. 每个镜头确定画面、运镜、时长、配音和字幕。
|
||||
2. 每个镜头确定景别/角度/运镜、画面场景与对白、人物动作细节、音效/BGM、转场。
|
||||
3. 检查所有镜头时长加起来接近目标时长,误差不超过2秒。
|
||||
4. image_index 必须在已上传图片范围内,第一张主图必须用在第一个镜头。
|
||||
|
||||
{GLOBAL_CONSTRAINTS}
|
||||
{fusion_instruction}
|
||||
|
||||
{global_constraints}
|
||||
|
||||
{negative_rules}
|
||||
|
||||
请严格按下面的标签格式输出,不要解释,不要用代码块:
|
||||
<clips> 下面每个镜头用一个 <clip> 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade、cut、zoom_in、slide_left、dissolve、wipe 之一、zoom 取 in、out 或 null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 <clip> 里面包含:
|
||||
<voice_text> 该镜头配音文本;
|
||||
<subtitle_text> 字幕文本,可与配音一致或更精简;
|
||||
<ken_burns> 用一个空标签,属性 start、end 写“x,y”坐标、ease 写缓动方式;不需要运镜时坐标相同。"""
|
||||
<clips> 下面每个镜头用一个 <clip> 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade/cut/zoom_in/slide_left/dissolve/wipe 之一、zoom 取 in/out/null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 <clip> 里面包含:
|
||||
<voice_text> 该镜头配音文本(纯口播文本,不加旁白标注);
|
||||
<subtitle_text> 字幕文本,可与配音一致或更精简;
|
||||
<shot_type_angle_movement> 景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜);
|
||||
<scene_and_dialogue> 画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔);
|
||||
<action_details> 人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示);
|
||||
<audio_bgm> 环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音);
|
||||
<transition> 硬切/淡入淡出/叠化(最后一镜写『结束』即可);
|
||||
<reference_image_index> 参考图片索引(0-based,对应第几张产品图,无则空);
|
||||
<ken_burns> 用一个空标签,属性 start、end 写"x,y"坐标、ease 写缓动方式;不需要运镜时坐标相同。"""
|
||||
|
||||
_STORYBOARD_USER = """目标时长:{duration}秒
|
||||
上传图片数量:{image_count}张(第1张是主图/封面)
|
||||
@@ -194,17 +210,35 @@ _STORYBOARD_EXAMPLE = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="3" bgm_note="日常、轻微烦躁">
|
||||
<voice_text>这油污我真的忍很久了</voice_text>
|
||||
<subtitle_text>这油污忍很久了</subtitle_text>
|
||||
<shot_type_angle_movement>近景俯拍45度,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>厨房台面,主妇皱眉看着灶台油污。对白:这油污我真的忍很久了</scene_and_dialogue>
|
||||
<action_details>右手拿着脏抹布,无奈摇头</action_details>
|
||||
<audio_bgm>轻快日常BGM,带一点烦躁感</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="zoom_in" zoom="in" duration_sec="6" bgm_note="轻快、出现转机">
|
||||
<voice_text>后来换了大公鸡头油污净,喷上等几分钟,一擦就干净</voice_text>
|
||||
<subtitle_text>喷上等几分钟,一擦就干净</subtitle_text>
|
||||
<shot_type_angle_movement>特写平视,固定镜头</shot_type_angle_movement>
|
||||
<scene_and_dialogue>手部特写,喷油污净在油污处。对白:后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</scene_and_dialogue>
|
||||
<action_details>左手拿产品瓶身,右手按压喷头,等待片刻后用抹布轻擦</action_details>
|
||||
<audio_bgm>轻快转折BGM,带清爽感</audio_bgm>
|
||||
<transition>淡入淡出</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="20,20" end="80,80" ease="ease-in-out"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="fade" zoom="null" duration_sec="4" bgm_note="温暖、推荐">
|
||||
<voice_text>39块钱625ml,厨房重油污的可以试一瓶</voice_text>
|
||||
<subtitle_text>39元625ml,可以试一瓶</subtitle_text>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
<shot_type_angle_movement>中景平视,缓慢拉镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>产品正面展示,明亮背景。对白:39块钱625ml,厨房重油污的可以试一瓶</scene_and_dialogue>
|
||||
<action_details>产品置于画面中央,轻微转动展示瓶身</action_details>
|
||||
<audio_bgm>温暖收尾BGM</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="50,50" end="20,20" ease="ease-in-out"/>
|
||||
</clip>
|
||||
</clips>"""
|
||||
|
||||
|
||||
@@ -92,7 +92,7 @@ class SharedSettings(BaseSettings):
|
||||
doubao_api_key: str = ""
|
||||
doubao_model: str = "doubao-seed-2-1-pro-260915" # 推理模型(Seed 2.1 Pro,深度思考+多模态;原 seed-1-6 已下线)
|
||||
doubao_fast_model: str = (
|
||||
"doubao-seed-2-1-lite-260915" # 快速模型(Seed 2.1 Lite,高 RPM,编导/审核/VLM lite;原 1-5-pro-32k 已 Retiring)
|
||||
"doubao-seed-2-1-pro-260915" # #2181: lite方舟侧100%超时,默认fast_model也走pro;方舟恢复lite后通过ENV DOUBAO_FAST_MODEL切回
|
||||
)
|
||||
doubao_base_url: str = "https://ark.cn-beijing.volces.com/api/v3"
|
||||
doubao_timeout: int = 45 # #2180: 方舟LLM高峰期响应6-8s,原30s太紧提到45s
|
||||
@@ -103,7 +103,7 @@ class SharedSettings(BaseSettings):
|
||||
doubao_vision_lite_model: str = (
|
||||
"doubao-seed-2-1-lite-260915" # 快速视觉(Seed 2.1 Lite 原生多模态;原 vision-lite-250315 不可用)
|
||||
)
|
||||
doubao_vision_use_lite: bool = True # viral-video 图片分析默认用 lite 提速
|
||||
doubao_vision_use_lite: bool = False # #2181: lite视觉模型100%超时,默认关闭走pro(25-38s稳定返回)
|
||||
doubao_embedding_model: str = "doubao-embedding-vision-251215" # 多模态向量化(原 large-text-240915 已 Retiring)
|
||||
doubao_video_model: str = "doubao-seedance-2-5-260628"
|
||||
doubao_video_timeout: int = 600 # 视频生成轮询总超时(秒)
|
||||
|
||||
@@ -585,7 +585,7 @@ class DoubaoClient:
|
||||
# #2172/#2174: 信任链——使用预热好的 Seedream t2i 文生图(纯模型生成人像,是方舟信任产物,
|
||||
# 不会触发肖像审核)。预热在 VLM 分析后由 daemon 线程后台完成,结果通过 pre_trusted_images 传入。
|
||||
# - 预热结果有效 → 替换原参考图,走 omni_ref 模式
|
||||
# - 预热结果不可用 → 直接用原图(若被400肖像拦截,#2166自动降级纯t2v),避免现场跑t2i阻塞渲染
|
||||
# - 预热结果不可用 → 直接用原图(上层 _step_render 已现场同步跑 Seedream t2i 兜底;若再被400拦截,下方自动降级纯t2v)
|
||||
# 信任链只作用于 doubao provider;DashScope(Wan) 保持原行为。
|
||||
trust_chain_applied = False
|
||||
if provider == "doubao" and getattr(self, "trust_chain_enabled", True) and pre_trusted_images:
|
||||
@@ -763,6 +763,26 @@ class DoubaoClient:
|
||||
# 保留第二次的错误信息
|
||||
sc, body = sc2, body2
|
||||
|
||||
# #2183: portrait_intercept / 真人肖像审核拦截 → 去掉所有参考图(含image_url首帧),纯 t2v 重试一次
|
||||
# 信任链预热或现场 t2i 都失败时的最后兜底,保证能出片
|
||||
if not task_id and sc == 400:
|
||||
_err_code_for_400, _ = _classify_video_error(sc, body, last_err)
|
||||
if _err_code_for_400 == "portrait_intercept" and (image_url or ref_imgs):
|
||||
logger.warning(
|
||||
"Seedance 创建因 portrait_intercept 失败,降级纯 t2v(移除所有参考图)重试: img=%d ref=%d",
|
||||
1 if image_url else 0,
|
||||
len(ref_imgs),
|
||||
)
|
||||
_t2v_payload = dict(create_payload)
|
||||
_t2v_payload["content"] = [{"type": "text", "text": prompt.strip()}]
|
||||
_t2v_payload["ratio"] = ratio or "9:16"
|
||||
task_id, last_err, sc3, body3 = _do_create(_t2v_payload)
|
||||
if task_id:
|
||||
sc, body = sc3, body3
|
||||
logger.info("[trust-chain] portrait_intercept 降级纯 t2v 成功 task_id=%s", task_id)
|
||||
else:
|
||||
sc, body = sc3, body3
|
||||
|
||||
if not task_id:
|
||||
err_code, user_msg = _classify_video_error(sc, body, last_err)
|
||||
self.last_video_error = {
|
||||
|
||||
@@ -57,7 +57,7 @@ if [ "$TARGET_ENV" = "staging" ]; then
|
||||
fi
|
||||
|
||||
# 共用 secrets 直接导出(如果存在)
|
||||
SHARED_SECRETS="OSS_ACCESS_KEY_ID OSS_ACCESS_KEY_SECRET COSYVOICE_API_KEY DASHSCOPE_API_KEY MEDIAKIT_API_KEY DOUBAO_API_KEY DOUBAO_MODEL DOUBAO_FAST_MODEL DOUBAO_BASE_URL DOUBAO_VISION_MODEL DOUBAO_VISION_LITE_MODEL DOUBAO_VISION_USE_LITE WECHAT_APP_ID WECHAT_APP_SECRET TIKHUB_API_KEY APIZERO_API_KEY GPU_WORKER_TOKEN"
|
||||
SHARED_SECRETS="OSS_ACCESS_KEY_ID OSS_ACCESS_KEY_SECRET COSYVOICE_API_KEY DASHSCOPE_API_KEY MEDIAKIT_API_KEY DOUBAO_API_KEY DOUBAO_MODEL DOUBAO_FAST_MODEL DOUBAO_BASE_URL DOUBAO_VISION_MODEL DOUBAO_VISION_LITE_MODEL DOUBAO_VISION_USE_LITE DOUBAO_IMAGE_MODEL DOUBAO_IMAGE_SIZE DOUBAO_IMAGE_TIMEOUT DOUBAO_FAST_MODEL DOUBAO_TIMEOUT DOUBAO_MAX_RETRIES WECHAT_APP_ID WECHAT_APP_SECRET TIKHUB_API_KEY APIZERO_API_KEY GPU_WORKER_TOKEN"
|
||||
for var in $SHARED_SECRETS; do
|
||||
value="${!var:-}"
|
||||
# 已经在环境中了,无需额外操作
|
||||
|
||||
@@ -17,7 +17,7 @@ def mock_settings():
|
||||
doubao_api_key="test-api-key",
|
||||
doubao_model="doubao-pro-32k",
|
||||
doubao_base_url="https://ark.example.com/api/v3",
|
||||
doubao_timeout=30,
|
||||
doubao_timeout=45,
|
||||
doubao_max_retries=2,
|
||||
)
|
||||
yield mock
|
||||
@@ -37,7 +37,7 @@ def client_without_key():
|
||||
doubao_api_key="",
|
||||
doubao_model="doubao-pro-32k",
|
||||
doubao_base_url="https://ark.example.com/api/v3",
|
||||
doubao_timeout=30,
|
||||
doubao_timeout=45,
|
||||
doubao_max_retries=2,
|
||||
)
|
||||
yield DoubaoClient()
|
||||
@@ -52,7 +52,7 @@ class TestDoubaoClientInit:
|
||||
assert client.api_key == "test-api-key"
|
||||
assert client.model == "doubao-pro-32k"
|
||||
assert client.base_url == "https://ark.example.com/api/v3"
|
||||
assert client.timeout == 30
|
||||
assert client.timeout == 45
|
||||
assert client.max_retries == 2
|
||||
|
||||
def test_base_url_strips_trailing_slash(self, mock_settings):
|
||||
|
||||
@@ -80,8 +80,8 @@ class TestSharedSettingsDefaults:
|
||||
def test_default_doubao_settings(self):
|
||||
s = SharedSettings()
|
||||
assert "doubao" in s.doubao_model
|
||||
assert s.doubao_timeout == 30
|
||||
assert s.doubao_max_retries == 2
|
||||
assert s.doubao_timeout == 45 # #2180 默认提到45s
|
||||
assert s.doubao_max_retries == 1
|
||||
|
||||
|
||||
class TestAPISettingsDefaults:
|
||||
|
||||
@@ -110,8 +110,8 @@ class TestSharedSettingsDefaults:
|
||||
def test_default_doubao_config(self):
|
||||
"""豆包默认配置"""
|
||||
s = self._make_settings()
|
||||
assert s.doubao_timeout == 30
|
||||
assert s.doubao_max_retries == 2
|
||||
assert s.doubao_timeout == 45 # #2180 默认提到45s
|
||||
assert s.doubao_max_retries == 1
|
||||
assert "volces.com" in s.doubao_base_url
|
||||
|
||||
def test_default_empty_api_keys(self):
|
||||
|
||||
@@ -403,33 +403,30 @@ class TestViralVideoPipeline:
|
||||
"""v1.6: _step_script_generation 返回 dict 形式的 CopyResult,含 voiceover_script + shots。"""
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_script_generation
|
||||
|
||||
mock_llm.return_value = {
|
||||
"overview": {"theme": "口红推荐", "total_duration": 15, "aspect_ratio": "9:16"},
|
||||
"scene_and_lighting": "明亮化妆台,柔和自然光",
|
||||
"shots": [
|
||||
{
|
||||
"time_range": "0-5秒",
|
||||
"shot_type_angle_movement": "近景平视,缓慢推镜",
|
||||
"scene_and_dialogue": "女主微笑展示口红:大家好,今天分享一款口红",
|
||||
"action_details": "手持口红特写",
|
||||
"audio_bgm": "轻快流行BGM",
|
||||
"transition": "硬切",
|
||||
"reference_image_index": 0,
|
||||
},
|
||||
{
|
||||
"time_range": "5-15秒",
|
||||
"shot_type_angle_movement": "特写,固定镜头",
|
||||
"scene_and_dialogue": "涂抹口红:颜色特别好看很显白",
|
||||
"action_details": "嘴唇涂抹特写",
|
||||
"audio_bgm": "轻快BGM继续",
|
||||
"transition": "结束",
|
||||
"reference_image_index": 1,
|
||||
},
|
||||
],
|
||||
"hard_constraints": ["无字幕无水印"],
|
||||
"negative_prompts": ["字幕", "水印"],
|
||||
"voiceover_script": "大家好,今天分享一款口红,颜色特别好看很显白。",
|
||||
}
|
||||
mock_llm.return_value = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="5" bgm_note="轻快流行BGM">
|
||||
<voice_text>大家好,今天分享一款口红</voice_text>
|
||||
<subtitle_text>大家好,今天分享一款口红</subtitle_text>
|
||||
<shot_type_angle_movement>近景平视,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>女主微笑展示口红:大家好,今天分享一款口红</scene_and_dialogue>
|
||||
<action_details>手持口红特写</action_details>
|
||||
<audio_bgm>轻快流行BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="fade" zoom="null" duration_sec="10" bgm_note="轻快BGM">
|
||||
<voice_text>颜色特别好看很显白</voice_text>
|
||||
<subtitle_text>颜色特别好看很显白</subtitle_text>
|
||||
<shot_type_angle_movement>特写,固定镜头</shot_type_angle_movement>
|
||||
<scene_and_dialogue>涂抹口红:颜色特别好看很显白</scene_and_dialogue>
|
||||
<action_details>嘴唇涂抹特写</action_details>
|
||||
<audio_bgm>轻快BGM继续</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>1</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
</clips>"""
|
||||
result = _step_script_generation(
|
||||
mock_job, {"intent": "推广口红", "key_messages": [], "tone": "亲切"}, {"products": []}
|
||||
)
|
||||
@@ -452,12 +449,13 @@ class TestViralVideoPipeline:
|
||||
assert result["voiceover_script"]
|
||||
assert len(result["shots"]) >= 1
|
||||
|
||||
@patch("packages.shared.ai_service.call_llm")
|
||||
def test_review_pass_v16(self, mock_llm, mock_job):
|
||||
@patch("packages.application.viral_video.reviewer.Reviewer.review")
|
||||
def test_review_pass_v16(self, mock_review, mock_job):
|
||||
"""v1.6 _step_review 接收 copy_result dict。"""
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_review
|
||||
from packages.application.viral_video.reviewer import ReviewIssue, ReviewResult
|
||||
|
||||
mock_llm.return_value = {"passed": True, "score": 90, "details": {}}
|
||||
mock_review.return_value = ReviewResult(passed=True, score=90, issues=[], rewrite_suggestions=[])
|
||||
cr = {"voiceover_script": "大家好", "shots": []}
|
||||
result = _step_review(mock_job, cr)
|
||||
assert result["passed"] is True
|
||||
|
||||
@@ -0,0 +1,302 @@
|
||||
"""#2040 接线集成测试:验证运行中的 viral_video 任务使用 prompt_loader 从 DB 读取模板。
|
||||
|
||||
mock LLM/Vision 调用,验证:
|
||||
1. image_analysis 走 loader 模板 + XML 解析
|
||||
2. intent_parsing 走 loader 模板 + XML 解析
|
||||
3. script_generation 走 storyboard 模板 + XML 解析,输出兼容 Seedance 的 copy_result
|
||||
4. review 走 Reviewer(review 模板)带自动重写
|
||||
5. 三档融合(ai_full / ai_polish / user_primary)注入不同 FUSION_INSTRUCTIONS
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path as _Path
|
||||
|
||||
_WORKER_ROOT = _Path(__file__).resolve().parents[2] / "apps" / "worker"
|
||||
if str(_WORKER_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(_WORKER_ROOT))
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from packages.domain.viral_video import ViralVideoJob
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def job():
|
||||
j = ViralVideoJob(
|
||||
user_id="u1",
|
||||
images=["https://img/1.jpg", "https://img/2.jpg"],
|
||||
industry="美妆",
|
||||
duration=15,
|
||||
user_copy_text="这款口红真的太绝了,显白又持久,姐妹们冲!",
|
||||
fusion_level="ai_polish",
|
||||
)
|
||||
return j
|
||||
|
||||
|
||||
# ── Mock LLM/Vision 返回的 XML 文本 ─────────────────────────────────
|
||||
|
||||
IMAGE_XML = """
|
||||
<analysis>
|
||||
<scene>室内桌面拍摄,柔和自然光</scene>
|
||||
<mood>清新温暖</mood>
|
||||
<product name="lipstick" brand="品牌X" category="唇部彩妆"
|
||||
appearance="管状红色膏体" packaging="黑色金属管"
|
||||
features="显白,持久,滋润" portrait_prompt="无人像"
|
||||
summary="品牌X红色口红">
|
||||
<text_on_package>品牌X,211</text_on_package>
|
||||
</product>
|
||||
</analysis>
|
||||
""".strip()
|
||||
|
||||
INTENT_XML = """
|
||||
<intent>
|
||||
<intent_summary>推广显白持久口红</intent_summary>
|
||||
<core_messages>
|
||||
<message must_keep="true">显白</message>
|
||||
<message must_keep="true">持久</message>
|
||||
</core_messages>
|
||||
<personal_brands>
|
||||
<brand text="品牌X" category="brand"/>
|
||||
</personal_brands>
|
||||
<emotion_tone>亲切自然</emotion_tone>
|
||||
<suggested_title>显白持久口红推荐</suggested_title>
|
||||
</intent>
|
||||
""".strip()
|
||||
|
||||
STORYBOARD_XML = """
|
||||
<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="5" bgm_note="轻快BGM">
|
||||
<voice_text>这款口红真的太绝了</voice_text>
|
||||
<subtitle_text>显白又持久</subtitle_text>
|
||||
<shot_type_angle_movement>近景俯拍45度,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>厨房台面,主妇展示口红。对白:这款口红真的太绝了</scene_and_dialogue>
|
||||
<action_details>右手持口红展示膏体</action_details>
|
||||
<audio_bgm>轻快BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
</clips>
|
||||
""".strip()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def invalidate_loader_cache():
|
||||
from packages.application.viral_video import prompt_loader as pl
|
||||
|
||||
pl.invalidate()
|
||||
yield
|
||||
pl.invalidate()
|
||||
|
||||
|
||||
# ── 1) 图片分析走模板 ───────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestImageAnalysisWiring:
|
||||
def test_uses_loader_template_and_xml_parse(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
with patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML) as mock_v:
|
||||
result = vv._analyze_single_image(0, "https://img/1.jpg", "vlm-lite", 15)
|
||||
|
||||
mock_v.assert_called_once()
|
||||
# 验证调用时传入了 system_prompt(说明走了 loader 渲染的模板)
|
||||
call_kwargs = mock_v.call_args.kwargs
|
||||
assert "system_prompt" in call_kwargs and call_kwargs["system_prompt"]
|
||||
# 结果包含从 XML 解析出的产品信息
|
||||
assert result["name"] == "lipstick"
|
||||
assert result["brand"] == "品牌X"
|
||||
assert "显白" in result["key_features"]
|
||||
assert result["text_on_package"] == ["品牌X", "211"]
|
||||
|
||||
|
||||
# ── 2) 意图解析走模板 ───────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestIntentParsingWiring:
|
||||
def test_uses_loader_and_parses_xml(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
img_result = {"products": [{"name": "lipstick", "brand": "品牌X", "key_features": ["显白", "持久"]}]}
|
||||
with patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML) as mock_llm:
|
||||
result = vv._step_intent_parsing(job, img_result)
|
||||
|
||||
mock_llm.assert_called_once()
|
||||
assert result["intent"] == "推广显白持久口红"
|
||||
assert "显白" in result["key_messages"]
|
||||
assert result["suggested_title"] == "显白持久口红推荐"
|
||||
|
||||
|
||||
# ── 3) 脚本生成:storyboard 模板 + XML 解析 + fusion_level 注入 ────
|
||||
|
||||
|
||||
class TestScriptGenerationWiring:
|
||||
@pytest.mark.parametrize("level", ["ai_full", "ai_polish", "user_primary"])
|
||||
def test_fusion_level_injected(self, job, level):
|
||||
"""三档融合水平被注入到 storyboard 模板的 system_prompt"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.prompts import FUSION_INSTRUCTIONS
|
||||
|
||||
job.fusion_level = level
|
||||
intent = {"intent": "推广", "key_messages": ["显白"], "tone": "亲切"}
|
||||
|
||||
captured_system = {}
|
||||
|
||||
def fake_call_llm(messages, **kw):
|
||||
captured_system["final"] = messages[0]["content"]
|
||||
return STORYBOARD_XML
|
||||
|
||||
with patch("packages.shared.ai_service.call_llm", side_effect=fake_call_llm):
|
||||
result = vv._step_script_generation(job, intent, {})
|
||||
|
||||
# fusion_level 对应的指令文本被注入到 system prompt 中
|
||||
assert FUSION_INSTRUCTIONS[level] in captured_system["final"], f"fusion_level {level} 指令未注入 system_prompt"
|
||||
# 输出保持 Seedance 兼容结构
|
||||
assert "overview" in result
|
||||
assert "shots" in result
|
||||
assert len(result["shots"]) >= 1
|
||||
assert result["shots"][0]["shot_type_angle_movement"]
|
||||
assert result["voiceover_script"]
|
||||
|
||||
def test_fallback_when_xml_and_json_unparseable(self, job):
|
||||
"""XML 解析失败且无法解析为 JSON 时,回退到兜底脚本"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
job.fusion_level = "ai_polish"
|
||||
intent = {"intent": "推广", "key_messages": [], "tone": "亲切"}
|
||||
with patch("packages.shared.ai_service.call_llm", return_value="not xml not json"):
|
||||
result = vv._step_script_generation(job, intent, {})
|
||||
assert isinstance(result, dict)
|
||||
assert "voiceover_script" in result
|
||||
assert "shots" in result
|
||||
|
||||
|
||||
# ── 4) Review 使用 Reviewer + 自动重写 ─────────────────────────────
|
||||
|
||||
|
||||
class TestReviewWiring:
|
||||
def test_pass_path(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
|
||||
copy_result = {
|
||||
"title": "口红推荐",
|
||||
"overview": {"theme": "口红推荐"},
|
||||
"voiceover_script": "这款口红显白又持久",
|
||||
"shots": [{"scene_and_dialogue": "展示口红"}],
|
||||
}
|
||||
job.intent_result = {"key_messages": ["显白", "持久"], "intent": "推广"}
|
||||
|
||||
pass_result = ReviewResult(passed=True, score=90, issues=[], rewrite_suggestions=[])
|
||||
with patch.object(Reviewer, "review", return_value=pass_result):
|
||||
out = vv._step_review(job, copy_result)
|
||||
assert out["passed"] is True
|
||||
|
||||
def test_rewrite_path(self, job):
|
||||
"""审核不通过时触发自动重写,并更新 job.copy_result"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
from packages.application.viral_video.schemas import FusionResult, ReviewIssue, ScriptSegment
|
||||
|
||||
copy_result = {
|
||||
"title": "原标题",
|
||||
"overview": {"theme": "原标题"},
|
||||
"voiceover_script": "这款口红绝了",
|
||||
"shots": [{"scene_and_dialogue": "展示"}],
|
||||
}
|
||||
job.intent_result = {"key_messages": ["显白"], "intent": "推广"}
|
||||
|
||||
fail_result = ReviewResult(
|
||||
passed=False,
|
||||
score=50,
|
||||
issues=[ReviewIssue(dimension="违规词", severity="high", location="开头", text="绝了")],
|
||||
rewrite_suggestions=["去掉夸大词"],
|
||||
)
|
||||
rewritten = FusionResult(
|
||||
title="新标题",
|
||||
hook="修改后钩子",
|
||||
script_segments=[ScriptSegment(text="修改后口播正文")],
|
||||
cta="行动号召",
|
||||
word_count=10,
|
||||
estimated_duration=10,
|
||||
)
|
||||
pass_after = ReviewResult(passed=True, score=88, issues=[], rewrite_suggestions=[])
|
||||
|
||||
with (
|
||||
patch.object(Reviewer, "review", side_effect=[fail_result, pass_after]),
|
||||
patch.object(Reviewer, "rewrite", return_value=rewritten),
|
||||
):
|
||||
out = vv._step_review(job, copy_result)
|
||||
|
||||
assert out["passed"] is True
|
||||
assert "rewritten_copy" in out
|
||||
assert job.generated_copy_text == "修改后口播正文"
|
||||
|
||||
|
||||
# ── 5) 端到端:每个 step 调用 loader 对应 prompt_type ──────────────
|
||||
|
||||
|
||||
class TestEndToEndLoaderUsed:
|
||||
def test_each_step_calls_loader(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video import prompt_loader as pl
|
||||
|
||||
called_types = []
|
||||
real_get = pl.get_template
|
||||
|
||||
def spy_get(prompt_type, **kwargs):
|
||||
called_types.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get),
|
||||
patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML),
|
||||
patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML),
|
||||
):
|
||||
# 1) image
|
||||
img_res = vv._analyze_single_image(0, "https://img/1.jpg", "vlm", 15)
|
||||
# 2) intent
|
||||
intent_res = vv._step_intent_parsing(job, {"products": [img_res]})
|
||||
|
||||
# 前两步分别调用了 image_analysis 和 intent_parsing
|
||||
assert "image_analysis" in called_types
|
||||
assert "intent_parsing" in called_types
|
||||
|
||||
# script 和 review 单独验证(需要不同的 LLM 返回)
|
||||
called_types_2 = []
|
||||
|
||||
def spy_get_2(prompt_type, **kwargs):
|
||||
called_types_2.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get_2),
|
||||
patch("packages.shared.ai_service.call_llm", return_value=STORYBOARD_XML),
|
||||
):
|
||||
copy_res = vv._step_script_generation(job, intent_res, {"products": [img_res]})
|
||||
assert "storyboard" in called_types_2
|
||||
|
||||
called_types_3 = []
|
||||
|
||||
def spy_get_3(prompt_type, **kwargs):
|
||||
called_types_3.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
|
||||
pass_result = ReviewResult(passed=True, score=90, issues=[], rewrite_suggestions=[])
|
||||
job.intent_result = intent_res
|
||||
job.copy_result = copy_res
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get_3),
|
||||
patch.object(Reviewer, "review", return_value=pass_result) as mock_review,
|
||||
):
|
||||
review_res = vv._step_review(job, copy_res)
|
||||
# review 步骤内部直接调用 Reviewer.review,该方法被 mock,因此 get_template 不会被调用;
|
||||
# 此处验证 Reviewer.review 被调用即可说明 review 步骤走通了。
|
||||
assert mock_review.called, "_step_review 未调用 Reviewer.review"
|
||||
assert isinstance(review_res, dict) and "passed" in review_res
|
||||
Reference in New Issue
Block a user