Compare commits
14 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 22b9019325 | |||
| 652d2cd270 | |||
| 019bbe6897 | |||
| f25fa9d978 | |||
| 2080ecc4f9 | |||
| bc45a10fff | |||
| e22a62802d | |||
| 70d4a21055 | |||
| d6e4a09628 | |||
| 213e89f93a | |||
| 79696e18f0 | |||
| 8ff9ab0cb4 | |||
| f10b5393fa | |||
| ba0e3fc67f |
@@ -28,10 +28,17 @@ depends_on = None
|
||||
|
||||
|
||||
def _tpl(prompt_type: str, version: int) -> dict:
|
||||
"""取模板:先按指定版本找,找不到则取该类型最新版本(兼容 v3→v4 升级)。"""
|
||||
# 先按指定版本找
|
||||
for t in DEFAULT_TEMPLATES:
|
||||
if t["prompt_type"] == prompt_type and t["version"] == version:
|
||||
return t
|
||||
raise RuntimeError("default template missing: %s v%s" % (prompt_type, version))
|
||||
# 找不到则取最新版本
|
||||
candidates = [t for t in DEFAULT_TEMPLATES if t["prompt_type"] == prompt_type]
|
||||
if candidates:
|
||||
latest = max(candidates, key=lambda x: x["version"])
|
||||
return latest
|
||||
raise RuntimeError("default template missing: %s" % prompt_type)
|
||||
|
||||
|
||||
def _upsert(bind, t: dict) -> None:
|
||||
|
||||
@@ -0,0 +1,235 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""storyboard v4 - 多角色对话 + 废除旁白 + visual 5要素
|
||||
|
||||
Revision ID: 108_storyboard_v4_multivoice
|
||||
Revises: 107
|
||||
Create Date: 2026-10-10
|
||||
|
||||
变更:
|
||||
1. storyboard v4: 废除旁白思维,所有voiceover必须是角色台词
|
||||
- 增加<speaker>标签,每镜必须标注说话人
|
||||
- visual强制5要素结构(景别/运镜/动作/环境/光线),每镜不少于30字
|
||||
- voiceover_script用[speaker:xxx]标记格式
|
||||
2. 旧版storyboard模板is_active设为false
|
||||
3. 检查image_analysis和review是否有active模板,没有则插入保底版本
|
||||
"""
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "108_storyboard_v4_multivoice"
|
||||
down_revision = "107"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
# ── storyboard v4 system_prompt ──────────────────────────────────────
|
||||
V4_STORYBOARD_SYSTEM = """你是一名懂短视频的编导和口播文案高手。你会拿到图片的真实观察、营销目的和用户参数,请一次性完成对营销意图的理解,并产出可直接拍摄/生成的分镜脚本。不要单独输出"意图解析",意图要直接体现在台词和分镜里。
|
||||
|
||||
## 核心设计原则(必须严格遵守)
|
||||
1. **废除旁白思维**:所有视频类型——无论对话短剧/口播带货/获客引流/品牌故事——voiceover 必须是人物说的话(第一人称或角色对白),绝对不能出现第三人称旁白解说。观众看的是人在演、在说。
|
||||
2. **严禁第三人称解说性台词**:如"接下来展示...""这款产品..."这类上帝视角描述禁止出现在 voiceover 中。
|
||||
3. **短剧类营销目的**(对话短剧/反转短剧/悬念短剧/情绪短片):双角色对话格式"甲:xxx 乙:xxx",镜头在角色间切换。
|
||||
4. **口播类**(口播带货/促销转化/功能演示/痛点解决/获客引流/账号涨粉/活动通知/场景种草):第一人称对镜头说话,像真人出镜。
|
||||
|
||||
## 输出格式(XML,严格按结构输出,不要输出额外解释)
|
||||
<script>
|
||||
<copy_display_markdown><![CDATA[直接展示给用户看的成片文案,用 Markdown 写成流畅叙述]]></copy_display_markdown>
|
||||
<clips>
|
||||
<clip index="1">
|
||||
<time_range>0-3秒</time_range>
|
||||
<speaker>说话人标识(如"店主""顾客""主播")</speaker>
|
||||
<voiceover>这一镜的角色台词(人物说的话,不是旁白)</voiceover>
|
||||
<visual>【景别】【镜头运动】【人物动作/表情】【环境/道具】【光线氛围】5要素结构,不少于30字</visual>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
</clip>
|
||||
</clips>
|
||||
<voiceover_script>把所有 clip 的 voiceover 连成完整台词稿,用 [speaker:xxx] 标记每个说话段落</voiceover_script>
|
||||
<theme>一句话主题</theme>
|
||||
<negative>【反套路化要求】
|
||||
禁止使用"家人们谁懂啊""绝绝子""宝子们""家人们""太绝了""yyds"等烂大街网络词;
|
||||
禁止固定模板化开头;语言要像真人朋友之间的分享,自然、具体、有信息量。</negative>
|
||||
</script>
|
||||
|
||||
## 写作要求
|
||||
1. **台词(voiceover)**:像真人面对镜头说话或角色对白,短句、口语化、有停顿有情绪,开头 3 秒给出钩子;不要书面腔,不要机械报参数。严禁第三人称解说。
|
||||
2. **说话人(speaker)**:每个 clip 必须标注说话人标识,如"店主""顾客""主播""我"等。短剧类必须有至少 2 个不同角色。
|
||||
3. **画面描述(visual)**:强制 5 要素结构——【景别】【镜头运动】【人物动作/表情】【环境/道具】【光线氛围】,每镜 visual 不少于 30 字,要具体到"闭眼听台词能想象出画面"。
|
||||
4. **voiceover_script 格式**:用 [speaker:xxx] 标记每个说话段落,如"[speaker:店主]你是不是也觉得...[speaker:顾客]是啊,怎么回事?"
|
||||
5. copy_display_markdown:直接展示给最终用户的文案,用 Markdown 写成自然、流畅、有感染力的成片文案。
|
||||
6. 内容必须来自图片观察与用户给出的信息,不编造卖点、不夸大、不使用绝对化用语和虚假承诺。
|
||||
7. reference_image_index 填本镜参考图片序号(从 0 开始),没有合适参考图填 -1。
|
||||
8. 分镜数量与时长匹配总时长,节奏紧凑。
|
||||
9. **口播字数硬约束**(必须严格遵守):按每秒约 2.5~3 个中文字(正常口播语速)计算:
|
||||
- 5秒视频:voiceover_script 总字数 12~15 字
|
||||
- 10秒视频:voiceover_script 总字数 25~30 字
|
||||
- 15秒视频:voiceover_script 总字数 35~45 字
|
||||
- 20秒视频:voiceover_script 总字数 50~60 字
|
||||
- 30秒视频:voiceover_script 总字数 75~90 字
|
||||
- 宁可少写也不要多写,超长会导致 TTS 音频超出视频时长限制
|
||||
10. **镜头数量硬约束**:5秒1~2镜、10秒3镜、15秒3~4镜、20秒4~5镜、30秒6~8镜
|
||||
11. **时间轴硬约束**:第一个clip从0秒开始,最后一个clip结束于total_duration秒,相邻clip首尾相接
|
||||
12. 必须严格按<marketing_purpose><target_audience><persona><viral_structure><language><industry>指定的参数写文案和分镜
|
||||
13. 镜头间动作衔接要自然,画面描述要具体到能直接拍摄/生成"""
|
||||
|
||||
V4_STORYBOARD_USER = """<marketing_purpose>{marketing_purpose}</marketing_purpose>
|
||||
<industry>{industry}</industry>
|
||||
<image_analysis>
|
||||
{image_summary}
|
||||
</image_analysis>
|
||||
<user_parameters>
|
||||
<theme_hint>{theme_hint}</theme_hint>
|
||||
<duration>{duration}秒</duration>
|
||||
<aspect_ratio>{aspect_ratio}</aspect_ratio>
|
||||
<tone>{tone}</tone>
|
||||
<target_audience>{target_audience}</target_audience>
|
||||
<persona>{persona_hint}</persona>
|
||||
<viral_structure>{viral_structure_hint}</viral_structure>
|
||||
<language>{language_hint}</language>
|
||||
<extra_requirements>{extra_requirements}</extra_requirements>
|
||||
</user_parameters>
|
||||
{video_style_section}
|
||||
请严格按 XML 结构输出分镜脚本。"""
|
||||
|
||||
V4_STORYBOARD_EXAMPLE = """<script>
|
||||
<copy_display_markdown><![CDATA[# 在御众堂,把松弛的自己一点点找回来
|
||||
产后妈妈最懂那种力不从心,推开门,暖光和一杯热茶先接住了你……]]></copy_display_markdown>
|
||||
<clips>
|
||||
<clip index="1">
|
||||
<time_range>0-3秒</time_range>
|
||||
<speaker>店主</speaker>
|
||||
<voiceover>生完娃,是不是连照镜子的勇气都没了?</voiceover>
|
||||
<visual>【中近景】【缓推】【妈妈疲惫看向镜子】【暖光店内环境】【柔和暖光】</visual>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
</clip>
|
||||
<clip index="2">
|
||||
<time_range>3-6秒</time_range>
|
||||
<speaker>顾客</speaker>
|
||||
<voiceover>是啊,怎么回事?</voiceover>
|
||||
<visual>【近景】【固定】【顾客表情惊讶】【店内休息区】【暖色调】</visual>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
</clip>
|
||||
</clips>
|
||||
<voiceover_script>[speaker:店主]生完娃,是不是连照镜子的勇气都没了?[speaker:顾客]是啊,怎么回事?</voiceover_script>
|
||||
<theme>产后妈妈走进御众堂重拾状态</theme>
|
||||
<negative>模糊、畸变、夸大疗效、绝对化用语</negative>
|
||||
</script>"""
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
conn = op.get_bind()
|
||||
|
||||
# 1. 查询 storyboard 当前最大 version
|
||||
result = conn.execute(
|
||||
text("SELECT MAX(version) FROM viral_video_prompt_templates WHERE prompt_type = :pt"),
|
||||
{"pt": "storyboard"},
|
||||
)
|
||||
max_version = result.scalar() or 0
|
||||
new_version = max_version + 1
|
||||
|
||||
# 2. 旧版 storyboard 模板 is_active 设为 false
|
||||
conn.execute(
|
||||
text("UPDATE viral_video_prompt_templates SET is_active = false WHERE prompt_type = :pt"),
|
||||
{"pt": "storyboard"},
|
||||
)
|
||||
|
||||
# 3. 防御性插入新版 storyboard 模板
|
||||
existing = conn.execute(
|
||||
text("SELECT id FROM viral_video_prompt_templates " "WHERE prompt_type = :pt AND version = :ver"),
|
||||
{"pt": "storyboard", "ver": new_version},
|
||||
).fetchone()
|
||||
|
||||
if not existing:
|
||||
conn.execute(
|
||||
text(
|
||||
"INSERT INTO viral_video_prompt_templates "
|
||||
"(name, prompt_type, version, system_prompt, user_prompt_template, example_output, is_active) "
|
||||
"VALUES (:name, :pt, :ver, :sys, :usr, :ex, :active)"
|
||||
),
|
||||
{
|
||||
"name": "编导分镜v4-多角色对话版",
|
||||
"pt": "storyboard",
|
||||
"ver": new_version,
|
||||
"sys": V4_STORYBOARD_SYSTEM,
|
||||
"usr": V4_STORYBOARD_USER,
|
||||
"ex": V4_STORYBOARD_EXAMPLE,
|
||||
"active": True,
|
||||
},
|
||||
)
|
||||
|
||||
# 4. 检查 image_analysis 是否有 active 模板,没有则插入保底
|
||||
ia_active = conn.execute(
|
||||
text("SELECT COUNT(*) FROM viral_video_prompt_templates WHERE prompt_type = :pt AND is_active = true"),
|
||||
{"pt": "image_analysis"},
|
||||
).scalar()
|
||||
|
||||
if ia_active == 0:
|
||||
# 插入保底 image_analysis 模板(简化版)
|
||||
ia_max = (
|
||||
conn.execute(
|
||||
text("SELECT MAX(version) FROM viral_video_prompt_templates WHERE prompt_type = :pt"),
|
||||
{"pt": "image_analysis"},
|
||||
).scalar()
|
||||
or 0
|
||||
)
|
||||
conn.execute(
|
||||
text(
|
||||
"INSERT INTO viral_video_prompt_templates "
|
||||
"(name, prompt_type, version, system_prompt, user_prompt_template, example_output, is_active) "
|
||||
"VALUES (:name, :pt, :ver, :sys, :usr, :ex, :active)"
|
||||
),
|
||||
{
|
||||
"name": "图片分析保底版",
|
||||
"pt": "image_analysis",
|
||||
"ver": ia_max + 1,
|
||||
"sys": "你是一名擅长观察和写作的品牌内容编导。分析图片并输出JSON。",
|
||||
"usr": "请分析这张图片。图片地址:{image_url}",
|
||||
"ex": '{"images": [{"type": "store", "name": "门店", "summary_markdown": "描述"}]}',
|
||||
"active": True,
|
||||
},
|
||||
)
|
||||
|
||||
# 5. 检查 review 是否有 active 模板
|
||||
review_active = conn.execute(
|
||||
text("SELECT COUNT(*) FROM viral_video_prompt_templates WHERE prompt_type = :pt AND is_active = true"),
|
||||
{"pt": "review"},
|
||||
).scalar()
|
||||
|
||||
if review_active == 0:
|
||||
rv_max = (
|
||||
conn.execute(
|
||||
text("SELECT MAX(version) FROM viral_video_prompt_templates WHERE prompt_type = :pt"),
|
||||
{"pt": "review"},
|
||||
).scalar()
|
||||
or 0
|
||||
)
|
||||
conn.execute(
|
||||
text(
|
||||
"INSERT INTO viral_video_prompt_templates "
|
||||
"(name, prompt_type, version, system_prompt, user_prompt_template, example_output, is_active) "
|
||||
"VALUES (:name, :pt, :ver, :sys, :usr, :ex, :active)"
|
||||
),
|
||||
{
|
||||
"name": "文案审核保底版",
|
||||
"pt": "review",
|
||||
"ver": rv_max + 1,
|
||||
"sys": "你是短视频广告合规审核专家。审核文案输出XML。",
|
||||
"usr": "请审核:{fusion_text}",
|
||||
"ex": "<review><passed>true</passed></review>",
|
||||
"active": True,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# 恢复旧版 storyboard 为 active
|
||||
conn = op.get_bind()
|
||||
conn.execute(
|
||||
text("UPDATE viral_video_prompt_templates SET is_active = true WHERE prompt_type = :pt AND version < :ver"),
|
||||
{"pt": "storyboard", "ver": 4},
|
||||
)
|
||||
# 删除新版
|
||||
conn.execute(
|
||||
text("DELETE FROM viral_video_prompt_templates WHERE prompt_type = :pt AND version >= :ver"),
|
||||
{"pt": "storyboard", "ver": 4},
|
||||
)
|
||||
@@ -101,6 +101,8 @@ export interface ImageAnalysisResult {
|
||||
export interface ShotScript {
|
||||
/** 时间区间,如 "0-3秒" */
|
||||
time_range?: string
|
||||
/** 说话人角色,如 "店主"、"顾客" */
|
||||
speaker?: string
|
||||
/** 景别/角度/运镜,如 "近景俯拍45度,缓慢推镜" */
|
||||
shot_type_angle_movement?: string
|
||||
/** 场景描述+对白 */
|
||||
|
||||
@@ -298,9 +298,33 @@ const UPSCALE_OPTIONS = [
|
||||
const VOICE_TIPS =
|
||||
"支持 MP3/WAV/M4A/AAC/OGG 格式,最大 10MB,时长 ≤30 秒。建议清晰人声、无背景音乐、环境安静;录音请保持距麦克风 15-20cm,音量适中。"
|
||||
|
||||
/** 解析 voiceover_script 中的 [speaker:xxx] 标记,渲染为带颜色的 React 片段 */
|
||||
function renderVoiceoverWithSpeakers(text: string) {
|
||||
if (!text) return null
|
||||
const parts = text.split(/\[speaker:([^\]]+)\]/)
|
||||
if (parts.length === 1) {
|
||||
// 没有 speaker 标记,直接显示
|
||||
return <span>{text}</span>
|
||||
}
|
||||
const elements: React.ReactNode[] = []
|
||||
for (let i = 1; i < parts.length; i += 2) {
|
||||
const speaker = parts[i]
|
||||
const segText = parts[i + 1] || ""
|
||||
const isMain = ["店主", "主播", "老板", "我", "主讲人"].includes(speaker)
|
||||
elements.push(
|
||||
<span key={i}>
|
||||
<strong style={{ color: isMain ? "#1890ff" : "#fa8c16" }}>{speaker}</strong>
|
||||
<span>{segText}</span>
|
||||
</span>,
|
||||
)
|
||||
}
|
||||
return <>{elements}</>
|
||||
}
|
||||
|
||||
// ── 分镜脚本数据模型(新后端 copy_result 结构,前端先 mock 展示) ──
|
||||
interface StoryboardShot {
|
||||
time_range: string
|
||||
speaker?: string
|
||||
shot_type_angle_movement: string
|
||||
scene_and_dialogue: string
|
||||
action_details: string
|
||||
@@ -333,6 +357,7 @@ function copyResultToStoryboard(cr: CopyResult | null | undefined): Storyboard |
|
||||
scene_and_lighting: cr.scene_and_lighting || "",
|
||||
shots: ((cr.shots ?? []) as ShotScript[]).map((s) => ({
|
||||
time_range: s.time_range || "",
|
||||
speaker: s.speaker || "",
|
||||
shot_type_angle_movement: s.shot_type_angle_movement || "",
|
||||
scene_and_dialogue: s.scene_and_dialogue || "",
|
||||
action_details: s.action_details || "",
|
||||
@@ -1924,7 +1949,7 @@ const ViralVideoPage: React.FC = () => {
|
||||
if (!locked) setEditingField("voiceover_script")
|
||||
}}
|
||||
>
|
||||
{sb.voiceover_script}
|
||||
{renderVoiceoverWithSpeakers(sb.voiceover_script)}
|
||||
</span>
|
||||
)}
|
||||
</p>
|
||||
|
||||
@@ -764,8 +764,9 @@ def _script_from_xml(raw: str, job: ViralVideoJob) -> dict | None:
|
||||
ref_idx = xp.attr_int(ref_raw, -1) if ref_raw not in (None, "") else -1
|
||||
if not isinstance(ref_idx, int) or ref_idx < 0:
|
||||
ref_idx = None
|
||||
speaker = xp.text_of(body, "speaker") or "主播"
|
||||
if voice:
|
||||
voice_parts.append(voice)
|
||||
voice_parts.append(f"[speaker:{speaker}]{voice}")
|
||||
shot = {
|
||||
"time_range": a.get("time_range") or xp.text_of(body, "time_range") or f"{i * 3}-{(i + 1) * 3}秒",
|
||||
"shot_type_angle_movement": visual or "中景平视,固定镜头",
|
||||
|
||||
@@ -10,7 +10,22 @@ source "${SCRIPT_DIR}/ci_env.sh"
|
||||
|
||||
|
||||
echo "=== Installing mypy ==="
|
||||
python3 -m pip install -q mypy
|
||||
# pip�容错: 默认�(阿里云)缺文件时fallback到清�/官方�(2026-10-10 librt-0.6.0 metadata 404)
|
||||
MYPY_SPEC="mypy<1.19"
|
||||
INSTALL_OK=0
|
||||
python3 -m pip install -q "$MYPY_SPEC" && INSTALL_OK=1 || true
|
||||
if [ "$INSTALL_OK" != "1" ]; then
|
||||
echo "WARN: default pip index failed, retry tsinghua mirror..."
|
||||
python3 -m pip install -q -i https://pypi.tuna.tsinghua.edu.cn/simple "$MYPY_SPEC" && INSTALL_OK=1 || true
|
||||
fi
|
||||
if [ "$INSTALL_OK" != "1" ]; then
|
||||
echo "WARN: tsinghua mirror failed, retry pypi.org..."
|
||||
python3 -m pip install -q -i https://pypi.org/simple "$MYPY_SPEC" && INSTALL_OK=1 || true
|
||||
fi
|
||||
if [ "$INSTALL_OK" != "1" ]; then
|
||||
echo "ERROR: pip install mypy failed on all indexes"
|
||||
exit 1
|
||||
fi
|
||||
mypy --version
|
||||
echo ""
|
||||
echo "=== Running mypy type check (hard gate mode) ==="
|
||||
|
||||
Reference in New Issue
Block a user