Compare commits
23 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 61c15eb987 | |||
| 0012ecad30 | |||
| e1994ada0a | |||
| f9f6c53ef4 | |||
| 9699a1fcde | |||
| 305e2bd9d5 | |||
| 30cc58441f | |||
| b3b5dbd459 | |||
| 9f5948dd2b | |||
| fbc1df36e0 | |||
| d36cc09be5 | |||
| fa4dbc6761 | |||
| 77133eb9a0 | |||
| 35207ae040 | |||
| 5af16f1aea | |||
| c232721fff | |||
| fc27d4e81e | |||
| 3c782f89d1 | |||
| 2b4f11036f | |||
| 2881da65cf | |||
| 1037e218bb | |||
| f0514d7487 | |||
| bf8f62ec5b |
@@ -0,0 +1,2 @@
|
||||
Mon Oct 5 04:09:11 PM CST 2026
|
||||
2198 lite/pro并行竞速 (commit 9699a1f) — CI rebuild trigger Mon Oct 5 08:09:11 AM UTC 2026
|
||||
@@ -176,7 +176,7 @@ export interface ViralVideoJob {
|
||||
fusion_level?: FusionLevel
|
||||
voice_id?: string
|
||||
voice_mode?: "global" | "per_video"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
bgm_preference?: string
|
||||
intent_result?: IntentResult
|
||||
intent_text?: string
|
||||
@@ -212,7 +212,7 @@ export interface GenerateViralVideoRequest {
|
||||
user_copy_text?: string
|
||||
fusion_level?: FusionLevel
|
||||
voice_id?: string
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
bgm_preference?: string
|
||||
industry?: string
|
||||
target_customer?: string
|
||||
@@ -244,7 +244,7 @@ export interface AnalyzeImagesRequest {
|
||||
/** TTS 音色 ID(STEP1 已选音色时传) */
|
||||
voice_id?: string
|
||||
/** 音色来源:preset | library | clone | upload */
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
/** Seedance 视频比例:9:16 | 16:9 | 1:1 */
|
||||
video_ratio?: string
|
||||
/** Seedance 模型 ID(空则使用服务端默认) */
|
||||
@@ -273,7 +273,7 @@ export interface GenerateCopyRequest {
|
||||
/** TTS 音色 ID(优先级高于 persona_id) */
|
||||
voice_id?: string
|
||||
/** 音色来源:preset | library | clone | upload */
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
/** Seedance 视频比例(9:16/16:9/1:1 等) */
|
||||
video_ratio?: string
|
||||
/** Seedance 模型 ID(空则使用服务端默认) */
|
||||
|
||||
@@ -18,6 +18,8 @@ export interface VoiceClone {
|
||||
language: string
|
||||
gender: string
|
||||
error_message: string | null
|
||||
/** CosyVoice 实际使用的音色 ID(status=ready 时由后端填充,用于 TTS 调用) */
|
||||
voice_id?: string | null
|
||||
created_at: string
|
||||
updated_at: string
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@ export const toVoiceClone = (profile: VoiceCloneProfile): VoiceClone => ({
|
||||
language: profile.language || "",
|
||||
gender: profile.gender || "",
|
||||
error_message: profile.error_message || null,
|
||||
voice_id: profile.voice_id,
|
||||
created_at: profile.created_at,
|
||||
updated_at: profile.updated_at,
|
||||
})
|
||||
|
||||
@@ -10,4 +10,4 @@
|
||||
* 功能流程不做积分预校验,直接走生成。
|
||||
* - true:展示完整积分系统 UI。
|
||||
*/
|
||||
export const ENABLE_CREDIT_SYSTEM = false
|
||||
export const ENABLE_CREDIT_SYSTEM = true
|
||||
|
||||
@@ -1834,3 +1834,117 @@
|
||||
color: #7c3aed;
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
/* ── v1.6 我的音色(默认主路径) ── */
|
||||
.vv-voice-section {
|
||||
margin-top: 4px;
|
||||
border: 1px solid #e5e7eb;
|
||||
border-radius: 8px;
|
||||
background: #fff;
|
||||
padding: 10px;
|
||||
}
|
||||
.vv-voice-section-head {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
margin-bottom: 8px;
|
||||
}
|
||||
.vv-voice-section-title {
|
||||
font-size: 13px;
|
||||
font-weight: 600;
|
||||
color: #1f2937;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
}
|
||||
.vv-voice-section-title .anticon {
|
||||
color: #7c3aed;
|
||||
}
|
||||
.vv-link-btn-sm {
|
||||
font-size: 12px;
|
||||
padding: 2px 6px;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 4px;
|
||||
}
|
||||
.vv-my-voice-list {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 4px;
|
||||
max-height: 220px;
|
||||
overflow-y: auto;
|
||||
}
|
||||
.vv-my-voice-item {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 10px;
|
||||
padding: 8px 10px;
|
||||
border-radius: 6px;
|
||||
cursor: pointer;
|
||||
transition: background 0.15s;
|
||||
}
|
||||
.vv-my-voice-item:hover {
|
||||
background: #f5f0ff;
|
||||
}
|
||||
.vv-my-voice-item.selected {
|
||||
background: #f5f0ff;
|
||||
}
|
||||
.vv-my-voice-item.selected .vv-voice-radio {
|
||||
border-color: #7c3aed;
|
||||
background: #7c3aed;
|
||||
box-shadow: inset 0 0 0 2px #fff;
|
||||
}
|
||||
.vv-voice-empty {
|
||||
padding: 16px 12px;
|
||||
text-align: center;
|
||||
color: #9ca3af;
|
||||
font-size: 12px;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
}
|
||||
.vv-voice-empty-ic {
|
||||
font-size: 28px;
|
||||
color: #d1d5db;
|
||||
}
|
||||
.vv-voice-empty-text {
|
||||
font-size: 12px;
|
||||
}
|
||||
.vv-voice-alt-row {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
margin-top: 8px;
|
||||
}
|
||||
.vv-voice-alt-btn {
|
||||
flex: 1;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
gap: 6px;
|
||||
padding: 8px 10px;
|
||||
background: #fafafe;
|
||||
border: 1px solid #e5e7eb;
|
||||
border-radius: 6px;
|
||||
color: #6b7280;
|
||||
font-size: 12px;
|
||||
cursor: pointer;
|
||||
transition: all 0.15s;
|
||||
}
|
||||
.vv-voice-alt-btn:hover {
|
||||
border-color: #7c3aed;
|
||||
color: #7c3aed;
|
||||
background: #f5f0ff;
|
||||
}
|
||||
.vv-voice-alt-btn.selected {
|
||||
background: #f5f0ff;
|
||||
border-color: #7c3aed;
|
||||
color: #7c3aed;
|
||||
}
|
||||
.vv-voice-panel-actions {
|
||||
display: flex;
|
||||
gap: 12px;
|
||||
margin-bottom: 6px;
|
||||
padding-bottom: 6px;
|
||||
border-bottom: 1px dashed #e5e7eb;
|
||||
}
|
||||
|
||||
@@ -23,10 +23,13 @@ import {
|
||||
CheckCircleFilled,
|
||||
EditOutlined,
|
||||
HistoryOutlined,
|
||||
UserOutlined,
|
||||
} from "@ant-design/icons"
|
||||
import { Select, Input, message } from "antd"
|
||||
import { uploadAssetDirect, getAssetLibraries, getAssetsByKind, type AssetItem } from "@/api/assets"
|
||||
import { fetchPresetVoices } from "@/api/voices"
|
||||
import { getVoiceClones, getVoiceClonePreview } from "@/api/voice-clone"
|
||||
import type { VoiceClone } from "@/api/voice-clone"
|
||||
import {
|
||||
FUSION_LEVELS,
|
||||
STYLE_STRENGTHS,
|
||||
@@ -94,13 +97,15 @@ type RefVideo = {
|
||||
} | null
|
||||
|
||||
type RefAudio = {
|
||||
source: "preset" | "library" | "upload" | "clone"
|
||||
source: "preset" | "library" | "upload" | "clone" | "my_voice"
|
||||
id?: string
|
||||
name: string
|
||||
url?: string
|
||||
/** 克隆音色的 CosyVoice voice_id(source=my_voice 时使用) */
|
||||
ttsVoiceId?: string
|
||||
} | null
|
||||
|
||||
type VoicePanel = "preset" | "upload" | "library" | "record" | "douyin" | null
|
||||
type VoicePanel = "my_voices" | "preset" | "upload" | "library" | "record" | "douyin" | null
|
||||
|
||||
type TabTask = {
|
||||
id: string
|
||||
@@ -485,6 +490,9 @@ const ViralVideoPage: React.FC = () => {
|
||||
{ id: string; name: string; url: string; desc?: string }[]
|
||||
>([])
|
||||
const [presetVoices, setPresetVoices] = useState<PresetVoice[]>([])
|
||||
/** 我的克隆音色(GET /voice-clones?status=ready) */
|
||||
const [myVoices, setMyVoices] = useState<VoiceClone[]>([])
|
||||
const [myVoicesLoading, setMyVoicesLoading] = useState(false)
|
||||
const [voicePickerOpen, setVoicePickerOpen] = useState(false)
|
||||
const [videoModels, setVideoModels] = useState<ViralVideoModel[]>(FALLBACK_VIDEO_MODELS)
|
||||
const [assetPicker, setAssetPicker] = useState<{
|
||||
@@ -514,8 +522,18 @@ const ViralVideoPage: React.FC = () => {
|
||||
[activeId],
|
||||
)
|
||||
|
||||
/* ── 加载音色 ── */
|
||||
/* ── 加载音色(我的克隆音色 + 上传的配音素材 + 预设音色) ── */
|
||||
const reloadMyVoices = useCallback(() => {
|
||||
setMyVoicesLoading(true)
|
||||
getVoiceClones({ status: "ready", limit: 50 })
|
||||
.then((list) => {
|
||||
setMyVoices(list.filter((v) => v.status === "ready"))
|
||||
})
|
||||
.catch(() => setMyVoices([]))
|
||||
.finally(() => setMyVoicesLoading(false))
|
||||
}, [])
|
||||
useEffect(() => {
|
||||
reloadMyVoices()
|
||||
getAssetsByKind("voice", { limit: 50 })
|
||||
.then((list) =>
|
||||
setVoiceAssets(
|
||||
@@ -523,7 +541,7 @@ const ViralVideoPage: React.FC = () => {
|
||||
id: a.id,
|
||||
name: a.name,
|
||||
url: a.file_url || "",
|
||||
desc: a.duration ? `${Math.round(a.duration)}s` : "我的音色",
|
||||
desc: a.duration ? `${Math.round(a.duration)}s` : "配音文件",
|
||||
})),
|
||||
),
|
||||
)
|
||||
@@ -544,7 +562,7 @@ const ViralVideoPage: React.FC = () => {
|
||||
// 接口失败兜底:弹窗内 MOCK_VOICES 会生效(voice_id 与后端 CosyVoice 一致)
|
||||
setPresetVoices([])
|
||||
})
|
||||
}, [])
|
||||
}, [reloadMyVoices])
|
||||
|
||||
/* ── 加载视频模型列表 ── */
|
||||
useEffect(() => {
|
||||
@@ -914,21 +932,36 @@ const ViralVideoPage: React.FC = () => {
|
||||
[setTask, uploadFile],
|
||||
)
|
||||
|
||||
const onCloneSuccess = (voice: { id?: string; name?: string }) => {
|
||||
const onCloneSuccess = (
|
||||
voice: VoiceClone | { id?: string; name?: string; voice_id?: string | null },
|
||||
) => {
|
||||
setCloneModalOpen(false)
|
||||
// 克隆成功后刷新我的音色列表
|
||||
reloadMyVoices()
|
||||
const vid = "voice_id" in voice ? voice.voice_id : undefined
|
||||
if (voice?.id) {
|
||||
setTask({
|
||||
refAudio: { source: "clone", id: voice.id, name: voice.name || "我录制的音色" },
|
||||
refAudio: {
|
||||
source: "my_voice",
|
||||
id: voice.id,
|
||||
name: voice.name || "我录制的音色",
|
||||
ttsVoiceId: vid || undefined,
|
||||
},
|
||||
voicePanel: null,
|
||||
})
|
||||
getAssetsByKind("voice", { limit: 50 })
|
||||
.then((list) =>
|
||||
setVoiceAssets(
|
||||
list.map((a) => ({ id: a.id, name: a.name, url: a.file_url || "", desc: "我的音色" })),
|
||||
list.map((a) => ({
|
||||
id: a.id,
|
||||
name: a.name,
|
||||
url: a.file_url || "",
|
||||
desc: a.duration ? `${Math.round(a.duration)}s` : "配音文件",
|
||||
})),
|
||||
),
|
||||
)
|
||||
.catch(() => {})
|
||||
message.success("录音已完成,已自动选中")
|
||||
message.success("音色克隆完成,已自动选中")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1014,7 +1047,12 @@ const ViralVideoPage: React.FC = () => {
|
||||
: undefined,
|
||||
fusion_level: task.fusionLevel,
|
||||
style_strength: task.refVideo?.ossUrl ? task.styleStrength : undefined,
|
||||
voice_id: task.refAudio?.source === "preset" ? task.refAudio.id : undefined,
|
||||
voice_id:
|
||||
task.refAudio?.source === "preset"
|
||||
? task.refAudio.id
|
||||
: task.refAudio?.source === "my_voice"
|
||||
? task.refAudio.ttsVoiceId || task.refAudio.id
|
||||
: undefined,
|
||||
voice_source: task.refAudio?.source || undefined,
|
||||
video_ratio: task.videoRatio,
|
||||
video_model: task.videoModel,
|
||||
@@ -1058,7 +1096,12 @@ const ViralVideoPage: React.FC = () => {
|
||||
style_strength: task.refVideo?.ossUrl ? task.styleStrength : undefined,
|
||||
user_copy_text: task.storyboard?.voiceover_script || task.userCopy || undefined,
|
||||
fusion_level: task.fusionLevel,
|
||||
voice_id: task.refAudio?.id || undefined,
|
||||
voice_id:
|
||||
task.refAudio?.source === "preset"
|
||||
? task.refAudio.id
|
||||
: task.refAudio?.source === "my_voice"
|
||||
? task.refAudio.ttsVoiceId || task.refAudio.id
|
||||
: undefined,
|
||||
voice_source: task.refAudio?.source,
|
||||
industry: task.industry || undefined,
|
||||
target_customer: task.targetCustomer || undefined,
|
||||
@@ -2033,43 +2076,119 @@ const ViralVideoPage: React.FC = () => {
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* 参考音频 2x2 */}
|
||||
{/* 配音选择(默认"我的音色") */}
|
||||
<div className="vv-subblock">
|
||||
<div className="vv-subblock-head">
|
||||
<span className="vv-subblock-title">参考音频</span>
|
||||
<span className="vv-subblock-title">AI 配音</span>
|
||||
<span className="vv-opt-tag">选填</span>
|
||||
<span className="vv-subblock-hint">最大 10MB · 最长 30 秒</span>
|
||||
<span className="vv-subblock-hint">
|
||||
默认使用我的克隆音色,也可选择内置音色或上传配音
|
||||
</span>
|
||||
</div>
|
||||
<div className="vv-audio-grid">
|
||||
|
||||
{/* 我的音色(主路径) */}
|
||||
<div className="vv-voice-section">
|
||||
<div className="vv-voice-section-head">
|
||||
<span className="vv-voice-section-title">
|
||||
<UserOutlined /> 我的音色
|
||||
</span>
|
||||
<button
|
||||
className="vv-link-btn vv-link-btn-sm"
|
||||
onClick={() => setCloneModalOpen(true)}
|
||||
>
|
||||
<PlusOutlined /> 录制新音色
|
||||
</button>
|
||||
</div>
|
||||
{myVoicesLoading ? (
|
||||
<div className="vv-muted vv-voice-empty">加载中…</div>
|
||||
) : myVoices.length === 0 ? (
|
||||
<div className="vv-voice-empty">
|
||||
<AudioMutedOutlined className="vv-voice-empty-ic" />
|
||||
<div className="vv-voice-empty-text">还没有克隆音色</div>
|
||||
<button
|
||||
className="vv-btn vv-btn-primary vv-btn-sm"
|
||||
onClick={() => setCloneModalOpen(true)}
|
||||
>
|
||||
去录制一个
|
||||
</button>
|
||||
</div>
|
||||
) : (
|
||||
<div className="vv-my-voice-list">
|
||||
{myVoices.map((v) => {
|
||||
const selected =
|
||||
task.refAudio?.source === "my_voice" && task.refAudio.id === v.id
|
||||
const playing = task.playingVoiceId === `my-${v.id}`
|
||||
return (
|
||||
<div
|
||||
key={v.id}
|
||||
className={`vv-my-voice-item ${selected ? "selected" : ""}`}
|
||||
onClick={() =>
|
||||
setTask({
|
||||
refAudio: {
|
||||
source: "my_voice",
|
||||
id: v.id,
|
||||
name: v.name,
|
||||
ttsVoiceId: v.voice_id || undefined,
|
||||
},
|
||||
voicePanel: null,
|
||||
})
|
||||
}
|
||||
>
|
||||
<div className="vv-voice-radio" />
|
||||
<div className="vv-voice-info">
|
||||
<div className="vv-voice-name">{v.name}</div>
|
||||
<div className="vv-voice-desc">
|
||||
AI 克隆声音
|
||||
{v.gender
|
||||
? ` · ${v.gender === "male" ? "男" : v.gender === "female" ? "女" : ""}`
|
||||
: ""}
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
className={`vv-voice-play ${playing ? "playing" : ""}`}
|
||||
onClick={(e) => {
|
||||
e.stopPropagation()
|
||||
// 克隆音色试听:优先 sample_url,否则调用 preview 接口
|
||||
if (v.sample_url) {
|
||||
toggleVoice({ id: `my-${v.id}`, url: v.sample_url })
|
||||
} else {
|
||||
message.info("正在获取试听音频…")
|
||||
getVoiceClonePreview(v.id)
|
||||
.then((r) => {
|
||||
if (r.audio_url)
|
||||
toggleVoice({ id: `my-${v.id}`, url: r.audio_url })
|
||||
})
|
||||
.catch(() => message.error("试听获取失败"))
|
||||
}
|
||||
}}
|
||||
>
|
||||
{playing ? <PauseCircleOutlined /> : <PlayCircleOutlined />}
|
||||
</button>
|
||||
</div>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* 次要选项:内置音色 / 上传配音 */}
|
||||
<div className="vv-voice-alt-row">
|
||||
<button
|
||||
className={`vv-audio-cell ${task.refAudio?.source === "preset" ? "selected" : ""}`}
|
||||
className={`vv-voice-alt-btn ${task.refAudio?.source === "preset" ? "selected" : ""}`}
|
||||
onClick={() => setVoicePickerOpen(true)}
|
||||
>
|
||||
<FileTextOutlined className="vv-audio-ic" />
|
||||
<span>选择内置音色</span>
|
||||
<FileTextOutlined /> 内置音色
|
||||
</button>
|
||||
<button
|
||||
className={`vv-audio-cell ${task.voicePanel === "upload" ? "active" : ""} ${task.refAudio?.source === "upload" ? "selected" : ""}`}
|
||||
onClick={() => voiceInputRef.current?.click()}
|
||||
className={`vv-voice-alt-btn ${task.refAudio?.source === "upload" || task.refAudio?.source === "library" ? "selected" : ""}`}
|
||||
onClick={() =>
|
||||
setTask({ voicePanel: task.voicePanel === "library" ? null : "library" })
|
||||
}
|
||||
>
|
||||
<UploadOutlined className="vv-audio-ic" />
|
||||
<span>本地上传</span>
|
||||
</button>
|
||||
<button
|
||||
className={`vv-audio-cell ${task.voicePanel === "library" ? "active" : ""} ${task.refAudio?.source === "library" ? "selected" : ""}`}
|
||||
onClick={() => setAssetPicker({ open: true, kind: "voice", multiple: false })}
|
||||
>
|
||||
<FolderOpenOutlined className="vv-audio-ic" />
|
||||
<span>从素材库选择</span>
|
||||
</button>
|
||||
<button
|
||||
className={`vv-audio-cell ${task.voicePanel === "record" ? "active" : ""} ${task.refAudio?.source === "clone" ? "selected" : ""}`}
|
||||
onClick={() => setCloneModalOpen(true)}
|
||||
>
|
||||
<AudioMutedOutlined className="vv-audio-ic" />
|
||||
<span>直接录音</span>
|
||||
<FolderOpenOutlined /> 上传配音
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<input
|
||||
ref={voiceInputRef}
|
||||
type="file"
|
||||
@@ -2083,7 +2202,14 @@ const ViralVideoPage: React.FC = () => {
|
||||
{task.refAudio && (
|
||||
<div className="vv-selected-audio">
|
||||
<SoundOutlined style={{ color: "#7c3aed", marginRight: 6 }} />
|
||||
<span style={{ flex: 1 }}>已选:{task.refAudio.name}</span>
|
||||
<span style={{ flex: 1 }}>
|
||||
已选:
|
||||
{task.refAudio.source === "my_voice"
|
||||
? `我的音色 · ${task.refAudio.name}`
|
||||
: task.refAudio.source === "preset"
|
||||
? `内置 · ${task.refAudio.name}`
|
||||
: task.refAudio.name}
|
||||
</span>
|
||||
<button className="vv-link-btn" onClick={() => setTask({ refAudio: null })}>
|
||||
移除
|
||||
</button>
|
||||
@@ -2091,16 +2217,32 @@ const ViralVideoPage: React.FC = () => {
|
||||
)}
|
||||
{task.voicePanel === "library" && (
|
||||
<div className="vv-voice-panel">
|
||||
<div className="vv-voice-panel-actions">
|
||||
<button
|
||||
className="vv-link-btn vv-link-btn-sm"
|
||||
onClick={() => voiceInputRef.current?.click()}
|
||||
>
|
||||
<UploadOutlined /> 本地上传音频文件
|
||||
</button>
|
||||
<button
|
||||
className="vv-link-btn vv-link-btn-sm"
|
||||
onClick={() =>
|
||||
setAssetPicker({ open: true, kind: "voice", multiple: false })
|
||||
}
|
||||
>
|
||||
<FolderOpenOutlined /> 从素材库选择
|
||||
</button>
|
||||
</div>
|
||||
{voiceAssets.length === 0 ? (
|
||||
<div className="vv-muted" style={{ padding: 12, textAlign: "center" }}>
|
||||
暂无已上传的音频
|
||||
暂无已上传的配音文件
|
||||
</div>
|
||||
) : (
|
||||
<div className="vv-voice-list">
|
||||
{voiceAssets.map((v) => (
|
||||
<div
|
||||
key={v.id}
|
||||
className={`vv-voice-item ${task.refAudio?.id === v.id && task.refAudio.source === "library" ? "selected" : ""}`}
|
||||
className={`vv-voice-item ${task.refAudio?.id === v.id && (task.refAudio.source === "library" || task.refAudio.source === "upload") ? "selected" : ""}`}
|
||||
onClick={() =>
|
||||
pickVoiceAsset({
|
||||
id: v.id,
|
||||
|
||||
@@ -331,24 +331,73 @@ def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict:
|
||||
|
||||
|
||||
def _is_vision_result_usable(result: dict) -> bool:
|
||||
"""判断 VLM 返回是否有效:name/summary 不能为未识别/无法判断/空,summary 要够长。"""
|
||||
"""判断 VLM 返回是否有效。
|
||||
#2198b: 判定条件放宽——有人像(portrait_prompt非空/非'无人像')即视为usable(爆款视频核心
|
||||
是要人物描述给信任链t2i用,product name/brand/features识别不准是次要的)。
|
||||
非人像场景才要求name+summary+features有效。
|
||||
"""
|
||||
if not isinstance(result, dict):
|
||||
return False
|
||||
# 有人像描述(爆款视频最核心需求,portrait_prompt给信任链t2i做参考)就视为usable
|
||||
pp = (result.get("portrait_prompt") or "").strip()
|
||||
if pp and pp not in ("无人像", "无法判断", "未识别"):
|
||||
return True
|
||||
# 非人像场景:要求name+summary有效
|
||||
name = (result.get("name") or "").strip()
|
||||
if not name or name in ("未识别", "无法判断", "未知"):
|
||||
return False
|
||||
summary = (result.get("summary") or "").strip()
|
||||
if len(summary) < 30 or summary in ("无法判断", "未识别"):
|
||||
if len(summary) < 5 or summary in ("无法判断", "未识别"):
|
||||
return False
|
||||
category = (result.get("category") or "").strip()
|
||||
if category == "非产品图":
|
||||
return True
|
||||
feats = result.get("key_features") or []
|
||||
if not isinstance(feats, list) or len(feats) == 0:
|
||||
if not isinstance(feats, list) or len(feats) == 0 or feats == ["无法判断"]:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _normalize_image_url(raw: str, idx: int) -> str:
|
||||
"""#2188: 将 job.images 中的 storage_key/相对路径/空值统一归一化为可公网访问 URL。
|
||||
- 以 http:// 或 https:// 开头 → 视为公网 URL
|
||||
- 其他 → 视为 storage_key,用 SharedStorageService.get_url() 转公网 URL
|
||||
- 空值/None/非字符串 → 抛 ValueError(上层 catch 后走 400 错误)
|
||||
返回前做 HTTP 可达性检查(GET+Range:0-1024 避免 OSS 签名 URL 对 HEAD 返回 403 的假阴性)。
|
||||
"""
|
||||
import requests as _req
|
||||
|
||||
if not raw or not isinstance(raw, str):
|
||||
raise ValueError(f"图片 #{idx} URL 为空或类型错误: {type(raw).__name__}={raw!r}")
|
||||
url = raw.strip()
|
||||
if not url:
|
||||
raise ValueError(f"图片 #{idx} URL 为空白字符串")
|
||||
# storage_key 判定:不以 http 开头
|
||||
if not url.startswith("http://") and not url.startswith("https://"):
|
||||
# 去掉可能的前导斜杠
|
||||
storage_key = url.lstrip("/")
|
||||
try:
|
||||
from packages.shared.storage import get_storage_service
|
||||
|
||||
_svc = get_storage_service()
|
||||
url = _svc.get_url(storage_key)
|
||||
except Exception as _e:
|
||||
raise ValueError(f"图片 #{idx} storage_key={storage_key!r} 转公网URL失败: {_e}") from _e
|
||||
logger.info("[爆款视频] 图片 #%d storage_key 已转公网 URL: %s", idx, url[:120])
|
||||
# #2194: 用 GET+Range 代替 HEAD。
|
||||
# Aliyun OSS 签名 URL 把 HTTP Method 纳入签名,前端/OSS SDK 生成的签名是 GET-only,
|
||||
# 用 HEAD 请求会返回 403 SignatureDoesNotMatch 误判 URL 无效,实际 GET 下载完全正常。
|
||||
# Range: bytes=0-1024 只取前1KB,开销极小。
|
||||
try:
|
||||
_r = _req.get(url, timeout=5, allow_redirects=True, stream=True, headers={"Range": "bytes=0-1024"})
|
||||
if _r.status_code >= 400:
|
||||
logger.warning("[爆款视频] 图片 #%d URL 可达性检查返回 %d: %s", idx, _r.status_code, url[:120])
|
||||
_r.close()
|
||||
except Exception as _e:
|
||||
logger.warning("[爆款视频] 图片 #%d URL 可达性检查异常: %s url=%s", idx, _e, url[:120])
|
||||
return url
|
||||
|
||||
|
||||
def _analyze_single_image(
|
||||
idx: int,
|
||||
img_url: str,
|
||||
@@ -385,19 +434,58 @@ def _analyze_single_image(
|
||||
image_urls=f"第1张:{img_url}",
|
||||
)
|
||||
|
||||
def _call(model: str, tmo: int):
|
||||
def _call(model: str, tmo: int, label: str):
|
||||
# #2194/#2198: max_retries=0 由外层 _step_image_analysis 统一设置(阶段前置0、阶段后恢复),
|
||||
# 子线程只读不改,避免嵌套并行竞速时多线程同时改 client.max_retries 产生竞态
|
||||
_t0 = time.time()
|
||||
try:
|
||||
return call_vision(
|
||||
image_url=img_url,
|
||||
prompt=user,
|
||||
model=model,
|
||||
max_tokens=2048,
|
||||
import json as _json
|
||||
|
||||
from packages.shared.ai_client import get_doubao_client as _gdc
|
||||
|
||||
_client = _gdc()
|
||||
_messages = [
|
||||
{"role": "system", "content": system},
|
||||
{"role": "user", "content": user},
|
||||
]
|
||||
raw = _client.vision_completion(
|
||||
messages=_messages,
|
||||
images=[img_url],
|
||||
temperature=0.3,
|
||||
max_tokens=1200,
|
||||
timeout=tmo,
|
||||
system_prompt=system,
|
||||
model=model,
|
||||
)
|
||||
_elapsed = time.time() - _t0
|
||||
logger.info(
|
||||
"[爆款视频] 图片 #%d VLM(%s/%s) 完成 elapsed=%.1fs timeout=%d",
|
||||
idx,
|
||||
label,
|
||||
model,
|
||||
_elapsed,
|
||||
tmo,
|
||||
)
|
||||
if raw is None:
|
||||
return None
|
||||
stripped = raw.strip()
|
||||
if stripped.startswith("```"):
|
||||
stripped = stripped.strip("`")
|
||||
if stripped.startswith("json"):
|
||||
stripped = stripped[4:].lstrip()
|
||||
try:
|
||||
return _json.loads(stripped)
|
||||
except (_json.JSONDecodeError, TypeError):
|
||||
return stripped
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片 #%d call_vision(%s) 异常 err=%s", idx, model, e)
|
||||
_elapsed = time.time() - _t0
|
||||
logger.warning(
|
||||
"[爆款视频] 图片 #%d call_vision(%s/%s) 异常 elapsed=%.1fs err=%s",
|
||||
idx,
|
||||
label,
|
||||
model,
|
||||
_elapsed,
|
||||
e,
|
||||
)
|
||||
return None
|
||||
|
||||
def _xml_to_product(nodes: list, raw_text: str) -> dict:
|
||||
@@ -422,28 +510,193 @@ def _analyze_single_image(
|
||||
_pose = _pa.get("pose", "无法判断") or "无法判断"
|
||||
_expr = _pa.get("expression", "无法判断") or "无法判断"
|
||||
_count = xp.attr_int(_pa.get("count"), 1)
|
||||
# #2185: VLM有时对外貌属性输出"无法判断",用通用兜底值确保portrait_prompt始终有完整外貌描述
|
||||
if _hair == "无法判断":
|
||||
_hair = "自然发型"
|
||||
if _skin == "无法判断":
|
||||
_skin = "自然"
|
||||
if _face == "无法判断":
|
||||
_face = "标准"
|
||||
if _outfit == "无法判断":
|
||||
_outfit = "日常服装"
|
||||
_parts = []
|
||||
if _gender != "无法判断":
|
||||
_g = _gender + ("性" if not _gender.endswith("性") else "")
|
||||
_parts.append(_g)
|
||||
else:
|
||||
_parts.append("成年人")
|
||||
if _age != "无法判断":
|
||||
_parts.append(_age)
|
||||
_parts.append("人物")
|
||||
if _hair != "无法判断":
|
||||
_parts.append(_hair)
|
||||
if _skin != "无法判断":
|
||||
_parts.append(f"{_skin}肤色")
|
||||
if _face != "无法判断":
|
||||
_parts.append(f"{_face}脸型")
|
||||
if _outfit != "无法判断":
|
||||
_parts.append(f"身着{_outfit}")
|
||||
_parts.append(_hair)
|
||||
_parts.append(f"{_skin}肤色")
|
||||
_parts.append(f"{_face}脸型")
|
||||
_parts.append(f"身着{_outfit}")
|
||||
if _pose != "无法判断":
|
||||
_parts.append(f"姿态{_pose}")
|
||||
if _expr != "无法判断":
|
||||
_parts.append(f"表情{_expr}")
|
||||
else:
|
||||
_parts.append("表情自然")
|
||||
# #2186: 智能回填——VLM有时省略hair/outfit等外貌属性,但product.name/features/colors里已有相关信息
|
||||
# 从product名字和features中提取服装关键词回填outfit
|
||||
if _outfit in ("日常服装", "无法判断"):
|
||||
for _ppn in product_nodes:
|
||||
_pn = (_ppn.get("attrs") or {}).get("name", "") or ""
|
||||
_pf = (_ppn.get("attrs") or {}).get("features", "") or ""
|
||||
_ptxt = _pn + " " + _pf
|
||||
# 服装关键词识别(常见上装/下装/裙装/套装)
|
||||
_cloth_kws = [
|
||||
# 衬衫/T恤类
|
||||
"衬衫",
|
||||
"T恤",
|
||||
"POLO衫",
|
||||
"polo衫",
|
||||
"Polo衫",
|
||||
"打底衫",
|
||||
"雪纺衫",
|
||||
"罩衫",
|
||||
"针织衫",
|
||||
# 毛衣/卫衣/针织类
|
||||
"毛衣",
|
||||
"卫衣",
|
||||
"帽衫",
|
||||
"针织",
|
||||
"毛衫",
|
||||
"开衫",
|
||||
# 外套/西装/夹克/风衣类
|
||||
"外套",
|
||||
"西装",
|
||||
"西服",
|
||||
"夹克",
|
||||
"皮衣",
|
||||
"皮夹克",
|
||||
"风衣",
|
||||
"大衣",
|
||||
"羽绒服",
|
||||
"棉服",
|
||||
"棉服",
|
||||
"马甲",
|
||||
"背心",
|
||||
"开衫外套",
|
||||
# 裙装
|
||||
"连衣裙",
|
||||
"半身裙",
|
||||
"短裙",
|
||||
"长裙",
|
||||
"百褶裙",
|
||||
"A字裙",
|
||||
"旗袍",
|
||||
"汉服",
|
||||
"JK裙",
|
||||
# 裤装
|
||||
"牛仔裤",
|
||||
"休闲裤",
|
||||
"西裤",
|
||||
"运动裤",
|
||||
"短裤",
|
||||
"阔腿裤",
|
||||
"打底裤",
|
||||
# 制服/套装
|
||||
"制服",
|
||||
"套装",
|
||||
"职业装",
|
||||
"工装",
|
||||
# 通用上装/下装词(兜底)
|
||||
"上衣",
|
||||
"短袖",
|
||||
"长袖",
|
||||
"无袖",
|
||||
"半袖",
|
||||
"吊带",
|
||||
"背心",
|
||||
"网纱",
|
||||
"雪纺",
|
||||
"真丝",
|
||||
"纯棉",
|
||||
"亚麻",
|
||||
]
|
||||
for _ckw in _cloth_kws:
|
||||
if _ckw in _ptxt:
|
||||
_ci = _ptxt.find(_ckw)
|
||||
# 向前找颜色/材质/款式形容词(白/黑/米/红/蓝/灰/棉/麻/长/短/厚/薄/长袖/短袖/翻领/圆领/V领/印花/条纹等)
|
||||
_start = max(0, _ci - 12)
|
||||
# 向后包含款式词(长袖/短袖/外套/套装/上衣等后续修饰)
|
||||
_end = min(len(_ptxt), _ci + len(_ckw) + 8)
|
||||
_outfit_extract = _ptxt[_start:_end].strip(" ,,。.、")
|
||||
# 仅清理明确的品牌/产品类前缀(不清理颜色/款式/尺寸形容词)
|
||||
_outfit_extract = re.sub(
|
||||
r"^(\S{0,4}牌|\S{0,3}品牌|\S{0,3}款|产品|商品|的)", "", _outfit_extract
|
||||
).strip()
|
||||
# 尾部清理:去掉残留的品牌字/型号字(如"标""ml""g""装"等单字杂字)
|
||||
_outfit_extract = re.sub(
|
||||
r"(标[0-9a-zA-Z]*|\d+\s*(?:ml|g|L|斤|件|个|瓶|盒|包|袋|装)|\s+\d+\s*)$",
|
||||
"",
|
||||
_outfit_extract,
|
||||
flags=re.IGNORECASE,
|
||||
).strip()
|
||||
if len(_outfit_extract) >= 2:
|
||||
_outfit = _outfit_extract
|
||||
break
|
||||
if _outfit not in ("日常服装", "无法判断"):
|
||||
break
|
||||
# 从color标签中提取头发颜色回填hair
|
||||
if _hair in ("自然发型", "无法判断"):
|
||||
_hair_color = ""
|
||||
_color_nodes = [n for n in nodes if n["tag"] == "color"]
|
||||
_hair_kws_map = {
|
||||
"黑": "黑色",
|
||||
"棕": "棕色",
|
||||
"金": "金色",
|
||||
"栗": "栗色",
|
||||
"红": "红色",
|
||||
"白": "白色",
|
||||
"灰": "灰色",
|
||||
"蓝": "蓝色",
|
||||
"黄": "黄色",
|
||||
"紫": "紫色",
|
||||
}
|
||||
for _cn in _color_nodes:
|
||||
_cname = (_cn.get("attrs") or {}).get("name", "") or ""
|
||||
# 小占比颜色更可能是发色(非主色的小面积色),且名称含头发/黑/棕/金等
|
||||
_ccov = 0.0
|
||||
try:
|
||||
_ccov = float((_cn.get("attrs") or {}).get("coverage", "0") or 0)
|
||||
except Exception:
|
||||
pass
|
||||
for _hk, _hv in _hair_kws_map.items():
|
||||
if _hk in _cname and _ccov < 0.3:
|
||||
_hair_color = _hv
|
||||
break
|
||||
if _hair_color:
|
||||
break
|
||||
if _hair_color:
|
||||
_hair = f"{_hair_color}头发"
|
||||
else:
|
||||
_hair = "自然发型"
|
||||
# 重新拼装_parts(回填后)
|
||||
_parts = []
|
||||
if _gender != "无法判断":
|
||||
_g = _gender + ("性" if not _gender.endswith("性") else "")
|
||||
_parts.append(_g)
|
||||
else:
|
||||
_parts.append("成年人")
|
||||
if _age != "无法判断":
|
||||
_parts.append(_age)
|
||||
_parts.append("人物")
|
||||
_parts.append(_hair)
|
||||
_parts.append(f"{_skin}肤色")
|
||||
_parts.append(f"{_face}脸型")
|
||||
_parts.append(f"身着{_outfit}")
|
||||
if _pose != "无法判断":
|
||||
_parts.append(f"姿态{_pose}")
|
||||
if _expr != "无法判断":
|
||||
_parts.append(f"表情{_expr}")
|
||||
else:
|
||||
_parts.append("表情自然")
|
||||
portrait_prompt = ",".join(_parts)
|
||||
logger.info(
|
||||
"[爆款视频] 图片 #%d 解析<people>: count=%d gender=%s age=%s hair=%s skin=%s face=%s outfit=%s pose=%s expr=%s → %s",
|
||||
"[爆款视频] 图片 #%d 解析<people>(回填后): count=%d gender=%s age=%s hair=%s skin=%s face=%s outfit=%s pose=%s expr=%s → %s",
|
||||
idx,
|
||||
_count,
|
||||
_gender,
|
||||
@@ -512,35 +765,102 @@ def _analyze_single_image(
|
||||
def _normalize(raw, source: str) -> dict:
|
||||
if raw is None:
|
||||
return _vision_fallback(idx, f"{source}_none")
|
||||
# VLM 偶尔直接返回 JSON 对象(不包裹```json),_call 里 json.loads 后已是 dict
|
||||
if isinstance(raw, dict):
|
||||
_prod = {
|
||||
"name": raw.get("name") or "未识别",
|
||||
"brand": raw.get("brand") or "无法判断",
|
||||
"category": raw.get("category") or "无法判断",
|
||||
"appearance": raw.get("appearance") or "无法判断",
|
||||
"packaging": raw.get("packaging") or "无法判断",
|
||||
"text_on_package": raw.get("text_on_package") or [],
|
||||
"key_features": raw.get("key_features") or raw.get("features") or ["无法判断"],
|
||||
"scene": raw.get("scene") or "通用",
|
||||
"mood": raw.get("mood") or "",
|
||||
"portrait_prompt": raw.get("portrait_prompt") or "无人像",
|
||||
"summary": raw.get("summary") or f"{raw.get('brand','')} {raw.get('name','')}",
|
||||
"_source": source,
|
||||
}
|
||||
return _prod
|
||||
if not isinstance(raw, str):
|
||||
return _vision_fallback(idx, f"{source}_badtype")
|
||||
nodes = xp.parse_tags(raw)
|
||||
if not nodes:
|
||||
logger.warning("[爆款视频] 图片 #%d XML 解析失败 source=%s", idx, source)
|
||||
return _vision_fallback(idx, f"{source}_xml_fail", {"_raw": raw[:500]})
|
||||
# 不是 XML 也不是 dict:尝试当作纯 JSON 字符串再解析一次
|
||||
try:
|
||||
import json as _j2
|
||||
_jd = _j2.loads(raw)
|
||||
if isinstance(_jd, dict):
|
||||
return _normalize(_jd, source)
|
||||
except Exception:
|
||||
pass
|
||||
logger.warning("[爆款视频] 图片 #%d XML/JSON 解析都失败 source=%s raw_head=%s", idx, source, raw[:200])
|
||||
return _vision_fallback(idx, f"{source}_parse_fail", {"_raw": raw[:500]})
|
||||
product = _xml_to_product(nodes, raw)
|
||||
product.setdefault("_source", source)
|
||||
product["raw"] = raw[:500]
|
||||
return product
|
||||
|
||||
first_raw = _call(vision_model, timeout)
|
||||
tag1 = vision_model.split("/")[-1] if "/" in vision_model else vision_model
|
||||
first_result = _normalize(first_raw, tag1)
|
||||
if _is_vision_result_usable(first_result):
|
||||
return first_result
|
||||
|
||||
if pro_fallback_model and pro_fallback_model != vision_model:
|
||||
pro_raw = _call(pro_fallback_model, 60) # #2180: pro VLM 实测也需25-38s,原25s太短,提到60s
|
||||
pro_result = _normalize(pro_raw, "pro_fallback")
|
||||
if _is_vision_result_usable(pro_result):
|
||||
pro_result["_fallback_used"] = True
|
||||
return pro_result
|
||||
return pro_result
|
||||
return first_result
|
||||
# #2198: lite/pro 并行竞速。同时发两个请求,先返回 usable 结果就用哪个,避免
|
||||
# 串行 lite超时→再发pro 累计80-100s的惩罚。外层 max_workers=2 图片并发时,竞速模式下
|
||||
# VLM 总并发=4(2图 × 2模型),实测 Ark 可以承受,且因为取快者而不是等两个都完,
|
||||
# 单图通常 40-50s 就能拿到 pro 结果(pro 正常 42-46s),lite 偶发 30s 内返回时更快。
|
||||
race_t0 = time.time()
|
||||
lite_tag = vision_model.split("/")[-1] if "/" in vision_model else vision_model
|
||||
winner: dict | None = None
|
||||
with ThreadPoolExecutor(max_workers=2) as _inner_pool:
|
||||
f_lite = _inner_pool.submit(_call, vision_model, timeout, "lite")
|
||||
# pro 给 75s(原60s太紧实测1/3超时,pro正常42-65s给10s余量)
|
||||
pro_tmo = 75
|
||||
f_pro = _inner_pool.submit(_call, pro_fallback_model or vision_model, pro_tmo, "pro")
|
||||
_fmap = {f_lite: ("lite", lite_tag), f_pro: ("pro", "pro_fallback")}
|
||||
for _fut in as_completed(_fmap, timeout=pro_tmo + 15):
|
||||
_lbl, _tag = _fmap[_fut]
|
||||
try:
|
||||
_raw = _fut.result()
|
||||
except Exception as _e:
|
||||
logger.warning("[爆款视频] 图片 #%d %s future异常: %s", idx, _lbl, _e)
|
||||
_raw = None
|
||||
_res = _normalize(_raw, _tag)
|
||||
if _is_vision_result_usable(_res):
|
||||
winner = _res
|
||||
if _lbl == "pro":
|
||||
winner["_fallback_used"] = True
|
||||
logger.info(
|
||||
"[爆款视频] 图片 #%d 竞速胜出=%s elapsed=%.1fs",
|
||||
idx,
|
||||
_lbl,
|
||||
time.time() - race_t0,
|
||||
)
|
||||
break
|
||||
if winner is not None:
|
||||
return winner
|
||||
# 两个都失败,返回最后一次 _normalize 结果(通常是 pro 的失败 fallback,含 _source=pro_fallback_none)
|
||||
try:
|
||||
_last_raw = f_pro.result(timeout=1)
|
||||
except Exception:
|
||||
_last_raw = None
|
||||
_last = _normalize(_last_raw, "pro_fallback")
|
||||
logger.warning(
|
||||
"[爆款视频] 图片 #%d lite/pro 竞速均失败 elapsed=%.1fs",
|
||||
idx,
|
||||
time.time() - race_t0,
|
||||
)
|
||||
return _last
|
||||
|
||||
|
||||
def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
"""步骤 1: 图片 VLM 分析 — 识别产品特征(v1.6 优化:并行 + lite 模型提速)。"""
|
||||
"""步骤 1: 图片 VLM 分析 — 识别产品特征(v1.6/#2198 优化:lite/pro 并行竞速)。
|
||||
#2188/#2194/#2198: (1) 所有图片 URL 先归一化(storage_key→公网URL+空值报400)
|
||||
(2) 爆款视频强制 lite-first,不依赖 .env USE_LITE 开关
|
||||
(3) max_tokens=1200,max_workers=min(2,n) 防方舟限流(竞速模式总并发=4)
|
||||
(4) lite/pro 并行竞速:单张图同时发 lite(30s) 和 pro(75s),
|
||||
谁先返回 usable 结果就用谁。单图最坏 75s(pro慢),典型 40-50s,
|
||||
3图2并发最坏约75s,比原串行 lite→pro 240s 改善70%+
|
||||
(5) 整个阶段统一关闭底层 httpx 重试(外层 max_retries=0,finally 恢复),
|
||||
子线程只读不改 client 属性避免竞态
|
||||
(6) 每张图 VLM 调用结束打印 elapsed 耗时日志便于排查
|
||||
"""
|
||||
try:
|
||||
from packages.shared.ai_service import call_vision # noqa: F401
|
||||
except ImportError:
|
||||
@@ -551,46 +871,64 @@ def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
logger.warning("[爆款视频] 任务无 images,跳过图片分析")
|
||||
return {"products": []}
|
||||
|
||||
# 选择视觉模型:lite 速度优先(默认),pro 作为降级备用
|
||||
# #2188 BUG1: URL 归一化 — storage_key→公网URL + 空值报400
|
||||
normalized_urls: list[str] = []
|
||||
for idx, raw in enumerate(job.images):
|
||||
try:
|
||||
normalized_urls.append(_normalize_image_url(raw, idx))
|
||||
except ValueError as _ve:
|
||||
# 空/非法URL:直接让任务失败,不默默走 fallback
|
||||
logger.error("[爆款视频] 图片 #%d URL 归一化失败: %s", idx, _ve)
|
||||
raise # 上层 celery 捕获后标记任务失败,避免"未识别·无法判断"误导
|
||||
|
||||
# #2188/#2198 BUG2: 爆款视频强制 lite-first(不依赖 .env 开关),lite/pro 并行竞速
|
||||
try:
|
||||
_s = get_shared_settings()
|
||||
if _s.doubao_vision_use_lite:
|
||||
vision_model = _s.doubao_vision_lite_model
|
||||
pro_model = _s.doubao_vision_model
|
||||
vision_timeout = 45 # #2180: 方舟 VLM 实测服务端处理24-38s,原15s必超时3次重试全挂,提到45s
|
||||
else:
|
||||
vision_model = _s.doubao_vision_model
|
||||
pro_model = None # 已经是 pro,不再降级
|
||||
vision_timeout = 60
|
||||
lite_model = _s.doubao_vision_lite_model
|
||||
pro_model = _s.doubao_vision_model
|
||||
except Exception:
|
||||
vision_model = "doubao-1-5-vision-lite-250315"
|
||||
pro_model = "doubao-1-5-vision-pro-250328"
|
||||
vision_timeout = 45
|
||||
lite_model = "doubao-seed-2-1-lite-260915"
|
||||
pro_model = "doubao-seed-2-1-pro-260915"
|
||||
vision_model = lite_model
|
||||
# #2198: lite 单次 30s 封顶(竞速快速路径,30s 还没出就等 pro),pro 75s(在 _analyze_single_image
|
||||
# 的内部竞速池里设置),外层不感知。单图最坏 75s(仅 pro 成功),典型 40-50s(pro 正常返回)。
|
||||
vision_timeout = 30
|
||||
|
||||
results: list[dict] = [None] * len(job.images) # type: ignore
|
||||
max_workers = min(4, max(1, len(job.images)))
|
||||
# #2194/#2198: 整个并行图片分析阶段统一把共享 client 的 max_retries 置 0,
|
||||
# 阶段结束 finally 恢复。子线程 _call 只读不改,避免竞态。
|
||||
from packages.shared.ai_client import get_doubao_client as _gdc_step
|
||||
|
||||
_step_client = _gdc_step()
|
||||
_step_orig_retries = _step_client.max_retries
|
||||
_step_client.max_retries = 0
|
||||
|
||||
results: list[dict] = [None] * len(normalized_urls) # type: ignore
|
||||
max_workers = min(2, max(1, len(normalized_urls))) # 并发≤2 防方舟限流(竞速模式下总并发=4)
|
||||
logger.info(
|
||||
"[爆款视频] 开始并行图片分析 n=%d model=%s pro_fallback=%s timeout=%d workers=%d",
|
||||
len(job.images),
|
||||
"[爆款视频] 开始并行竞速图片分析 n=%d lite=%s(%ds) pro=%s(75s) img_workers=%d",
|
||||
len(normalized_urls),
|
||||
vision_model,
|
||||
pro_model,
|
||||
vision_timeout,
|
||||
pro_model,
|
||||
max_workers,
|
||||
)
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as pool:
|
||||
future_to_idx = {
|
||||
pool.submit(
|
||||
_analyze_single_image, idx, url, vision_model, vision_timeout, pro_fallback_model=pro_model
|
||||
): idx
|
||||
for idx, url in enumerate(job.images)
|
||||
}
|
||||
for fut in as_completed(future_to_idx):
|
||||
idx = future_to_idx[fut]
|
||||
try:
|
||||
results[idx] = fut.result()
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片 #%d future 异常 err=%s", idx, e, exc_info=True)
|
||||
results[idx] = _vision_fallback(idx, "future_exception", {"_error": str(e)[:200]})
|
||||
try:
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as pool:
|
||||
future_to_idx = {
|
||||
pool.submit(
|
||||
_analyze_single_image, idx, url, vision_model, vision_timeout, pro_fallback_model=pro_model
|
||||
): idx
|
||||
for idx, url in enumerate(normalized_urls)
|
||||
}
|
||||
for fut in as_completed(future_to_idx):
|
||||
idx = future_to_idx[fut]
|
||||
try:
|
||||
results[idx] = fut.result()
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片 #%d future 异常 err=%s", idx, e, exc_info=True)
|
||||
results[idx] = _vision_fallback(idx, "future_exception", {"_error": str(e)[:200]})
|
||||
finally:
|
||||
_step_client.max_retries = _step_orig_retries
|
||||
|
||||
return {"products": results}
|
||||
|
||||
@@ -1221,15 +1559,79 @@ def _step_review(job: ViralVideoJob, copy_result: dict) -> dict:
|
||||
return {"passed": True, "score": 75, "details": {}, "issues": []}
|
||||
|
||||
|
||||
def _resolve_tts_voice_id(job: ViralVideoJob) -> str:
|
||||
"""#2188: 根据 voice_source + voice_id 解析真正传给 CosyVoice 的 voice_id。
|
||||
|
||||
- voice_source 在 ("my_voice", "clone"):voice_id 是 VoiceCloneProfile.id,
|
||||
需要从 DB 查 profile.voice_id(CosyVoice 返回的音色 ID)。
|
||||
- 其他/空:voice_id 直接视为 CosyVoice preset 音色名(longxiaochun_v3 等)。
|
||||
- 任何解析失败都回退默认 longxiaochun_v3,保证任务不崩。
|
||||
"""
|
||||
default_voice = "longxiaochun_v3"
|
||||
raw_voice_id = (getattr(job, "voice_id", "") or "").strip()
|
||||
voice_source = (getattr(job, "voice_source", "") or "").strip().lower()
|
||||
|
||||
if not raw_voice_id:
|
||||
return default_voice
|
||||
|
||||
# 克隆音色:前端传 profile.id,需查 DB 取 cosyvoice_voice_id
|
||||
if voice_source in ("my_voice", "clone"):
|
||||
session = None
|
||||
try:
|
||||
from packages.adapters.sqlalchemy_impl.voice_clone_profile_repository import (
|
||||
SQLAlchemyVoiceCloneProfileRepository,
|
||||
)
|
||||
|
||||
session = SessionLocal()
|
||||
_repo = SQLAlchemyVoiceCloneProfileRepository(session)
|
||||
_profile = _repo.get(raw_voice_id)
|
||||
if _profile and _profile.status.value == "ready" and (_profile.voice_id or "").strip():
|
||||
logger.info(
|
||||
"[爆款视频] 克隆音色解析: profile_id=%s → cosyvoice_voice_id=%s",
|
||||
raw_voice_id,
|
||||
_profile.voice_id,
|
||||
)
|
||||
return _profile.voice_id.strip()
|
||||
# 找不到/未就绪/无voice_id
|
||||
if _profile is None:
|
||||
logger.warning("[爆款视频] 克隆音色 profile_id=%s 不存在,回退默认音色", raw_voice_id)
|
||||
elif _profile.status.value != "ready":
|
||||
logger.warning(
|
||||
"[爆款视频] 克隆音色 profile_id=%s 状态=%s 未就绪,回退默认音色",
|
||||
raw_voice_id,
|
||||
_profile.status.value,
|
||||
)
|
||||
else:
|
||||
logger.warning("[爆款视频] 克隆音色 profile_id=%s 就绪但 voice_id 为空,回退默认音色", raw_voice_id)
|
||||
return default_voice
|
||||
except Exception as _e:
|
||||
logger.warning("[爆款视频] 克隆音色解析异常 profile_id=%s err=%s,回退默认音色", raw_voice_id, _e)
|
||||
return default_voice
|
||||
finally:
|
||||
if session is not None:
|
||||
try:
|
||||
session.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# preset/my-voice 直传/空 source:voice_id 就是 CosyVoice 音色名
|
||||
return raw_voice_id or default_voice
|
||||
|
||||
|
||||
def _step_tts(job: ViralVideoJob, voiceover_script: str):
|
||||
"""步骤 5: CosyVoice 整段配音 → 返回本地 MP3 Path;失败返回 None。"""
|
||||
"""步骤 5: CosyVoice 整段配音 → 返回本地 MP3 Path;失败返回 None。
|
||||
|
||||
#2188: 支持 voice_source='my_voice'/'clone',先把前端传的 profile.id
|
||||
解析成 CosyVoice 真正的克隆音色 voice_id,再走同一条 synthesize 链路。
|
||||
"""
|
||||
try:
|
||||
from pathlib import Path as _Path
|
||||
|
||||
from apps.worker.services.tts_service_factory import get_tts_service
|
||||
|
||||
tts_service = get_tts_service()
|
||||
voice_id = (getattr(job, "voice_id", "") or "").strip()
|
||||
# #2188: 根据 voice_source 解析实际传给 CosyVoice 的 voice_id
|
||||
voice_id = _resolve_tts_voice_id(job)
|
||||
text = (voiceover_script or "").strip()
|
||||
if not text:
|
||||
logger.warning("[爆款视频] voiceover_script 为空,跳过 TTS")
|
||||
@@ -1237,21 +1639,19 @@ def _step_tts(job: ViralVideoJob, voiceover_script: str):
|
||||
try:
|
||||
result = tts_service.synthesize(
|
||||
text=text,
|
||||
voice_id=voice_id or "longxiaochun_v3",
|
||||
voice_id=voice_id,
|
||||
format="mp3",
|
||||
)
|
||||
except TypeError:
|
||||
try:
|
||||
result = tts_service.synthesize(text=text, voice_id=voice_id or "longxiaochun_v3")
|
||||
result = tts_service.synthesize(text=text, voice_id=voice_id)
|
||||
except TypeError:
|
||||
result = tts_service.synthesize(text=text)
|
||||
if result is None:
|
||||
return None
|
||||
p = _Path(result) if not isinstance(result, _Path) else result
|
||||
if p.exists() and p.stat().st_size > 0:
|
||||
logger.info(
|
||||
"[爆款视频] TTS 合成完成: voice=%s path=%s size=%d", voice_id or "longxiaochun_v3", p, p.stat().st_size
|
||||
)
|
||||
logger.info("[爆款视频] TTS 合成完成: voice=%s path=%s size=%d", voice_id, p, p.stat().st_size)
|
||||
return p
|
||||
logger.warning("[爆款视频] TTS 返回路径不存在或空文件: %s", p)
|
||||
return None
|
||||
|
||||
@@ -57,7 +57,18 @@ _IMAGE_ANALYSIS_SYSTEM = f"""你是电商商品视觉分析师,负责从商品
|
||||
<quality> 用一个标签,属性 resolution、lighting、composition、blur 描述画质。
|
||||
<key_selling_points> 下面每个卖点用一个 <point> 标签。
|
||||
|
||||
看不到或无法判断的内容,属性值填“无法判断”,布尔值填 false,不要留空标签。"""
|
||||
【人物属性硬性要求(has_person=true时必须遵守)】
|
||||
hair/skin_tone/face_shape/outfit四项绝对禁止填“无法判断”,必须基于图片可见特征给出具体中文描述:
|
||||
- hair:必须描述发型+发色,如“黑色齐肩直发”“棕色微卷中长发”“深棕色短发”
|
||||
- skin_tone:必须描述肤色,如“暖调自然肤色”“白皙肤色”“小麦色”
|
||||
- face_shape:必须描述脸型,如“鹅蛋脸”“圆脸”“瓜子脸”“方脸”
|
||||
- outfit:必须描述可见穿着,如“米色翻领衬衫”“白色T恤”“黑色连衣裙”
|
||||
即使局部被遮挡也要根据可见部分合理推断;确实看不清时按最接近的直观印象描述。
|
||||
|
||||
其他非人物属性看不到或无法判断时填“无法判断”,布尔值填false,不要留空标签。
|
||||
|
||||
【有人物场景输出参考(女性手持商品示例,必须写全10个属性,禁止省略)】
|
||||
<people has_person="true" count="1" gender="女" age_range="青年" hair="黑色齐肩直发" skin_tone="暖调自然肤色" face_shape="鹅蛋脸" outfit="米色翻领衬衫" pose="正面半身,手持商品" expression="面带微笑"/>"""
|
||||
|
||||
_IMAGE_ANALYSIS_USER = """请分析以下商品图片,共 {image_count} 张。
|
||||
所属行业:{industry}
|
||||
|
||||
@@ -103,7 +103,7 @@ class SharedSettings(BaseSettings):
|
||||
doubao_vision_lite_model: str = (
|
||||
"doubao-seed-2-1-lite-260915" # 快速视觉(Seed 2.1 Lite 原生多模态;原 vision-lite-250315 不可用)
|
||||
)
|
||||
doubao_vision_use_lite: bool = False # #2181: lite视觉模型100%超时,默认关闭走pro(25-38s稳定返回)
|
||||
doubao_vision_use_lite: bool = True # #2188: lite恢复稳定,爆款视频默认lite-first提速(20-30s)
|
||||
doubao_embedding_model: str = "doubao-embedding-vision-251215" # 多模态向量化(原 large-text-240915 已 Retiring)
|
||||
doubao_video_model: str = "doubao-seedance-2-5-260628"
|
||||
doubao_video_timeout: int = 600 # 视频生成轮询总超时(秒)
|
||||
|
||||
Reference in New Issue
Block a user