Compare commits
46 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 9344314eac | |||
| 3f49867384 | |||
| dd420c556f | |||
| 2d823a9255 | |||
| ed24c7cd68 | |||
| 69f88434bd | |||
| 599388d9e0 | |||
| 9ffe909dc0 | |||
| 6243196408 | |||
| 3d8f2c2ce2 | |||
| 3eef497dfe | |||
| 61c15eb987 | |||
| 0012ecad30 | |||
| e1994ada0a | |||
| f9f6c53ef4 | |||
| 9699a1fcde | |||
| 305e2bd9d5 | |||
| 30cc58441f | |||
| b3b5dbd459 | |||
| 9f5948dd2b | |||
| fbc1df36e0 | |||
| d36cc09be5 | |||
| fa4dbc6761 | |||
| 77133eb9a0 | |||
| 35207ae040 | |||
| 5af16f1aea | |||
| c232721fff | |||
| fc27d4e81e | |||
| 3c782f89d1 | |||
| 2b4f11036f | |||
| 2881da65cf | |||
| 1037e218bb | |||
| f0514d7487 | |||
| bf8f62ec5b | |||
| cb911f5eba | |||
| f6fa3ea653 | |||
| 6a7709ba43 | |||
| 2c70dd4c29 | |||
| 5548e78eee | |||
| 44bd96b148 | |||
| dda67cd10f | |||
| 233d0272a9 | |||
| 631dd643c0 | |||
| 50489b05d6 | |||
| 252ea71d50 | |||
| d5f6e9499d |
@@ -0,0 +1,2 @@
|
||||
Mon Oct 5 04:09:11 PM CST 2026
|
||||
2198 lite/pro并行竞速 (commit 9699a1f) — CI rebuild trigger Mon Oct 5 08:09:11 AM UTC 2026
|
||||
@@ -176,7 +176,7 @@ export interface ViralVideoJob {
|
||||
fusion_level?: FusionLevel
|
||||
voice_id?: string
|
||||
voice_mode?: "global" | "per_video"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
bgm_preference?: string
|
||||
intent_result?: IntentResult
|
||||
intent_text?: string
|
||||
@@ -212,7 +212,7 @@ export interface GenerateViralVideoRequest {
|
||||
user_copy_text?: string
|
||||
fusion_level?: FusionLevel
|
||||
voice_id?: string
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
bgm_preference?: string
|
||||
industry?: string
|
||||
target_customer?: string
|
||||
@@ -244,7 +244,7 @@ export interface AnalyzeImagesRequest {
|
||||
/** TTS 音色 ID(STEP1 已选音色时传) */
|
||||
voice_id?: string
|
||||
/** 音色来源:preset | library | clone | upload */
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
/** Seedance 视频比例:9:16 | 16:9 | 1:1 */
|
||||
video_ratio?: string
|
||||
/** Seedance 模型 ID(空则使用服务端默认) */
|
||||
@@ -273,7 +273,7 @@ export interface GenerateCopyRequest {
|
||||
/** TTS 音色 ID(优先级高于 persona_id) */
|
||||
voice_id?: string
|
||||
/** 音色来源:preset | library | clone | upload */
|
||||
voice_source?: "preset" | "library" | "clone" | "upload"
|
||||
voice_source?: "preset" | "library" | "clone" | "upload" | "my_voice"
|
||||
/** Seedance 视频比例(9:16/16:9/1:1 等) */
|
||||
video_ratio?: string
|
||||
/** Seedance 模型 ID(空则使用服务端默认) */
|
||||
|
||||
@@ -18,6 +18,8 @@ export interface VoiceClone {
|
||||
language: string
|
||||
gender: string
|
||||
error_message: string | null
|
||||
/** CosyVoice 实际使用的音色 ID(status=ready 时由后端填充,用于 TTS 调用) */
|
||||
voice_id?: string | null
|
||||
created_at: string
|
||||
updated_at: string
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@ export const toVoiceClone = (profile: VoiceCloneProfile): VoiceClone => ({
|
||||
language: profile.language || "",
|
||||
gender: profile.gender || "",
|
||||
error_message: profile.error_message || null,
|
||||
voice_id: profile.voice_id,
|
||||
created_at: profile.created_at,
|
||||
updated_at: profile.updated_at,
|
||||
})
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
/* DurationWheelPicker —— 弹层式滚轮选择器(样式与表单一致) */
|
||||
|
||||
/* 触发按钮:外观复用 .vv-select 风格 */
|
||||
.dw-trigger {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
width: 100%;
|
||||
height: 36px;
|
||||
padding: 0 12px;
|
||||
background: #fff;
|
||||
border: 1px solid #e0e0e8;
|
||||
border-radius: 8px;
|
||||
font-size: 13px;
|
||||
color: #1f2937;
|
||||
cursor: pointer;
|
||||
box-sizing: border-box;
|
||||
transition: all 0.15s;
|
||||
user-select: none;
|
||||
}
|
||||
.dw-trigger:hover {
|
||||
border-color: #c0c0d0;
|
||||
}
|
||||
.dw-trigger-open,
|
||||
.dw-trigger:focus-within {
|
||||
border-color: #7c3aed !important;
|
||||
box-shadow: 0 0 0 2px rgba(124, 58, 237, 0.12);
|
||||
}
|
||||
.dw-trigger-disabled {
|
||||
opacity: 0.5;
|
||||
pointer-events: none;
|
||||
cursor: not-allowed;
|
||||
}
|
||||
.dw-trigger-val {
|
||||
flex: 1;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
}
|
||||
.dw-trigger-placeholder {
|
||||
color: #9ca3af;
|
||||
}
|
||||
.dw-trigger-arrow {
|
||||
font-size: 10px;
|
||||
color: #9ca3af;
|
||||
margin-left: 8px;
|
||||
transition: transform 0.2s;
|
||||
}
|
||||
.dw-trigger-arrow-up {
|
||||
transform: rotate(180deg);
|
||||
}
|
||||
|
||||
/* 弹层容器 */
|
||||
.dw-popup {
|
||||
padding: 8px;
|
||||
min-width: 140px;
|
||||
}
|
||||
|
||||
/* 滚轮 */
|
||||
.dw-picker {
|
||||
position: relative;
|
||||
width: 100%;
|
||||
overflow: hidden;
|
||||
border-radius: 8px;
|
||||
background: #fafafe;
|
||||
border: 1px solid #e5e7eb;
|
||||
}
|
||||
.dw-picker-list {
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
list-style: none;
|
||||
height: 100%;
|
||||
overflow-y: scroll;
|
||||
scroll-snap-type: y mandatory;
|
||||
-webkit-overflow-scrolling: touch;
|
||||
scrollbar-width: none;
|
||||
}
|
||||
.dw-picker-list::-webkit-scrollbar {
|
||||
display: none;
|
||||
}
|
||||
.dw-picker-item {
|
||||
display: flex;
|
||||
align-items: baseline;
|
||||
justify-content: center;
|
||||
gap: 3px;
|
||||
scroll-snap-align: center;
|
||||
cursor: pointer;
|
||||
font-size: 15px;
|
||||
color: #9ca3af;
|
||||
font-weight: 400;
|
||||
transition:
|
||||
color 0.15s,
|
||||
transform 0.15s,
|
||||
font-weight 0.15s;
|
||||
}
|
||||
.dw-picker-item-val {
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
.dw-picker-item-unit {
|
||||
font-size: 13px;
|
||||
color: inherit;
|
||||
}
|
||||
.dw-picker-item-active {
|
||||
color: #7c3aed;
|
||||
font-weight: 600;
|
||||
}
|
||||
.dw-picker-item-active .dw-picker-item-val {
|
||||
font-size: 18px;
|
||||
}
|
||||
.dw-picker-item-active .dw-picker-item-unit {
|
||||
font-size: 14px;
|
||||
}
|
||||
|
||||
/* 中心选中条 */
|
||||
.dw-picker-mask {
|
||||
position: absolute;
|
||||
left: 6px;
|
||||
right: 6px;
|
||||
pointer-events: none;
|
||||
background: #f5f0ff;
|
||||
border-radius: 6px;
|
||||
z-index: 1;
|
||||
}
|
||||
.dw-picker-mask::before,
|
||||
.dw-picker-mask::after {
|
||||
content: "";
|
||||
position: absolute;
|
||||
left: 0;
|
||||
right: 0;
|
||||
height: 1px;
|
||||
background: #d8c4ff;
|
||||
}
|
||||
.dw-picker-mask::before {
|
||||
top: 0;
|
||||
}
|
||||
.dw-picker-mask::after {
|
||||
bottom: 0;
|
||||
}
|
||||
|
||||
/* 上下渐变 */
|
||||
.dw-picker-fade {
|
||||
position: absolute;
|
||||
left: 0;
|
||||
right: 0;
|
||||
height: 40%;
|
||||
pointer-events: none;
|
||||
z-index: 2;
|
||||
}
|
||||
.dw-picker-fade-top {
|
||||
top: 0;
|
||||
background: linear-gradient(to bottom, #fafafe 25%, rgba(250, 250, 254, 0));
|
||||
}
|
||||
.dw-picker-fade-bottom {
|
||||
bottom: 0;
|
||||
background: linear-gradient(to top, #fafafe 25%, rgba(250, 250, 254, 0));
|
||||
}
|
||||
|
||||
/* 弹层按钮区 */
|
||||
.dw-popup-actions {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
justify-content: flex-end;
|
||||
margin-top: 8px;
|
||||
}
|
||||
.dw-popup-actions .ant-btn {
|
||||
border-radius: 6px;
|
||||
}
|
||||
.dw-popup-actions .ant-btn-primary {
|
||||
background: #7c3aed;
|
||||
}
|
||||
.dw-popup-actions .ant-btn-primary:hover {
|
||||
background: #6d28d9 !important;
|
||||
}
|
||||
|
||||
/* 覆盖 antd Popover 默认内边距 */
|
||||
.dw-popover .ant-popover-inner {
|
||||
padding: 0 !important;
|
||||
overflow: hidden;
|
||||
}
|
||||
.dw-popover .ant-popover-arrow {
|
||||
display: none;
|
||||
}
|
||||
@@ -0,0 +1,180 @@
|
||||
/**
|
||||
* DurationWheelPicker —— 竖屏滚轮式时长选择器(弹层版)
|
||||
*
|
||||
* 设计:
|
||||
* - 外观是和其他表单 Select 一致的输入框(白色底+1px灰边+紫色focus ring)
|
||||
* - 点击输入框弹出 Popover,内部是滚轮 picker(原生 scroll-snap,零依赖)
|
||||
* - 滚轮样式:白底容器,选中行 #7c3aed 紫字加粗+浅紫背景条
|
||||
* - 支持触摸/鼠标滚轮/点击;松手吸附;底部"确认/取消"按钮
|
||||
* - 默认范围 15–30 秒,步长 1 秒
|
||||
*/
|
||||
import React, { useEffect, useMemo, useRef, useState, useCallback } from "react"
|
||||
import { Popover, Button } from "antd"
|
||||
import { DownOutlined } from "@ant-design/icons"
|
||||
import "./DurationWheelPicker.css"
|
||||
|
||||
export interface DurationWheelPickerProps {
|
||||
value?: number
|
||||
min?: number
|
||||
max?: number
|
||||
step?: number
|
||||
unit?: string
|
||||
onChange?: (value: number) => void
|
||||
placeholder?: string
|
||||
disabled?: boolean
|
||||
/** 弹层宽度,默认 160px */
|
||||
popupWidth?: number
|
||||
/** 弹层内滚轮高度,默认 180px */
|
||||
wheelHeight?: number
|
||||
}
|
||||
|
||||
const ITEM_HEIGHT = 36
|
||||
|
||||
const DurationWheelPicker: React.FC<DurationWheelPickerProps> = ({
|
||||
value = 20,
|
||||
min = 15,
|
||||
max = 30,
|
||||
step = 1,
|
||||
unit = "秒",
|
||||
onChange,
|
||||
placeholder = "请选择时长",
|
||||
disabled = false,
|
||||
popupWidth = 160,
|
||||
wheelHeight = 180,
|
||||
}) => {
|
||||
const options = useMemo(() => {
|
||||
const arr: number[] = []
|
||||
for (let v = min; v <= max; v += step) arr.push(v)
|
||||
return arr
|
||||
}, [min, max, step])
|
||||
|
||||
const [open, setOpen] = useState(false)
|
||||
// 弹层内暂存值,点确认才提交
|
||||
const [draft, setDraft] = useState<number>(value)
|
||||
const listRef = useRef<HTMLUListElement>(null)
|
||||
const scrollTimerRef = useRef<ReturnType<typeof setTimeout> | null>(null)
|
||||
|
||||
useEffect(() => {
|
||||
if (open) {
|
||||
setDraft(value)
|
||||
// 下一帧滚到当前值
|
||||
requestAnimationFrame(() => scrollToValue(value, false))
|
||||
}
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [open])
|
||||
|
||||
const scrollToValue = useCallback(
|
||||
(v: number, smooth = true) => {
|
||||
const list = listRef.current
|
||||
if (!list) return
|
||||
const idx = options.indexOf(v)
|
||||
if (idx < 0) return
|
||||
list.scrollTo({ top: idx * ITEM_HEIGHT, behavior: smooth ? "smooth" : "auto" })
|
||||
},
|
||||
[options],
|
||||
)
|
||||
|
||||
const handleScroll = () => {
|
||||
if (scrollTimerRef.current) clearTimeout(scrollTimerRef.current)
|
||||
scrollTimerRef.current = setTimeout(() => {
|
||||
const list = listRef.current
|
||||
if (!list) return
|
||||
const idx = Math.round(list.scrollTop / ITEM_HEIGHT)
|
||||
const clamped = Math.max(0, Math.min(options.length - 1, idx))
|
||||
const targetTop = clamped * ITEM_HEIGHT
|
||||
if (Math.abs(list.scrollTop - targetTop) > 1) {
|
||||
list.scrollTo({ top: targetTop, behavior: "smooth" })
|
||||
}
|
||||
setDraft(options[clamped])
|
||||
}, 100)
|
||||
}
|
||||
|
||||
const handleConfirm = () => {
|
||||
onChange?.(draft)
|
||||
setOpen(false)
|
||||
}
|
||||
|
||||
const handleCancel = () => {
|
||||
setOpen(false)
|
||||
}
|
||||
|
||||
const handleItemClick = (v: number) => {
|
||||
setDraft(v)
|
||||
scrollToValue(v, true)
|
||||
}
|
||||
|
||||
const maskTop = wheelHeight / 2 - ITEM_HEIGHT / 2
|
||||
|
||||
const wheel = (
|
||||
<div className="dw-popup">
|
||||
<div
|
||||
className="dw-picker"
|
||||
style={{ height: wheelHeight, width: popupWidth - 24 /* padding */ }}
|
||||
>
|
||||
<div className="dw-picker-mask" style={{ top: maskTop, height: ITEM_HEIGHT }} aria-hidden />
|
||||
<div className="dw-picker-fade dw-picker-fade-top" aria-hidden />
|
||||
<div className="dw-picker-fade dw-picker-fade-bottom" aria-hidden />
|
||||
<ul
|
||||
ref={listRef}
|
||||
className="dw-picker-list"
|
||||
onScroll={handleScroll}
|
||||
style={{
|
||||
paddingTop: wheelHeight / 2 - ITEM_HEIGHT / 2,
|
||||
paddingBottom: wheelHeight / 2 - ITEM_HEIGHT / 2,
|
||||
}}
|
||||
>
|
||||
{options.map((v) => {
|
||||
const isActive = v === draft
|
||||
return (
|
||||
<li
|
||||
key={v}
|
||||
className={`dw-picker-item${isActive ? " dw-picker-item-active" : ""}`}
|
||||
style={{ height: ITEM_HEIGHT, lineHeight: `${ITEM_HEIGHT}px` }}
|
||||
onClick={() => handleItemClick(v)}
|
||||
aria-selected={isActive}
|
||||
role="option"
|
||||
>
|
||||
<span className="dw-picker-item-val">{v}</span>
|
||||
<span className="dw-picker-item-unit">{unit}</span>
|
||||
</li>
|
||||
)
|
||||
})}
|
||||
</ul>
|
||||
</div>
|
||||
<div className="dw-popup-actions">
|
||||
<Button size="small" onClick={handleCancel}>
|
||||
取消
|
||||
</Button>
|
||||
<Button size="small" type="primary" onClick={handleConfirm}>
|
||||
确认
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
|
||||
return (
|
||||
<Popover
|
||||
open={!disabled && open}
|
||||
onOpenChange={(v) => setOpen(v)}
|
||||
content={wheel}
|
||||
trigger="click"
|
||||
placement="bottomLeft"
|
||||
overlayClassName="dw-popover"
|
||||
overlayStyle={{ padding: 0 }}
|
||||
overlayInnerStyle={{ padding: 0, borderRadius: 10 }}
|
||||
destroyTooltipOnHide
|
||||
>
|
||||
<div
|
||||
className={`dw-trigger${disabled ? " dw-trigger-disabled" : ""}${open ? " dw-trigger-open" : ""}`}
|
||||
style={{ height: 36 }}
|
||||
>
|
||||
<span className={`dw-trigger-val${value != null ? "" : " dw-trigger-placeholder"}`}>
|
||||
{value != null ? `${value}${unit}` : placeholder}
|
||||
</span>
|
||||
<DownOutlined className={`dw-trigger-arrow${open ? " dw-trigger-arrow-up" : ""}`} />
|
||||
</div>
|
||||
</Popover>
|
||||
)
|
||||
}
|
||||
|
||||
export default DurationWheelPicker
|
||||
@@ -10,4 +10,4 @@
|
||||
* 功能流程不做积分预校验,直接走生成。
|
||||
* - true:展示完整积分系统 UI。
|
||||
*/
|
||||
export const ENABLE_CREDIT_SYSTEM = false
|
||||
export const ENABLE_CREDIT_SYSTEM = true
|
||||
|
||||
@@ -1834,3 +1834,117 @@
|
||||
color: #7c3aed;
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
/* ── v1.6 我的音色(默认主路径) ── */
|
||||
.vv-voice-section {
|
||||
margin-top: 4px;
|
||||
border: 1px solid #e5e7eb;
|
||||
border-radius: 8px;
|
||||
background: #fff;
|
||||
padding: 10px;
|
||||
}
|
||||
.vv-voice-section-head {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
margin-bottom: 8px;
|
||||
}
|
||||
.vv-voice-section-title {
|
||||
font-size: 13px;
|
||||
font-weight: 600;
|
||||
color: #1f2937;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
}
|
||||
.vv-voice-section-title .anticon {
|
||||
color: #7c3aed;
|
||||
}
|
||||
.vv-link-btn-sm {
|
||||
font-size: 12px;
|
||||
padding: 2px 6px;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 4px;
|
||||
}
|
||||
.vv-my-voice-list {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 4px;
|
||||
max-height: 220px;
|
||||
overflow-y: auto;
|
||||
}
|
||||
.vv-my-voice-item {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 10px;
|
||||
padding: 8px 10px;
|
||||
border-radius: 6px;
|
||||
cursor: pointer;
|
||||
transition: background 0.15s;
|
||||
}
|
||||
.vv-my-voice-item:hover {
|
||||
background: #f5f0ff;
|
||||
}
|
||||
.vv-my-voice-item.selected {
|
||||
background: #f5f0ff;
|
||||
}
|
||||
.vv-my-voice-item.selected .vv-voice-radio {
|
||||
border-color: #7c3aed;
|
||||
background: #7c3aed;
|
||||
box-shadow: inset 0 0 0 2px #fff;
|
||||
}
|
||||
.vv-voice-empty {
|
||||
padding: 16px 12px;
|
||||
text-align: center;
|
||||
color: #9ca3af;
|
||||
font-size: 12px;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
}
|
||||
.vv-voice-empty-ic {
|
||||
font-size: 28px;
|
||||
color: #d1d5db;
|
||||
}
|
||||
.vv-voice-empty-text {
|
||||
font-size: 12px;
|
||||
}
|
||||
.vv-voice-alt-row {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
margin-top: 8px;
|
||||
}
|
||||
.vv-voice-alt-btn {
|
||||
flex: 1;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
gap: 6px;
|
||||
padding: 8px 10px;
|
||||
background: #fafafe;
|
||||
border: 1px solid #e5e7eb;
|
||||
border-radius: 6px;
|
||||
color: #6b7280;
|
||||
font-size: 12px;
|
||||
cursor: pointer;
|
||||
transition: all 0.15s;
|
||||
}
|
||||
.vv-voice-alt-btn:hover {
|
||||
border-color: #7c3aed;
|
||||
color: #7c3aed;
|
||||
background: #f5f0ff;
|
||||
}
|
||||
.vv-voice-alt-btn.selected {
|
||||
background: #f5f0ff;
|
||||
border-color: #7c3aed;
|
||||
color: #7c3aed;
|
||||
}
|
||||
.vv-voice-panel-actions {
|
||||
display: flex;
|
||||
gap: 12px;
|
||||
margin-bottom: 6px;
|
||||
padding-bottom: 6px;
|
||||
border-bottom: 1px dashed #e5e7eb;
|
||||
}
|
||||
|
||||
@@ -23,10 +23,13 @@ import {
|
||||
CheckCircleFilled,
|
||||
EditOutlined,
|
||||
HistoryOutlined,
|
||||
UserOutlined,
|
||||
} from "@ant-design/icons"
|
||||
import { Select, Input, message } from "antd"
|
||||
import { uploadAssetDirect, getAssetLibraries, getAssetsByKind, type AssetItem } from "@/api/assets"
|
||||
import { fetchPresetVoices } from "@/api/voices"
|
||||
import { getVoiceClones, getVoiceClonePreview } from "@/api/voice-clone"
|
||||
import type { VoiceClone } from "@/api/voice-clone"
|
||||
import {
|
||||
FUSION_LEVELS,
|
||||
STYLE_STRENGTHS,
|
||||
@@ -94,13 +97,15 @@ type RefVideo = {
|
||||
} | null
|
||||
|
||||
type RefAudio = {
|
||||
source: "preset" | "library" | "upload" | "clone"
|
||||
source: "preset" | "library" | "upload" | "clone" | "my_voice"
|
||||
id?: string
|
||||
name: string
|
||||
url?: string
|
||||
/** 克隆音色的 CosyVoice voice_id(source=my_voice 时使用) */
|
||||
ttsVoiceId?: string
|
||||
} | null
|
||||
|
||||
type VoicePanel = "preset" | "upload" | "library" | "record" | "douyin" | null
|
||||
type VoicePanel = "my_voices" | "preset" | "upload" | "library" | "record" | "douyin" | null
|
||||
|
||||
type TabTask = {
|
||||
id: string
|
||||
@@ -213,12 +218,12 @@ const PURPOSES = [
|
||||
"悬念短剧",
|
||||
"情绪短片",
|
||||
]
|
||||
const DURATIONS = [15, 20, 30, 45, 60]
|
||||
const RATIOS = [
|
||||
{ v: "9:16", label: "9:16 竖屏(抖音/视频号)" },
|
||||
{ v: "16:9", label: "16:9 横屏(B站/YouTube)" },
|
||||
{ v: "1:1", label: "1:1 方形(小红书)" },
|
||||
]
|
||||
const DURATIONS = Array.from({ length: 16 }, (_, i) => 15 + i)
|
||||
/** 兜底模型列表(接口未返回时使用,字段与 ViralVideoModel 对齐;后端返回后自动覆盖) */
|
||||
const FALLBACK_VIDEO_MODELS: ViralVideoModel[] = [
|
||||
{
|
||||
@@ -432,7 +437,7 @@ const emptyTask = (id: string, title: string): TabTask => ({
|
||||
language: "中文(普通话)",
|
||||
viralStructure: STRUCTURES[0],
|
||||
marketingPurpose: "",
|
||||
duration: 15,
|
||||
duration: 20,
|
||||
persona: "",
|
||||
videoRatio: "9:16",
|
||||
videoModel: "seedance-2.5",
|
||||
@@ -485,6 +490,9 @@ const ViralVideoPage: React.FC = () => {
|
||||
{ id: string; name: string; url: string; desc?: string }[]
|
||||
>([])
|
||||
const [presetVoices, setPresetVoices] = useState<PresetVoice[]>([])
|
||||
/** 我的克隆音色(GET /voice-clones?status=ready) */
|
||||
const [myVoices, setMyVoices] = useState<VoiceClone[]>([])
|
||||
const [myVoicesLoading, setMyVoicesLoading] = useState(false)
|
||||
const [voicePickerOpen, setVoicePickerOpen] = useState(false)
|
||||
const [videoModels, setVideoModels] = useState<ViralVideoModel[]>(FALLBACK_VIDEO_MODELS)
|
||||
const [assetPicker, setAssetPicker] = useState<{
|
||||
@@ -514,8 +522,18 @@ const ViralVideoPage: React.FC = () => {
|
||||
[activeId],
|
||||
)
|
||||
|
||||
/* ── 加载音色 ── */
|
||||
/* ── 加载音色(我的克隆音色 + 上传的配音素材 + 预设音色) ── */
|
||||
const reloadMyVoices = useCallback(() => {
|
||||
setMyVoicesLoading(true)
|
||||
getVoiceClones({ status: "ready", limit: 50 })
|
||||
.then((list) => {
|
||||
setMyVoices(list.filter((v) => v.status === "ready"))
|
||||
})
|
||||
.catch(() => setMyVoices([]))
|
||||
.finally(() => setMyVoicesLoading(false))
|
||||
}, [])
|
||||
useEffect(() => {
|
||||
reloadMyVoices()
|
||||
getAssetsByKind("voice", { limit: 50 })
|
||||
.then((list) =>
|
||||
setVoiceAssets(
|
||||
@@ -523,7 +541,7 @@ const ViralVideoPage: React.FC = () => {
|
||||
id: a.id,
|
||||
name: a.name,
|
||||
url: a.file_url || "",
|
||||
desc: a.duration ? `${Math.round(a.duration)}s` : "我的音色",
|
||||
desc: a.duration ? `${Math.round(a.duration)}s` : "配音文件",
|
||||
})),
|
||||
),
|
||||
)
|
||||
@@ -544,7 +562,7 @@ const ViralVideoPage: React.FC = () => {
|
||||
// 接口失败兜底:弹窗内 MOCK_VOICES 会生效(voice_id 与后端 CosyVoice 一致)
|
||||
setPresetVoices([])
|
||||
})
|
||||
}, [])
|
||||
}, [reloadMyVoices])
|
||||
|
||||
/* ── 加载视频模型列表 ── */
|
||||
useEffect(() => {
|
||||
@@ -914,21 +932,36 @@ const ViralVideoPage: React.FC = () => {
|
||||
[setTask, uploadFile],
|
||||
)
|
||||
|
||||
const onCloneSuccess = (voice: { id?: string; name?: string }) => {
|
||||
const onCloneSuccess = (
|
||||
voice: VoiceClone | { id?: string; name?: string; voice_id?: string | null },
|
||||
) => {
|
||||
setCloneModalOpen(false)
|
||||
// 克隆成功后刷新我的音色列表
|
||||
reloadMyVoices()
|
||||
const vid = "voice_id" in voice ? voice.voice_id : undefined
|
||||
if (voice?.id) {
|
||||
setTask({
|
||||
refAudio: { source: "clone", id: voice.id, name: voice.name || "我录制的音色" },
|
||||
refAudio: {
|
||||
source: "my_voice",
|
||||
id: voice.id,
|
||||
name: voice.name || "我录制的音色",
|
||||
ttsVoiceId: vid || undefined,
|
||||
},
|
||||
voicePanel: null,
|
||||
})
|
||||
getAssetsByKind("voice", { limit: 50 })
|
||||
.then((list) =>
|
||||
setVoiceAssets(
|
||||
list.map((a) => ({ id: a.id, name: a.name, url: a.file_url || "", desc: "我的音色" })),
|
||||
list.map((a) => ({
|
||||
id: a.id,
|
||||
name: a.name,
|
||||
url: a.file_url || "",
|
||||
desc: a.duration ? `${Math.round(a.duration)}s` : "配音文件",
|
||||
})),
|
||||
),
|
||||
)
|
||||
.catch(() => {})
|
||||
message.success("录音已完成,已自动选中")
|
||||
message.success("音色克隆完成,已自动选中")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1014,7 +1047,12 @@ const ViralVideoPage: React.FC = () => {
|
||||
: undefined,
|
||||
fusion_level: task.fusionLevel,
|
||||
style_strength: task.refVideo?.ossUrl ? task.styleStrength : undefined,
|
||||
voice_id: task.refAudio?.source === "preset" ? task.refAudio.id : undefined,
|
||||
voice_id:
|
||||
task.refAudio?.source === "preset"
|
||||
? task.refAudio.id
|
||||
: task.refAudio?.source === "my_voice"
|
||||
? task.refAudio.ttsVoiceId || task.refAudio.id
|
||||
: undefined,
|
||||
voice_source: task.refAudio?.source || undefined,
|
||||
video_ratio: task.videoRatio,
|
||||
video_model: task.videoModel,
|
||||
@@ -1058,7 +1096,12 @@ const ViralVideoPage: React.FC = () => {
|
||||
style_strength: task.refVideo?.ossUrl ? task.styleStrength : undefined,
|
||||
user_copy_text: task.storyboard?.voiceover_script || task.userCopy || undefined,
|
||||
fusion_level: task.fusionLevel,
|
||||
voice_id: task.refAudio?.id || undefined,
|
||||
voice_id:
|
||||
task.refAudio?.source === "preset"
|
||||
? task.refAudio.id
|
||||
: task.refAudio?.source === "my_voice"
|
||||
? task.refAudio.ttsVoiceId || task.refAudio.id
|
||||
: undefined,
|
||||
voice_source: task.refAudio?.source,
|
||||
industry: task.industry || undefined,
|
||||
target_customer: task.targetCustomer || undefined,
|
||||
@@ -2033,43 +2076,119 @@ const ViralVideoPage: React.FC = () => {
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* 参考音频 2x2 */}
|
||||
{/* 配音选择(默认"我的音色") */}
|
||||
<div className="vv-subblock">
|
||||
<div className="vv-subblock-head">
|
||||
<span className="vv-subblock-title">参考音频</span>
|
||||
<span className="vv-subblock-title">AI 配音</span>
|
||||
<span className="vv-opt-tag">选填</span>
|
||||
<span className="vv-subblock-hint">最大 10MB · 最长 30 秒</span>
|
||||
<span className="vv-subblock-hint">
|
||||
默认使用我的克隆音色,也可选择内置音色或上传配音
|
||||
</span>
|
||||
</div>
|
||||
<div className="vv-audio-grid">
|
||||
|
||||
{/* 我的音色(主路径) */}
|
||||
<div className="vv-voice-section">
|
||||
<div className="vv-voice-section-head">
|
||||
<span className="vv-voice-section-title">
|
||||
<UserOutlined /> 我的音色
|
||||
</span>
|
||||
<button
|
||||
className="vv-link-btn vv-link-btn-sm"
|
||||
onClick={() => setCloneModalOpen(true)}
|
||||
>
|
||||
<PlusOutlined /> 录制新音色
|
||||
</button>
|
||||
</div>
|
||||
{myVoicesLoading ? (
|
||||
<div className="vv-muted vv-voice-empty">加载中…</div>
|
||||
) : myVoices.length === 0 ? (
|
||||
<div className="vv-voice-empty">
|
||||
<AudioMutedOutlined className="vv-voice-empty-ic" />
|
||||
<div className="vv-voice-empty-text">还没有克隆音色</div>
|
||||
<button
|
||||
className="vv-btn vv-btn-primary vv-btn-sm"
|
||||
onClick={() => setCloneModalOpen(true)}
|
||||
>
|
||||
去录制一个
|
||||
</button>
|
||||
</div>
|
||||
) : (
|
||||
<div className="vv-my-voice-list">
|
||||
{myVoices.map((v) => {
|
||||
const selected =
|
||||
task.refAudio?.source === "my_voice" && task.refAudio.id === v.id
|
||||
const playing = task.playingVoiceId === `my-${v.id}`
|
||||
return (
|
||||
<div
|
||||
key={v.id}
|
||||
className={`vv-my-voice-item ${selected ? "selected" : ""}`}
|
||||
onClick={() =>
|
||||
setTask({
|
||||
refAudio: {
|
||||
source: "my_voice",
|
||||
id: v.id,
|
||||
name: v.name,
|
||||
ttsVoiceId: v.voice_id || undefined,
|
||||
},
|
||||
voicePanel: null,
|
||||
})
|
||||
}
|
||||
>
|
||||
<div className="vv-voice-radio" />
|
||||
<div className="vv-voice-info">
|
||||
<div className="vv-voice-name">{v.name}</div>
|
||||
<div className="vv-voice-desc">
|
||||
AI 克隆声音
|
||||
{v.gender
|
||||
? ` · ${v.gender === "male" ? "男" : v.gender === "female" ? "女" : ""}`
|
||||
: ""}
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
className={`vv-voice-play ${playing ? "playing" : ""}`}
|
||||
onClick={(e) => {
|
||||
e.stopPropagation()
|
||||
// 克隆音色试听:优先 sample_url,否则调用 preview 接口
|
||||
if (v.sample_url) {
|
||||
toggleVoice({ id: `my-${v.id}`, url: v.sample_url })
|
||||
} else {
|
||||
message.info("正在获取试听音频…")
|
||||
getVoiceClonePreview(v.id)
|
||||
.then((r) => {
|
||||
if (r.audio_url)
|
||||
toggleVoice({ id: `my-${v.id}`, url: r.audio_url })
|
||||
})
|
||||
.catch(() => message.error("试听获取失败"))
|
||||
}
|
||||
}}
|
||||
>
|
||||
{playing ? <PauseCircleOutlined /> : <PlayCircleOutlined />}
|
||||
</button>
|
||||
</div>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* 次要选项:内置音色 / 上传配音 */}
|
||||
<div className="vv-voice-alt-row">
|
||||
<button
|
||||
className={`vv-audio-cell ${task.refAudio?.source === "preset" ? "selected" : ""}`}
|
||||
className={`vv-voice-alt-btn ${task.refAudio?.source === "preset" ? "selected" : ""}`}
|
||||
onClick={() => setVoicePickerOpen(true)}
|
||||
>
|
||||
<FileTextOutlined className="vv-audio-ic" />
|
||||
<span>选择内置音色</span>
|
||||
<FileTextOutlined /> 内置音色
|
||||
</button>
|
||||
<button
|
||||
className={`vv-audio-cell ${task.voicePanel === "upload" ? "active" : ""} ${task.refAudio?.source === "upload" ? "selected" : ""}`}
|
||||
onClick={() => voiceInputRef.current?.click()}
|
||||
className={`vv-voice-alt-btn ${task.refAudio?.source === "upload" || task.refAudio?.source === "library" ? "selected" : ""}`}
|
||||
onClick={() =>
|
||||
setTask({ voicePanel: task.voicePanel === "library" ? null : "library" })
|
||||
}
|
||||
>
|
||||
<UploadOutlined className="vv-audio-ic" />
|
||||
<span>本地上传</span>
|
||||
</button>
|
||||
<button
|
||||
className={`vv-audio-cell ${task.voicePanel === "library" ? "active" : ""} ${task.refAudio?.source === "library" ? "selected" : ""}`}
|
||||
onClick={() => setAssetPicker({ open: true, kind: "voice", multiple: false })}
|
||||
>
|
||||
<FolderOpenOutlined className="vv-audio-ic" />
|
||||
<span>从素材库选择</span>
|
||||
</button>
|
||||
<button
|
||||
className={`vv-audio-cell ${task.voicePanel === "record" ? "active" : ""} ${task.refAudio?.source === "clone" ? "selected" : ""}`}
|
||||
onClick={() => setCloneModalOpen(true)}
|
||||
>
|
||||
<AudioMutedOutlined className="vv-audio-ic" />
|
||||
<span>直接录音</span>
|
||||
<FolderOpenOutlined /> 上传配音
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<input
|
||||
ref={voiceInputRef}
|
||||
type="file"
|
||||
@@ -2083,7 +2202,14 @@ const ViralVideoPage: React.FC = () => {
|
||||
{task.refAudio && (
|
||||
<div className="vv-selected-audio">
|
||||
<SoundOutlined style={{ color: "#7c3aed", marginRight: 6 }} />
|
||||
<span style={{ flex: 1 }}>已选:{task.refAudio.name}</span>
|
||||
<span style={{ flex: 1 }}>
|
||||
已选:
|
||||
{task.refAudio.source === "my_voice"
|
||||
? `我的音色 · ${task.refAudio.name}`
|
||||
: task.refAudio.source === "preset"
|
||||
? `内置 · ${task.refAudio.name}`
|
||||
: task.refAudio.name}
|
||||
</span>
|
||||
<button className="vv-link-btn" onClick={() => setTask({ refAudio: null })}>
|
||||
移除
|
||||
</button>
|
||||
@@ -2091,16 +2217,32 @@ const ViralVideoPage: React.FC = () => {
|
||||
)}
|
||||
{task.voicePanel === "library" && (
|
||||
<div className="vv-voice-panel">
|
||||
<div className="vv-voice-panel-actions">
|
||||
<button
|
||||
className="vv-link-btn vv-link-btn-sm"
|
||||
onClick={() => voiceInputRef.current?.click()}
|
||||
>
|
||||
<UploadOutlined /> 本地上传音频文件
|
||||
</button>
|
||||
<button
|
||||
className="vv-link-btn vv-link-btn-sm"
|
||||
onClick={() =>
|
||||
setAssetPicker({ open: true, kind: "voice", multiple: false })
|
||||
}
|
||||
>
|
||||
<FolderOpenOutlined /> 从素材库选择
|
||||
</button>
|
||||
</div>
|
||||
{voiceAssets.length === 0 ? (
|
||||
<div className="vv-muted" style={{ padding: 12, textAlign: "center" }}>
|
||||
暂无已上传的音频
|
||||
暂无已上传的配音文件
|
||||
</div>
|
||||
) : (
|
||||
<div className="vv-voice-list">
|
||||
{voiceAssets.map((v) => (
|
||||
<div
|
||||
key={v.id}
|
||||
className={`vv-voice-item ${task.refAudio?.id === v.id && task.refAudio.source === "library" ? "selected" : ""}`}
|
||||
className={`vv-voice-item ${task.refAudio?.id === v.id && (task.refAudio.source === "library" || task.refAudio.source === "upload") ? "selected" : ""}`}
|
||||
onClick={() =>
|
||||
pickVoiceAsset({
|
||||
id: v.id,
|
||||
@@ -2289,7 +2431,7 @@ const ViralVideoPage: React.FC = () => {
|
||||
options={PURPOSES.map((i) => ({ value: i, label: i }))}
|
||||
/>
|
||||
</div>
|
||||
<div className="vv-form-row" style={{ gridColumn: "1 / -1" }}>
|
||||
<div className="vv-form-row">
|
||||
<label className="vv-label">文案视频时长</label>
|
||||
<Select
|
||||
className="vv-select"
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,4 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 图片分析:火山OCR专用API + doubao-lite强约束JSON并行,单次pro VLM兜底。"""
|
||||
|
||||
from .fast_path import analyze_image_v2, analyze_images_v2 # noqa: F401
|
||||
@@ -0,0 +1,307 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""把 fast_json VLM 输出 + OCR 文本组装为与旧 _normalize() 完全一致的 dict。
|
||||
|
||||
目标:下游(信任链t2i/intent_parsing/script_generation)零改动。
|
||||
必出字段:name, brand, category, appearance, packaging, text_on_package,
|
||||
key_features, scene, mood, portrait_prompt, summary, _source
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
# ---------- portrait_prompt 模板 ----------
|
||||
# 目标:60-100 字的人物穿搭描述,用于 Seedream 纯文生图。要求具体、风格化、视觉细节丰富。
|
||||
# 旧 VLM 输出格式参考:"一位25岁左右的亚洲女性,身穿白色V领短袖T恤,黑色高腰阔腿裤,
|
||||
# 搭配银色项链,长发披肩,表情自信,街拍风格,阳光明媚的城市街头"
|
||||
|
||||
|
||||
def _join_parts(*parts: str | None) -> str:
|
||||
return "".join(p for p in parts if p)
|
||||
|
||||
|
||||
_AGE_PREFIX = {
|
||||
"青年": "年轻",
|
||||
"中年": "中年",
|
||||
"老年": "老年",
|
||||
}
|
||||
# gender 后缀
|
||||
_GENDER_WORD = {"男": "男性", "女": "女性"}
|
||||
|
||||
|
||||
def _person_subject(fj: dict[str, Any]) -> str:
|
||||
"""人物主语:年轻女性 / 中年男性 / 少女 / 小男孩 / 人物 等。"""
|
||||
gender = fj.get("gender") or ""
|
||||
age = fj.get("age_range") or ""
|
||||
gw = _GENDER_WORD.get(gender, "")
|
||||
if age == "儿童":
|
||||
if gender == "女":
|
||||
return "小女孩"
|
||||
if gender == "男":
|
||||
return "小男孩"
|
||||
return "儿童"
|
||||
if age == "青少年":
|
||||
if gender == "女":
|
||||
return "少女"
|
||||
if gender == "男":
|
||||
return "少年"
|
||||
return "青少年"
|
||||
prefix = _AGE_PREFIX.get(age, "")
|
||||
if gw:
|
||||
return f"{prefix}{gw}" if prefix else gw
|
||||
return f"{prefix}人物" if prefix else "人物"
|
||||
|
||||
|
||||
def _build_wear_sentence(fj: dict[str, Any]) -> str:
|
||||
"""穿搭段:上装+下装/连衣裙,带颜色+材质+图案。"""
|
||||
upper = fj.get("upper_wear") or ""
|
||||
upper_color = fj.get("upper_color") or ""
|
||||
lower = fj.get("lower_wear") or ""
|
||||
lower_color = fj.get("lower_color") or ""
|
||||
dress_color = fj.get("dress_color") or ""
|
||||
material = fj.get("material") or ""
|
||||
pattern = fj.get("pattern") or ""
|
||||
|
||||
is_dress = ("连衣裙" in upper) or ("裙" in upper and not lower)
|
||||
if is_dress:
|
||||
c = dress_color or upper_color
|
||||
wear = f"{c}{upper}" if c else upper
|
||||
if material and material not in wear:
|
||||
wear = f"{material}{wear}"
|
||||
if pattern and pattern not in wear and pattern != "纯色":
|
||||
wear += f",{pattern}图案"
|
||||
return f"身穿{wear}"
|
||||
|
||||
parts: list[str] = []
|
||||
if upper:
|
||||
up = f"{upper_color}{upper}" if upper_color else upper
|
||||
if material and material not in up:
|
||||
up = f"{material}{up}"
|
||||
if pattern and pattern != "纯色" and pattern not in up:
|
||||
up += f"({pattern})"
|
||||
parts.append(f"上身{up}" if up else "")
|
||||
if lower:
|
||||
lo = f"{lower_color}{lower}" if lower_color else lower
|
||||
parts.append(f"下身{lo}" if lo else "")
|
||||
return ",".join(p for p in parts if p)
|
||||
|
||||
|
||||
def _build_portrait_prompt(fj: dict[str, Any]) -> str:
|
||||
"""组装最终 portrait_prompt(目标 60-100 字,用于 Seedream 纯文生图)。"""
|
||||
if not fj.get("has_person"):
|
||||
# 非人像:用商品+场景+mood 拼一段
|
||||
name = fj.get("product_name") or "商品"
|
||||
brand = fj.get("brand") or ""
|
||||
colors = fj.get("colors") or []
|
||||
style = fj.get("style") or ""
|
||||
scene = fj.get("scene") or ""
|
||||
mood = fj.get("mood") or ""
|
||||
pieces = []
|
||||
if brand:
|
||||
pieces.append(brand)
|
||||
pieces.append(name)
|
||||
if colors:
|
||||
pieces.append("、".join(colors[:3]) + "配色")
|
||||
if style:
|
||||
pieces.append(style + "风格")
|
||||
if mood:
|
||||
pieces.append(mood + "氛围")
|
||||
if scene and scene not in ("通用",):
|
||||
pieces.append(scene + "场景")
|
||||
pieces.append("产品特写")
|
||||
prompt = ",".join(p for p in pieces if p)
|
||||
return prompt if len(prompt) >= 10 else "产品展示图,特写镜头"
|
||||
|
||||
subject = _person_subject(fj)
|
||||
wear = _build_wear_sentence(fj)
|
||||
|
||||
accessories = fj.get("accessories") or []
|
||||
if isinstance(accessories, str):
|
||||
accessories = [accessories]
|
||||
acc_str = ""
|
||||
if accessories:
|
||||
acc_str = ",佩戴" + "、".join(str(a) for a in accessories if a)
|
||||
|
||||
hairstyle = fj.get("hairstyle") or ""
|
||||
expression = fj.get("expression") or ""
|
||||
pose = fj.get("pose") or ""
|
||||
style = fj.get("style") or ""
|
||||
scene = fj.get("scene") or ""
|
||||
mood = fj.get("mood") or ""
|
||||
|
||||
detail_parts: list[str] = []
|
||||
if hairstyle:
|
||||
detail_parts.append(hairstyle)
|
||||
if expression and expression not in ("自然", "平静"):
|
||||
detail_parts.append(f"神情{expression}")
|
||||
if pose and pose not in ("站立",):
|
||||
detail_parts.append(pose)
|
||||
|
||||
style_parts: list[str] = []
|
||||
if style:
|
||||
style_parts.append(style)
|
||||
if mood:
|
||||
style_parts.append(mood)
|
||||
if scene and scene not in ("通用",):
|
||||
style_parts.append(scene)
|
||||
|
||||
pieces = [f"一位{subject}"]
|
||||
if wear:
|
||||
pieces.append(wear)
|
||||
if acc_str:
|
||||
pieces.append(acc_str.lstrip(","))
|
||||
if detail_parts:
|
||||
pieces.append(",".join(detail_parts))
|
||||
if style_parts:
|
||||
# 风格词之间不用逗号,用空格紧凑
|
||||
pieces.append("".join(style_parts) + "风格")
|
||||
else:
|
||||
pieces.append("人像写真")
|
||||
|
||||
full = ",".join(p for p in pieces if p)
|
||||
# 过短补充镜头词
|
||||
if len(full) < 40:
|
||||
full += ",自然光线下人像特写,画面清晰"
|
||||
# 过长截断
|
||||
if len(full) > 120:
|
||||
full = full[:120].rstrip(",") + "。"
|
||||
return full
|
||||
|
||||
|
||||
# ---------- 商品字段 ----------
|
||||
|
||||
|
||||
def _infer_name(fj: dict[str, Any], ocr_texts: list[str]) -> str:
|
||||
pname = fj.get("product_name")
|
||||
if pname and pname != "未识别":
|
||||
return str(pname)
|
||||
# 人物图 → name 用穿搭主件
|
||||
if fj.get("has_person"):
|
||||
up = fj.get("upper_wear") or ""
|
||||
if "连衣裙" in up:
|
||||
return up
|
||||
return up or "人物穿搭"
|
||||
if ocr_texts:
|
||||
# 商品名可能是 OCR 最长的一行(品牌/产品名)
|
||||
return max(ocr_texts, key=len)
|
||||
return "未识别"
|
||||
|
||||
|
||||
def _infer_brand(fj: dict[str, Any], ocr_texts: list[str]) -> str:
|
||||
brand = fj.get("brand")
|
||||
if brand:
|
||||
return str(brand)
|
||||
# OCR 里短的、纯字母/汉字短串可能是 brand
|
||||
for t in ocr_texts:
|
||||
if 1 < len(t) <= 12:
|
||||
return t
|
||||
return "无法判断"
|
||||
|
||||
|
||||
def _infer_category(fj: dict[str, Any]) -> str:
|
||||
cat = fj.get("category")
|
||||
if cat:
|
||||
return str(cat)
|
||||
if fj.get("has_person"):
|
||||
return "服饰"
|
||||
return "非产品图"
|
||||
|
||||
|
||||
def _build_appearance(fj: dict[str, Any]) -> str:
|
||||
"""外观描述:颜色+款式+材质+图案 拼成一段。"""
|
||||
parts: list[str] = []
|
||||
for key, _label in [
|
||||
("upper_color", "主色"),
|
||||
("upper_wear", "款式"),
|
||||
("material", "材质"),
|
||||
("pattern", "图案"),
|
||||
]:
|
||||
v = fj.get(key)
|
||||
if v and v not in ("无法判断", "未知", "纯色"):
|
||||
parts.append(str(v))
|
||||
if not parts:
|
||||
if fj.get("has_person"):
|
||||
return "人像穿搭整体造型"
|
||||
return "无法判断"
|
||||
return "、".join(parts)
|
||||
|
||||
|
||||
def _build_key_features(fj: dict[str, Any], ocr_texts: list[str]) -> list[str]:
|
||||
feats: list[str] = []
|
||||
for key in (
|
||||
"upper_wear",
|
||||
"lower_wear",
|
||||
"upper_color",
|
||||
"lower_color",
|
||||
"dress_color",
|
||||
"material",
|
||||
"pattern",
|
||||
"style",
|
||||
"accessories",
|
||||
):
|
||||
v = fj.get(key)
|
||||
if not v:
|
||||
continue
|
||||
if isinstance(v, list):
|
||||
feats.extend(str(x) for x in v if x)
|
||||
elif isinstance(v, str) and v not in ("无法判断", "未知", "纯色"):
|
||||
feats.append(v)
|
||||
if ocr_texts:
|
||||
feats.append(f"画面文字: {'/'.join(ocr_texts[:3])}")
|
||||
# 去重
|
||||
out: list[str] = []
|
||||
seen: set[str] = set()
|
||||
for f in feats:
|
||||
f = f.strip()
|
||||
if f and f not in seen and len(f) <= 30:
|
||||
seen.add(f)
|
||||
out.append(f)
|
||||
return out[:6] if out else ["无法判断"]
|
||||
|
||||
|
||||
def assemble_result(
|
||||
idx: int,
|
||||
fast_json: dict[str, Any] | None,
|
||||
ocr_texts: list[str],
|
||||
) -> dict[str, Any]:
|
||||
"""把 fast_json 结果 + OCR 文本组装成下游兼容的 product dict。"""
|
||||
fj = fast_json or {}
|
||||
ocr_texts = ocr_texts or []
|
||||
|
||||
portrait_prompt = _build_portrait_prompt(fj)
|
||||
name = _infer_name(fj, ocr_texts)
|
||||
brand = _infer_brand(fj, ocr_texts)
|
||||
category = _infer_category(fj)
|
||||
appearance = _build_appearance(fj)
|
||||
key_features = _build_key_features(fj, ocr_texts)
|
||||
scene = fj.get("scene") or "通用"
|
||||
mood = fj.get("mood") or ""
|
||||
packaging = "无法判断" # 包装细节专用API无,保留占位
|
||||
text_on_package = ocr_texts[:8]
|
||||
summary = _build_summary(fj, name, brand, category)
|
||||
|
||||
return {
|
||||
"name": name,
|
||||
"brand": brand,
|
||||
"category": category,
|
||||
"appearance": appearance,
|
||||
"packaging": packaging,
|
||||
"text_on_package": text_on_package,
|
||||
"key_features": key_features,
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": portrait_prompt,
|
||||
"summary": summary,
|
||||
"_source": "v2_fast_json",
|
||||
}
|
||||
|
||||
|
||||
def _build_summary(fj: dict, name: str, brand: str, category: str) -> str:
|
||||
if fj.get("has_person"):
|
||||
up = fj.get("upper_wear") or "穿搭"
|
||||
style = fj.get("style") or ""
|
||||
base = f"{style}{up}" if style and style not in up else up
|
||||
return base
|
||||
if brand != "无法判断" and name != brand:
|
||||
return f"{brand} {name}"
|
||||
return name
|
||||
@@ -0,0 +1,144 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 图片分析主路径:每图并行 OCR(火山专用API)+ lite JSON VLM,失败时单次 pro VLM 兜底。
|
||||
|
||||
设计原则(灵应10-05要求):
|
||||
- 主力路径简洁:单图2路并行,外层N图全并发
|
||||
- 兜底简单:单次 pro VLM 调用,无竞速/重试/复杂超时
|
||||
- 输出 dict 格式与旧版完全一致,下游零改动
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from typing import Any
|
||||
|
||||
from . import assembler, ocr_volc, vlm_fallback, vlm_fast_json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# 可通过环境变量调参(有默认值,无需配置即可跑)
|
||||
_IMG_WORKERS = int(os.environ.get("VISION_V2_IMG_WORKERS", "8"))
|
||||
_FAST_TIMEOUT = float(os.environ.get("VISION_V2_FAST_TIMEOUT", "6"))
|
||||
_FAST_JSON_TIMEOUT = float(os.environ.get("VISION_V2_FAST_JSON_TIMEOUT", "5"))
|
||||
_OCR_TIMEOUT = float(os.environ.get("VISION_V2_OCR_TIMEOUT", "5"))
|
||||
_PRO_TIMEOUT = float(os.environ.get("VISION_V2_PRO_TIMEOUT", "45"))
|
||||
|
||||
_FALLBACK_RESULT = {
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "非产品图",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": "通用",
|
||||
"mood": "",
|
||||
"portrait_prompt": "无法判断",
|
||||
"summary": "未识别",
|
||||
}
|
||||
|
||||
|
||||
def _is_usable(r: dict[str, Any]) -> bool:
|
||||
"""结果可用判定:portrait_prompt 是核心,有效就算 usable。"""
|
||||
pp = (r.get("portrait_prompt") or "").strip()
|
||||
if pp and pp not in ("无人像", "无法判断", "未识别"):
|
||||
return True
|
||||
name = (r.get("name") or "").strip()
|
||||
if name and name not in ("未识别", "无法判断", "未知"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
"""单张图片 V2 分析。"""
|
||||
t0 = time.time()
|
||||
|
||||
# 第1层:OCR + lite JSON VLM 并行
|
||||
fj_result: dict[str, Any] | None = None
|
||||
ocr_result: list[str] = []
|
||||
with ThreadPoolExecutor(max_workers=2) as pool:
|
||||
f_fj = pool.submit(vlm_fast_json.call_fast_json, img_url, timeout=_FAST_JSON_TIMEOUT)
|
||||
f_ocr = pool.submit(ocr_volc.call_ocr, img_url, timeout=_OCR_TIMEOUT)
|
||||
try:
|
||||
for fut in as_completed([f_fj, f_ocr], timeout=_FAST_TIMEOUT):
|
||||
try:
|
||||
res = fut.result(timeout=1)
|
||||
except Exception as e:
|
||||
logger.warning("[vision.v2] 图片 #%d 子任务异常: %s", idx, e)
|
||||
continue
|
||||
if fut is f_fj and isinstance(res, dict):
|
||||
fj_result = res
|
||||
elif fut is f_ocr and isinstance(res, list):
|
||||
ocr_result = res
|
||||
except TimeoutError:
|
||||
# fast 整体超时,取消还没跑完的子任务,继续走 pro 兜底
|
||||
for f in (f_fj, f_ocr):
|
||||
if not f.done():
|
||||
f.cancel()
|
||||
logger.warning("[vision.v2] 图片 #%d fast路径超时(%.0fs),走pro兜底", idx, _FAST_TIMEOUT)
|
||||
|
||||
fast_elapsed = time.time() - t0
|
||||
|
||||
# 组装 fast 结果
|
||||
if fj_result:
|
||||
assembled = assembler.assemble_result(idx, fj_result, ocr_result)
|
||||
if _is_usable(assembled):
|
||||
assembled["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
logger.info(
|
||||
"[vision.v2] 图片 #%d fast命中 elapsed=%.2fs pp=%s",
|
||||
idx,
|
||||
fast_elapsed,
|
||||
(assembled.get("portrait_prompt") or "")[:40],
|
||||
)
|
||||
return assembled
|
||||
|
||||
# 第2层:pro VLM 单次兜底
|
||||
pro_t0 = time.time()
|
||||
pro_result = vlm_fallback.call_pro_vlm(img_url, idx, timeout=_PRO_TIMEOUT)
|
||||
if pro_result and _is_usable(pro_result):
|
||||
pro_result["_fallback_used"] = True
|
||||
pro_result["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
pro_result["_pro_elapsed"] = round(time.time() - pro_t0, 2)
|
||||
if ocr_result and not pro_result.get("text_on_package"):
|
||||
pro_result["text_on_package"] = ocr_result[:8]
|
||||
logger.info("[vision.v2] 图片 #%d pro兜底命中 total=%.2fs", idx, time.time() - t0)
|
||||
return pro_result
|
||||
|
||||
# 最终:返回最小可用结果
|
||||
logger.warning("[vision.v2] 图片 #%d 全路径失败 elapsed=%.2fs", idx, time.time() - t0)
|
||||
out = dict(_FALLBACK_RESULT)
|
||||
out["_source"] = "v2_all_failed"
|
||||
out["text_on_package"] = ocr_result[:8]
|
||||
out["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
return out
|
||||
|
||||
|
||||
def analyze_images_v2(img_urls: list[str]) -> list[dict[str, Any]]:
|
||||
"""批量图片 V2 分析,外层全并发。"""
|
||||
if not img_urls:
|
||||
return []
|
||||
workers = min(_IMG_WORKERS, len(img_urls), 16)
|
||||
results: list[dict[str, Any] | None] = [None] * len(img_urls)
|
||||
|
||||
logger.info("[vision.v2] 开始图片分析 n=%d workers=%d fast_timeout=%.0fs", len(img_urls), workers, _FAST_TIMEOUT)
|
||||
t0 = time.time()
|
||||
with ThreadPoolExecutor(max_workers=workers) as pool:
|
||||
future_to_idx = {pool.submit(analyze_image_v2, idx, url): idx for idx, url in enumerate(img_urls)}
|
||||
for fut in as_completed(future_to_idx):
|
||||
idx = future_to_idx[fut]
|
||||
try:
|
||||
results[idx] = fut.result()
|
||||
except Exception as e:
|
||||
logger.warning("[vision.v2] 图片 #%d future异常: %s", idx, e, exc_info=True)
|
||||
r = dict(_FALLBACK_RESULT)
|
||||
r["_source"] = "v2_future_exception"
|
||||
results[idx] = r
|
||||
|
||||
elapsed = time.time() - t0
|
||||
succ = sum(1 for r in results if r and _is_usable(r))
|
||||
fb = sum(1 for r in results if r and r.get("_fallback_used"))
|
||||
logger.info("[vision.v2] 完成 n=%d usable=%d pro_fallback=%d elapsed=%.2fs", len(img_urls), succ, fb, elapsed)
|
||||
return [r for r in results if r is not None]
|
||||
@@ -0,0 +1,109 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""火山引擎 AI MediaKit OCR(同步)调用封装。
|
||||
|
||||
接口:POST {mediakit_base_url}/tools-sync/ocr
|
||||
鉴权:Bearer {mediakit_api_key}
|
||||
请求体:{"image_url": "<公网可访问URL>"} (部分版本也支持 image_base64)
|
||||
响应:{"code":0,"data":{"texts":[{"text":"...","bbox":[x,y,w,h],...},...],...}}
|
||||
|
||||
目标:识别商品包装/Logo/水印上的文字,作为 fast_json VLM 的补充。
|
||||
返回值:识别到的文本字符串列表(失败返回 [])。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_TIMEOUT = 8 # OCR 秒级返回,8s 绰绰有余
|
||||
|
||||
|
||||
def call_ocr(img_url: str, *, timeout: int = DEFAULT_TIMEOUT) -> list[str]:
|
||||
"""调用 MediaKit 同步 OCR,返回去重后的纯文本列表。
|
||||
|
||||
不做重试(外层降级逻辑负责)。失败/未配置返回空列表,不抛异常。
|
||||
"""
|
||||
t0 = time.time()
|
||||
try:
|
||||
import httpx
|
||||
|
||||
from packages.shared.mediakit_client import get_mediakit_client
|
||||
|
||||
client = get_mediakit_client()
|
||||
if not client.is_available:
|
||||
logger.info("[vision.v2] mediakit 未配置,跳过 OCR")
|
||||
return []
|
||||
|
||||
url = f"{client.base_url}/tools-sync/ocr"
|
||||
headers = {
|
||||
"Authorization": f"Bearer {client.api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload: dict[str, Any] = {"image_url": img_url}
|
||||
# 部分文档版本用 image_base64,但公网 URL 场景下 image_url 最简
|
||||
resp = httpx.post(url, headers=headers, json=payload, timeout=timeout)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code != 200:
|
||||
logger.warning(
|
||||
"[vision.v2] OCR HTTP %d elapsed=%.1fs body=%s",
|
||||
resp.status_code,
|
||||
elapsed,
|
||||
resp.text[:200],
|
||||
)
|
||||
return []
|
||||
data = resp.json()
|
||||
# 兼容几种可能的响应结构
|
||||
code = data.get("code", data.get("status", 0))
|
||||
if code not in (0, "OK", "success", 200):
|
||||
logger.warning("[vision.v2] OCR 业务错误 code=%s elapsed=%.1fs resp=%s", code, elapsed, str(data)[:200])
|
||||
return []
|
||||
texts = _extract_texts(data)
|
||||
# 去重 + 过滤空
|
||||
seen: set[str] = set()
|
||||
out: list[str] = []
|
||||
for t in texts:
|
||||
t = (t or "").strip()
|
||||
if t and t not in seen and len(t) <= 100: # 过滤过长的误识别
|
||||
seen.add(t)
|
||||
out.append(t)
|
||||
logger.info("[vision.v2] OCR 完成 elapsed=%.1fs n=%d texts=%s", elapsed, len(out), out[:5])
|
||||
return out
|
||||
except Exception as e:
|
||||
elapsed = time.time() - t0
|
||||
logger.warning("[vision.v2] OCR 异常 elapsed=%.1fs err=%s", elapsed, e, exc_info=True)
|
||||
return []
|
||||
|
||||
|
||||
def _extract_texts(data: dict) -> list[str]:
|
||||
"""从 OCR 响应中抽取文本,兼容多种结构。"""
|
||||
out: list[str] = []
|
||||
# 常见结构1: data.texts = [{"text": "..."}, ...]
|
||||
d = data.get("data") or data
|
||||
if isinstance(d, dict):
|
||||
for key in ("texts", "lines", "words", "items", "result"):
|
||||
items = d.get(key)
|
||||
if isinstance(items, list):
|
||||
for it in items:
|
||||
if isinstance(it, dict):
|
||||
txt = it.get("text") or it.get("content") or it.get("word")
|
||||
if txt:
|
||||
out.append(str(txt))
|
||||
elif isinstance(it, str):
|
||||
out.append(it)
|
||||
break
|
||||
# 结构2: data.text = "..."
|
||||
if not out:
|
||||
t = d.get("text")
|
||||
if isinstance(t, str):
|
||||
out.append(t)
|
||||
# 结构3: data.ocr_text / data.content
|
||||
if not out:
|
||||
for key in ("ocr_text", "content", "raw_text"):
|
||||
v = d.get(key)
|
||||
if isinstance(v, str) and v.strip():
|
||||
out.append(v)
|
||||
break
|
||||
return out
|
||||
@@ -0,0 +1,226 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""VLM 兜底:专用API路径失败时的最后一道防线,单次调用 doubao-seed-2.1-pro。
|
||||
|
||||
设计原则:简单、直接、无竞速、无复杂超时逻辑。只在 fast_json 结果不可用时调用。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_PRO_MODEL = "doubao-seed-2-1-pro-260915"
|
||||
DEFAULT_TIMEOUT = 45
|
||||
DEFAULT_MAX_TOKENS = 800
|
||||
|
||||
|
||||
def _strip_code_fence(s: str) -> str:
|
||||
s = s.strip()
|
||||
if s.startswith("```"):
|
||||
lines = s.split("\n")
|
||||
if lines and lines[0].startswith("```"):
|
||||
lines = lines[1:]
|
||||
if lines and lines[-1].strip().startswith("```"):
|
||||
lines = lines[:-1]
|
||||
s = "\n".join(lines).strip()
|
||||
return s
|
||||
|
||||
|
||||
def _xml_text(tag: str, xml: str) -> str:
|
||||
m = re.search(rf"<{tag}[^>]*>(.*?)</{tag}>", xml, re.S)
|
||||
return (m.group(1) if m else "").strip()
|
||||
|
||||
|
||||
def _xml_attr(tag: str, attr: str, xml: str) -> str:
|
||||
m = re.search(rf"<{tag}[^>]*\b{attr}\s*=\s*[\"']([^\"']*)[\"']", xml)
|
||||
return (m.group(1) if m else "").strip()
|
||||
|
||||
|
||||
def _xml_to_product(raw: str, idx: int) -> dict[str, Any]:
|
||||
"""解析 VLM 输出的 XML 格式(简化版)。"""
|
||||
scene = _xml_text("scene", raw) or "通用"
|
||||
mood = _xml_text("mood", raw) or ""
|
||||
|
||||
portrait_prompt = "无人像"
|
||||
p_has = _xml_attr("people", "has_person", raw)
|
||||
if p_has and p_has.lower() != "false":
|
||||
gender = _xml_attr("people", "gender", raw) or ""
|
||||
age = _xml_attr("people", "age_range", raw) or ""
|
||||
outfit = _xml_attr("people", "outfit", raw) or ""
|
||||
hair = _xml_attr("people", "hair", raw) or "自然发型"
|
||||
pose = _xml_attr("people", "pose", raw) or ""
|
||||
expr = _xml_attr("people", "expression", raw) or "自然"
|
||||
parts: list[str] = []
|
||||
if gender:
|
||||
parts.append(gender + ("性" if not gender.endswith("性") else ""))
|
||||
if age:
|
||||
parts.append(age)
|
||||
parts.append("人物")
|
||||
parts.append(hair)
|
||||
if outfit:
|
||||
parts.append(f"身着{outfit}")
|
||||
if pose:
|
||||
parts.append(f"姿态{pose}")
|
||||
parts.append(f"表情{expr}")
|
||||
portrait_prompt = ",".join(parts)
|
||||
|
||||
m = re.search(r"<product[^>]*>(.*?)</product>", raw, re.S)
|
||||
if m:
|
||||
pbody = m.group(1)
|
||||
name = _xml_attr("product", "name", raw) or _xml_text("name", pbody) or "未识别"
|
||||
brand = _xml_attr("product", "brand", raw) or _xml_text("brand", pbody) or "无法判断"
|
||||
category = _xml_attr("product", "category", raw) or _xml_text("category", pbody) or "无法判断"
|
||||
appearance = _xml_attr("product", "appearance", raw) or _xml_text("appearance", pbody) or "无法判断"
|
||||
packaging = _xml_attr("product", "packaging", raw) or _xml_text("packaging", pbody) or "无法判断"
|
||||
feat = _xml_attr("product", "features", raw) or _xml_text("features", pbody) or ""
|
||||
feat_list = [x.strip() for x in re.split(r"[,,;;]", feat) if x.strip()] if feat else ["无法判断"]
|
||||
top_text = _xml_attr("product", "text_on_package", raw) or _xml_text("text_on_package", pbody) or ""
|
||||
text_list = [x.strip() for x in re.split(r"[,,;;]", top_text) if x.strip()] if top_text else []
|
||||
summary = _xml_attr("product", "summary", raw) or _xml_text("summary", pbody) or f"{brand} {name}"
|
||||
pp_attr = _xml_attr("product", "portrait_prompt", raw)
|
||||
if pp_attr and pp_attr != "无人像":
|
||||
portrait_prompt = pp_attr
|
||||
return {
|
||||
"name": name,
|
||||
"brand": brand,
|
||||
"category": category,
|
||||
"appearance": appearance,
|
||||
"packaging": packaging,
|
||||
"text_on_package": text_list,
|
||||
"key_features": feat_list,
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": portrait_prompt,
|
||||
"summary": summary,
|
||||
"_source": "vlm_pro_xml",
|
||||
}
|
||||
|
||||
if portrait_prompt != "无人像":
|
||||
return {
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "无法判断",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": portrait_prompt,
|
||||
"summary": "未识别",
|
||||
"_source": "vlm_pro_no_product",
|
||||
}
|
||||
return {
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "无法判断",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": "无人像",
|
||||
"summary": "未识别",
|
||||
"_source": "vlm_pro_no_tag",
|
||||
}
|
||||
|
||||
|
||||
def call_pro_vlm(
|
||||
img_url: str,
|
||||
idx: int,
|
||||
*,
|
||||
model: str | None = None,
|
||||
timeout: int = DEFAULT_TIMEOUT,
|
||||
) -> dict[str, Any] | None:
|
||||
"""单次调用 pro VLM,解析后返回 product dict;失败返回 None。"""
|
||||
t0 = time.time()
|
||||
try:
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template,
|
||||
render_system_prompt,
|
||||
render_user_prompt,
|
||||
)
|
||||
from packages.shared.ai_client import get_doubao_client
|
||||
except ImportError as e:
|
||||
logger.warning("[vision.vlm] 导入失败: %s", e)
|
||||
return None
|
||||
|
||||
try:
|
||||
template = get_template("image_analysis")
|
||||
system = render_system_prompt(template)
|
||||
user = render_user_prompt(template, image_count=1, industry="通用", image_urls=f"第1张:{img_url}")
|
||||
except Exception as e:
|
||||
logger.warning("[vision.vlm] 模板加载失败: %s", e)
|
||||
return None
|
||||
|
||||
client = get_doubao_client()
|
||||
if not client.is_available:
|
||||
return None
|
||||
|
||||
use_model = model or DEFAULT_PRO_MODEL
|
||||
_orig_retries = client.max_retries
|
||||
client.max_retries = 0
|
||||
try:
|
||||
raw = client.vision_completion(
|
||||
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
images=[img_url],
|
||||
temperature=0.3,
|
||||
max_tokens=DEFAULT_MAX_TOKENS,
|
||||
timeout=timeout,
|
||||
model=use_model,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("[vision.vlm] 图片 #%d pro VLM 调用失败 elapsed=%.1fs err=%s", idx, time.time() - t0, e)
|
||||
client.max_retries = _orig_retries
|
||||
return None
|
||||
client.max_retries = _orig_retries
|
||||
|
||||
elapsed = time.time() - t0
|
||||
if not raw:
|
||||
logger.warning("[vision.vlm] 图片 #%d pro VLM 返回空 elapsed=%.1fs", idx, elapsed)
|
||||
return None
|
||||
|
||||
text = _strip_code_fence(raw)
|
||||
l, r = text.find("{"), text.rfind("}")
|
||||
if l >= 0 and r > l:
|
||||
try:
|
||||
obj = json.loads(text[l : r + 1])
|
||||
if isinstance(obj, dict):
|
||||
logger.info("[vision.vlm] 图片 #%d pro VLM JSON 完成 elapsed=%.1fs", idx, elapsed)
|
||||
return {
|
||||
"name": obj.get("name") or "未识别",
|
||||
"brand": obj.get("brand") or "无法判断",
|
||||
"category": obj.get("category") or "无法判断",
|
||||
"appearance": obj.get("appearance") or "无法判断",
|
||||
"packaging": obj.get("packaging") or "无法判断",
|
||||
"text_on_package": obj.get("text_on_package") or [],
|
||||
"key_features": obj.get("key_features") or obj.get("features") or ["无法判断"],
|
||||
"scene": obj.get("scene") or "通用",
|
||||
"mood": obj.get("mood") or "",
|
||||
"portrait_prompt": obj.get("portrait_prompt") or "无人像",
|
||||
"summary": obj.get("summary") or f"{obj.get('brand','')} {obj.get('name','')}",
|
||||
"_source": "vlm_pro_json",
|
||||
}
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
|
||||
try:
|
||||
result = _xml_to_product(text, idx)
|
||||
result["_fallback_used"] = True
|
||||
result["_pro_elapsed"] = round(elapsed, 2)
|
||||
logger.info(
|
||||
"[vision.vlm] 图片 #%d pro VLM XML 完成 elapsed=%.2fs pp=%s",
|
||||
idx,
|
||||
elapsed,
|
||||
(result.get("portrait_prompt") or "")[:40],
|
||||
)
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.warning("[vision.vlm] 图片 #%d 解析失败 elapsed=%.1fs err=%s head=%s", idx, elapsed, e, raw[:200])
|
||||
return None
|
||||
@@ -0,0 +1,148 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""doubao-seed-2.1-lite 强约束 JSON-only 调用。
|
||||
|
||||
目标:替代"人体属性/商品检测/图像标签"三个火山不存在的专用云端 API。
|
||||
设计要点:
|
||||
- system prompt 极致精简,只给字段 schema 和强约束(禁止自然语言、禁止 markdown)
|
||||
- max_tokens=350(比旧 VLM 的 1200 小很多,降低延迟)
|
||||
- temperature=0.1(极低,稳定输出 JSON)
|
||||
- timeout=8s(够快,失败则由外层走 pro VLM 兜底)
|
||||
- 期望返回纯 JSON object(无 ```json 包裹、无解释文字)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# 极简 system prompt:只给字段定义 + 硬性输出要求
|
||||
_FAST_SYSTEM = (
|
||||
"你是图片结构化识别器。严格按下方 JSON schema 返回一个对象,不要任何解释、"
|
||||
"不要markdown、不要代码块、不要前后缀文字。字段值不确定时填 null 或空数组。\n"
|
||||
"{\n"
|
||||
' "has_person": true/false, // 图中是否有人\n'
|
||||
' "gender": "男"/"女"/null,\n'
|
||||
' "age_range": "儿童"/"青少年"/"青年"/"中年"/"老年"/null,\n'
|
||||
' "upper_wear": "上装款式,如T恤/衬衫/卫衣/毛衣/西装/夹克/连衣裙/吊带/背心/外套等",\n'
|
||||
' "upper_color": "上装主色",\n'
|
||||
' "lower_wear": "下装款式,如牛仔裤/休闲裤/短裙/长裙/短裤/西裤/运动裤等;穿连衣裙时填null",\n'
|
||||
' "lower_color": "下装主色",\n'
|
||||
' "dress_color": "连衣裙主色(穿连衣裙时填)",\n'
|
||||
' "accessories": ["眼镜"/"帽子"/"项链"/"耳环"/"背包"/"手表"等数组],\n'
|
||||
' "hairstyle": "发型,如短发/长发/马尾/卷发/丸子头/光头等",\n'
|
||||
' "expression": "表情,如微笑/严肃/酷/开心等",\n'
|
||||
' "pose": "姿势,如站立/坐姿/侧身/行走等",\n'
|
||||
' "scene": "场景,如室内/街拍/户外/办公室/家居/海边/雪景/森林等",\n'
|
||||
' "style": "风格,如休闲/商务/运动/复古/潮流/甜美/酷飒/优雅/街头/法式等",\n'
|
||||
' "has_product": true/false, // 是否有明确商品展示\n'
|
||||
' "category": "产品类目:服饰/鞋包/美妆/数码/食品/家居/配饰/母婴/非产品图",\n'
|
||||
' "product_name": "产品名称,非产品图填null",\n'
|
||||
' "brand": "品牌或文字标识,无则null",\n'
|
||||
' "material": "材质,如棉质/牛仔/皮革/真丝/针织/涤纶等",\n'
|
||||
' "pattern": "图案,如纯色/条纹/波点/格子/印花/碎花/Logo等",\n'
|
||||
' "colors": ["主色数组"],\n'
|
||||
' "mood": "整体氛围/情绪,如清新/活力/高级/温暖/冷峻/甜美/复古等"\n'
|
||||
"}"
|
||||
)
|
||||
|
||||
_FAST_USER = "识别这张图片的人物穿搭与主体信息,只返回JSON对象。"
|
||||
|
||||
# 默认模型
|
||||
DEFAULT_LITE_MODEL = "doubao-seed-2-1-lite-260915"
|
||||
DEFAULT_TIMEOUT = 8
|
||||
DEFAULT_MAX_TOKENS = 350
|
||||
|
||||
|
||||
def _strip_code_fence(s: str) -> str:
|
||||
"""剥离 ```json ... ``` 包裹(即使要求纯 JSON,模型偶尔仍会包代码块)。"""
|
||||
s = s.strip()
|
||||
if s.startswith("```"):
|
||||
lines = s.split("\n")
|
||||
# 去掉首行 ```json
|
||||
if lines and lines[0].startswith("```"):
|
||||
lines = lines[1:]
|
||||
# 去掉尾行 ```
|
||||
if lines and lines[-1].strip().startswith("```"):
|
||||
lines = lines[:-1]
|
||||
s = "\n".join(lines).strip()
|
||||
return s
|
||||
|
||||
|
||||
def call_fast_json(
|
||||
img_url: str,
|
||||
*,
|
||||
model: str | None = None,
|
||||
timeout: int = DEFAULT_TIMEOUT,
|
||||
max_tokens: int = DEFAULT_MAX_TOKENS,
|
||||
) -> dict[str, Any] | None:
|
||||
"""调用 lite VLM 返回结构化 dict;失败/非 JSON 返回 None。
|
||||
|
||||
注意:不做重试(外层竞速/降级逻辑负责),max_retries=0 由外层统一设置。
|
||||
"""
|
||||
t0 = time.time()
|
||||
try:
|
||||
from packages.shared.ai_client import get_doubao_client
|
||||
|
||||
client = get_doubao_client()
|
||||
if not client.is_available:
|
||||
logger.warning("[vision.v2] doubao client 不可用,跳过 fast_json")
|
||||
return None
|
||||
|
||||
use_model = model or DEFAULT_LITE_MODEL
|
||||
# 强制不重试:lite 是快速路径,失败直接走外层 pro 兜底
|
||||
_orig_retries = client.max_retries
|
||||
client.max_retries = 0
|
||||
try:
|
||||
raw = client.vision_completion(
|
||||
messages=[
|
||||
{"role": "system", "content": _FAST_SYSTEM},
|
||||
{"role": "user", "content": _FAST_USER},
|
||||
],
|
||||
images=[img_url],
|
||||
temperature=0.1,
|
||||
max_tokens=max_tokens,
|
||||
timeout=timeout,
|
||||
model=use_model,
|
||||
)
|
||||
finally:
|
||||
client.max_retries = _orig_retries
|
||||
elapsed = time.time() - t0
|
||||
if raw is None:
|
||||
logger.warning("[vision.v2] fast_json 返回 None elapsed=%.1fs model=%s", elapsed, use_model)
|
||||
return None
|
||||
|
||||
text = _strip_code_fence(raw)
|
||||
# 截到第一个 { 和最后一个 } 之间,容忍前后偶发文字
|
||||
l = text.find("{")
|
||||
r = text.rfind("}")
|
||||
if l >= 0 and r > l:
|
||||
text = text[l : r + 1]
|
||||
try:
|
||||
obj = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json JSON 解析失败 elapsed=%.1fs head=%s",
|
||||
elapsed,
|
||||
raw[:200],
|
||||
)
|
||||
return None
|
||||
if not isinstance(obj, dict):
|
||||
logger.warning("[vision.v2] fast_json 非 dict: %s", type(obj))
|
||||
return None
|
||||
logger.info(
|
||||
"[vision.v2] fast_json 完成 model=%s elapsed=%.1fs has_person=%s has_product=%s category=%s",
|
||||
use_model,
|
||||
elapsed,
|
||||
obj.get("has_person"),
|
||||
obj.get("has_product"),
|
||||
obj.get("category"),
|
||||
)
|
||||
return obj
|
||||
except Exception as e:
|
||||
elapsed = time.time() - t0
|
||||
logger.warning("[vision.v2] fast_json 异常 elapsed=%.1fs err=%s", elapsed, e, exc_info=True)
|
||||
return None
|
||||
@@ -50,14 +50,25 @@ _IMAGE_ANALYSIS_SYSTEM = f"""你是电商商品视觉分析师,负责从商品
|
||||
请严格按下面的标签格式输出,标签名一个都不能改,不要输出任何解释,不要用代码块:
|
||||
<products> 下面每个产品用一个 <product> 标签,属性 name 是产品名、features 是外观特征、position 是 main 或 secondary、image_index 是第几张图(从0开始)。
|
||||
<colors> 下面每个主要颜色用一个 <color> 标签,属性 hex 是色值、name 是颜色名、coverage 是占比小数。
|
||||
<people> 用一个标签,属性 has_person、count、gender、age_range、pose、expression 分别描述人物情况。
|
||||
<people> 用一个标签,属性 has_person、count、gender、age_range、hair(发型发色)、skin_tone(肤色)、face_shape(脸型)、outfit(穿着)、pose(姿态)、expression(表情)分别描述人物外貌。有人物时属性尽量具体(如hair="黑色长直发"、outfit="白色衬衫"),无人像时除has_person=false外其他填"无法判断"。
|
||||
<mood> 标签写画面整体情绪氛围。
|
||||
<visible_text> 下面每处可见文字用一个 <text_item> 标签,属性 text 是文字内容、position 是位置。
|
||||
<scene> 标签写场景描述。
|
||||
<quality> 用一个标签,属性 resolution、lighting、composition、blur 描述画质。
|
||||
<key_selling_points> 下面每个卖点用一个 <point> 标签。
|
||||
|
||||
看不到或无法判断的内容,属性值填“无法判断”,布尔值填 false,不要留空标签。"""
|
||||
【人物属性硬性要求(has_person=true时必须遵守)】
|
||||
hair/skin_tone/face_shape/outfit四项绝对禁止填“无法判断”,必须基于图片可见特征给出具体中文描述:
|
||||
- hair:必须描述发型+发色,如“黑色齐肩直发”“棕色微卷中长发”“深棕色短发”
|
||||
- skin_tone:必须描述肤色,如“暖调自然肤色”“白皙肤色”“小麦色”
|
||||
- face_shape:必须描述脸型,如“鹅蛋脸”“圆脸”“瓜子脸”“方脸”
|
||||
- outfit:必须描述可见穿着,如“米色翻领衬衫”“白色T恤”“黑色连衣裙”
|
||||
即使局部被遮挡也要根据可见部分合理推断;确实看不清时按最接近的直观印象描述。
|
||||
|
||||
其他非人物属性看不到或无法判断时填“无法判断”,布尔值填false,不要留空标签。
|
||||
|
||||
【有人物场景输出参考(女性手持商品示例,必须写全10个属性,禁止省略)】
|
||||
<people has_person="true" count="1" gender="女" age_range="青年" hair="黑色齐肩直发" skin_tone="暖调自然肤色" face_shape="鹅蛋脸" outfit="米色翻领衬衫" pose="正面半身,手持商品" expression="面带微笑"/>"""
|
||||
|
||||
_IMAGE_ANALYSIS_USER = """请分析以下商品图片,共 {image_count} 张。
|
||||
所属行业:{industry}
|
||||
@@ -73,7 +84,7 @@ _IMAGE_ANALYSIS_EXAMPLE = """<products>
|
||||
<color hex="#D32F2F" name="红色" coverage="0.4"/>
|
||||
<color hex="#FFFFFF" name="白色" coverage="0.5"/>
|
||||
</colors>
|
||||
<people has_person="false" count="0" gender="无法判断" age_range="无法判断" pose="无法判断" expression="无法判断"/>
|
||||
<people has_person="false" count="0" gender="无法判断" age_range="无法判断" hair="无法判断" skin_tone="无法判断" face_shape="无法判断" outfit="无法判断" pose="无法判断" expression="无法判断"/>
|
||||
<mood>干净、实用</mood>
|
||||
<visible_text>
|
||||
<text_item text="多功能油污净" position="瓶身正面"/>
|
||||
@@ -132,7 +143,10 @@ _FUSION_SYSTEM = """你负责为短视频生成营销文案。请按思维链分
|
||||
<hook> 开头3秒钩子,5到15字。
|
||||
<body_points> 每个要点用一个 <point> 标签,属性 elaboration 是展开说明、image_index 是对应第几张图(从0开始),标签内容写要点。
|
||||
<cta> 口语化的行动号召。
|
||||
<script_segments> 每段配音用一个 <segment> 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案。
|
||||
<script_segments> 每段配音用一个 <segment> 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案(纯口播文本,不加旁白标注、不加镜头标注、不加"主播:"之类前缀)。
|
||||
<voiceover_script> 把所有 segment 的配音文案按顺序自然拼接成一段完整的纯口播文本(无标记、无括号、无前缀),长度要适配 {duration} 秒,约 {approx_chars} 字。
|
||||
<overview_theme> 视频主题(一句话概括)。
|
||||
<scene_and_lighting> 整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)。
|
||||
<word_count> 配音总字数,只写数字。
|
||||
<estimated_duration> 预计时长秒数,只写数字。
|
||||
|
||||
@@ -161,25 +175,38 @@ _FUSION_EXAMPLE = """<title>厨房重油污,别再用洗洁精硬擦了</title
|
||||
<segment duration_sec="6" image_index="0">后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</segment>
|
||||
<segment duration_sec="4" image_index="0">39块钱625ml,厨房重油污的可以试一瓶</segment>
|
||||
</script_segments>
|
||||
<voiceover_script>这油污我真的忍很久了,用洗洁精擦半天都没用。后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净。39块钱625ml,厨房重油污的可以试一瓶。</voiceover_script>
|
||||
<overview_theme>厨房油污清洁好物分享</overview_theme>
|
||||
<scene_and_lighting>简洁明亮的厨房台面场景,自然光从窗户洒入,色调温暖柔和,突出产品白色瓶身与去油污对比效果。</scene_and_lighting>
|
||||
<word_count>58</word_count>
|
||||
<estimated_duration>13</estimated_duration>"""
|
||||
|
||||
# ── 模板4:编导级分镜(LLM)────────────────────────────────────────────
|
||||
_STORYBOARD_SYSTEM = f"""你是短视频编导,负责把文案拆成可拍摄的分镜。
|
||||
_STORYBOARD_SYSTEM = """你是短视频编导,负责把文案拆成可拍摄的分镜,为 Seedance 2.5 视频模型写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。
|
||||
|
||||
工作方式:
|
||||
1. 按文案的 script_segments 顺序分配镜头。
|
||||
2. 每个镜头确定画面、运镜、时长、配音和字幕。
|
||||
2. 每个镜头确定景别/角度/运镜、画面场景与对白、人物动作细节、音效/BGM、转场。
|
||||
3. 检查所有镜头时长加起来接近目标时长,误差不超过2秒。
|
||||
4. image_index 必须在已上传图片范围内,第一张主图必须用在第一个镜头。
|
||||
|
||||
{GLOBAL_CONSTRAINTS}
|
||||
{fusion_instruction}
|
||||
|
||||
{global_constraints}
|
||||
|
||||
{negative_rules}
|
||||
|
||||
请严格按下面的标签格式输出,不要解释,不要用代码块:
|
||||
<clips> 下面每个镜头用一个 <clip> 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade、cut、zoom_in、slide_left、dissolve、wipe 之一、zoom 取 in、out 或 null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 <clip> 里面包含:
|
||||
<voice_text> 该镜头配音文本;
|
||||
<subtitle_text> 字幕文本,可与配音一致或更精简;
|
||||
<ken_burns> 用一个空标签,属性 start、end 写“x,y”坐标、ease 写缓动方式;不需要运镜时坐标相同。"""
|
||||
<clips> 下面每个镜头用一个 <clip> 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade/cut/zoom_in/slide_left/dissolve/wipe 之一、zoom 取 in/out/null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 <clip> 里面包含:
|
||||
<voice_text> 该镜头配音文本(纯口播文本,不加旁白标注);
|
||||
<subtitle_text> 字幕文本,可与配音一致或更精简;
|
||||
<shot_type_angle_movement> 景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜);
|
||||
<scene_and_dialogue> 画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔);
|
||||
<action_details> 人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示);
|
||||
<audio_bgm> 环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音);
|
||||
<transition> 硬切/淡入淡出/叠化(最后一镜写『结束』即可);
|
||||
<reference_image_index> 参考图片索引(0-based,对应第几张产品图,无则空);
|
||||
<ken_burns> 用一个空标签,属性 start、end 写"x,y"坐标、ease 写缓动方式;不需要运镜时坐标相同。"""
|
||||
|
||||
_STORYBOARD_USER = """目标时长:{duration}秒
|
||||
上传图片数量:{image_count}张(第1张是主图/封面)
|
||||
@@ -194,17 +221,35 @@ _STORYBOARD_EXAMPLE = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="3" bgm_note="日常、轻微烦躁">
|
||||
<voice_text>这油污我真的忍很久了</voice_text>
|
||||
<subtitle_text>这油污忍很久了</subtitle_text>
|
||||
<shot_type_angle_movement>近景俯拍45度,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>厨房台面,主妇皱眉看着灶台油污。对白:这油污我真的忍很久了</scene_and_dialogue>
|
||||
<action_details>右手拿着脏抹布,无奈摇头</action_details>
|
||||
<audio_bgm>轻快日常BGM,带一点烦躁感</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="zoom_in" zoom="in" duration_sec="6" bgm_note="轻快、出现转机">
|
||||
<voice_text>后来换了大公鸡头油污净,喷上等几分钟,一擦就干净</voice_text>
|
||||
<subtitle_text>喷上等几分钟,一擦就干净</subtitle_text>
|
||||
<shot_type_angle_movement>特写平视,固定镜头</shot_type_angle_movement>
|
||||
<scene_and_dialogue>手部特写,喷油污净在油污处。对白:后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</scene_and_dialogue>
|
||||
<action_details>左手拿产品瓶身,右手按压喷头,等待片刻后用抹布轻擦</action_details>
|
||||
<audio_bgm>轻快转折BGM,带清爽感</audio_bgm>
|
||||
<transition>淡入淡出</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="20,20" end="80,80" ease="ease-in-out"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="fade" zoom="null" duration_sec="4" bgm_note="温暖、推荐">
|
||||
<voice_text>39块钱625ml,厨房重油污的可以试一瓶</voice_text>
|
||||
<subtitle_text>39元625ml,可以试一瓶</subtitle_text>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
<shot_type_angle_movement>中景平视,缓慢拉镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>产品正面展示,明亮背景。对白:39块钱625ml,厨房重油污的可以试一瓶</scene_and_dialogue>
|
||||
<action_details>产品置于画面中央,轻微转动展示瓶身</action_details>
|
||||
<audio_bgm>温暖收尾BGM</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="50,50" end="20,20" ease="ease-in-out"/>
|
||||
</clip>
|
||||
</clips>"""
|
||||
|
||||
|
||||
@@ -103,7 +103,7 @@ class SharedSettings(BaseSettings):
|
||||
doubao_vision_lite_model: str = (
|
||||
"doubao-seed-2-1-lite-260915" # 快速视觉(Seed 2.1 Lite 原生多模态;原 vision-lite-250315 不可用)
|
||||
)
|
||||
doubao_vision_use_lite: bool = False # #2181: lite视觉模型100%超时,默认关闭走pro(25-38s稳定返回)
|
||||
doubao_vision_use_lite: bool = True # #2188: lite恢复稳定,爆款视频默认lite-first提速(20-30s)
|
||||
doubao_embedding_model: str = "doubao-embedding-vision-251215" # 多模态向量化(原 large-text-240915 已 Retiring)
|
||||
doubao_video_model: str = "doubao-seedance-2-5-260628"
|
||||
doubao_video_timeout: int = 600 # 视频生成轮询总超时(秒)
|
||||
|
||||
@@ -585,7 +585,7 @@ class DoubaoClient:
|
||||
# #2172/#2174: 信任链——使用预热好的 Seedream t2i 文生图(纯模型生成人像,是方舟信任产物,
|
||||
# 不会触发肖像审核)。预热在 VLM 分析后由 daemon 线程后台完成,结果通过 pre_trusted_images 传入。
|
||||
# - 预热结果有效 → 替换原参考图,走 omni_ref 模式
|
||||
# - 预热结果不可用 → 直接用原图(若被400肖像拦截,#2166自动降级纯t2v),避免现场跑t2i阻塞渲染
|
||||
# - 预热结果不可用 → 直接用原图(上层 _step_render 已现场同步跑 Seedream t2i 兜底;若再被400拦截,下方自动降级纯t2v)
|
||||
# 信任链只作用于 doubao provider;DashScope(Wan) 保持原行为。
|
||||
trust_chain_applied = False
|
||||
if provider == "doubao" and getattr(self, "trust_chain_enabled", True) and pre_trusted_images:
|
||||
@@ -763,6 +763,26 @@ class DoubaoClient:
|
||||
# 保留第二次的错误信息
|
||||
sc, body = sc2, body2
|
||||
|
||||
# #2183: portrait_intercept / 真人肖像审核拦截 → 去掉所有参考图(含image_url首帧),纯 t2v 重试一次
|
||||
# 信任链预热或现场 t2i 都失败时的最后兜底,保证能出片
|
||||
if not task_id and sc == 400:
|
||||
_err_code_for_400, _ = _classify_video_error(sc, body, last_err)
|
||||
if _err_code_for_400 == "portrait_intercept" and (image_url or ref_imgs):
|
||||
logger.warning(
|
||||
"Seedance 创建因 portrait_intercept 失败,降级纯 t2v(移除所有参考图)重试: img=%d ref=%d",
|
||||
1 if image_url else 0,
|
||||
len(ref_imgs),
|
||||
)
|
||||
_t2v_payload = dict(create_payload)
|
||||
_t2v_payload["content"] = [{"type": "text", "text": prompt.strip()}]
|
||||
_t2v_payload["ratio"] = ratio or "9:16"
|
||||
task_id, last_err, sc3, body3 = _do_create(_t2v_payload)
|
||||
if task_id:
|
||||
sc, body = sc3, body3
|
||||
logger.info("[trust-chain] portrait_intercept 降级纯 t2v 成功 task_id=%s", task_id)
|
||||
else:
|
||||
sc, body = sc3, body3
|
||||
|
||||
if not task_id:
|
||||
err_code, user_msg = _classify_video_error(sc, body, last_err)
|
||||
self.last_video_error = {
|
||||
|
||||
@@ -403,33 +403,30 @@ class TestViralVideoPipeline:
|
||||
"""v1.6: _step_script_generation 返回 dict 形式的 CopyResult,含 voiceover_script + shots。"""
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_script_generation
|
||||
|
||||
mock_llm.return_value = {
|
||||
"overview": {"theme": "口红推荐", "total_duration": 15, "aspect_ratio": "9:16"},
|
||||
"scene_and_lighting": "明亮化妆台,柔和自然光",
|
||||
"shots": [
|
||||
{
|
||||
"time_range": "0-5秒",
|
||||
"shot_type_angle_movement": "近景平视,缓慢推镜",
|
||||
"scene_and_dialogue": "女主微笑展示口红:大家好,今天分享一款口红",
|
||||
"action_details": "手持口红特写",
|
||||
"audio_bgm": "轻快流行BGM",
|
||||
"transition": "硬切",
|
||||
"reference_image_index": 0,
|
||||
},
|
||||
{
|
||||
"time_range": "5-15秒",
|
||||
"shot_type_angle_movement": "特写,固定镜头",
|
||||
"scene_and_dialogue": "涂抹口红:颜色特别好看很显白",
|
||||
"action_details": "嘴唇涂抹特写",
|
||||
"audio_bgm": "轻快BGM继续",
|
||||
"transition": "结束",
|
||||
"reference_image_index": 1,
|
||||
},
|
||||
],
|
||||
"hard_constraints": ["无字幕无水印"],
|
||||
"negative_prompts": ["字幕", "水印"],
|
||||
"voiceover_script": "大家好,今天分享一款口红,颜色特别好看很显白。",
|
||||
}
|
||||
mock_llm.return_value = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="5" bgm_note="轻快流行BGM">
|
||||
<voice_text>大家好,今天分享一款口红</voice_text>
|
||||
<subtitle_text>大家好,今天分享一款口红</subtitle_text>
|
||||
<shot_type_angle_movement>近景平视,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>女主微笑展示口红:大家好,今天分享一款口红</scene_and_dialogue>
|
||||
<action_details>手持口红特写</action_details>
|
||||
<audio_bgm>轻快流行BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="fade" zoom="null" duration_sec="10" bgm_note="轻快BGM">
|
||||
<voice_text>颜色特别好看很显白</voice_text>
|
||||
<subtitle_text>颜色特别好看很显白</subtitle_text>
|
||||
<shot_type_angle_movement>特写,固定镜头</shot_type_angle_movement>
|
||||
<scene_and_dialogue>涂抹口红:颜色特别好看很显白</scene_and_dialogue>
|
||||
<action_details>嘴唇涂抹特写</action_details>
|
||||
<audio_bgm>轻快BGM继续</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>1</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
</clips>"""
|
||||
result = _step_script_generation(
|
||||
mock_job, {"intent": "推广口红", "key_messages": [], "tone": "亲切"}, {"products": []}
|
||||
)
|
||||
@@ -452,12 +449,13 @@ class TestViralVideoPipeline:
|
||||
assert result["voiceover_script"]
|
||||
assert len(result["shots"]) >= 1
|
||||
|
||||
@patch("packages.shared.ai_service.call_llm")
|
||||
def test_review_pass_v16(self, mock_llm, mock_job):
|
||||
@patch("packages.application.viral_video.reviewer.Reviewer.review")
|
||||
def test_review_pass_v16(self, mock_review, mock_job):
|
||||
"""v1.6 _step_review 接收 copy_result dict。"""
|
||||
from apps.worker.worker_app.tasks.viral_video import _step_review
|
||||
from packages.application.viral_video.reviewer import ReviewIssue, ReviewResult
|
||||
|
||||
mock_llm.return_value = {"passed": True, "score": 90, "details": {}}
|
||||
mock_review.return_value = ReviewResult(passed=True, score=90, issues=[], rewrite_suggestions=[])
|
||||
cr = {"voiceover_script": "大家好", "shots": []}
|
||||
result = _step_review(mock_job, cr)
|
||||
assert result["passed"] is True
|
||||
|
||||
@@ -0,0 +1,302 @@
|
||||
"""#2040 接线集成测试:验证运行中的 viral_video 任务使用 prompt_loader 从 DB 读取模板。
|
||||
|
||||
mock LLM/Vision 调用,验证:
|
||||
1. image_analysis 走 loader 模板 + XML 解析
|
||||
2. intent_parsing 走 loader 模板 + XML 解析
|
||||
3. script_generation 走 storyboard 模板 + XML 解析,输出兼容 Seedance 的 copy_result
|
||||
4. review 走 Reviewer(review 模板)带自动重写
|
||||
5. 三档融合(ai_full / ai_polish / user_primary)注入不同 FUSION_INSTRUCTIONS
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path as _Path
|
||||
|
||||
_WORKER_ROOT = _Path(__file__).resolve().parents[2] / "apps" / "worker"
|
||||
if str(_WORKER_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(_WORKER_ROOT))
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from packages.domain.viral_video import ViralVideoJob
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def job():
|
||||
j = ViralVideoJob(
|
||||
user_id="u1",
|
||||
images=["https://img/1.jpg", "https://img/2.jpg"],
|
||||
industry="美妆",
|
||||
duration=15,
|
||||
user_copy_text="这款口红真的太绝了,显白又持久,姐妹们冲!",
|
||||
fusion_level="ai_polish",
|
||||
)
|
||||
return j
|
||||
|
||||
|
||||
# ── Mock LLM/Vision 返回的 XML 文本 ─────────────────────────────────
|
||||
|
||||
IMAGE_XML = """
|
||||
<analysis>
|
||||
<scene>室内桌面拍摄,柔和自然光</scene>
|
||||
<mood>清新温暖</mood>
|
||||
<product name="lipstick" brand="品牌X" category="唇部彩妆"
|
||||
appearance="管状红色膏体" packaging="黑色金属管"
|
||||
features="显白,持久,滋润" portrait_prompt="无人像"
|
||||
summary="品牌X红色口红">
|
||||
<text_on_package>品牌X,211</text_on_package>
|
||||
</product>
|
||||
</analysis>
|
||||
""".strip()
|
||||
|
||||
INTENT_XML = """
|
||||
<intent>
|
||||
<intent_summary>推广显白持久口红</intent_summary>
|
||||
<core_messages>
|
||||
<message must_keep="true">显白</message>
|
||||
<message must_keep="true">持久</message>
|
||||
</core_messages>
|
||||
<personal_brands>
|
||||
<brand text="品牌X" category="brand"/>
|
||||
</personal_brands>
|
||||
<emotion_tone>亲切自然</emotion_tone>
|
||||
<suggested_title>显白持久口红推荐</suggested_title>
|
||||
</intent>
|
||||
""".strip()
|
||||
|
||||
STORYBOARD_XML = """
|
||||
<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="5" bgm_note="轻快BGM">
|
||||
<voice_text>这款口红真的太绝了</voice_text>
|
||||
<subtitle_text>显白又持久</subtitle_text>
|
||||
<shot_type_angle_movement>近景俯拍45度,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>厨房台面,主妇展示口红。对白:这款口红真的太绝了</scene_and_dialogue>
|
||||
<action_details>右手持口红展示膏体</action_details>
|
||||
<audio_bgm>轻快BGM</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
</clips>
|
||||
""".strip()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def invalidate_loader_cache():
|
||||
from packages.application.viral_video import prompt_loader as pl
|
||||
|
||||
pl.invalidate()
|
||||
yield
|
||||
pl.invalidate()
|
||||
|
||||
|
||||
# ── 1) 图片分析走模板 ───────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestImageAnalysisWiring:
|
||||
def test_uses_loader_template_and_xml_parse(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
with patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML) as mock_v:
|
||||
result = vv._analyze_single_image(0, "https://img/1.jpg", "vlm-lite", 15)
|
||||
|
||||
mock_v.assert_called_once()
|
||||
# 验证调用时传入了 system_prompt(说明走了 loader 渲染的模板)
|
||||
call_kwargs = mock_v.call_args.kwargs
|
||||
assert "system_prompt" in call_kwargs and call_kwargs["system_prompt"]
|
||||
# 结果包含从 XML 解析出的产品信息
|
||||
assert result["name"] == "lipstick"
|
||||
assert result["brand"] == "品牌X"
|
||||
assert "显白" in result["key_features"]
|
||||
assert result["text_on_package"] == ["品牌X", "211"]
|
||||
|
||||
|
||||
# ── 2) 意图解析走模板 ───────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestIntentParsingWiring:
|
||||
def test_uses_loader_and_parses_xml(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
img_result = {"products": [{"name": "lipstick", "brand": "品牌X", "key_features": ["显白", "持久"]}]}
|
||||
with patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML) as mock_llm:
|
||||
result = vv._step_intent_parsing(job, img_result)
|
||||
|
||||
mock_llm.assert_called_once()
|
||||
assert result["intent"] == "推广显白持久口红"
|
||||
assert "显白" in result["key_messages"]
|
||||
assert result["suggested_title"] == "显白持久口红推荐"
|
||||
|
||||
|
||||
# ── 3) 脚本生成:storyboard 模板 + XML 解析 + fusion_level 注入 ────
|
||||
|
||||
|
||||
class TestScriptGenerationWiring:
|
||||
@pytest.mark.parametrize("level", ["ai_full", "ai_polish", "user_primary"])
|
||||
def test_fusion_level_injected(self, job, level):
|
||||
"""三档融合水平被注入到 storyboard 模板的 system_prompt"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.prompts import FUSION_INSTRUCTIONS
|
||||
|
||||
job.fusion_level = level
|
||||
intent = {"intent": "推广", "key_messages": ["显白"], "tone": "亲切"}
|
||||
|
||||
captured_system = {}
|
||||
|
||||
def fake_call_llm(messages, **kw):
|
||||
captured_system["final"] = messages[0]["content"]
|
||||
return STORYBOARD_XML
|
||||
|
||||
with patch("packages.shared.ai_service.call_llm", side_effect=fake_call_llm):
|
||||
result = vv._step_script_generation(job, intent, {})
|
||||
|
||||
# fusion_level 对应的指令文本被注入到 system prompt 中
|
||||
assert FUSION_INSTRUCTIONS[level] in captured_system["final"], f"fusion_level {level} 指令未注入 system_prompt"
|
||||
# 输出保持 Seedance 兼容结构
|
||||
assert "overview" in result
|
||||
assert "shots" in result
|
||||
assert len(result["shots"]) >= 1
|
||||
assert result["shots"][0]["shot_type_angle_movement"]
|
||||
assert result["voiceover_script"]
|
||||
|
||||
def test_fallback_when_xml_and_json_unparseable(self, job):
|
||||
"""XML 解析失败且无法解析为 JSON 时,回退到兜底脚本"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
job.fusion_level = "ai_polish"
|
||||
intent = {"intent": "推广", "key_messages": [], "tone": "亲切"}
|
||||
with patch("packages.shared.ai_service.call_llm", return_value="not xml not json"):
|
||||
result = vv._step_script_generation(job, intent, {})
|
||||
assert isinstance(result, dict)
|
||||
assert "voiceover_script" in result
|
||||
assert "shots" in result
|
||||
|
||||
|
||||
# ── 4) Review 使用 Reviewer + 自动重写 ─────────────────────────────
|
||||
|
||||
|
||||
class TestReviewWiring:
|
||||
def test_pass_path(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
|
||||
copy_result = {
|
||||
"title": "口红推荐",
|
||||
"overview": {"theme": "口红推荐"},
|
||||
"voiceover_script": "这款口红显白又持久",
|
||||
"shots": [{"scene_and_dialogue": "展示口红"}],
|
||||
}
|
||||
job.intent_result = {"key_messages": ["显白", "持久"], "intent": "推广"}
|
||||
|
||||
pass_result = ReviewResult(passed=True, score=90, issues=[], rewrite_suggestions=[])
|
||||
with patch.object(Reviewer, "review", return_value=pass_result):
|
||||
out = vv._step_review(job, copy_result)
|
||||
assert out["passed"] is True
|
||||
|
||||
def test_rewrite_path(self, job):
|
||||
"""审核不通过时触发自动重写,并更新 job.copy_result"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
from packages.application.viral_video.schemas import FusionResult, ReviewIssue, ScriptSegment
|
||||
|
||||
copy_result = {
|
||||
"title": "原标题",
|
||||
"overview": {"theme": "原标题"},
|
||||
"voiceover_script": "这款口红绝了",
|
||||
"shots": [{"scene_and_dialogue": "展示"}],
|
||||
}
|
||||
job.intent_result = {"key_messages": ["显白"], "intent": "推广"}
|
||||
|
||||
fail_result = ReviewResult(
|
||||
passed=False,
|
||||
score=50,
|
||||
issues=[ReviewIssue(dimension="违规词", severity="high", location="开头", text="绝了")],
|
||||
rewrite_suggestions=["去掉夸大词"],
|
||||
)
|
||||
rewritten = FusionResult(
|
||||
title="新标题",
|
||||
hook="修改后钩子",
|
||||
script_segments=[ScriptSegment(text="修改后口播正文")],
|
||||
cta="行动号召",
|
||||
word_count=10,
|
||||
estimated_duration=10,
|
||||
)
|
||||
pass_after = ReviewResult(passed=True, score=88, issues=[], rewrite_suggestions=[])
|
||||
|
||||
with (
|
||||
patch.object(Reviewer, "review", side_effect=[fail_result, pass_after]),
|
||||
patch.object(Reviewer, "rewrite", return_value=rewritten),
|
||||
):
|
||||
out = vv._step_review(job, copy_result)
|
||||
|
||||
assert out["passed"] is True
|
||||
assert "rewritten_copy" in out
|
||||
assert job.generated_copy_text == "修改后口播正文"
|
||||
|
||||
|
||||
# ── 5) 端到端:每个 step 调用 loader 对应 prompt_type ──────────────
|
||||
|
||||
|
||||
class TestEndToEndLoaderUsed:
|
||||
def test_each_step_calls_loader(self, job):
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
from packages.application.viral_video import prompt_loader as pl
|
||||
|
||||
called_types = []
|
||||
real_get = pl.get_template
|
||||
|
||||
def spy_get(prompt_type, **kwargs):
|
||||
called_types.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get),
|
||||
patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML),
|
||||
patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML),
|
||||
):
|
||||
# 1) image
|
||||
img_res = vv._analyze_single_image(0, "https://img/1.jpg", "vlm", 15)
|
||||
# 2) intent
|
||||
intent_res = vv._step_intent_parsing(job, {"products": [img_res]})
|
||||
|
||||
# 前两步分别调用了 image_analysis 和 intent_parsing
|
||||
assert "image_analysis" in called_types
|
||||
assert "intent_parsing" in called_types
|
||||
|
||||
# script 和 review 单独验证(需要不同的 LLM 返回)
|
||||
called_types_2 = []
|
||||
|
||||
def spy_get_2(prompt_type, **kwargs):
|
||||
called_types_2.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get_2),
|
||||
patch("packages.shared.ai_service.call_llm", return_value=STORYBOARD_XML),
|
||||
):
|
||||
copy_res = vv._step_script_generation(job, intent_res, {"products": [img_res]})
|
||||
assert "storyboard" in called_types_2
|
||||
|
||||
called_types_3 = []
|
||||
|
||||
def spy_get_3(prompt_type, **kwargs):
|
||||
called_types_3.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
from packages.application.viral_video.reviewer import Reviewer, ReviewResult
|
||||
|
||||
pass_result = ReviewResult(passed=True, score=90, issues=[], rewrite_suggestions=[])
|
||||
job.intent_result = intent_res
|
||||
job.copy_result = copy_res
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get_3),
|
||||
patch.object(Reviewer, "review", return_value=pass_result) as mock_review,
|
||||
):
|
||||
review_res = vv._step_review(job, copy_res)
|
||||
# review 步骤内部直接调用 Reviewer.review,该方法被 mock,因此 get_template 不会被调用;
|
||||
# 此处验证 Reviewer.review 被调用即可说明 review 步骤走通了。
|
||||
assert mock_review.called, "_step_review 未调用 Reviewer.review"
|
||||
assert isinstance(review_res, dict) and "passed" in review_res
|
||||
Reference in New Issue
Block a user