Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e672c17eb2 | |||
| fb0e4989cd | |||
| 702f09e6b7 | |||
| 12b0d15473 | |||
| 9344314eac | |||
| 3f49867384 | |||
| dd420c556f | |||
| 2d823a9255 |
@@ -30,7 +30,6 @@ import { uploadAssetDirect, getAssetLibraries, getAssetsByKind, type AssetItem }
|
||||
import { fetchPresetVoices } from "@/api/voices"
|
||||
import { getVoiceClones, getVoiceClonePreview } from "@/api/voice-clone"
|
||||
import type { VoiceClone } from "@/api/voice-clone"
|
||||
import DurationWheelPicker from "@/components/common/DurationWheelPicker"
|
||||
import {
|
||||
FUSION_LEVELS,
|
||||
STYLE_STRENGTHS,
|
||||
@@ -224,6 +223,7 @@ const RATIOS = [
|
||||
{ v: "16:9", label: "16:9 横屏(B站/YouTube)" },
|
||||
{ v: "1:1", label: "1:1 方形(小红书)" },
|
||||
]
|
||||
const DURATIONS = Array.from({ length: 16 }, (_, i) => 15 + i)
|
||||
/** 兜底模型列表(接口未返回时使用,字段与 ViralVideoModel 对齐;后端返回后自动覆盖) */
|
||||
const FALLBACK_VIDEO_MODELS: ViralVideoModel[] = [
|
||||
{
|
||||
@@ -2433,12 +2433,12 @@ const ViralVideoPage: React.FC = () => {
|
||||
</div>
|
||||
<div className="vv-form-row">
|
||||
<label className="vv-label">文案视频时长</label>
|
||||
<DurationWheelPicker
|
||||
<Select
|
||||
className="vv-select"
|
||||
style={{ width: "100%" }}
|
||||
value={task.duration}
|
||||
min={15}
|
||||
max={30}
|
||||
step={1}
|
||||
onChange={(v) => setTask({ duration: v })}
|
||||
options={DURATIONS.map((n) => ({ value: n, label: `${n}秒` }))}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -24,7 +24,6 @@ from __future__ import annotations
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
@@ -347,6 +346,7 @@ def _normalize_image_url(raw: str, idx: int) -> str:
|
||||
storage_key = url.lstrip("/")
|
||||
try:
|
||||
from packages.shared.storage import get_storage_service
|
||||
|
||||
url = get_storage_service().get_url(storage_key)
|
||||
except Exception as _e:
|
||||
raise ValueError(f"图片 #{idx} storage_key={storage_key!r} 转公网URL失败: {_e}") from _e
|
||||
@@ -386,6 +386,7 @@ def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
# 整个阶段关闭底层 httpx 重试,避免线程里出现不可控等待
|
||||
try:
|
||||
from packages.shared.ai_client import get_doubao_client as _gdc
|
||||
|
||||
_cli = _gdc()
|
||||
_orig_retries = _cli.max_retries
|
||||
_cli.max_retries = 0
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 图片分析:火山OCR专用API + doubao-lite强约束JSON并行,单次pro VLM兜底。"""
|
||||
|
||||
from .fast_path import analyze_image_v2, analyze_images_v2 # noqa: F401
|
||||
|
||||
@@ -22,15 +22,22 @@ logger = logging.getLogger(__name__)
|
||||
# 可通过环境变量调参(有默认值,无需配置即可跑)
|
||||
_IMG_WORKERS = int(os.environ.get("VISION_V2_IMG_WORKERS", "8"))
|
||||
_FAST_TIMEOUT = float(os.environ.get("VISION_V2_FAST_TIMEOUT", "8"))
|
||||
_FAST_JSON_TIMEOUT = float(os.environ.get("VISION_V2_FAST_JSON_TIMEOUT", "6"))
|
||||
_FAST_JSON_TIMEOUT = float(os.environ.get("VISION_V2_FAST_JSON_TIMEOUT", "8"))
|
||||
_OCR_TIMEOUT = float(os.environ.get("VISION_V2_OCR_TIMEOUT", "6"))
|
||||
_PRO_TIMEOUT = float(os.environ.get("VISION_V2_PRO_TIMEOUT", "45"))
|
||||
|
||||
_FALLBACK_RESULT = {
|
||||
"name": "未识别", "brand": "无法判断", "category": "非产品图",
|
||||
"appearance": "无法判断", "packaging": "无法判断", "text_on_package": [],
|
||||
"key_features": ["无法判断"], "scene": "通用", "mood": "",
|
||||
"portrait_prompt": "无法判断", "summary": "未识别",
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "非产品图",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": "通用",
|
||||
"mood": "",
|
||||
"portrait_prompt": "无法判断",
|
||||
"summary": "未识别",
|
||||
}
|
||||
|
||||
|
||||
@@ -55,16 +62,23 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
with ThreadPoolExecutor(max_workers=2) as pool:
|
||||
f_fj = pool.submit(vlm_fast_json.call_fast_json, img_url, timeout=_FAST_JSON_TIMEOUT)
|
||||
f_ocr = pool.submit(ocr_volc.call_ocr, img_url, timeout=_OCR_TIMEOUT)
|
||||
for fut in as_completed([f_fj, f_ocr], timeout=_FAST_TIMEOUT + 2):
|
||||
try:
|
||||
res = fut.result(timeout=1)
|
||||
except Exception as e:
|
||||
logger.warning("[vision.v2] 图片 #%d 子任务异常: %s", idx, e)
|
||||
continue
|
||||
if fut is f_fj and isinstance(res, dict):
|
||||
fj_result = res
|
||||
elif fut is f_ocr and isinstance(res, list):
|
||||
ocr_result = res
|
||||
try:
|
||||
for fut in as_completed([f_fj, f_ocr], timeout=_FAST_TIMEOUT):
|
||||
try:
|
||||
res = fut.result(timeout=1)
|
||||
except Exception as e:
|
||||
logger.warning("[vision.v2] 图片 #%d 子任务异常: %s", idx, e)
|
||||
continue
|
||||
if fut is f_fj and isinstance(res, dict):
|
||||
fj_result = res
|
||||
elif fut is f_ocr and isinstance(res, list):
|
||||
ocr_result = res
|
||||
except TimeoutError:
|
||||
# fast 整体超时,取消还没跑完的子任务,继续走 pro 兜底
|
||||
for f in (f_fj, f_ocr):
|
||||
if not f.done():
|
||||
f.cancel()
|
||||
logger.warning("[vision.v2] 图片 #%d fast路径超时(%.0fs),走pro兜底", idx, _FAST_TIMEOUT)
|
||||
|
||||
fast_elapsed = time.time() - t0
|
||||
|
||||
@@ -75,7 +89,9 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
assembled["_fast_elapsed"] = round(fast_elapsed, 2)
|
||||
logger.info(
|
||||
"[vision.v2] 图片 #%d fast命中 elapsed=%.2fs pp=%s",
|
||||
idx, fast_elapsed, (assembled.get("portrait_prompt") or "")[:40],
|
||||
idx,
|
||||
fast_elapsed,
|
||||
(assembled.get("portrait_prompt") or "")[:40],
|
||||
)
|
||||
return assembled
|
||||
|
||||
@@ -88,11 +104,11 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
pro_result["_pro_elapsed"] = round(time.time() - pro_t0, 2)
|
||||
if ocr_result and not pro_result.get("text_on_package"):
|
||||
pro_result["text_on_package"] = ocr_result[:8]
|
||||
logger.info("[vision.v2] 图片 #%d pro兜底命中 total=%.2fs", idx, time.time()-t0)
|
||||
logger.info("[vision.v2] 图片 #%d pro兜底命中 total=%.2fs", idx, time.time() - t0)
|
||||
return pro_result
|
||||
|
||||
# 最终:返回最小可用结果
|
||||
logger.warning("[vision.v2] 图片 #%d 全路径失败 elapsed=%.2fs", idx, time.time()-t0)
|
||||
logger.warning("[vision.v2] 图片 #%d 全路径失败 elapsed=%.2fs", idx, time.time() - t0)
|
||||
out = dict(_FALLBACK_RESULT)
|
||||
out["_source"] = "v2_all_failed"
|
||||
out["text_on_package"] = ocr_result[:8]
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
|
||||
设计原则:简单、直接、无竞速、无复杂超时逻辑。只在 fast_json 结果不可用时调用。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
@@ -85,26 +86,48 @@ def _xml_to_product(raw: str, idx: int) -> dict[str, Any]:
|
||||
if pp_attr and pp_attr != "无人像":
|
||||
portrait_prompt = pp_attr
|
||||
return {
|
||||
"name": name, "brand": brand, "category": category,
|
||||
"appearance": appearance, "packaging": packaging, "text_on_package": text_list,
|
||||
"key_features": feat_list, "scene": scene, "mood": mood,
|
||||
"portrait_prompt": portrait_prompt, "summary": summary,
|
||||
"name": name,
|
||||
"brand": brand,
|
||||
"category": category,
|
||||
"appearance": appearance,
|
||||
"packaging": packaging,
|
||||
"text_on_package": text_list,
|
||||
"key_features": feat_list,
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": portrait_prompt,
|
||||
"summary": summary,
|
||||
"_source": "vlm_pro_xml",
|
||||
}
|
||||
|
||||
if portrait_prompt != "无人像":
|
||||
return {
|
||||
"name": "未识别", "brand": "无法判断", "category": "无法判断",
|
||||
"appearance": "无法判断", "packaging": "无法判断", "text_on_package": [],
|
||||
"key_features": ["无法判断"], "scene": scene, "mood": mood,
|
||||
"portrait_prompt": portrait_prompt, "summary": "未识别",
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "无法判断",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": portrait_prompt,
|
||||
"summary": "未识别",
|
||||
"_source": "vlm_pro_no_product",
|
||||
}
|
||||
return {
|
||||
"name": "未识别", "brand": "无法判断", "category": "无法判断",
|
||||
"appearance": "无法判断", "packaging": "无法判断", "text_on_package": [],
|
||||
"key_features": ["无法判断"], "scene": scene, "mood": mood,
|
||||
"portrait_prompt": "无人像", "summary": "未识别", "_source": "vlm_pro_no_tag",
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "无法判断",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": "无人像",
|
||||
"summary": "未识别",
|
||||
"_source": "vlm_pro_no_tag",
|
||||
}
|
||||
|
||||
|
||||
@@ -141,6 +164,8 @@ def call_pro_vlm(
|
||||
return None
|
||||
|
||||
use_model = model or DEFAULT_PRO_MODEL
|
||||
_orig_retries = client.max_retries
|
||||
client.max_retries = 0
|
||||
try:
|
||||
raw = client.vision_completion(
|
||||
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
@@ -151,8 +176,10 @@ def call_pro_vlm(
|
||||
model=use_model,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("[vision.vlm] 图片 #%d pro VLM 调用失败 elapsed=%.1fs err=%s", idx, time.time()-t0, e)
|
||||
logger.warning("[vision.vlm] 图片 #%d pro VLM 调用失败 elapsed=%.1fs err=%s", idx, time.time() - t0, e)
|
||||
client.max_retries = _orig_retries
|
||||
return None
|
||||
client.max_retries = _orig_retries
|
||||
|
||||
elapsed = time.time() - t0
|
||||
if not raw:
|
||||
@@ -163,7 +190,7 @@ def call_pro_vlm(
|
||||
l, r = text.find("{"), text.rfind("}")
|
||||
if l >= 0 and r > l:
|
||||
try:
|
||||
obj = json.loads(text[l:r+1])
|
||||
obj = json.loads(text[l : r + 1])
|
||||
if isinstance(obj, dict):
|
||||
logger.info("[vision.vlm] 图片 #%d pro VLM JSON 完成 elapsed=%.1fs", idx, elapsed)
|
||||
return {
|
||||
@@ -189,7 +216,9 @@ def call_pro_vlm(
|
||||
result["_pro_elapsed"] = round(elapsed, 2)
|
||||
logger.info(
|
||||
"[vision.vlm] 图片 #%d pro VLM XML 完成 elapsed=%.2fs pp=%s",
|
||||
idx, elapsed, (result.get("portrait_prompt") or "")[:40],
|
||||
idx,
|
||||
elapsed,
|
||||
(result.get("portrait_prompt") or "")[:40],
|
||||
)
|
||||
return result
|
||||
except Exception as e:
|
||||
|
||||
@@ -81,28 +81,92 @@ def call_fast_json(
|
||||
) -> dict[str, Any] | None:
|
||||
"""调用 lite VLM 返回结构化 dict;失败/非 JSON 返回 None。
|
||||
|
||||
注意:不做重试(外层竞速/降级逻辑负责),max_retries=0 由外层统一设置。
|
||||
直接用 httpx 发最小 payload(关闭 thinking),不走 ai_client 包装:
|
||||
- 关闭 thinking/推理链(reasoning_tokens 是延迟主因,单次要10-12s)
|
||||
- 单次调用不重试(失败由外层走 pro 兜底)
|
||||
- 温度=0.1 稳定输出 JSON
|
||||
"""
|
||||
t0 = time.time()
|
||||
try:
|
||||
from packages.shared.ai_client import get_doubao_client
|
||||
import httpx
|
||||
|
||||
client = get_doubao_client()
|
||||
if not client.is_available:
|
||||
logger.warning("[vision.v2] doubao client 不可用,跳过 fast_json")
|
||||
try:
|
||||
from packages.shared import get_shared_settings
|
||||
|
||||
settings = get_shared_settings()
|
||||
api_key = settings.doubao_api_key
|
||||
base_url = (settings.doubao_base_url or "https://ark.cn-beijing.volces.com/api/v3").rstrip("/")
|
||||
if not api_key:
|
||||
logger.warning("[vision.v2] doubao api_key 未配置,跳过 fast_json")
|
||||
return None
|
||||
|
||||
use_model = model or DEFAULT_LITE_MODEL
|
||||
raw = client.vision_completion(
|
||||
messages=[
|
||||
url = f"{base_url}/chat/completions"
|
||||
payload: dict[str, Any] = {
|
||||
"model": use_model,
|
||||
"messages": [
|
||||
{"role": "system", "content": _FAST_SYSTEM},
|
||||
{"role": "user", "content": _FAST_USER},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "image_url", "image_url": {"url": img_url}},
|
||||
{"type": "text", "text": _FAST_USER},
|
||||
],
|
||||
},
|
||||
],
|
||||
images=[img_url],
|
||||
temperature=0.1,
|
||||
max_tokens=max_tokens,
|
||||
"temperature": 0.1,
|
||||
"max_tokens": max_tokens,
|
||||
"stream": False,
|
||||
}
|
||||
# 关键:关闭 thinking(避免产生 reasoning_tokens 拖慢响应)
|
||||
# 方舟/豆包 2.x 模型支持 thinking.type=disabled
|
||||
try:
|
||||
payload["thinking"] = {"type": "disabled"}
|
||||
except Exception:
|
||||
pass
|
||||
# 部分模型用 reasoning_effort 控制思考深度
|
||||
payload["reasoning_effort"] = "low"
|
||||
|
||||
resp = httpx.post(
|
||||
url,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||
json=payload,
|
||||
timeout=timeout,
|
||||
model=use_model,
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code != 200:
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json HTTP %d elapsed=%.1fs body=%s", resp.status_code, elapsed, resp.text[:200]
|
||||
)
|
||||
# 如果400说明不支持thinking参数,降级重试一次
|
||||
if resp.status_code == 400 and "thinking" in resp.text.lower():
|
||||
payload.pop("thinking", None)
|
||||
payload.pop("reasoning_effort", None)
|
||||
time.time()
|
||||
resp = httpx.post(
|
||||
url,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||
json=payload,
|
||||
timeout=timeout,
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code != 200:
|
||||
logger.warning("[vision.v2] fast_json 降级后 HTTP %d elapsed=%.1fs", resp.status_code, elapsed)
|
||||
return None
|
||||
else:
|
||||
return None
|
||||
data = resp.json()
|
||||
raw = (data.get("choices") or [{}])[0].get("message", {}).get("content")
|
||||
if raw is None:
|
||||
logger.warning("[vision.v2] fast_json 返回 None elapsed=%.1fs", elapsed)
|
||||
return None
|
||||
usage = data.get("usage") or {}
|
||||
logger.info(
|
||||
"[vision.v2] fast_json 直连完成 model=%s elapsed=%.1fs in=%d out=%d reasoning=%d",
|
||||
use_model,
|
||||
elapsed,
|
||||
usage.get("prompt_tokens", 0),
|
||||
usage.get("completion_tokens", 0),
|
||||
usage.get("reasoning_tokens", 0),
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if raw is None:
|
||||
|
||||
Reference in New Issue
Block a user