Compare commits
4 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7adbb7d331 | |||
| 65343473d8 | |||
| d7fa9d9e8d | |||
| 90004cced4 |
@@ -3,7 +3,7 @@
|
||||
V2 图片分析(10-05):火山OCR专用API + doubao-lite强约束JSON并行,单图<3s,8图<15s;pro VLM单次兜底。输出字段兼容旧格式,下游信任链/t2i零改动。
|
||||
|
||||
流水线步骤:
|
||||
1. _step_image_analysis 图片分析(V2: OCR+lite VLM并行 + pro兜底)
|
||||
1. _step_image_analysis 图片分析(V2: OCR+qwen3.8-flash并行 + qwen3.7-plus兜底)
|
||||
1.5 _step_video_analysis 参考视频风格分析(可选)
|
||||
2. _step_intent_parsing 用户文案意图解析
|
||||
3. _step_script_generation 编导分镜脚本生成(融合原 copy_fusion+storyboard+review,输出 copy_result 结构 + voiceover_script)
|
||||
@@ -358,10 +358,10 @@ def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
"""步骤 1: 图片分析(V2 主路径)。
|
||||
|
||||
架构:
|
||||
- 主力:火山 MediaKit OCR(专用API)+ doubao-seed-2.1-lite 强约束 JSON(弥补火山云端缺失的
|
||||
人体属性/商品检测/图像标签专用HTTP API),每图2路并行,目标<3s;
|
||||
- 主力:火山 MediaKit OCR(专用API,未配置时自动跳过)+ qwen3.8-flash 强约束 JSON,每图2路并行,目标<3s;
|
||||
- 外层全并发(workers=8),目标8图<15s;
|
||||
- 兜底:fast 结果不可用时单次调用 doubao-seed-2.1-pro VLM(简单、无竞速)。
|
||||
- 兜底:fast 结果不可用时单次调用 qwen3.7-plus(简单、无竞速)。
|
||||
- 唯一后端:阿里云百炼 DashScope,API Key 从环境变量 DASHSCOPE_API_KEY 读取。
|
||||
输出 dict 字段(name/brand/category/appearance/key_features/scene/mood/portrait_prompt/summary/_source)
|
||||
与旧版格式完全一致,下游信任链/t2i/intent_parsing/script_generation 零改动。
|
||||
"""
|
||||
@@ -383,26 +383,8 @@ def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
logger.error("[爆款视频] vision 模块导入失败: %s", e)
|
||||
return {"products": [_vision_fallback(0, f"vision_import_error:{e}")]}
|
||||
|
||||
# 整个阶段关闭底层 httpx 重试,避免线程里出现不可控等待
|
||||
try:
|
||||
from packages.shared.ai_client import get_doubao_client as _gdc
|
||||
|
||||
_cli = _gdc()
|
||||
_orig_retries = _cli.max_retries
|
||||
_cli.max_retries = 0
|
||||
except Exception:
|
||||
_cli = None
|
||||
_orig_retries = 0
|
||||
|
||||
try:
|
||||
results = _aiv2(normalized_urls)
|
||||
finally:
|
||||
if _cli is not None:
|
||||
try:
|
||||
_cli.max_retries = _orig_retries
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# V2 内部 httpx 直连 dashscope,单次调用无重试,无需调整全局 client
|
||||
results = _aiv2(normalized_urls)
|
||||
return {"products": list(results)}
|
||||
|
||||
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""V2 图片分析主路径:每图并行 OCR(火山专用API)+ lite JSON VLM,失败时单次 pro VLM 兜底。
|
||||
"""V2 图片分析主路径:每图并行 OCR(火山MediaKit,未配置时自动跳过)+ qwen3.8-flash JSON VLM,
|
||||
失败时单次 qwen3.7-plus 兜底。
|
||||
|
||||
设计原则(灵应10-05要求):
|
||||
- 主力路径简洁:单图2路并行,外层N图全并发
|
||||
- 兜底简单:单次 pro VLM 调用,无竞速/重试/复杂超时
|
||||
架构(灵应10-05确认):
|
||||
- 唯一后端:阿里云百炼 DashScope,qwen3.8-flash 做快速路径、qwen3.7-plus 做兜底
|
||||
- 主力:单图2路并行(OCR + fast VLM),外层N图全并发(workers=8)
|
||||
- 兜底:单次 pro VLM 调用,无竞速/重试/复杂超时
|
||||
- 输出 dict 格式与旧版完全一致,下游零改动
|
||||
"""
|
||||
|
||||
@@ -19,12 +21,12 @@ from . import assembler, ocr_volc, vlm_fallback, vlm_fast_json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# 可通过环境变量调参(有默认值,无需配置即可跑)
|
||||
# 超时(可通过环境变量覆盖)
|
||||
_IMG_WORKERS = int(os.environ.get("VISION_V2_IMG_WORKERS", "8"))
|
||||
_FAST_TIMEOUT = float(os.environ.get("VISION_V2_FAST_TIMEOUT", "8"))
|
||||
_FAST_JSON_TIMEOUT = float(os.environ.get("VISION_V2_FAST_JSON_TIMEOUT", "8"))
|
||||
_FAST_TIMEOUT = float(os.environ.get("VISION_V2_FAST_TIMEOUT", "12"))
|
||||
_FAST_JSON_TIMEOUT = float(os.environ.get("VISION_V2_FAST_JSON_TIMEOUT", "12"))
|
||||
_OCR_TIMEOUT = float(os.environ.get("VISION_V2_OCR_TIMEOUT", "6"))
|
||||
_PRO_TIMEOUT = float(os.environ.get("VISION_V2_PRO_TIMEOUT", "45"))
|
||||
_PRO_TIMEOUT = float(os.environ.get("VISION_V2_PRO_TIMEOUT", "25"))
|
||||
|
||||
_FALLBACK_RESULT = {
|
||||
"name": "未识别",
|
||||
@@ -42,7 +44,6 @@ _FALLBACK_RESULT = {
|
||||
|
||||
|
||||
def _is_usable(r: dict[str, Any]) -> bool:
|
||||
"""结果可用判定:portrait_prompt 是核心,有效就算 usable。"""
|
||||
pp = (r.get("portrait_prompt") or "").strip()
|
||||
if pp and pp not in ("无人像", "无法判断", "未识别"):
|
||||
return True
|
||||
@@ -53,10 +54,8 @@ def _is_usable(r: dict[str, Any]) -> bool:
|
||||
|
||||
|
||||
def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
"""单张图片 V2 分析。"""
|
||||
t0 = time.time()
|
||||
|
||||
# 第1层:OCR + lite JSON VLM 并行
|
||||
fj_result: dict[str, Any] | None = None
|
||||
ocr_result: list[str] = []
|
||||
with ThreadPoolExecutor(max_workers=2) as pool:
|
||||
@@ -74,7 +73,6 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
elif fut is f_ocr and isinstance(res, list):
|
||||
ocr_result = res
|
||||
except TimeoutError:
|
||||
# fast 整体超时,取消还没跑完的子任务,继续走 pro 兜底
|
||||
for f in (f_fj, f_ocr):
|
||||
if not f.done():
|
||||
f.cancel()
|
||||
@@ -82,7 +80,6 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
|
||||
fast_elapsed = time.time() - t0
|
||||
|
||||
# 组装 fast 结果
|
||||
if fj_result:
|
||||
assembled = assembler.assemble_result(idx, fj_result, ocr_result)
|
||||
if _is_usable(assembled):
|
||||
@@ -95,7 +92,6 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
)
|
||||
return assembled
|
||||
|
||||
# 第2层:pro VLM 单次兜底
|
||||
pro_t0 = time.time()
|
||||
pro_result = vlm_fallback.call_pro_vlm(img_url, idx, timeout=_PRO_TIMEOUT)
|
||||
if pro_result and _is_usable(pro_result):
|
||||
@@ -107,7 +103,6 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
logger.info("[vision.v2] 图片 #%d pro兜底命中 total=%.2fs", idx, time.time() - t0)
|
||||
return pro_result
|
||||
|
||||
# 最终:返回最小可用结果
|
||||
logger.warning("[vision.v2] 图片 #%d 全路径失败 elapsed=%.2fs", idx, time.time() - t0)
|
||||
out = dict(_FALLBACK_RESULT)
|
||||
out["_source"] = "v2_all_failed"
|
||||
@@ -117,13 +112,18 @@ def analyze_image_v2(idx: int, img_url: str) -> dict[str, Any]:
|
||||
|
||||
|
||||
def analyze_images_v2(img_urls: list[str]) -> list[dict[str, Any]]:
|
||||
"""批量图片 V2 分析,外层全并发。"""
|
||||
if not img_urls:
|
||||
return []
|
||||
workers = min(_IMG_WORKERS, len(img_urls), 16)
|
||||
results: list[dict[str, Any] | None] = [None] * len(img_urls)
|
||||
|
||||
logger.info("[vision.v2] 开始图片分析 n=%d workers=%d fast_timeout=%.0fs", len(img_urls), workers, _FAST_TIMEOUT)
|
||||
logger.info(
|
||||
"[vision.v2] 开始图片分析 n=%d workers=%d fast_timeout=%.0fs pro_timeout=%.0fs",
|
||||
len(img_urls),
|
||||
workers,
|
||||
_FAST_TIMEOUT,
|
||||
_PRO_TIMEOUT,
|
||||
)
|
||||
t0 = time.time()
|
||||
with ThreadPoolExecutor(max_workers=workers) as pool:
|
||||
future_to_idx = {pool.submit(analyze_image_v2, idx, url): idx for idx, url in enumerate(img_urls)}
|
||||
|
||||
@@ -1,22 +1,46 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""VLM 兜底:专用API路径失败时的最后一道防线,单次调用 doubao-seed-2.1-pro。
|
||||
"""V2 pro 兜底:qwen3.7-plus(阿里云百炼/DashScope)单次调用。
|
||||
|
||||
设计原则:简单、直接、无竞速、无复杂超时逻辑。只在 fast_json 结果不可用时调用。
|
||||
fast_json 结果不可用时单次调用,无竞速、无重试、无复杂超时逻辑。
|
||||
直接 httpx 发精简 JSON-only prompt(比旧版 prompt_loader XML 模板短很多,降低延迟)。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import os
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_PRO_MODEL = "doubao-seed-2-1-pro-260915"
|
||||
DEFAULT_TIMEOUT = 45
|
||||
DEFAULT_MAX_TOKENS = 800
|
||||
_BASE_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1"
|
||||
_PRO_MODEL = "qwen3.7-plus"
|
||||
_DEFAULT_TIMEOUT = 25
|
||||
_DEFAULT_MAX_TOKENS = 800
|
||||
|
||||
_PRO_SYSTEM = (
|
||||
"你是图片分析助手。仔细观察图片,严格按JSON schema返回一个对象,不要任何解释、"
|
||||
"不要markdown、不要代码块、不要前后缀文字。字段值不确定时填null或空数组。\n"
|
||||
"{\n"
|
||||
' "has_person": true/false,\n'
|
||||
' "gender": "男"/"女"/null,\n'
|
||||
' "age_range": "儿童"/"青少年"/"青年"/"中年"/"老年"/null,\n'
|
||||
' "outfit": "人物穿搭描述,60字以内(例:白色T恤+牛仔裤)",\n'
|
||||
' "hair": "发型",\n'
|
||||
' "pose": "姿态",\n'
|
||||
' "expression": "表情",\n'
|
||||
' "scene": "场景",\n'
|
||||
' "mood": "氛围",\n'
|
||||
' "has_product": true/false,\n'
|
||||
' "category": "服饰/鞋包/美妆/数码/食品/家居/配饰/母婴/非产品图",\n'
|
||||
' "product_name": "产品名称,非产品图填null",\n'
|
||||
' "brand": "品牌或文字标识,无则null",\n'
|
||||
' "key_features": ["特征数组"]\n'
|
||||
"}"
|
||||
)
|
||||
_PRO_USER = "分析这张图片,返回符合schema的JSON。"
|
||||
|
||||
|
||||
def _strip_code_fence(s: str) -> str:
|
||||
@@ -31,196 +55,121 @@ def _strip_code_fence(s: str) -> str:
|
||||
return s
|
||||
|
||||
|
||||
def _xml_text(tag: str, xml: str) -> str:
|
||||
m = re.search(rf"<{tag}[^>]*>(.*?)</{tag}>", xml, re.S)
|
||||
return (m.group(1) if m else "").strip()
|
||||
|
||||
|
||||
def _xml_attr(tag: str, attr: str, xml: str) -> str:
|
||||
m = re.search(rf"<{tag}[^>]*\b{attr}\s*=\s*[\"']([^\"']*)[\"']", xml)
|
||||
return (m.group(1) if m else "").strip()
|
||||
|
||||
|
||||
def _xml_to_product(raw: str, idx: int) -> dict[str, Any]:
|
||||
"""解析 VLM 输出的 XML 格式(简化版)。"""
|
||||
scene = _xml_text("scene", raw) or "通用"
|
||||
mood = _xml_text("mood", raw) or ""
|
||||
|
||||
portrait_prompt = "无人像"
|
||||
p_has = _xml_attr("people", "has_person", raw)
|
||||
if p_has and p_has.lower() != "false":
|
||||
gender = _xml_attr("people", "gender", raw) or ""
|
||||
age = _xml_attr("people", "age_range", raw) or ""
|
||||
outfit = _xml_attr("people", "outfit", raw) or ""
|
||||
hair = _xml_attr("people", "hair", raw) or "自然发型"
|
||||
pose = _xml_attr("people", "pose", raw) or ""
|
||||
expr = _xml_attr("people", "expression", raw) or "自然"
|
||||
parts: list[str] = []
|
||||
if gender:
|
||||
parts.append(gender + ("性" if not gender.endswith("性") else ""))
|
||||
if age:
|
||||
parts.append(age)
|
||||
parts.append("人物")
|
||||
def _assemble_pp(obj: dict[str, Any]) -> str:
|
||||
if not obj.get("has_person", False):
|
||||
return "无人像"
|
||||
parts: list[str] = []
|
||||
gender = obj.get("gender")
|
||||
age = obj.get("age_range")
|
||||
if gender:
|
||||
parts.append(gender + ("性" if not gender.endswith("性") else ""))
|
||||
if age:
|
||||
parts.append(age)
|
||||
parts.append("人物")
|
||||
hair = obj.get("hair")
|
||||
if hair:
|
||||
parts.append(hair)
|
||||
if outfit:
|
||||
parts.append(f"身着{outfit}")
|
||||
if pose:
|
||||
parts.append(f"姿态{pose}")
|
||||
outfit = obj.get("outfit")
|
||||
if outfit:
|
||||
parts.append(f"身着{outfit}")
|
||||
pose = obj.get("pose")
|
||||
if pose:
|
||||
parts.append(f"姿态{pose}")
|
||||
expr = obj.get("expression")
|
||||
if expr:
|
||||
parts.append(f"表情{expr}")
|
||||
portrait_prompt = ",".join(parts)
|
||||
|
||||
m = re.search(r"<product[^>]*>(.*?)</product>", raw, re.S)
|
||||
if m:
|
||||
pbody = m.group(1)
|
||||
name = _xml_attr("product", "name", raw) or _xml_text("name", pbody) or "未识别"
|
||||
brand = _xml_attr("product", "brand", raw) or _xml_text("brand", pbody) or "无法判断"
|
||||
category = _xml_attr("product", "category", raw) or _xml_text("category", pbody) or "无法判断"
|
||||
appearance = _xml_attr("product", "appearance", raw) or _xml_text("appearance", pbody) or "无法判断"
|
||||
packaging = _xml_attr("product", "packaging", raw) or _xml_text("packaging", pbody) or "无法判断"
|
||||
feat = _xml_attr("product", "features", raw) or _xml_text("features", pbody) or ""
|
||||
feat_list = [x.strip() for x in re.split(r"[,,;;]", feat) if x.strip()] if feat else ["无法判断"]
|
||||
top_text = _xml_attr("product", "text_on_package", raw) or _xml_text("text_on_package", pbody) or ""
|
||||
text_list = [x.strip() for x in re.split(r"[,,;;]", top_text) if x.strip()] if top_text else []
|
||||
summary = _xml_attr("product", "summary", raw) or _xml_text("summary", pbody) or f"{brand} {name}"
|
||||
pp_attr = _xml_attr("product", "portrait_prompt", raw)
|
||||
if pp_attr and pp_attr != "无人像":
|
||||
portrait_prompt = pp_attr
|
||||
return {
|
||||
"name": name,
|
||||
"brand": brand,
|
||||
"category": category,
|
||||
"appearance": appearance,
|
||||
"packaging": packaging,
|
||||
"text_on_package": text_list,
|
||||
"key_features": feat_list,
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": portrait_prompt,
|
||||
"summary": summary,
|
||||
"_source": "vlm_pro_xml",
|
||||
}
|
||||
|
||||
if portrait_prompt != "无人像":
|
||||
return {
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "无法判断",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": portrait_prompt,
|
||||
"summary": "未识别",
|
||||
"_source": "vlm_pro_no_product",
|
||||
}
|
||||
return {
|
||||
"name": "未识别",
|
||||
"brand": "无法判断",
|
||||
"category": "无法判断",
|
||||
"appearance": "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": "无人像",
|
||||
"summary": "未识别",
|
||||
"_source": "vlm_pro_no_tag",
|
||||
}
|
||||
return ",".join(parts) if parts else "无人像"
|
||||
|
||||
|
||||
def call_pro_vlm(
|
||||
img_url: str,
|
||||
idx: int,
|
||||
*,
|
||||
model: str | None = None,
|
||||
timeout: int = DEFAULT_TIMEOUT,
|
||||
timeout: int = _DEFAULT_TIMEOUT,
|
||||
) -> dict[str, Any] | None:
|
||||
"""单次调用 pro VLM,解析后返回 product dict;失败返回 None。"""
|
||||
"""单次调用 qwen3.7-plus,解析后返回 product dict;失败返回 None。"""
|
||||
t0 = time.time()
|
||||
try:
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template,
|
||||
render_system_prompt,
|
||||
render_user_prompt,
|
||||
)
|
||||
from packages.shared.ai_client import get_doubao_client
|
||||
except ImportError as e:
|
||||
logger.warning("[vision.vlm] 导入失败: %s", e)
|
||||
import httpx
|
||||
|
||||
api_key = os.environ.get("DASHSCOPE_API_KEY")
|
||||
if not api_key:
|
||||
logger.warning("[vision.v2] DASHSCOPE_API_KEY 未配置,跳过 pro 兜底")
|
||||
return None
|
||||
|
||||
url = f"{_BASE_URL}/chat/completions"
|
||||
payload: dict[str, Any] = {
|
||||
"model": _PRO_MODEL,
|
||||
"messages": [
|
||||
{"role": "system", "content": _PRO_SYSTEM},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "image_url", "image_url": {"url": img_url}},
|
||||
{"type": "text", "text": _PRO_USER},
|
||||
],
|
||||
},
|
||||
],
|
||||
"temperature": 0.3,
|
||||
"max_tokens": _DEFAULT_MAX_TOKENS,
|
||||
"stream": False,
|
||||
"enable_thinking": False,
|
||||
"response_format": {"type": "json_object"},
|
||||
}
|
||||
try:
|
||||
template = get_template("image_analysis")
|
||||
system = render_system_prompt(template)
|
||||
user = render_user_prompt(template, image_count=1, industry="通用", image_urls=f"第1张:{img_url}")
|
||||
except Exception as e:
|
||||
logger.warning("[vision.vlm] 模板加载失败: %s", e)
|
||||
return None
|
||||
|
||||
client = get_doubao_client()
|
||||
if not client.is_available:
|
||||
return None
|
||||
|
||||
use_model = model or DEFAULT_PRO_MODEL
|
||||
_orig_retries = client.max_retries
|
||||
client.max_retries = 0
|
||||
try:
|
||||
raw = client.vision_completion(
|
||||
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
images=[img_url],
|
||||
temperature=0.3,
|
||||
max_tokens=DEFAULT_MAX_TOKENS,
|
||||
r = httpx.post(
|
||||
url,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||
json=payload,
|
||||
timeout=timeout,
|
||||
model=use_model,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("[vision.vlm] 图片 #%d pro VLM 调用失败 elapsed=%.1fs err=%s", idx, time.time() - t0, e)
|
||||
client.max_retries = _orig_retries
|
||||
return None
|
||||
client.max_retries = _orig_retries
|
||||
|
||||
elapsed = time.time() - t0
|
||||
if not raw:
|
||||
logger.warning("[vision.vlm] 图片 #%d pro VLM 返回空 elapsed=%.1fs", idx, elapsed)
|
||||
return None
|
||||
|
||||
text = _strip_code_fence(raw)
|
||||
l, r = text.find("{"), text.rfind("}")
|
||||
if l >= 0 and r > l:
|
||||
try:
|
||||
obj = json.loads(text[l : r + 1])
|
||||
if isinstance(obj, dict):
|
||||
logger.info("[vision.vlm] 图片 #%d pro VLM JSON 完成 elapsed=%.1fs", idx, elapsed)
|
||||
return {
|
||||
"name": obj.get("name") or "未识别",
|
||||
"brand": obj.get("brand") or "无法判断",
|
||||
"category": obj.get("category") or "无法判断",
|
||||
"appearance": obj.get("appearance") or "无法判断",
|
||||
"packaging": obj.get("packaging") or "无法判断",
|
||||
"text_on_package": obj.get("text_on_package") or [],
|
||||
"key_features": obj.get("key_features") or obj.get("features") or ["无法判断"],
|
||||
"scene": obj.get("scene") or "通用",
|
||||
"mood": obj.get("mood") or "",
|
||||
"portrait_prompt": obj.get("portrait_prompt") or "无人像",
|
||||
"summary": obj.get("summary") or f"{obj.get('brand','')} {obj.get('name','')}",
|
||||
"_source": "vlm_pro_json",
|
||||
}
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
|
||||
try:
|
||||
result = _xml_to_product(text, idx)
|
||||
result["_fallback_used"] = True
|
||||
result["_pro_elapsed"] = round(elapsed, 2)
|
||||
elapsed = time.time() - t0
|
||||
if r.status_code != 200:
|
||||
logger.warning("[vision.v2] pro HTTP %d elapsed=%.1fs body=%s", r.status_code, elapsed, r.text[:200])
|
||||
return None
|
||||
data = r.json()
|
||||
raw = (data.get("choices") or [{}])[0].get("message", {}).get("content")
|
||||
if not raw:
|
||||
logger.warning("[vision.v2] pro 返回空 elapsed=%.1fs", elapsed)
|
||||
return None
|
||||
usage = data.get("usage") or {}
|
||||
logger.info(
|
||||
"[vision.vlm] 图片 #%d pro VLM XML 完成 elapsed=%.2fs pp=%s",
|
||||
"[vision.v2] pro 完成 idx=%d model=%s elapsed=%.1fs in=%d out=%d",
|
||||
idx,
|
||||
_PRO_MODEL,
|
||||
elapsed,
|
||||
(result.get("portrait_prompt") or "")[:40],
|
||||
usage.get("prompt_tokens", 0),
|
||||
usage.get("completion_tokens", 0),
|
||||
)
|
||||
return result
|
||||
text = _strip_code_fence(raw)
|
||||
l, r_pos = text.find("{"), text.rfind("}")
|
||||
if l < 0 or r_pos <= l:
|
||||
logger.warning("[vision.v2] pro 无JSON elapsed=%.1fs head=%s", elapsed, raw[:200])
|
||||
return None
|
||||
obj = json.loads(text[l : r_pos + 1])
|
||||
if not isinstance(obj, dict):
|
||||
return None
|
||||
scene = obj.get("scene") or "通用"
|
||||
mood = obj.get("mood") or ""
|
||||
pp = _assemble_pp(obj)
|
||||
has_person = obj.get("has_person", False)
|
||||
has_product = obj.get("has_product", False)
|
||||
name = obj.get("product_name") or "未识别"
|
||||
brand = obj.get("brand") or "无法判断"
|
||||
category = obj.get("category") or ("非产品图" if has_person and not has_product else "无法判断")
|
||||
return {
|
||||
"name": name,
|
||||
"brand": brand,
|
||||
"category": category,
|
||||
"appearance": obj.get("outfit") or "无法判断",
|
||||
"packaging": "无法判断",
|
||||
"text_on_package": [],
|
||||
"key_features": obj.get("key_features") or ["无法判断"],
|
||||
"scene": scene,
|
||||
"mood": mood,
|
||||
"portrait_prompt": pp,
|
||||
"summary": f"{brand} {name}" if name != "未识别" else "未识别",
|
||||
"_source": "vlm_pro",
|
||||
}
|
||||
except Exception as e:
|
||||
logger.warning("[vision.vlm] 图片 #%d 解析失败 elapsed=%.1fs err=%s head=%s", idx, elapsed, e, raw[:200])
|
||||
logger.warning("[vision.v2] pro 异常 idx=%d elapsed=%.1fs err=%s", idx, time.time() - t0, e, exc_info=True)
|
||||
return None
|
||||
|
||||
@@ -1,35 +1,43 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""doubao-seed-2.1-lite 强约束 JSON-only 调用。
|
||||
"""V2 快速路径:qwen3.8-flash(阿里云百炼/DashScope)强约束 JSON-only 调用。
|
||||
|
||||
目标:替代"人体属性/商品检测/图像标签"三个火山不存在的专用云端 API。
|
||||
设计要点:
|
||||
- 直接用 httpx 发最小 payload 到 DashScope OpenAI 兼容 endpoint,不走 ai_client 包装
|
||||
- enable_thinking=false 关闭推理链(reasoning 是延迟主因)
|
||||
- system prompt 极致精简,只给字段 schema 和强约束(禁止自然语言、禁止 markdown)
|
||||
- max_tokens=350(比旧 VLM 的 1200 小很多,降低延迟)
|
||||
- temperature=0.1(极低,稳定输出 JSON)
|
||||
- timeout=8s(够快,失败则由外层走 pro VLM 兜底)
|
||||
- 期望返回纯 JSON object(无 ```json 包裹、无解释文字)
|
||||
- max_tokens=350、temperature=0.1(稳定输出 JSON)
|
||||
- timeout=12s(失败由外层走 pro 兜底)
|
||||
- API Key 从环境变量 DASHSCOPE_API_KEY 读取
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# DashScope OpenAI 兼容 endpoint
|
||||
_BASE_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1"
|
||||
_FAST_MODEL = "qwen3.8-flash"
|
||||
_DEFAULT_TIMEOUT = 12
|
||||
_DEFAULT_MAX_TOKENS = 350
|
||||
|
||||
# 极简 system prompt:只给字段定义 + 硬性输出要求
|
||||
_FAST_SYSTEM = (
|
||||
"你是图片结构化识别器。严格按下方 JSON schema 返回一个对象,不要任何解释、"
|
||||
"不要markdown、不要代码块、不要前后缀文字。字段值不确定时填 null 或空数组。\n"
|
||||
"{\n"
|
||||
' "has_person": true/false, // 图中是否有人\n'
|
||||
' "has_person": true/false,\n'
|
||||
' "gender": "男"/"女"/null,\n'
|
||||
' "age_range": "儿童"/"青少年"/"青年"/"中年"/"老年"/null,\n'
|
||||
' "upper_wear": "上装款式,如T恤/衬衫/卫衣/毛衣/西装/夹克/连衣裙/吊带/背心/外套等",\n'
|
||||
' "upper_color": "上装主色",\n'
|
||||
' "lower_wear": "下装款式,如牛仔裤/休闲裤/短裙/长裙/短裤/西裤/运动裤等;穿连衣裙时填null",\n'
|
||||
' "lower_wear": "下装款式;穿连衣裙时填null",\n'
|
||||
' "lower_color": "下装主色",\n'
|
||||
' "dress_color": "连衣裙主色(穿连衣裙时填)",\n'
|
||||
' "accessories": ["眼镜"/"帽子"/"项链"/"耳环"/"背包"/"手表"等数组],\n'
|
||||
@@ -38,7 +46,7 @@ _FAST_SYSTEM = (
|
||||
' "pose": "姿势,如站立/坐姿/侧身/行走等",\n'
|
||||
' "scene": "场景,如室内/街拍/户外/办公室/家居/海边/雪景/森林等",\n'
|
||||
' "style": "风格,如休闲/商务/运动/复古/潮流/甜美/酷飒/优雅/街头/法式等",\n'
|
||||
' "has_product": true/false, // 是否有明确商品展示\n'
|
||||
' "has_product": true/false,\n'
|
||||
' "category": "产品类目:服饰/鞋包/美妆/数码/食品/家居/配饰/母婴/非产品图",\n'
|
||||
' "product_name": "产品名称,非产品图填null",\n'
|
||||
' "brand": "品牌或文字标识,无则null",\n'
|
||||
@@ -51,21 +59,17 @@ _FAST_SYSTEM = (
|
||||
|
||||
_FAST_USER = "识别这张图片的人物穿搭与主体信息,只返回JSON对象。"
|
||||
|
||||
# 默认模型
|
||||
DEFAULT_LITE_MODEL = "doubao-seed-2-1-lite-260915"
|
||||
DEFAULT_TIMEOUT = 8
|
||||
DEFAULT_MAX_TOKENS = 350
|
||||
|
||||
def _api_key() -> str | None:
|
||||
return os.environ.get("DASHSCOPE_API_KEY")
|
||||
|
||||
|
||||
def _strip_code_fence(s: str) -> str:
|
||||
"""剥离 ```json ... ``` 包裹(即使要求纯 JSON,模型偶尔仍会包代码块)。"""
|
||||
s = s.strip()
|
||||
if s.startswith("```"):
|
||||
lines = s.split("\n")
|
||||
# 去掉首行 ```json
|
||||
if lines and lines[0].startswith("```"):
|
||||
lines = lines[1:]
|
||||
# 去掉尾行 ```
|
||||
if lines and lines[-1].strip().startswith("```"):
|
||||
lines = lines[:-1]
|
||||
s = "\n".join(lines).strip()
|
||||
@@ -75,57 +79,38 @@ def _strip_code_fence(s: str) -> str:
|
||||
def call_fast_json(
|
||||
img_url: str,
|
||||
*,
|
||||
model: str | None = None,
|
||||
timeout: int = DEFAULT_TIMEOUT,
|
||||
max_tokens: int = DEFAULT_MAX_TOKENS,
|
||||
timeout: int = _DEFAULT_TIMEOUT,
|
||||
max_tokens: int = _DEFAULT_MAX_TOKENS,
|
||||
) -> dict[str, Any] | None:
|
||||
"""调用 lite VLM 返回结构化 dict;失败/非 JSON 返回 None。
|
||||
|
||||
直接用 httpx 发最小 payload(关闭 thinking),不走 ai_client 包装:
|
||||
- 关闭 thinking/推理链(reasoning_tokens 是延迟主因,单次要10-12s)
|
||||
- 单次调用不重试(失败由外层走 pro 兜底)
|
||||
- 温度=0.1 稳定输出 JSON
|
||||
"""
|
||||
"""调用 qwen3.8-flash 返回结构化 dict;失败/非 JSON 返回 None。"""
|
||||
t0 = time.time()
|
||||
import httpx
|
||||
|
||||
api_key = _api_key()
|
||||
if not api_key:
|
||||
logger.warning("[vision.v2] DASHSCOPE_API_KEY 未配置,跳过 fast_json")
|
||||
return None
|
||||
|
||||
url = f"{_BASE_URL}/chat/completions"
|
||||
payload: dict[str, Any] = {
|
||||
"model": _FAST_MODEL,
|
||||
"messages": [
|
||||
{"role": "system", "content": _FAST_SYSTEM},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "image_url", "image_url": {"url": img_url}},
|
||||
{"type": "text", "text": _FAST_USER},
|
||||
],
|
||||
},
|
||||
],
|
||||
"temperature": 0.1,
|
||||
"max_tokens": max_tokens,
|
||||
"stream": False,
|
||||
"enable_thinking": False,
|
||||
"response_format": {"type": "json_object"},
|
||||
}
|
||||
try:
|
||||
from packages.shared import get_shared_settings
|
||||
|
||||
settings = get_shared_settings()
|
||||
api_key = settings.doubao_api_key
|
||||
base_url = (settings.doubao_base_url or "https://ark.cn-beijing.volces.com/api/v3").rstrip("/")
|
||||
if not api_key:
|
||||
logger.warning("[vision.v2] doubao api_key 未配置,跳过 fast_json")
|
||||
return None
|
||||
|
||||
use_model = model or DEFAULT_LITE_MODEL
|
||||
url = f"{base_url}/chat/completions"
|
||||
payload: dict[str, Any] = {
|
||||
"model": use_model,
|
||||
"messages": [
|
||||
{"role": "system", "content": _FAST_SYSTEM},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "image_url", "image_url": {"url": img_url}},
|
||||
{"type": "text", "text": _FAST_USER},
|
||||
],
|
||||
},
|
||||
],
|
||||
"temperature": 0.1,
|
||||
"max_tokens": max_tokens,
|
||||
"stream": False,
|
||||
}
|
||||
# 关键:关闭 thinking(避免产生 reasoning_tokens 拖慢响应)
|
||||
# 方舟/豆包 2.x 模型支持 thinking.type=disabled
|
||||
try:
|
||||
payload["thinking"] = {"type": "disabled"}
|
||||
except Exception:
|
||||
pass
|
||||
# 部分模型用 reasoning_effort 控制思考深度
|
||||
payload["reasoning_effort"] = "low"
|
||||
|
||||
resp = httpx.post(
|
||||
url,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||
@@ -133,67 +118,54 @@ def call_fast_json(
|
||||
timeout=timeout,
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code == 400 and "enable_thinking" in resp.text[:300].lower():
|
||||
# 极少数 endpoint 版本不识别 enable_thinking,重试一次不带
|
||||
logger.warning("[vision.v2] fast_json HTTP 400 thinking 参数不兼容,重试 elapsed=%.1fs", elapsed)
|
||||
payload.pop("enable_thinking", None)
|
||||
resp = httpx.post(
|
||||
url,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||
json=payload,
|
||||
timeout=timeout,
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code != 200:
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json HTTP %d elapsed=%.1fs body=%s", resp.status_code, elapsed, resp.text[:200]
|
||||
)
|
||||
# 如果400说明不支持thinking参数,降级重试一次
|
||||
if resp.status_code == 400 and "thinking" in resp.text.lower():
|
||||
payload.pop("thinking", None)
|
||||
payload.pop("reasoning_effort", None)
|
||||
time.time()
|
||||
resp = httpx.post(
|
||||
url,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||
json=payload,
|
||||
timeout=timeout,
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code != 200:
|
||||
logger.warning("[vision.v2] fast_json 降级后 HTTP %d elapsed=%.1fs", resp.status_code, elapsed)
|
||||
return None
|
||||
else:
|
||||
return None
|
||||
return None
|
||||
data = resp.json()
|
||||
raw = (data.get("choices") or [{}])[0].get("message", {}).get("content")
|
||||
if raw is None:
|
||||
logger.warning("[vision.v2] fast_json 返回 None elapsed=%.1fs", elapsed)
|
||||
if not raw:
|
||||
logger.warning("[vision.v2] fast_json 返回空 elapsed=%.1fs", elapsed)
|
||||
return None
|
||||
usage = data.get("usage") or {}
|
||||
reasoning_tokens = usage.get("reasoning_tokens", 0)
|
||||
ctd = usage.get("completion_tokens_details") or {}
|
||||
if not reasoning_tokens:
|
||||
reasoning_tokens = ctd.get("reasoning_tokens", 0)
|
||||
logger.info(
|
||||
"[vision.v2] fast_json 直连完成 model=%s elapsed=%.1fs in=%d out=%d reasoning=%d",
|
||||
use_model,
|
||||
"[vision.v2] fast_json 完成 model=%s elapsed=%.1fs in=%d out=%d reasoning=%d",
|
||||
_FAST_MODEL,
|
||||
elapsed,
|
||||
usage.get("prompt_tokens", 0),
|
||||
usage.get("completion_tokens", 0),
|
||||
usage.get("reasoning_tokens", 0),
|
||||
reasoning_tokens,
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if raw is None:
|
||||
logger.warning("[vision.v2] fast_json 返回 None elapsed=%.1fs model=%s", elapsed, use_model)
|
||||
return None
|
||||
|
||||
text = _strip_code_fence(raw)
|
||||
# 截到第一个 { 和最后一个 } 之间,容忍前后偶发文字
|
||||
l = text.find("{")
|
||||
r = text.rfind("}")
|
||||
l, r = text.find("{"), text.rfind("}")
|
||||
if l >= 0 and r > l:
|
||||
text = text[l : r + 1]
|
||||
try:
|
||||
obj = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json JSON 解析失败 elapsed=%.1fs head=%s",
|
||||
elapsed,
|
||||
raw[:200],
|
||||
)
|
||||
logger.warning("[vision.v2] fast_json JSON 解析失败 elapsed=%.1fs head=%s", elapsed, raw[:200])
|
||||
return None
|
||||
if not isinstance(obj, dict):
|
||||
logger.warning("[vision.v2] fast_json 非 dict: %s", type(obj))
|
||||
return None
|
||||
logger.info(
|
||||
"[vision.v2] fast_json 完成 model=%s elapsed=%.1fs has_person=%s has_product=%s category=%s",
|
||||
use_model,
|
||||
"[vision.v2] fast_json 完成 elapsed=%.1fs has_person=%s has_product=%s category=%s",
|
||||
elapsed,
|
||||
obj.get("has_person"),
|
||||
obj.get("has_product"),
|
||||
|
||||
Reference in New Issue
Block a user