Compare commits

...

11 Commits

Author SHA1 Message Date
xiaoxia a9f596fca3 feat(tts): style语气风格参数全链路接入 + 语速/音量/音调透传修复 (#2005)
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 3s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m2s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 2m5s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m19s
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 5m58s
AI Code Review / AI Code Review (pull_request) Successful in 6m48s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 8m10s
CI/CD Pipeline / CI Gate (pull_request) Successful in 1s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 0s
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 2s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Build Staging Web Image (push) Successful in 2m33s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m43s
CI/CD Pipeline / Build Staging API Image (push) Successful in 7m32s
CI/CD Pipeline / Validate - Style (push) Successful in 36m8s
CI/CD Pipeline / Unit Tests (push) Successful in 41m59s
CI/CD Pipeline / Integration Tests (push) Successful in 42m40s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 46m4s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 49m6s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 7m56s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m36s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m47s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 20m52s
CI/CD Pipeline / Validate - Security (push) Successful in 2h17m39s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-09-21 04:23:07 +08:00
xiaoxia 9ba036abb3 feat(#2001): 爆款标题样式配置面板后端支持 (#2004)
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 9s
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 14s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 12s
CI/CD Pipeline / Check push changed paths (push) Successful in 7s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m33s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m0s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 51s
CI/CD Pipeline / Validate - Style (push) Successful in 5m19s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 31s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 5m1s
AI Code Review / AI Code Review (pull_request) Successful in 6m41s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 2m15s
CI/CD Pipeline / Build Staging API Image (push) Successful in 1m10s
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m3s
CI/CD Pipeline / CI Gate (pull_request) Successful in 3s
CI/CD Pipeline / Integration Tests (push) Successful in 4m26s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 48s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m3s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 5m40s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m53s
CI/CD Pipeline / Validate - Security (push) Successful in 10m58s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m20s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m40s
CI/CD Pipeline / Unit Tests (push) Successful in 12m39s
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
API Base Image Build / Build API Base Image (push) Successful in 31m37s
Worker Base Image Build / Build Worker Base Image (push) Successful in 38m53s
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-09-21 03:01:24 +08:00
saas-backend-agent 3424e55a32 feat(gpu): #1978 MuseTalk v1.5 + GFPGAN face enhancement + ffmpeg pipe optimization (#2003)
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 3s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 47s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 24s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 1m22s
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (push) Successful in 34s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 26s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 24s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m56s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m6s
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 1s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
AI Code Review / AI Code Review (pull_request) Successful in 7m53s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 9m6s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 7m53s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m31s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 2m50s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m58s
CI/CD Pipeline / Integration Tests (push) Successful in 41m58s
CI/CD Pipeline / Validate - Style (push) Successful in 42m28s
CI/CD Pipeline / Unit Tests (push) Failing after 50m8s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 55m52s
CI/CD Pipeline / Validate - Security (push) Successful in 1h27m56s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
- MuseTalk v1.5 完整推理pipeline(DWPose+SFD GPU人脸检测)
- GFPGANv1.4 人脸超分增强(FP16, 可通过 MUSE_USE_GFPGAN=0 关闭)
- ffmpeg rawvideo pipe 编码替代PNG磁盘IO,省~2s
- 异步推理架构 + /cancel 端点 + 线程锁并发控制
- 长音频驱动架构,禁止ffmpeg预循环视频(避免16x性能退化)

# Conflicts:
#	deploy/gpu_worker/musetalk_server.py
2026-09-21 00:09:57 +08:00
xiaoxia 8c715474f4 feat(web): #2001 爆款标题样式配置面板升级(P0+P1) (#2002)
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 2s
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check push changed paths (push) Successful in 4s
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 16s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 19s
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 19s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m49s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 3m53s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Build Staging API Image (push) Successful in 33s
AI Code Review / AI Code Review (pull_request) Successful in 6m43s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 2s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (push) Successful in 41s
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 36s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m55s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 8m8s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m38s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m48s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m45s
CI/CD Pipeline / Validate - Style (push) Successful in 27m21s
CI/CD Pipeline / Integration Tests (push) Successful in 26m59s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 31m35s
CI/CD Pipeline / Unit Tests (push) Failing after 35m54s
CI/CD Pipeline / Validate - Security (push) Has been cancelled
CI/CD Pipeline / Build Production API Image (push) Has been cancelled
CI/CD Pipeline / Build Production Web Image (push) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
CI/CD Pipeline / Canary Release to Production (push) Has been cancelled
CI/CD Pipeline / CI Gate (push) Has been cancelled
2026-09-20 23:09:04 +08:00
ying 93cb3e12a0 feat(gpu): add GFPGAN face enhancement + ffmpeg pipe optimization
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1s
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m43s
AI Code Review / AI Code Review (pull_request) Successful in 7m33s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 6m26s
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 15s
CI/CD Pipeline / PR Build Web Image (pull_request) Has been skipped
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 10s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 10m40s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Validate - Style (pull_request) Successful in 22m19s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 18m23s
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Successful in 26m31s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 29m43s
ACR Cleanup / ACR Image Cleanup (pull_request_target) Successful in 10s
Preview Cleanup / Cleanup Preview Environment (pull_request) Successful in 7m54s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 1h28m58s
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 1s
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
- Add GFPGANv1.4 face super-resolution (FP16, +170MB VRAM, ~64ms/frame)
  - Controllable via MUSE_USE_GFPGAN env var (default enabled)
  - Enhances generated face region before blending with original frame
- Replace PNG disk I/O with ffmpeg stdin pipe encoding (~2s faster)
  - Frames streamed directly to ffmpeg as raw BGR24, no temp files
  - Eliminates output_frames_dir/ PNG write/read cycle
- Total warm inference: 54s for 5s video (was 56s with PNG, 48s without GFPGAN)
- VRAM usage: ~3.3GB with MuseTalk+GFPGAN loaded (RTX 2060 6GB OK)
2026-09-20 22:51:02 +08:00
ying e7bf85ca86 fix(gpu): #1978 MuseTalk v1.5 推理修复与Worker稳定性补丁
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 3s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m48s
CI/CD Pipeline / PR Build Web Image (pull_request) Has been skipped
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m4s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 2m27s
AI Code Review / AI Code Review (pull_request) Successful in 7m54s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 8m56s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 10m41s
CI/CD Pipeline / Validate - Style (pull_request) Successful in 39m59s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 40m7s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 44m57s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 1h5m19s
CI/CD Pipeline / Build Production API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (pull_request) Has been cancelled
CI/CD Pipeline / Production Browser E2E (pull_request) Has been cancelled
CI/CD Pipeline / Canary Release to Production (pull_request) Has been cancelled
CI/CD Pipeline / CI Gate (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Security (pull_request) Has been cancelled
musetalk_server.py:
- v3 API兼容性:支持新的请求格式(audio_url/video_url字段)
- v4 正确pipeline:使用vae.get_latents_for_unet()生成8ch输入
- v4.4 VRAM优化:batch_size=2 + CPU offload + PYTORCH_CUDA_ALLOC_CONF=max_split_size_mb:128
- RTX2060 6GB可稳定运行,22s视频推理约204s(0.11x实时)

gpu_worker.py:
- 修复健康检查响应格式适配musetalk_server
- REQUEST_TIMEOUT提升到900s,与服务端GPU_TASK_TIMEOUT_SECONDS对齐
- TASK_MAX_RETRY降到1,避免与服务端MAX_ATTEMPTS=3相乘放大重试
- 增加TASK_HEARTBEAT_INTERVAL=30s,推理期间独立线程心跳防超时
- 增加MIN_VIDEO_DURATION_SECONDS=3s前置拦截,避免1s视频division by zero
2026-09-20 21:29:31 +08:00
xiaoxia 38ffa0b98b fix(1978): register lipsync_gpu task in worker celery imports
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 4s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 5s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 4s
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 6s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 43s
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m7s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 1m7s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 38s
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (push) Successful in 57s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 40s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 2s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m58s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Failing after 47s
CI/CD Pipeline / Staging E2E Tests (push) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (push) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (push) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 3m25s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m19s
CI/CD Pipeline / Integration Tests (push) Successful in 3m27s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Validate - Style (push) Successful in 4m15s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 5m7s
AI Code Review / AI Code Review (pull_request) Successful in 6m49s
CI/CD Pipeline / Validate - Security (push) Successful in 7m56s
CI/CD Pipeline / Unit Tests (push) Failing after 10m23s
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
Worker celery_app.conf.imports was missing app.tasks.lipsync_gpu,
so lipsync_gpu_process_async.apply_async() messages were sent to
broker but never consumed by any worker, leaving lipsync_jobs stuck
at status=processing forever.
2026-09-20 13:44:10 +08:00
xiaoxia c9876d70e4 fix(1978): fix xiaoxia-gpu-worker.service bad unit file & log path permissions
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 2s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 4s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m58s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 2m45s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 4m9s
CI/CD Pipeline / Integration Tests (push) Successful in 4m19s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 2m25s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m23s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 4m32s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 3m41s
CI/CD Pipeline / CI Gate (pull_request) Successful in 2s
CI/CD Pipeline / Build Staging API Image (push) Successful in 4m37s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Validate - Style (push) Successful in 4m56s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Failing after 39s
CI/CD Pipeline / Staging E2E Tests (push) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (push) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (push) Has been skipped
CI/CD Pipeline / Build Production API Image (push) Has been cancelled
CI/CD Pipeline / Build Production Web Image (push) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
CI/CD Pipeline / Canary Release to Production (push) Has been cancelled
CI/CD Pipeline / CI Gate (push) Has been cancelled
CI/CD Pipeline / Validate - Security (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Unit Tests (push) Has been cancelled
AI Code Review / AI Code Review (pull_request) Has been cancelled
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been cancelled
PR Automation / Auto Approve on CI Green (pull_request) Has been cancelled
- User=%i -> User=ying (%%i is template-only specifier, not valid here)
- After=musetalk.service -> After=musetalk-worker.service (match actual service name)
- Log path from /tmp/ to $HOME/ to avoid permission issues for ying user
2026-09-20 13:36:36 +08:00
xiaoxia 8413713315 feat(web): TTS配音情感风格选择器 (#2000)
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 3s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 10s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m32s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m12s
CI/CD Pipeline / Integration Tests (push) Successful in 3m12s
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 3m38s
CI/CD Pipeline / Validate - Style (push) Successful in 3m57s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 3m53s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 2m33s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 4m0s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 4m0s
CI/CD Pipeline / CI Gate (pull_request) Successful in 1s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 1m6s
CI/CD Pipeline / Build Staging API Image (push) Successful in 3m50s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 38s
AI Code Review / AI Code Review (pull_request) Successful in 6m50s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m57s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 2m44s
CI/CD Pipeline / Validate - Security (push) Successful in 8m7s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 9m7s
CI/CD Pipeline / Unit Tests (push) Failing after 9m46s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Staging E2E Tests (push) Failing after 23m36s
feat(web): add TTS emotion style selector (#2000)
2026-09-20 12:52:47 +08:00
xiaoxia 9d8c6260e3 perf: async GPU lipsync inference + fix 16x performance regression (#1997) (#1998)
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 6s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 37s
CI/CD Pipeline / Build Staging API Image (push) Successful in 53s
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 30s
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 1m13s
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m36s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 45s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 48s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m5s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (push) Successful in 3m48s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 4m10s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 4m8s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m23s
CI/CD Pipeline / CI Gate (pull_request) Successful in 2s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 2m2s
CI/CD Pipeline / Validate - Style (push) Successful in 4m55s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m52s
AI Code Review / AI Code Review (pull_request) Successful in 6m37s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m3s
CI/CD Pipeline / Validate - Security (push) Successful in 8m45s
CI/CD Pipeline / Unit Tests (push) Failing after 10m3s
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
2026-09-20 12:19:35 +08:00
xiaoxia 0764a7820c Merge pull request 'feat(deploy): GPU节点自动部署配置文件入库(musetalk-worker+gpu-poll+setup/update脚本)' (#1999) from feat/gpu-node-auto-deploy into develop
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 6s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 2m17s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 19s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m59s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 23s
CI/CD Pipeline / Build Staging API Image (push) Successful in 26s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 52s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (push) Successful in 23s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 1s
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 21s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
AI Code Review / AI Code Review (pull_request) Successful in 7m55s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 11m31s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 7m56s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 3m1s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m39s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m20s
CI/CD Pipeline / Integration Tests (push) Successful in 41m51s
CI/CD Pipeline / Validate - Style (push) Successful in 42m31s
CI/CD Pipeline / Unit Tests (push) Failing after 47m47s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 55m51s
CI/CD Pipeline / Validate - Security (push) Successful in 1h10m22s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
2026-09-20 11:02:08 +08:00
78 changed files with 4721 additions and 520 deletions
@@ -0,0 +1,26 @@
"""#2001 爆款标题样式面板升级: ai_avatar_render_jobs 新增 cover_title_config
Revision ID: 083_cover_title_config
Revises: 082_atom_clip_ai_tags
Create Date: 2026-09-20
"""
import sqlalchemy as sa
from alembic import op
revision = "083_cover_title_config"
down_revision = "082_atom_clip_ai_tags"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.add_column(
"ai_avatar_render_jobs",
sa.Column("cover_title_config", sa.JSON(), nullable=False, server_default=sa.text("'{}'")),
)
def downgrade() -> None:
op.drop_column("ai_avatar_render_jobs", "cover_title_config")
@@ -0,0 +1,26 @@
"""lipsync_jobs 新增 style 字段(TTS 语气风格)
Revision ID: 084_lipsync_jobs_style
Revises: 083_cover_title_config
Create Date: 2026-09-21
"""
import sqlalchemy as sa
from alembic import op
revision = "084_lipsync_jobs_style"
down_revision = "083_cover_title_config"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.add_column(
"lipsync_jobs",
sa.Column("style", sa.String(length=32), nullable=False, server_default=""),
)
def downgrade() -> None:
op.drop_column("lipsync_jobs", "style")
@@ -63,6 +63,7 @@ def create_render_job(
b_roll_segments=[s.model_dump() for s in body.b_roll_segments], b_roll_segments=[s.model_dump() for s in body.b_roll_segments],
title_config=body.title_config, title_config=body.title_config,
cover_config=body.cover_config, cover_config=body.cover_config,
cover_title_config=body.cover_title_config,
project_id=body.project_id, project_id=body.project_id,
) )
except AiAvatarRenderError as exc: except AiAvatarRenderError as exc:
+4
View File
@@ -111,6 +111,8 @@ def create_lipsync_job(
voice_id=body.voice_id, voice_id=body.voice_id,
script_text=body.script_text, script_text=body.script_text,
speed=body.speed, speed=body.speed,
style=body.style or "",
volume=body.volume if body.volume is not None else 50,
emotion=body.emotion, emotion=body.emotion,
enable_video_loop=body.enable_video_loop, enable_video_loop=body.enable_video_loop,
project_id=body.project_id, project_id=body.project_id,
@@ -211,6 +213,8 @@ def preview_tts(
voice_id=body.voice_id, voice_id=body.voice_id,
script_text=body.script_text, script_text=body.script_text,
speed=body.speed, speed=body.speed,
style=body.style or "",
volume=body.volume if body.volume is not None else 50,
emotion=body.emotion, emotion=body.emotion,
) )
except MediaKitError as exc: except MediaKitError as exc:
+6
View File
@@ -207,6 +207,9 @@ def synthesize(
synthesis_meta = { synthesis_meta = {
"speed": request.speed, "speed": request.speed,
"emotion": request.emotion or "", "emotion": request.emotion or "",
"style": request.style or "",
"volume": request.volume if request.volume is not None else 50,
"pitch": request.pitch if request.pitch is not None else 1.0,
"language": request.language or "zh-CN", "language": request.language or "zh-CN",
} }
if request.metadata_: if request.metadata_:
@@ -654,6 +657,9 @@ def preview_tts(
text=request.text, text=request.text,
voice_id=actual_voice_id, voice_id=actual_voice_id,
speed=request.speed, speed=request.speed,
style=request.style or "",
volume=request.volume if request.volume is not None else 50,
pitch=request.pitch,
emotion=request.emotion, emotion=request.emotion,
language=getattr(request, "language", "zh-CN"), language=getattr(request, "language", "zh-CN"),
) )
+7 -1
View File
@@ -53,9 +53,14 @@ class CreateAiAvatarRenderRequest(BaseModel):
script_id: str = Field("", description="文案 ID(选自文案库时传;手动输入文案直生场景可留空)") script_id: str = Field("", description="文案 ID(选自文案库时传;手动输入文案直生场景可留空)")
b_roll_segments: list[BRollSegment] = Field(default_factory=list, description="B-roll 片段列表") b_roll_segments: list[BRollSegment] = Field(default_factory=list, description="B-roll 片段列表")
title_config: dict[str, Any] = Field( title_config: dict[str, Any] = Field(
default_factory=dict, description="标题配置(可含 title_image_dataurl:前端 Canvas 渲染的标题 PNG dataURL" default_factory=dict,
description="标题配置(可含 title_image_dataurl:前端 Canvas 渲染的标题 PNG dataURL;含 line_overrides 逐行样式)",
) )
cover_config: dict[str, Any] = Field(default_factory=dict, description="封面配置") cover_config: dict[str, Any] = Field(default_factory=dict, description="封面配置")
cover_title_config: dict[str, Any] = Field(
default_factory=dict,
description="封面独立标题配置(#2001),结构同 title_config;为空时封面不叠标题",
)
project_id: str = Field("", description="项目 ID") project_id: str = Field("", description="项目 ID")
@field_validator("lipsync_job_id") @field_validator("lipsync_job_id")
@@ -83,6 +88,7 @@ class AiAvatarRenderJobResponse(BaseModel):
b_roll_segments: list[dict[str, Any]] b_roll_segments: list[dict[str, Any]]
title_config: dict[str, Any] title_config: dict[str, Any]
cover_config: dict[str, Any] cover_config: dict[str, Any]
cover_title_config: dict[str, Any] = Field(default_factory=dict, description="封面独立标题配置")
status: str status: str
progress: int progress: int
output_video_url: str output_video_url: str
+13 -2
View File
@@ -29,6 +29,7 @@ class LipsyncJobResponse(BaseModel):
voice_id: str = "" voice_id: str = ""
script_text: str = "" script_text: str = ""
speed: float = 1.0 speed: float = 1.0
style: str = ""
emotion: str = "" emotion: str = ""
mediakit_task_id: str mediakit_task_id: str
status: str status: str
@@ -67,9 +68,14 @@ class CreateLipsyncJobRequest(BaseModel):
voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID") voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID")
script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)") script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
style: Optional[str] = Field(
None,
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
)
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
emotion: str = Field( emotion: str = Field(
"", "",
description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;空为默认自然)", description="[deprecated] 旧情绪参数,内部映射为 style",
) )
enable_video_loop: bool = Field( enable_video_loop: bool = Field(
@@ -123,10 +129,15 @@ class AiAvatarTtsPreviewRequest(BaseModel):
voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID") voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID")
script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案") script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
style: Optional[str] = Field(
None,
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
)
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
emotion: str = Field( emotion: str = Field(
"neutral", "neutral",
max_length=32, max_length=32,
description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;默认 neutral", description="[deprecated] 旧情绪参数,内部映射为 style;默认 neutral",
) )
+14 -3
View File
@@ -16,9 +16,15 @@ class TTSSynthesizeRequest(BaseModel):
output_name: str = Field("", description="输出文件名") output_name: str = Field("", description="输出文件名")
language: str = Field("zh-CN", description="语言") language: str = Field("zh-CN", description="语言")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速")
style: Optional[str] = Field(
None,
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
)
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
pitch: Optional[float] = Field(None, ge=0.5, le=2.0, description="音调(0.5-2.0),默认 1.0")
emotion: str = Field( emotion: str = Field(
"", "",
description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等;通过 instruction 自然语言指令控制)", description="[deprecated] 旧情绪参数,内部映射为 style;新接入请使用 style",
) )
voice_model: str = Field("", description="语音模型名称") voice_model: str = Field("", description="语音模型名称")
voice_clone_profile_id: str = Field("", description="关联的音色克隆档案 ID") voice_clone_profile_id: str = Field("", description="关联的音色克隆档案 ID")
@@ -113,9 +119,14 @@ class TTSPreviewRequest(BaseModel):
text: str = Field(..., min_length=1, max_length=200, description="合成文本,限制 200 字") text: str = Field(..., min_length=1, max_length=200, description="合成文本,限制 200 字")
voice_id: str = Field(..., min_length=1, description="音色 ID") voice_id: str = Field(..., min_length=1, description="音色 ID")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速")
emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)") style: Optional[str] = Field(
None,
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
)
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
emotion: str = Field("", description="[deprecated] 旧情绪参数,内部映射为 style")
language: str = Field("zh-CN", description="语言(zh-CN/en-US 等)") language: str = Field("zh-CN", description="语言(zh-CN/en-US 等)")
pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(预留,当前未使用)") pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(0.5-2.0),默认 1.0")
class TTSPreviewResponse(BaseModel): class TTSPreviewResponse(BaseModel):
@@ -61,6 +61,7 @@ class AiAvatarRenderService:
b_roll_segments: list[dict[str, Any]] | None = None, b_roll_segments: list[dict[str, Any]] | None = None,
title_config: dict[str, Any], title_config: dict[str, Any],
cover_config: dict[str, Any], cover_config: dict[str, Any],
cover_title_config: dict[str, Any] | None = None,
project_id: str = "", project_id: str = "",
) -> AiAvatarRenderJob: ) -> AiAvatarRenderJob:
"""创建渲染任务. """创建渲染任务.
@@ -112,6 +113,7 @@ class AiAvatarRenderService:
b_roll_segments=[s if isinstance(s, dict) else s.model_dump() for s in (b_roll_segments or [])], b_roll_segments=[s if isinstance(s, dict) else s.model_dump() for s in (b_roll_segments or [])],
title_config=title_config, title_config=title_config,
cover_config=cover_config, cover_config=cover_config,
cover_title_config=cover_title_config or {},
status="pending", status="pending",
) )
self.db.add(job) self.db.add(job)
+12 -1
View File
@@ -111,6 +111,8 @@ class LipsyncService:
script_text: str, script_text: str,
speed: float, speed: float,
emotion: str, emotion: str,
style: str = "",
volume: int = 50,
) -> str: ) -> str:
"""TTS 直生:调 CosyVoice 合成音频并转存 OSS,返回可公网访问的音频 URL. """TTS 直生:调 CosyVoice 合成音频并转存 OSS,返回可公网访问的音频 URL.
@@ -124,7 +126,9 @@ class LipsyncService:
text=script_text, text=script_text,
voice_id=actual_voice_id, voice_id=actual_voice_id,
speed=speed, speed=speed,
emotion=emotion, # normalize 在 CosyVoiceService 内部完成 style=style,
volume=volume,
emotion=emotion,
language="zh", language="zh",
) )
except CosyVoiceError as exc: except CosyVoiceError as exc:
@@ -424,6 +428,8 @@ class LipsyncService:
voice_id: str = "", voice_id: str = "",
script_text: str = "", script_text: str = "",
speed: float = 1.0, speed: float = 1.0,
style: str = "",
volume: int = 50,
emotion: str = "", emotion: str = "",
enable_video_loop: bool = True, enable_video_loop: bool = True,
project_id: str = "", project_id: str = "",
@@ -472,6 +478,7 @@ class LipsyncService:
voice_id=voice_id or "", voice_id=voice_id or "",
script_text=script_text or "", script_text=script_text or "",
speed=speed, speed=speed,
style=style or "",
emotion=emotion or "", emotion=emotion or "",
# 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing # 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing
status="tts_processing" if is_tts_mode else "pending", status="tts_processing" if is_tts_mode else "pending",
@@ -493,6 +500,8 @@ class LipsyncService:
voice_id, voice_id,
script_text, script_text,
speed, speed,
style or "",
volume,
emotion or "", emotion or "",
) )
) )
@@ -527,6 +536,8 @@ class LipsyncService:
voice_id: str, voice_id: str,
script_text: str, script_text: str,
speed: float = 1.0, speed: float = 1.0,
style: str = "",
volume: int = 50,
emotion: str = "neutral", emotion: str = "neutral",
) -> dict: ) -> dict:
"""同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算. """同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算.
+5 -1
View File
@@ -82,7 +82,9 @@ def tts_synthesize_and_submit(
voice_id: str, voice_id: str,
script_text: str, script_text: str,
speed: float, speed: float,
emotion: str, style: str = "",
volume: int = 50,
emotion: str = "",
): ):
"""异步执行 TTS 合成 + OSS 转存 + MediaKit 提交. """异步执行 TTS 合成 + OSS 转存 + MediaKit 提交.
@@ -159,6 +161,8 @@ def tts_synthesize_and_submit(
text=script_text, text=script_text,
voice_id=voice_id, voice_id=voice_id,
speed=speed, speed=speed,
style=style,
volume=volume,
emotion=emotion, emotion=emotion,
language="zh", language="zh",
) )
+13 -2
View File
@@ -7,8 +7,19 @@ export interface GenerateCoverTitleConfig {
font_color?: string font_color?: string
position?: string position?: string
bold?: boolean bold?: boolean
stroke?: boolean italic?: boolean
shadow?: boolean stroke?: boolean | { enabled?: boolean; width?: number; color?: string }
shadow?:
| boolean
| { enabled?: boolean; offset_x?: number; offset_y?: number; blur?: number; color?: string }
line_height?: number
margin_top?: number
max_chars_per_line?: number
background?: { enabled?: boolean; color?: string; padding?: number; radius?: number }
line_overrides?: Array<Record<string, unknown>>
cover_title_config?: Record<string, unknown>
pos_x?: number
pos_y?: number
} }
export interface GenerateCoverRequest { export interface GenerateCoverRequest {
+29 -3
View File
@@ -81,7 +81,7 @@ export interface CreateGenerationTaskRequest {
tts_voice_source?: "preset" | "clone" tts_voice_source?: "preset" | "clone"
/** #1970:智能降重开关(默认 true) */ /** #1970:智能降重开关(默认 true) */
dedup_enabled?: boolean dedup_enabled?: boolean
/** 标题烧录配置 */ /** 标题烧录配置(#2001 扩展:描边/阴影参数/行距/自动换行/背景/逐行/封面) */
title_config?: { title_config?: {
text?: string text?: string
font?: string font?: string
@@ -89,8 +89,34 @@ export interface CreateGenerationTaskRequest {
font_color?: string font_color?: string
position?: string position?: string
bold?: boolean bold?: boolean
stroke?: boolean italic?: boolean
shadow?: boolean stroke?: boolean | { enabled?: boolean; width?: number; color?: string }
shadow?:
| boolean
| {
enabled?: boolean
offset_x?: number
offset_y?: number
blur?: number
color?: string
}
line_height?: number
margin_top?: number
max_chars_per_line?: number
background?: { enabled?: boolean; color?: string; padding?: number; radius?: number }
line_overrides?: Array<{
line_index: number
text?: string
size?: number
color?: string
bold?: boolean
italic?: boolean
stroke?: boolean
highlights?: Array<{ word: string; color?: string; bold?: boolean; scale?: number }>
}>
cover_title_config?: Record<string, unknown>
pos_x?: number
pos_y?: number
} }
/** 关联的草稿 ID(编辑流程数据链路用) */ /** 关联的草稿 ID(编辑流程数据链路用) */
source_edit_plan_id?: string source_edit_plan_id?: string
@@ -54,6 +54,8 @@ export interface SegmentTtsConfig {
pitch: number pitch: number
volume: number volume: number
subtitle_sync: boolean subtitle_sync: boolean
/** 配音风格预设(natural/excited/professional/sweet/news/livestream */
style?: string
} }
/** 片段裁剪配置 */ /** 片段裁剪配置 */
+3
View File
@@ -18,6 +18,9 @@ export type {
TTSPreviewResponse, TTSPreviewResponse,
} from "./types" } from "./types"
export type { TtsStyle, TtsStyleOption } from "./styles"
export { TTS_STYLE_OPTIONS, DEFAULT_TTS_STYLE, getTtsStyle } from "./styles"
// API 函数 // API 函数
export { export {
synthesizeSpeech, synthesizeSpeech,
+71
View File
@@ -0,0 +1,71 @@
/**
* TTS 配音风格预设(情感/语气风格)
* - key:传给后端的 style 标识,便于后端按策略合成
* - 未传 style 时后端默认自然亲切
*
* 注:与原 emotionCosyVoice 7 种基础情绪枚举)解耦;
* style 是更高层的"说话风格预设",后端可能映射到 emotion + speed + prompt 组合。
*/
export interface TtsStyleOption {
/** 传给后端的风格标识 */
value: string
/** 展示名 */
label: string
/** emoji 图标 */
emoji: string
/** 给用户/后端的风格描述(prompt 风格) */
description: string
}
export const TTS_STYLE_OPTIONS: readonly TtsStyleOption[] = [
{
value: "natural",
label: "自然亲切",
emoji: "😊",
description: "亲切自然,像朋友聊天",
},
{
value: "excited",
label: "激动兴奋",
emoji: "🤩",
description: "激动兴奋,语速稍快,充满活力",
},
{
value: "professional",
label: "沉稳专业",
emoji: "🧑‍💼",
description: "沉稳专业,语速适中,正式可靠",
},
{
value: "sweet",
label: "温柔甜美",
emoji: "🌸",
description: "温柔甜美,语速轻柔",
},
{
value: "news",
label: "新闻播报",
emoji: "📰",
description: "字正腔圆,严肃正式",
},
{
value: "livestream",
label: "直播带货",
emoji: "🎤",
description: "热情有感染力,有节奏感",
},
] as const
export type TtsStyle = (typeof TTS_STYLE_OPTIONS)[number]["value"]
/** 默认风格:自然亲切 */
export const DEFAULT_TTS_STYLE: TtsStyle = "natural"
/** 根据 value 查找风格选项(容错:找不到回退 natural) */
export function getTtsStyle(value: string | null | undefined): TtsStyleOption {
return (
(TTS_STYLE_OPTIONS as readonly TtsStyleOption[]).find((o) => o.value === value) ??
(TTS_STYLE_OPTIONS as readonly TtsStyleOption[])[0]
)
}
+4
View File
@@ -17,6 +17,8 @@ export interface TTSSynthesizeRequest {
output_name?: string output_name?: string
language?: string language?: string
emotion?: string emotion?: string
/** 配音风格预设(自然亲切/激动兴奋/沉稳专业/温柔甜美/新闻播报/直播带货),不传默认 natural */
style?: string
speed?: number speed?: number
voice_model?: string voice_model?: string
voice_clone_profile_id?: string voice_clone_profile_id?: string
@@ -106,6 +108,8 @@ export interface TTSPreviewRequest {
pitch?: number pitch?: number
language?: string language?: string
emotion?: string // 情绪参数:neutral/happy/sad/angry/surprised/fearful/disgusted(后端 normalize_emotion() 兼容旧 natural/excited/calm/friendly 与中文标签) emotion?: string // 情绪参数:neutral/happy/sad/angry/surprised/fearful/disgusted(后端 normalize_emotion() 兼容旧 natural/excited/calm/friendly 与中文标签)
/** 配音风格预设 */
style?: string
} }
/** TTS 试听响应 */ /** TTS 试听响应 */
+361
View File
@@ -0,0 +1,361 @@
/**
* 标题样式相关常量(#2001
* - 字体列表(新增4款爆款字体)
* - 色板(常用标题字色/描边色/背景色)
* - 预设样式方案(10 个,含抖音爆款黄)
*/
import type { TitleStyleConfig } from "./types"
/* ── 字体选项(#2001:新增优设标题黑/阿里普惠体Bold/抖音美好体/思源黑体Heavy ── */
export interface FontOption {
value: string
label: string
/** CSS font-family 栈 */
family: string
/** 爆款/常用标签 */
tag?: "hot" | "new"
}
export const FONT_OPTIONS: FontOption[] = [
{
value: "优设标题黑",
label: "优设标题黑",
family:
'"YouShe Title Black","YouSheBiaoTiHei","Source Han Sans SC Heavy","Noto Sans SC","PingFang SC",sans-serif',
tag: "hot",
},
{
value: "阿里普惠体Bold",
label: "阿里普惠体Bold",
family:
'"Alibaba PuHuiTi Bold","Alibaba PuHuiTi","Source Han Sans SC","PingFang SC",sans-serif',
tag: "hot",
},
{
value: "抖音美好体",
label: "抖音美好体",
family: '"Douyin Sans","DouyinSans","Source Han Sans SC","PingFang SC",sans-serif',
tag: "hot",
},
{
value: "思源黑体Heavy",
label: "思源黑体Heavy",
family:
'"Source Han Sans SC Heavy","Noto Sans SC","Source Han Sans CN Heavy","PingFang SC",sans-serif',
tag: "new",
},
{
value: "思源黑体",
label: "思源黑体",
family: '"Source Han Sans SC","Noto Sans SC","PingFang SC","Microsoft YaHei",sans-serif',
},
{
value: "思源宋体",
label: "思源宋体",
family: '"Source Han Serif SC","Noto Serif SC","Songti SC","SimSun",serif',
},
{
value: "苹方",
label: "苹方",
family: '"PingFang SC",-apple-system,"Helvetica Neue",sans-serif',
},
{
value: "微软雅黑",
label: "微软雅黑",
family: '"Microsoft YaHei","PingFang SC",sans-serif',
},
{
value: "楷体",
label: "楷体",
family: '"KaiTi","STKaiti","DFKai-SB",serif',
},
]
/** 根据中文名取 font-family 栈(找不到回退思源黑体) */
export function getFontFamily(font: string): string {
const f = FONT_OPTIONS.find((x) => x.value === font)
if (f) return f.family
return FONT_OPTIONS[4].family // 思源黑体
}
/* ── 色板 ── */
/** 标题字色(常用爆款色) */
export const TITLE_COLOR_PALETTE: string[] = [
"#ffffff",
"#000000",
"#ffd700", // 抖音黄
"#ff2d55", // 抖音红
"#ff4081",
"#00e5ff",
"#d4a843",
"#ffa500",
"#52c41a",
"#1890ff",
"#7c3aed",
"#ff6b35",
]
/** 描边色(黑/白/灰为主) */
export const STROKE_COLOR_PALETTE: string[] = [
"#000000",
"#ffffff",
"#333333",
"#555555",
"#8b0000",
"#001f3f",
]
/** 背景色(带透明度) */
export const BG_COLOR_PALETTE: string[] = [
"rgba(0,0,0,0.5)",
"rgba(0,0,0,0.7)",
"rgba(0,0,0,0.3)",
"rgba(255,215,0,0.9)",
"rgba(255,45,85,0.85)",
"rgba(124,58,237,0.85)",
"rgba(24,144,255,0.85)",
"rgba(82,196,26,0.85)",
]
/* ── 预设样式方案(10 个,含抖音爆款黄) ── */
export interface TitlePreset {
key: string
label: string
emoji: string
/** 应用时覆盖到 TitleStyleConfig 的字段(其他字段保持当前值) */
style: Partial<TitleStyleConfig>
}
const BASE: Partial<TitleStyleConfig> = {
line_overrides: [],
cover_title_config: null,
}
export const TITLE_PRESETS: TitlePreset[] = [
{
key: "douyin_hot",
label: "抖音爆款黄",
emoji: "🔥",
style: {
...BASE,
font: "优设标题黑",
size: 80,
color: "#ffd700",
bold: true,
italic: false,
stroke: true,
stroke_width: 8,
stroke_color: "#000000",
shadow: true,
shadow_offset_x: 3,
shadow_offset_y: 3,
shadow_blur: 6,
shadow_color: "rgba(0,0,0,0.6)",
bg_enabled: false,
line_height: 1.25,
max_chars_per_line: 8,
},
},
{
key: "classic_white",
label: "经典白字黑描边",
emoji: "⚪",
style: {
...BASE,
font: "思源黑体Heavy",
size: 56,
color: "#ffffff",
bold: true,
italic: false,
stroke: true,
stroke_width: 5,
stroke_color: "#000000",
shadow: false,
bg_enabled: false,
line_height: 1.2,
max_chars_per_line: 10,
},
},
{
key: "red_bold",
label: "醒目红字",
emoji: "🔴",
style: {
...BASE,
font: "优设标题黑",
size: 72,
color: "#ff2d55",
bold: true,
italic: false,
stroke: true,
stroke_width: 6,
stroke_color: "#ffffff",
shadow: true,
shadow_offset_x: 2,
shadow_offset_y: 2,
shadow_blur: 5,
shadow_color: "rgba(0,0,0,0.5)",
bg_enabled: false,
line_height: 1.2,
max_chars_per_line: 9,
},
},
{
key: "black_gold",
label: "黑金质感",
emoji: "🟡",
style: {
...BASE,
font: "思源宋体",
size: 52,
color: "#d4a843",
bold: true,
italic: false,
stroke: false,
shadow: true,
shadow_offset_x: 2,
shadow_offset_y: 2,
shadow_blur: 8,
shadow_color: "rgba(0,0,0,0.8)",
bg_enabled: false,
line_height: 1.25,
max_chars_per_line: 10,
},
},
{
key: "neon_blue",
label: "霓虹发光",
emoji: "💙",
style: {
...BASE,
font: "阿里普惠体Bold",
size: 60,
color: "#00e5ff",
bold: true,
italic: false,
stroke: false,
shadow: true,
shadow_offset_x: 0,
shadow_offset_y: 0,
shadow_blur: 16,
shadow_color: "#00e5ff",
bg_enabled: false,
line_height: 1.2,
max_chars_per_line: 10,
},
},
{
key: "bg_black",
label: "黑底白字",
emoji: "⬛",
style: {
...BASE,
font: "思源黑体Heavy",
size: 52,
color: "#ffffff",
bold: true,
italic: false,
stroke: false,
shadow: false,
bg_enabled: true,
bg_color: "rgba(0,0,0,0.7)",
bg_padding: 16,
bg_radius: 8,
line_height: 1.3,
max_chars_per_line: 10,
},
},
{
key: "bg_yellow",
label: "黄底黑字",
emoji: "🟨",
style: {
...BASE,
font: "抖音美好体",
size: 56,
color: "#000000",
bold: true,
italic: false,
stroke: false,
shadow: false,
bg_enabled: true,
bg_color: "rgba(255,215,0,0.95)",
bg_padding: 14,
bg_radius: 6,
line_height: 1.2,
max_chars_per_line: 9,
},
},
{
key: "sweet_pink",
label: "温柔甜美粉",
emoji: "🌸",
style: {
...BASE,
font: "阿里普惠体Bold",
size: 50,
color: "#ff4081",
bold: false,
italic: false,
stroke: true,
stroke_width: 4,
stroke_color: "#ffffff",
shadow: true,
shadow_offset_x: 2,
shadow_offset_y: 2,
shadow_blur: 4,
shadow_color: "rgba(255,64,129,0.4)",
bg_enabled: false,
line_height: 1.3,
max_chars_per_line: 11,
},
},
{
key: "business_dark",
label: "商务深色",
emoji: "💼",
style: {
...BASE,
font: "思源黑体",
size: 44,
color: "#ffffff",
bold: false,
italic: false,
stroke: false,
shadow: true,
shadow_offset_x: 1,
shadow_offset_y: 1,
shadow_blur: 3,
shadow_color: "rgba(0,0,0,0.8)",
bg_enabled: true,
bg_color: "rgba(24,144,255,0.85)",
bg_padding: 12,
bg_radius: 4,
line_height: 1.3,
max_chars_per_line: 12,
},
},
{
key: "minimal_clean",
label: "极简无描边",
emoji: "✨",
style: {
...BASE,
font: "苹方",
size: 48,
color: "#ffffff",
bold: true,
italic: false,
stroke: false,
shadow: false,
bg_enabled: false,
line_height: 1.3,
max_chars_per_line: 10,
},
},
]
/** 根据 key 获取预设 */
export function getTitlePreset(key: string): TitlePreset | undefined {
return TITLE_PRESETS.find((p) => p.key === key)
}
+118
View File
@@ -0,0 +1,118 @@
/**
* 共享标题样式配置类型(#2001 爆款标题样式配置面板升级)
*
* 设计原则:
* 1. 向后兼容:保留旧的 bold/stroke/shadow 布尔字段,新增细粒度字段
* stroke_width/stroke_color/shadow_offset_x-y-blur-color/bg_enabled-color-padding-radius/line_height/margin_top/max_chars_per_line)。
* 2. 后端契约:字段名使用 snake_case,与 title_config dict 直接对齐。
* 3. line_overrides 支持逐行覆盖(选中某行单独设置颜色/字号/关键词高亮/加粗/斜体)。
* 4. cover_title_config 为封面独立标题样式,null 表示封面沿用主标题样式。
*/
/** 关键词高亮配置 */
export interface TitleKeywordHighlight {
/** 要高亮的词 */
word: string
/** 高亮颜色(可选,默认主色反转) */
color?: string
/** 是否加粗(默认 true */
bold?: boolean
/** 额外字号放大倍数(1.0=不变,1.3=放大 30% */
scale?: number
}
/** 单行覆盖配置 */
export interface TitleLineOverride {
/** 行索引(0-based,按 / 或自动换行后的行序) */
line_index: number
/** 覆盖后的文字(可选,默认沿用原行) */
text?: string
/** 覆盖字号(可选) */
size?: number
/** 覆盖字色(可选) */
color?: string
/** 覆盖加粗(可选) */
bold?: boolean
/** 覆盖斜体(可选) */
italic?: boolean
/** 覆盖描边开关(可选) */
stroke?: boolean
/** 关键词高亮列表 */
highlights?: TitleKeywordHighlight[]
}
/**
* 标题样式配置(不含 title 文字本身,不含 auto_subtitle)。
*
* cover_title_config 使用 Partial<Omit<...,"cover_title_config">> 递归避免无限类型。
*/
export interface TitleStyleConfig {
/* ── 基础 ── */
font: string
size: number
color: string
bold: boolean
italic: boolean
position: "top" | "center" | "bottom" | "custom"
pos_x?: number
pos_y?: number
/* ── 排版(P0 ── */
/** 行距倍数(默认 1.2 */
line_height: number
/** 顶部边距(position=top 时距画面顶部距离,px @720p,默认 24 */
margin_top: number
/** 每行最大字符数(4-20,超出自动换行;0=不自动换行,使用 / 手动分行) */
max_chars_per_line: number
/* ── 描边参数化(P0) ── */
stroke: boolean
stroke_width: number
stroke_color: string
/* ── 阴影参数化(P1) ── */
shadow: boolean
shadow_offset_x: number
shadow_offset_y: number
shadow_blur: number
shadow_color: string
/* ── 背景色块(P1) ── */
bg_enabled: boolean
bg_color: string
bg_padding: number
bg_radius: number
/* ── 逐行独立样式(P1) ── */
line_overrides: TitleLineOverride[]
/* ── 封面独立标题配置(P1):null=沿用主标题样式 ── */
cover_title_config: null | Partial<Omit<TitleStyleConfig, "cover_title_config">>
}
/** 默认样式(经典白字黑描边,保持老版本观感) */
export const DEFAULT_TITLE_STYLE: TitleStyleConfig = {
font: "思源黑体",
size: 48,
color: "#ffffff",
bold: true,
italic: false,
position: "bottom",
line_height: 1.2,
margin_top: 24,
max_chars_per_line: 0,
stroke: true,
stroke_width: 4,
stroke_color: "#000000",
shadow: false,
shadow_offset_x: 2,
shadow_offset_y: 2,
shadow_blur: 4,
shadow_color: "rgba(0,0,0,0.8)",
bg_enabled: false,
bg_color: "rgba(0,0,0,0.5)",
bg_padding: 12,
bg_radius: 8,
line_overrides: [],
cover_title_config: null,
}
@@ -0,0 +1,151 @@
/**
* TTS 配音风格选择器
* - 6 种预设风格卡片(自然亲切 / 激动兴奋 / 沉稳专业 / 温柔甜美 / 新闻播报 / 直播带货)
* - 卡片单选,选中高亮紫色
* - 默认 natural
*
* 复用方式:
* <TtsStyleSelector value={style} onChange={setStyle} />
* <TtsStyleSelector value={style} onChange={setStyle} compact /> // 紧凑模式(小尺寸)
*/
import React from "react"
import { TTS_STYLE_OPTIONS, DEFAULT_TTS_STYLE, type TtsStyle } from "@/api/tts/styles"
export interface TtsStyleSelectorProps {
value?: TtsStyle | string
onChange: (style: TtsStyle) => void
/** 紧凑模式(小卡片),适合与其他参数并排 */
compact?: boolean
/** 是否显示"配音风格"标签 */
showLabel?: boolean
}
const TtsStyleSelector: React.FC<TtsStyleSelectorProps> = ({
value,
onChange,
compact = false,
showLabel = true,
}) => {
const current = value || DEFAULT_TTS_STYLE
if (compact) {
return (
<div>
{showLabel && (
<div
style={{
fontSize: 13,
color: "var(--text-secondary, #6b7280)",
marginBottom: 6,
}}
>
</div>
)}
<div
style={{
display: "grid",
gridTemplateColumns: "repeat(3, 1fr)",
gap: 6,
}}
>
{TTS_STYLE_OPTIONS.map((opt) => {
const selected = current === opt.value
return (
<button
type="button"
key={opt.value}
onClick={() => onChange(opt.value as TtsStyle)}
title={opt.description}
style={{
padding: "6px 4px",
borderRadius: 6,
border: selected ? "2px solid #7c3aed" : "1px solid #e5e7eb",
background: selected ? "#faf5ff" : "#fff",
color: selected ? "#6d28d9" : "#374151",
cursor: "pointer",
fontSize: 12,
fontWeight: selected ? 600 : 400,
textAlign: "center",
transition: "all 0.15s",
lineHeight: 1.3,
}}
>
<span style={{ marginRight: 3 }}>{opt.emoji}</span>
{opt.label}
</button>
)
})}
</div>
</div>
)
}
return (
<div>
{showLabel && (
<div
style={{
fontSize: 13,
color: "var(--text-secondary, #6b7280)",
marginBottom: 8,
fontWeight: 500,
}}
>
</div>
)}
<div
style={{
display: "grid",
gridTemplateColumns: "repeat(3, 1fr)",
gap: 8,
}}
>
{TTS_STYLE_OPTIONS.map((opt) => {
const selected = current === opt.value
return (
<button
type="button"
key={opt.value}
onClick={() => onChange(opt.value as TtsStyle)}
title={opt.description}
style={{
padding: "10px 8px",
borderRadius: 8,
border: selected ? "2px solid #7c3aed" : "1px solid #e5e7eb",
background: selected ? "#faf5ff" : "#fff",
color: selected ? "#6d28d9" : "#111",
cursor: "pointer",
textAlign: "center",
transition: "all 0.15s",
display: "flex",
flexDirection: "column",
alignItems: "center",
gap: 4,
}}
>
<span style={{ fontSize: 22, lineHeight: 1 }}>{opt.emoji}</span>
<span style={{ fontSize: 13, fontWeight: selected ? 600 : 500 }}>{opt.label}</span>
<span
style={{
fontSize: 10,
color: "#9ca3af",
lineHeight: 1.2,
maxWidth: "100%",
overflow: "hidden",
textOverflow: "ellipsis",
whiteSpace: "nowrap",
}}
>
{opt.description}
</span>
</button>
)
})}
</div>
</div>
)
}
export default TtsStyleSelector
+9 -12
View File
@@ -30,11 +30,7 @@ import {
} from "./api/aiAvatar" } from "./api/aiAvatar"
import { getOrCreateDefaultProject } from "@/api/projects" import { getOrCreateDefaultProject } from "@/api/projects"
import type { RenderJob, SentenceTiming } from "./types" import type { RenderJob, SentenceTiming } from "./types"
import { import { buildTitleConfigPayload, buildCoverConfigPayload } from "./utils/contract"
normalizeEmotion,
buildTitleConfigPayload,
buildCoverConfigPayload,
} from "./utils/contract"
import { renderTitleToPngDataUrl, getVideoResolution } from "./utils/titleCanvas" import { renderTitleToPngDataUrl, getVideoResolution } from "./utils/titleCanvas"
/** 面板折叠状态 */ /** 面板折叠状态 */
@@ -94,7 +90,7 @@ const AiAvatarPage: React.FC = () => {
state.resetTtsPreview() state.resetTtsPreview()
} }
// eslint-disable-next-line react-hooks/exhaustive-deps // eslint-disable-next-line react-hooks/exhaustive-deps
}, [state.scriptText, state.selectedVoice?.voice_id, state.speed, state.emotion]) }, [state.scriptText, state.selectedVoice?.voice_id, state.speed, state.style])
const _clearTtsProgressTimer = useCallback(() => { const _clearTtsProgressTimer = useCallback(() => {
if (ttsProgressTimerRef.current) { if (ttsProgressTimerRef.current) {
@@ -148,7 +144,7 @@ const AiAvatarPage: React.FC = () => {
voice_id: state.selectedVoice!.voice_id, voice_id: state.selectedVoice!.voice_id,
script_text: state.scriptText, script_text: state.scriptText,
speed: state.speed, speed: state.speed,
emotion: normalizeEmotion(state.emotion), style: state.style,
}) })
_clearTtsProgressTimer() _clearTtsProgressTimer()
setTtsProgress(100) setTtsProgress(100)
@@ -175,7 +171,7 @@ const AiAvatarPage: React.FC = () => {
}) })
} }
// eslint-disable-next-line react-hooks/exhaustive-deps // eslint-disable-next-line react-hooks/exhaustive-deps
}, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.emotion]) }, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.style])
const handleRetryTts = useCallback(() => { const handleRetryTts = useCallback(() => {
handleGenerateTts() handleGenerateTts()
@@ -254,7 +250,7 @@ const AiAvatarPage: React.FC = () => {
script_text: state.scriptText, script_text: state.scriptText,
video_url: videoUrl, video_url: videoUrl,
speed: state.speed, speed: state.speed,
emotion: normalizeEmotion(state.emotion), style: state.style,
} }
} }
const job = await createLipsyncJob(payload) const job = await createLipsyncJob(payload)
@@ -298,7 +294,8 @@ const AiAvatarPage: React.FC = () => {
state.selectedVoice, state.selectedVoice,
state.scriptText, state.scriptText,
state.speed, state.speed,
state.emotion,
state.style,
state.ttsPreview, state.ttsPreview,
]) ])
@@ -598,8 +595,8 @@ const AiAvatarPage: React.FC = () => {
onVoiceSourceChange={state.setVoiceSource} onVoiceSourceChange={state.setVoiceSource}
selectedVoice={state.selectedVoice} selectedVoice={state.selectedVoice}
onSelectVoice={state.setSelectedVoice} onSelectVoice={state.setSelectedVoice}
emotion={state.emotion} style={state.style}
onEmotionChange={state.setEmotion} onStyleChange={state.setStyle}
speed={state.speed} speed={state.speed}
onSpeedChange={state.setSpeed} onSpeedChange={state.setSpeed}
language={state.language} language={state.language}
@@ -41,6 +41,8 @@ export const createLipsyncJob = async (data: {
speed?: number speed?: number
/** 情绪英文枚举:neutral/happy/sad/angry/surprised/fearful/disgustedTTS 直生模式用;前端经 normalizeEmotion 归一化) */ /** 情绪英文枚举:neutral/happy/sad/angry/surprised/fearful/disgustedTTS 直生模式用;前端经 normalizeEmotion 归一化) */
emotion?: string emotion?: string
/** 配音风格预设(natural/excited/professional/sweet/news/livestream */
style?: string
enable_video_loop?: boolean enable_video_loop?: boolean
project_id?: string project_id?: string
}): Promise<LipsyncJob> => { }): Promise<LipsyncJob> => {
@@ -55,6 +57,7 @@ export const previewTts = async (data: {
script_text: string script_text: string
speed?: number speed?: number
emotion?: string emotion?: string
style?: string
}): Promise<{ }): Promise<{
audio_url: string audio_url: string
duration: number duration: number
@@ -29,15 +29,7 @@ function formatTime(seconds: number): string {
return `${m}:${s.toString().padStart(2, "0")}` return `${m}:${s.toString().padStart(2, "0")}`
} }
/** 字体名 → CSS font-family 映射(与 titleCanvas 字体链对齐) */ import { getFontFamily as getFontFamilyByKey } from "@/components/title/constants"
const FONT_FAMILY_MAP: Record<string, string> = {
:
"'Noto Sans CJK SC', 'Source Han Sans CN', 'PingFang SC', 'Microsoft YaHei', sans-serif",
: "'Noto Serif SC', 'Source Han Serif SC', 'SimSun', serif",
: "KaiTi, 'STKaiti', serif",
: "'Heiti SC', 'SimHei', 'Microsoft YaHei', sans-serif",
}
const getFontFamily = (font: string): string => FONT_FAMILY_MAP[font] || FONT_FAMILY_MAP["思源黑体"]
export function PanelLipsyncPreview({ export function PanelLipsyncPreview({
lipsyncJob, lipsyncJob,
@@ -83,63 +75,105 @@ export function PanelLipsyncPreview({
const previewScale = containerWidth > 0 ? containerWidth / 720 : 0.35 const previewScale = containerWidth > 0 ? containerWidth / 720 : 0.35
const ps = useCallback((v: number) => Math.round(v * previewScale * 100) / 100, [previewScale]) const ps = useCallback((v: number) => Math.round(v * previewScale * 100) / 100, [previewScale])
/** 标题叠加样式(字号/padding/描边/阴影均按 previewScale 缩放,保持与成片视觉一致 */ /** 标题叠加样式(新字段全支持:描边宽色/阴影参数化/背景块/行距/顶部边距/自动换行 */
const titleOverlayStyle: React.CSSProperties | null = const titleOverlayData =
titleConfig?.title && containerWidth > 0 titleConfig?.title && containerWidth > 0
? (() => { ? (() => {
const c = titleConfig as AiAvatarTitleConfig & {
stroke_width?: number
stroke_color?: string
shadow_offset_x?: number
shadow_offset_y?: number
shadow_blur?: number
shadow_color?: string
line_height?: number
margin_top?: number
max_chars_per_line?: number
bg_enabled?: boolean
bg_color?: string
bg_padding?: number
bg_radius?: number
cover_title_config?: Record<string, unknown> | null
line_overrides?: unknown[]
}
const baseSize = titleConfig.size || 48 const baseSize = titleConfig.size || 48
const fontSize = ps(baseSize) const fontSize = ps(baseSize)
// 描边宽度基准 ≈ size * 0.06,最小 1.5px @720p const strokeW = c.stroke ? ps(c.stroke_width ?? 4) : 0
const strokeW = Math.max(ps(1.5), +(baseSize * 0.06 * previewScale).toFixed(2)) const strokeC = c.stroke_color || "#000000"
// 阴影按比例缩放 const shBlur = ps(c.shadow_blur ?? 4)
const shadowBlur = ps(4) const shOffX = ps(c.shadow_offset_x ?? 2)
const shadowOffsetY = ps(2) const shOffY = ps(c.shadow_offset_y ?? 2)
// padding / top 边距按比例(基准 8px 对应预览小窗,成片基准 16px,这里 8px 对应约 0.33 缩放) const shColor = c.shadow_color || "rgba(0,0,0,0.8)"
const padV = ps(16) * 0.5 // ≈ 8px in ~240px container const lh = c.line_height ?? 1.2
const padH = ps(24) * 0.5 const mTop = ps(c.margin_top ?? 24)
const bgPad = ps(c.bg_padding ?? 12)
const bgR = ps(c.bg_radius ?? 8)
const maxChars = c.max_chars_per_line ?? 0
const rawText = titleConfig.title || ""
const lines = (() => {
const manual = rawText
.split(/[/]/)
.map((l) => l.trim())
.filter(Boolean)
if (!maxChars || maxChars <= 0) return manual
const out: string[] = []
manual.forEach((seg) => {
for (let i = 0; i < seg.length; i += maxChars) out.push(seg.slice(i, i + maxChars))
})
return out
})()
const padV = ps(16) * 0.5
const textShadow = titleConfig.shadow
? `${shOffX}px ${shOffY}px ${shBlur}px ${shColor}`
: undefined
const style: React.CSSProperties = { const style: React.CSSProperties = {
position: "absolute", position: "absolute",
color: titleConfig.color || "#ffffff", color: titleConfig.color || "#ffffff",
fontFamily: getFontFamily(titleConfig.font || "思源黑体"), fontFamily: getFontFamilyByKey(titleConfig.font || "source_sans_sc"),
fontSize: `${fontSize}px`, fontSize: `${fontSize}px`,
fontWeight: titleConfig.bold ? 700 : 400, fontWeight: titleConfig.bold ? 700 : 400,
fontStyle: titleConfig.italic ? "italic" : "normal", fontStyle: titleConfig.italic ? "italic" : "normal",
textAlign: "center", textAlign: "center",
width: "90%", lineHeight: lh,
lineHeight: 1.2, WebkitTextStroke:
padding: `${ps(4)}px ${padH}px`, titleConfig.stroke && strokeW > 0 ? `${strokeW}px ${strokeC}` : undefined,
textShadow: titleConfig.shadow paintOrder: "stroke fill",
? `0 ${shadowOffsetY}px ${shadowBlur}px rgba(0,0,0,0.8), 0 0 ${ps(2)}px rgba(0,0,0,0.5)` textShadow,
: undefined,
WebkitTextStroke: titleConfig.stroke ? `${strokeW}px #000` : undefined,
boxSizing: "border-box",
wordBreak: "break-word",
whiteSpace: "pre-wrap", whiteSpace: "pre-wrap",
padding: c.bg_enabled ? `${bgPad}px ${bgPad}px` : 0,
background: c.bg_enabled ? c.bg_color || "rgba(0,0,0,0.5)" : "transparent",
borderRadius: c.bg_enabled ? `${bgR}px` : 0,
boxSizing: "border-box",
display: "inline-block",
maxWidth: "94%",
}
const wrap: React.CSSProperties = {
position: "absolute",
left: "50%",
width: "100%",
display: "flex",
justifyContent: "center",
pointerEvents: onTitlePositionChange ? "auto" : "none",
} }
if ( if (
titleConfig.position === "custom" && titleConfig.position === "custom" &&
titleConfig.pos_x != null && titleConfig.pos_x != null &&
titleConfig.pos_y != null titleConfig.pos_y != null
) { ) {
style.left = `${titleConfig.pos_x}%` wrap.left = `${titleConfig.pos_x}%`
style.top = `${titleConfig.pos_y}%` wrap.top = `${titleConfig.pos_y}%`
style.transform = "translateX(-50%) translateY(-50%)" wrap.transform = "translate(-50%, -50%)"
} else if (titleConfig.position === "top") { } else if (titleConfig.position === "top") {
style.left = "50%" wrap.top = `${padV + mTop}px`
style.top = padV wrap.transform = "translateX(-50%)"
style.transform = "translateX(-50%)"
} else if (titleConfig.position === "bottom") { } else if (titleConfig.position === "bottom") {
style.left = "50%" wrap.bottom = `${padV}px`
style.bottom = padV wrap.transform = "translateX(-50%)"
style.transform = "translateX(-50%)"
} else { } else {
style.left = "50%" wrap.top = "50%"
style.top = "50%" wrap.transform = "translate(-50%, -50%)"
style.transform = "translateX(-50%) translateY(-50%)"
} }
return style return { style, wrap, lines }
})() })()
: null : null
@@ -252,25 +286,23 @@ export function PanelLipsyncPreview({
{isDone && lipsyncJob?.output_video_url ? ( {isDone && lipsyncJob?.output_video_url ? (
<div style={{ position: "relative", width: "100%", height: "100%" }}> <div style={{ position: "relative", width: "100%", height: "100%" }}>
<video src={lipsyncJob.output_video_url} controls /> <video src={lipsyncJob.output_video_url} controls />
{titleOverlayStyle && ( {titleOverlayData && (
<div <div
ref={titleDragRef} ref={titleDragRef}
style={{ style={{
...titleOverlayStyle, ...titleOverlayData.wrap,
cursor: onTitlePositionChange ? "grab" : "default", cursor: onTitlePositionChange ? "grab" : "default",
pointerEvents: onTitlePositionChange ? "auto" : "none",
}} }}
onPointerDown={handleTitlePointerDown} onPointerDown={handleTitlePointerDown}
onPointerMove={handleTitlePointerMove} onPointerMove={handleTitlePointerMove}
onPointerUp={handleTitlePointerUp} onPointerUp={handleTitlePointerUp}
onPointerCancel={handleTitlePointerUp} onPointerCancel={handleTitlePointerUp}
> >
{titleConfig!.title.split(/[/]/).map((part, i) => ( <div style={titleOverlayData.style}>
<span key={i}> {titleOverlayData.lines.map((part: string, i: number) => (
{i > 0 && <br />} <div key={i}>{part}</div>
{part} ))}
</span> </div>
))}
</div> </div>
)} )}
</div> </div>
@@ -13,7 +13,8 @@ import TitleStylePanel from "@/pages/generate/components/title/TitleStylePanel"
import TitleLibraryAutoComplete from "@/pages/generate/components/title/TitleLibraryAutoComplete" import TitleLibraryAutoComplete from "@/pages/generate/components/title/TitleLibraryAutoComplete"
import type { TitleOption } from "@/pages/generate/components/title/TitleLibraryAutoComplete" import type { TitleOption } from "@/pages/generate/components/title/TitleLibraryAutoComplete"
import type { TitleSettings } from "@/pages/generate/types" import type { TitleSettings } from "@/pages/generate/types"
import { POSITION_OPTIONS, FONT_OPTIONS, TITLE_PRESETS } from "@/pages/generate/constants" import { POSITION_OPTIONS } from "@/pages/generate/constants"
import { FONT_OPTIONS, TITLE_PRESETS } from "@/components/title/constants"
import type { AiAvatarTitleConfig } from "../types" import type { AiAvatarTitleConfig } from "../types"
// #1894: 标题数据源切换到文案库,取 script.title 作为候选 // #1894: 标题数据源切换到文案库,取 script.title 作为候选
import { getScripts } from "@/api/scripts" import { getScripts } from "@/api/scripts"
@@ -49,9 +50,60 @@ const PanelTitleConfig: React.FC<PanelTitleConfigProps> = ({ titleConfig, onUpda
.catch(() => setTitleOptions([])) .catch(() => setTitleOptions([]))
}, []) }, [])
/** AiAvatarTitleConfig → TitleSettings(补齐 aiAutoSelect / 自由坐标字段) */ /** AiAvatarTitleConfig (snake_case) → TitleSettings (camelCase) */
const titleSettings: TitleSettings = useMemo( const titleSettings: TitleSettings = useMemo(() => {
() => ({ const c = titleConfig as AiAvatarTitleConfig & {
stroke_width?: number
stroke_color?: string
shadow_offset_x?: number
shadow_offset_y?: number
shadow_blur?: number
shadow_color?: string
line_height?: number
margin_top?: number
max_chars_per_line?: number
bg_enabled?: boolean
bg_color?: string
bg_padding?: number
bg_radius?: number
cover_title_config?: {
title?: string
font?: string
size?: number
font_size?: number
color?: string
font_color?: string
bold?: boolean
italic?: boolean
position?: string
stroke?: { enabled: boolean; width?: number; color?: string } | boolean
stroke_width?: number
stroke_color?: string
shadow?:
| {
enabled: boolean
offset_x?: number
offset_y?: number
blur?: number
color?: string
}
| boolean
shadow_offset_x?: number
shadow_offset_y?: number
shadow_blur?: number
shadow_color?: string
background?: { enabled: boolean; color?: string; padding?: number; radius?: number }
bg_enabled?: boolean
bg_color?: string
bg_padding?: number
bg_radius?: number
line_height?: number
margin_top?: number
max_chars_per_line?: number
} | null
line_overrides?: unknown[]
}
return {
aiAutoSelect: false, aiAutoSelect: false,
title: titleConfig.title, title: titleConfig.title,
position: titleConfig.position, position: titleConfig.position,
@@ -64,24 +116,219 @@ const PanelTitleConfig: React.FC<PanelTitleConfigProps> = ({ titleConfig, onUpda
color: titleConfig.color, color: titleConfig.color,
posX: null, posX: null,
posY: null, posY: null,
}), lineHeight: c.line_height ?? 1.2,
[titleConfig], marginTop: c.margin_top ?? 24,
) maxCharsPerLine: c.max_chars_per_line ?? 0,
strokeWidth: c.stroke_width ?? 4,
strokeColor: c.stroke_color ?? "#000000",
shadowOffsetX: c.shadow_offset_x ?? 2,
shadowOffsetY: c.shadow_offset_y ?? 2,
shadowBlur: c.shadow_blur ?? 4,
shadowColor: c.shadow_color ?? "rgba(0,0,0,0.8)",
bgEnabled: !!c.bg_enabled,
bgColor: c.bg_color ?? "rgba(0,0,0,0.5)",
bgPadding: c.bg_padding ?? 12,
bgRadius: c.bg_radius ?? 8,
lineOverrides: Array.isArray(c.line_overrides) ? c.line_overrides : [],
coverTitle: (() => {
const ct = c.cover_title_config as
| null
| (AiAvatarTitleConfig & {
font_size?: number
font_color?: string
stroke?: { enabled?: boolean; width?: number; color?: string } | boolean
stroke_width?: number
stroke_color?: string
shadow?:
| {
enabled?: boolean
offset_x?: number
offset_y?: number
blur?: number
color?: string
}
| boolean
shadow_offset_x?: number
shadow_offset_y?: number
shadow_blur?: number
shadow_color?: string
background?: { enabled?: boolean; color?: string; padding?: number; radius?: number }
bg_enabled?: boolean
bg_color?: string
bg_padding?: number
bg_radius?: number
})
if (!ct) return null
const ctStroke = ct.stroke as
{ enabled?: boolean; width?: number; color?: string } | boolean | undefined
const ctShadow = ct.shadow as
| {
enabled?: boolean
offset_x?: number
offset_y?: number
blur?: number
color?: string
}
| boolean
| undefined
const ctBg = ct.background as
{ enabled?: boolean; color?: string; padding?: number; radius?: number } | undefined
return {
title: ct.title,
font: ct.font,
size: ct.font_size ?? ct.size,
color: ct.font_color ?? ct.color,
bold: ct.bold,
italic: ct.italic,
position: ct.position,
stroke:
typeof ctStroke === "object" && ctStroke ? ctStroke.enabled !== false : !!ctStroke,
strokeWidth:
(typeof ctStroke === "object" && ctStroke ? ctStroke.width : undefined) ??
ct.stroke_width ??
4,
strokeColor:
(typeof ctStroke === "object" && ctStroke ? ctStroke.color : undefined) ??
ct.stroke_color ??
"#000000",
shadow:
typeof ctShadow === "object" && ctShadow ? ctShadow.enabled !== false : !!ctShadow,
shadowOffsetX:
(typeof ctShadow === "object" && ctShadow ? ctShadow.offset_x : undefined) ??
ct.shadow_offset_x ??
2,
shadowOffsetY:
(typeof ctShadow === "object" && ctShadow ? ctShadow.offset_y : undefined) ??
ct.shadow_offset_y ??
2,
shadowBlur:
(typeof ctShadow === "object" && ctShadow ? ctShadow.blur : undefined) ??
ct.shadow_blur ??
4,
shadowColor:
(typeof ctShadow === "object" && ctShadow ? ctShadow.color : undefined) ??
ct.shadow_color ??
"rgba(0,0,0,0.8)",
bgEnabled: ctBg?.enabled ?? !!ct.bg_enabled,
bgColor: ctBg?.color ?? ct.bg_color ?? "rgba(0,0,0,0.5)",
bgPadding: ctBg?.padding ?? ct.bg_padding ?? 12,
bgRadius: ctBg?.radius ?? ct.bg_radius ?? 8,
}
})(),
}
}, [titleConfig])
/** 应用预设:与智能剪辑一致,只覆盖 color/bold/italic/stroke/shadow,不改变字号 */ /** 应用预设:覆盖新细粒度字段(颜色/描边/阴影/字号/字体等) */
const handleApplyPreset = (presetKey: string) => { const handleApplyPreset = (presetKey: string) => {
const preset = TITLE_PRESETS.find((p) => p.key === presetKey) const preset = TITLE_PRESETS.find((p) => p.key === presetKey)
if (!preset) return if (!preset) return
setActivePreset(presetKey) setActivePreset(presetKey)
const st = preset.style || {}
onUpdate({ onUpdate({
color: preset.style.color, font: st.font,
bold: preset.style.bold, size: st.size,
italic: preset.style.italic, color: st.color,
stroke: preset.style.stroke, bold: st.bold,
shadow: preset.style.shadow, italic: st.italic,
stroke: st.stroke,
stroke_width: st.stroke_width,
stroke_color: st.stroke_color,
shadow: st.shadow,
shadow_offset_x: st.shadow_offset_x,
shadow_offset_y: st.shadow_offset_y,
shadow_blur: st.shadow_blur,
shadow_color: st.shadow_color,
bg_enabled: st.bg_enabled,
bg_color: st.bg_color,
bg_padding: st.bg_padding,
bg_radius: st.bg_radius,
line_overrides: [],
cover_title_config: null,
}) })
} }
/** 字段 patch 透传:TitleStylePanel 的 onUpdateStylecamelCase → snake_case */
const handleUpdateStyle = (patch: Partial<TitleSettings>) => {
const snake: Record<string, unknown> = {}
const map: Record<string, string> = {
lineHeight: "line_height",
marginTop: "margin_top",
maxCharsPerLine: "max_chars_per_line",
strokeWidth: "stroke_width",
strokeColor: "stroke_color",
shadowOffsetX: "shadow_offset_x",
shadowOffsetY: "shadow_offset_y",
shadowBlur: "shadow_blur",
shadowColor: "shadow_color",
bgEnabled: "bg_enabled",
bgColor: "bg_color",
bgPadding: "bg_padding",
bgRadius: "bg_radius",
lineOverrides: "line_overrides",
coverTitle: "cover_title_config",
}
Object.entries(patch).forEach(([k, v]) => {
if (k === "coverTitle" && v && typeof v === "object") {
const ct = v as {
title?: string
font?: string
size?: number
color?: string
bold?: boolean
italic?: boolean
position?: string
stroke?: boolean
strokeWidth?: number
strokeColor?: string
shadow?: boolean
shadowOffsetX?: number
shadowOffsetY?: number
shadowBlur?: number
shadowColor?: string
bgEnabled?: boolean
bgColor?: string
bgPadding?: number
bgRadius?: number
lineHeight?: number
marginTop?: number
maxCharsPerLine?: number
}
snake.cover_title_config = {
title: ct.title,
font: ct.font,
font_size: ct.size,
font_color: ct.color,
bold: ct.bold,
italic: ct.italic,
position: ct.position,
stroke: ct.stroke
? { enabled: true, width: ct.strokeWidth ?? 4, color: ct.strokeColor ?? "#000" }
: { enabled: false },
shadow: ct.shadow
? {
enabled: true,
offset_x: ct.shadowOffsetX ?? 2,
offset_y: ct.shadowOffsetY ?? 2,
blur: ct.shadowBlur ?? 4,
color: ct.shadowColor ?? "rgba(0,0,0,0.8)",
}
: { enabled: false },
background: ct.bgEnabled
? { enabled: true, color: ct.bgColor, padding: ct.bgPadding, radius: ct.bgRadius }
: { enabled: false },
line_height: ct.lineHeight,
margin_top: ct.marginTop,
max_chars_per_line: ct.maxCharsPerLine,
}
} else if (map[k]) {
snake[map[k]] = v
} else {
snake[k] = v
}
})
onUpdate(snake)
}
return ( return (
<div className="aa-title-config"> <div className="aa-title-config">
{/* 主标题输入 — TextArea 多行 + 标题库选择 */} {/* 主标题输入 — TextArea 多行 + 标题库选择 */}
@@ -125,8 +372,13 @@ const PanelTitleConfig: React.FC<PanelTitleConfigProps> = ({ titleConfig, onUpda
onToggleStroke={() => onUpdate({ stroke: !titleConfig.stroke })} onToggleStroke={() => onUpdate({ stroke: !titleConfig.stroke })}
onToggleShadow={() => onUpdate({ shadow: !titleConfig.shadow })} onToggleShadow={() => onUpdate({ shadow: !titleConfig.shadow })}
onApplyPreset={handleApplyPreset} onApplyPreset={handleApplyPreset}
onUpdateStyle={handleUpdateStyle}
showCoverToggle
previewWidth={280}
activePreset={activePreset} activePreset={activePreset}
titlePresets={TITLE_PRESETS} titlePresets={
TITLE_PRESETS as unknown as React.ComponentProps<typeof TitleStylePanel>["titlePresets"]
}
POSITION_OPTIONS={POSITION_OPTIONS} POSITION_OPTIONS={POSITION_OPTIONS}
FONT_OPTIONS={FONT_OPTIONS} FONT_OPTIONS={FONT_OPTIONS}
/> />
@@ -1,18 +1,17 @@
/** /**
* AI数字人 — 配音库面板(面板3) * AI数字人 — 配音库面板(面板3)
* 音色来源切换(系统预设 / 我的音色)、音色选择与试听、情绪/语速/语言参数 * 音色来源切换(系统预设 / 我的音色)、音色选择与试听、风格/语速/语言参数
*/ */
import { useEffect, useRef, useState } from "react" import { useEffect, useRef, useState } from "react"
import { message } from "antd" import { message } from "antd"
import { fetchVoices } from "@/api/voices/voices" import { fetchVoices } from "@/api/voices/voices"
import { previewTts } from "@/api/tts" import { previewTts } from "@/api/tts"
import { normalizeEmotion } from "../utils/contract" import TtsStyleSelector from "@/components/voice/TtsStyleSelector"
import type { TtsStyle } from "@/api/tts/styles"
import type { UnifiedVoiceItem } from "@/api/voices/types" import type { UnifiedVoiceItem } from "@/api/voices/types"
import { import {
type VoiceSource, type VoiceSource,
type VoiceEmotion,
type VoiceLanguage, type VoiceLanguage,
VOICE_EMOTION_OPTIONS,
PRESET_VOICE_LANGUAGE_OPTIONS, PRESET_VOICE_LANGUAGE_OPTIONS,
CLONE_VOICE_LANGUAGE_OPTIONS, CLONE_VOICE_LANGUAGE_OPTIONS,
} from "../types" } from "../types"
@@ -22,8 +21,8 @@ interface PanelVoiceSelectorProps {
onVoiceSourceChange: (source: VoiceSource) => void onVoiceSourceChange: (source: VoiceSource) => void
selectedVoice: UnifiedVoiceItem | null selectedVoice: UnifiedVoiceItem | null
onSelectVoice: (voice: UnifiedVoiceItem) => void onSelectVoice: (voice: UnifiedVoiceItem) => void
emotion: VoiceEmotion style: TtsStyle
onEmotionChange: (e: VoiceEmotion) => void onStyleChange: (s: TtsStyle) => void
speed: number speed: number
onSpeedChange: (s: number) => void onSpeedChange: (s: number) => void
language: VoiceLanguage language: VoiceLanguage
@@ -35,8 +34,8 @@ export function PanelVoiceSelector({
onVoiceSourceChange, onVoiceSourceChange,
selectedVoice, selectedVoice,
onSelectVoice, onSelectVoice,
emotion, style,
onEmotionChange, onStyleChange,
speed, speed,
onSpeedChange, onSpeedChange,
language, language,
@@ -139,31 +138,30 @@ export function PanelVoiceSelector({
/* 克隆音色:preview_url/audio_url 通常为空,需走 POST /tts/preview /* 克隆音色:preview_url/audio_url 通常为空,需走 POST /tts/preview
* 现合成示例文案再播放,对齐配音库 useAudioPlayer 行为 */ * 现合成示例文案再播放,对齐配音库 useAudioPlayer 行为 */
if (voice.type === "clone") { if (voice.type === "clone") {
const cached = previewCacheRef.current.get(voice.voice_clone_profile_id || voice.id) const cacheKey = `${voice.voice_clone_profile_id || voice.id}::${style}`
const cached = previewCacheRef.current.get(cacheKey)
if (cached) { if (cached) {
playAudioUrl(voice.id, cached) playAudioUrl(voice.id, cached)
return return
} }
const targetId = voice.voice_clone_profile_id || voice.id const targetId = voice.voice_clone_profile_id || voice.id
// DEBUG: 打印请求参数,帮助定位 /tts/preview 失败原因
setPreviewingId(voice.id) setPreviewingId(voice.id)
try { try {
const res = await previewTts({ const res = await previewTts({
text: VOICE_PREVIEW_TEXT, text: VOICE_PREVIEW_TEXT,
voice_id: targetId, voice_id: targetId,
speed: speed, // 透传用户选择的语速(#1822) speed: speed, // 透传用户选择的语速(#1822)
emotion: normalizeEmotion(emotion), // 情绪中文→英文枚举 style,
}) })
if (!res.audio_url) { if (!res.audio_url) {
setPreviewingId(null) setPreviewingId(null)
message.error("合成试听失败:未返回音频") message.error("合成试听失败:未返回音频")
return return
} }
previewCacheRef.current.set(targetId, res.audio_url) previewCacheRef.current.set(cacheKey, res.audio_url)
playAudioUrl(voice.id, res.audio_url) playAudioUrl(voice.id, res.audio_url)
} catch (err) { } catch (err) {
setPreviewingId(null) setPreviewingId(null)
// DEBUG: 打印详细错误信息
console.error("[AI数字人-克隆试听] previewTts 失败:", { console.error("[AI数字人-克隆试听] previewTts 失败:", {
status: (err as { response?: { status?: number } })?.response?.status, status: (err as { response?: { status?: number } })?.response?.status,
data: (err as { response?: { data?: unknown } })?.response?.data, data: (err as { response?: { data?: unknown } })?.response?.data,
@@ -272,23 +270,6 @@ export function PanelVoiceSelector({
{/* 配音参数 */} {/* 配音参数 */}
<div className="aa-voice-params"> <div className="aa-voice-params">
<div className="aa-voice-params__row"> <div className="aa-voice-params__row">
<div className="aa-voice-params__field">
<label className="aa-label" htmlFor="aa-voice-emotion">
</label>
<select
id="aa-voice-emotion"
className="aa-select"
value={emotion}
onChange={(e) => onEmotionChange(e.target.value as VoiceEmotion)}
>
{VOICE_EMOTION_OPTIONS.map((opt) => (
<option key={opt.value} value={opt.value}>
{opt.label}
</option>
))}
</select>
</div>
<div className="aa-voice-params__field"> <div className="aa-voice-params__field">
<label className="aa-label" htmlFor="aa-voice-language"> <label className="aa-label" htmlFor="aa-voice-language">
@@ -324,6 +305,9 @@ export function PanelVoiceSelector({
onChange={(e) => handleSpeedChange(e.target.value)} onChange={(e) => handleSpeedChange(e.target.value)}
/> />
</div> </div>
<div className="aa-voice-params__field">
<TtsStyleSelector value={style} onChange={onStyleChange} compact />
</div>
</div> </div>
</div> </div>
) )
@@ -6,7 +6,6 @@ import type { AssetItem } from "@/api/assets"
import type { UnifiedVoiceItem } from "@/api/voices/types" import type { UnifiedVoiceItem } from "@/api/voices/types"
import { import {
type VoiceSource, type VoiceSource,
type VoiceEmotion,
type VoiceLanguage, type VoiceLanguage,
type Script, type Script,
type LipsyncJob, type LipsyncJob,
@@ -17,6 +16,7 @@ import {
DEFAULT_TITLE_CONFIG, DEFAULT_TITLE_CONFIG,
DEFAULT_COVER_CONFIG, DEFAULT_COVER_CONFIG,
} from "../types" } from "../types"
import { DEFAULT_TTS_STYLE, type TtsStyle } from "@/api/tts/styles"
const DEFAULT_TTS_PREVIEW: TtsPreviewResult = { const DEFAULT_TTS_PREVIEW: TtsPreviewResult = {
audioUrl: null, audioUrl: null,
@@ -34,7 +34,7 @@ export function useAiAvatar() {
/* ── 面板2:配音库 ── */ /* ── 面板2:配音库 ── */
const [voiceSource, setVoiceSource] = useState<VoiceSource>("preset") const [voiceSource, setVoiceSource] = useState<VoiceSource>("preset")
const [selectedVoice, setSelectedVoice] = useState<UnifiedVoiceItem | null>(null) const [selectedVoice, setSelectedVoice] = useState<UnifiedVoiceItem | null>(null)
const [emotion, setEmotion] = useState<VoiceEmotion>("neutral") const [style, setStyle] = useState<TtsStyle>(DEFAULT_TTS_STYLE)
const [speed, setSpeed] = useState(1.0) const [speed, setSpeed] = useState(1.0)
const [language, setLanguage] = useState<VoiceLanguage>("zh") const [language, setLanguage] = useState<VoiceLanguage>("zh")
@@ -113,8 +113,8 @@ export function useAiAvatar() {
setVoiceSource, setVoiceSource,
selectedVoice, selectedVoice,
setSelectedVoice, setSelectedVoice,
emotion, style,
setEmotion, setStyle,
speed, speed,
setSpeed, setSpeed,
language, language,
+53 -2
View File
@@ -100,7 +100,7 @@ export interface BRollSegment {
pip_scale: number pip_scale: number
} }
/* ── 标题配置 ── */ /* ── 标题配置(#2001 升级:细粒度描边/阴影/背景/排版/逐行/封面独立标题) ── */
export interface AiAvatarTitleConfig { export interface AiAvatarTitleConfig {
title: string title: string
position: string position: string
@@ -115,6 +115,42 @@ export interface AiAvatarTitleConfig {
/** 自定义位置坐标(position=custom 时生效,百分比 0-100 */ /** 自定义位置坐标(position=custom 时生效,百分比 0-100 */
pos_x?: number pos_x?: number
pos_y?: number pos_y?: number
/* ── 排版 ── */
line_height: number
margin_top: number
max_chars_per_line: number
/* ── 描边参数化 ── */
stroke_width: number
stroke_color: string
/* ── 阴影参数化 ── */
shadow_offset_x: number
shadow_offset_y: number
shadow_blur: number
shadow_color: string
/* ── 背景色块 ── */
bg_enabled: boolean
bg_color: string
bg_padding: number
bg_radius: number
/* ── 逐行覆盖 ── */
line_overrides: Array<{
line_index: number
text?: string
size?: number
color?: string
bold?: boolean
italic?: boolean
stroke?: boolean
highlights?: Array<{ word: string; color?: string; bold?: boolean; scale?: number }>
}>
/* ── 封面独立标题(null=沿用主标题) ── */
cover_title_config: null | Partial<AiAvatarTitleConfig>
} }
/* ── 封面配置 ── */ /* ── 封面配置 ── */
@@ -149,12 +185,27 @@ export const DEFAULT_TITLE_CONFIG: AiAvatarTitleConfig = {
size: 48, size: 48,
bold: true, bold: true,
italic: false, italic: false,
stroke: false, stroke: true,
shadow: false, shadow: false,
color: "#ffffff", color: "#ffffff",
auto_subtitle: true, auto_subtitle: true,
pos_x: undefined, pos_x: undefined,
pos_y: undefined, pos_y: undefined,
line_height: 1.2,
margin_top: 24,
max_chars_per_line: 0,
stroke_width: 4,
stroke_color: "#000000",
shadow_offset_x: 2,
shadow_offset_y: 2,
shadow_blur: 4,
shadow_color: "rgba(0,0,0,0.8)",
bg_enabled: false,
bg_color: "rgba(0,0,0,0.5)",
bg_padding: 12,
bg_radius: 8,
line_overrides: [],
cover_title_config: null,
} }
export const DEFAULT_COVER_CONFIG: AiAvatarCoverConfig = { export const DEFAULT_COVER_CONFIG: AiAvatarCoverConfig = {
+86 -3
View File
@@ -67,6 +67,37 @@ export function buildTitleConfigPayload(
const text = (cfg.title || "").trim() const text = (cfg.title || "").trim()
if (!text) return {} if (!text) return {}
const position = cfg.position || "bottom" const position = cfg.position || "bottom"
const anyCfg = cfg as AiAvatarTitleConfig & {
stroke_width?: number
stroke_color?: string
shadow_offset_x?: number
shadow_offset_y?: number
shadow_blur?: number
shadow_color?: string
line_height?: number
margin_top?: number
max_chars_per_line?: number
bg_enabled?: boolean
bg_color?: string
bg_padding?: number
bg_radius?: number
line_overrides?: unknown[]
cover_title_config?: Record<string, unknown> | null
}
const strokeWidth = anyCfg.stroke_width != null ? anyCfg.stroke_width : 4
const strokeColor = anyCfg.stroke_color || "#000000"
const shadowOffsetX = anyCfg.shadow_offset_x != null ? anyCfg.shadow_offset_x : 2
const shadowOffsetY = anyCfg.shadow_offset_y != null ? anyCfg.shadow_offset_y : 2
const shadowBlur = anyCfg.shadow_blur != null ? anyCfg.shadow_blur : 4
const shadowColor = anyCfg.shadow_color || "rgba(0,0,0,0.8)"
const lineHeight = anyCfg.line_height != null ? anyCfg.line_height : 1.2
const marginTop = anyCfg.margin_top != null ? anyCfg.margin_top : 24
const maxCharsPerLine = anyCfg.max_chars_per_line ?? 0
const bgEnabled = !!anyCfg.bg_enabled
const bgColor = anyCfg.bg_color || "rgba(0,0,0,0.5)"
const bgPadding = anyCfg.bg_padding != null ? anyCfg.bg_padding : 12
const bgRadius = anyCfg.bg_radius != null ? anyCfg.bg_radius : 8
const payload: Record<string, unknown> = { const payload: Record<string, unknown> = {
text, text,
enabled: true, enabled: true,
@@ -75,16 +106,68 @@ export function buildTitleConfigPayload(
font_color: cfg.color || "#ffffff", font_color: cfg.color || "#ffffff",
position, position,
bold: !!cfg.bold, bold: !!cfg.bold,
stroke: cfg.stroke ? { enabled: true, width: 2, color: "#000000" } : { enabled: false }, italic: !!cfg.italic,
shadow: cfg.shadow stroke: cfg.stroke
? { enabled: true, color: "#000000", offset_x: 2, offset_y: 2 } ? { enabled: true, width: strokeWidth, color: strokeColor }
: { enabled: false }, : { enabled: false },
shadow: cfg.shadow
? {
enabled: true,
color: shadowColor,
offset_x: shadowOffsetX,
offset_y: shadowOffsetY,
blur: shadowBlur,
}
: { enabled: false },
line_height: lineHeight,
margin_top: marginTop,
max_chars_per_line: maxCharsPerLine,
background: bgEnabled
? { enabled: true, color: bgColor, padding: bgPadding, radius: bgRadius }
: { enabled: false },
line_overrides: Array.isArray(anyCfg.line_overrides) ? anyCfg.line_overrides : [],
} }
// 自定义坐标(custom 位置) // 自定义坐标(custom 位置)
if (position === "custom" && typeof cfg.pos_x === "number" && typeof cfg.pos_y === "number") { if (position === "custom" && typeof cfg.pos_x === "number" && typeof cfg.pos_y === "number") {
payload.pos_x = cfg.pos_x payload.pos_x = cfg.pos_x
payload.pos_y = cfg.pos_y payload.pos_y = cfg.pos_y
} }
// 封面独立标题配置
if (anyCfg.cover_title_config) {
const ctc = anyCfg.cover_title_config
payload.cover_title_config = {
title: ctc.title,
font: ctc.font,
font_size: ctc.size,
font_color: ctc.color,
position: ctc.position,
bold: ctc.bold,
italic: ctc.italic,
stroke: ctc.stroke
? { enabled: true, width: ctc.stroke_width ?? 4, color: ctc.stroke_color ?? "#000000" }
: { enabled: false },
shadow: ctc.shadow
? {
enabled: true,
color: ctc.shadow_color ?? shadowColor,
offset_x: ctc.shadow_offset_x ?? 2,
offset_y: ctc.shadow_offset_y ?? 2,
blur: ctc.shadow_blur ?? 4,
}
: { enabled: false },
line_height: ctc.line_height ?? lineHeight,
margin_top: ctc.margin_top ?? marginTop,
max_chars_per_line: ctc.max_chars_per_line ?? maxCharsPerLine,
background: ctc.bg_enabled
? {
enabled: true,
color: ctc.bg_color ?? bgColor,
padding: ctc.bg_padding ?? bgPadding,
radius: ctc.bg_radius ?? bgRadius,
}
: { enabled: false },
}
}
// 前端 Canvas 渲染好的 PNG dataURL(所见即所得,后端优先 overlay 此图片图层) // 前端 Canvas 渲染好的 PNG dataURL(所见即所得,后端优先 overlay 此图片图层)
if (titleImageDataUrl) { if (titleImageDataUrl) {
payload.title_image_dataurl = titleImageDataUrl payload.title_image_dataurl = titleImageDataUrl
+211 -76
View File
@@ -9,37 +9,78 @@
* 按 videoWidth / 720 得到 scale,所有长度类参数乘以 scale, * 按 videoWidth / 720 得到 scale,所有长度类参数乘以 scale,
* 保证 1080p / 4K 成片里标题视觉大小与预览一致。 * 保证 1080p / 4K 成片里标题视觉大小与预览一致。
*/ */
import { getFontFamily } from "@/components/title/constants"
import type { AiAvatarTitleConfig } from "../types" import type { AiAvatarTitleConfig } from "../types"
export interface RenderTitlePngOptions { export interface RenderTitlePngOptions {
/** 标题配置 */
titleConfig: AiAvatarTitleConfig titleConfig: AiAvatarTitleConfig
/** 视频宽度(像素),默认 720 */
videoWidth?: number videoWidth?: number
/** 视频高度(像素),默认 1280 */
videoHeight?: number videoHeight?: number
useCoverTitle?: boolean
} }
/** function autoWrapLines(rawTitle: string, maxCharsPerLine: number): string[] {
* 将标题渲染为透明背景 PNG 的 dataURLdata:image/png;base64,... const manual = rawTitle
* Canvas 尺寸与视频一致,保证叠加时 1:1 像素对齐。
*
* 标题为空时返回 null。
*/
export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | null {
const { titleConfig, videoWidth = 720, videoHeight = 1280 } = opts
if (!titleConfig) return null
const rawTitle = (titleConfig.title || "").trim()
if (!rawTitle) return null
// 按 / 或 分割为多行
const lines = rawTitle
.split(/[/]/) .split(/[/]/)
.map((l) => l.trim()) .map((l) => l.trim())
.filter((l) => l.length > 0) .filter((l) => l.length > 0)
if (!maxCharsPerLine || maxCharsPerLine <= 0) return manual
const out: string[] = []
manual.forEach((seg) => {
for (let i = 0; i < seg.length; i += maxCharsPerLine) {
out.push(seg.slice(i, i + maxCharsPerLine))
}
})
return out
}
export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | null {
const { titleConfig, videoWidth = 720, videoHeight = 1280, useCoverTitle } = opts
if (!titleConfig) return null
type TitleCfgExt = AiAvatarTitleConfig & {
stroke_width?: number
stroke_color?: string
shadow_offset_x?: number
shadow_offset_y?: number
shadow_blur?: number
shadow_color?: string
line_height?: number
margin_top?: number
max_chars_per_line?: number
bg_enabled?: boolean
bg_color?: string
bg_padding?: number
bg_radius?: number
line_overrides?: Array<{
line_index: number
text?: string
size?: number
color?: string
bold?: boolean
italic?: boolean
stroke?: boolean
highlights?: Array<{ word: string; color?: string; bold?: boolean; scale?: number }>
}>
cover_title_config?: Partial<AiAvatarTitleConfig> | null
pos_x?: number
pos_y?: number
}
const cfg: TitleCfgExt =
useCoverTitle && titleConfig.cover_title_config
? ({
...(titleConfig as TitleCfgExt),
...(titleConfig.cover_title_config as object),
} as TitleCfgExt)
: (titleConfig as TitleCfgExt)
const rawTitle = (cfg.title || "").trim()
if (!rawTitle) return null
const maxCharsPerLine = cfg.max_chars_per_line ?? 0
const lines = autoWrapLines(rawTitle, maxCharsPerLine)
if (lines.length === 0) return null if (lines.length === 0) return null
// 分辨率缩放系数:基准 720p,所有长度类参数乘以 scale
const scale = videoWidth / 720 const scale = videoWidth / 720
const r = (v: number) => Math.round(v * scale) const r = (v: number) => Math.round(v * scale)
@@ -49,86 +90,183 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
const ctx = canvas.getContext("2d") const ctx = canvas.getContext("2d")
if (!ctx) return null if (!ctx) return null
const baseSize = Math.max(12, Math.round(titleConfig.size || 48)) const baseSize = Math.max(12, Math.round(cfg.size || 48))
const size = r(baseSize) const size = r(baseSize)
const bold = !!titleConfig.bold const bold = !!cfg.bold
const italic = !!titleConfig.italic const italic = !!cfg.italic
const color = titleConfig.color || "#ffffff" const color = cfg.color || "#ffffff"
const stroke = !!titleConfig.stroke const stroke = !!cfg.stroke
const shadow = !!titleConfig.shadow const shadow = !!cfg.shadow
// 字体族 fallback 链:优先中文字体 const strokeWidthBase = cfg.stroke_width != null ? cfg.stroke_width : 4
const fontFamily = const strokeColor = cfg.stroke_color || "#000000"
'"Noto Sans CJK SC","Source Han Sans CN","PingFang SC","Microsoft YaHei",sans-serif' const shadowOffsetXBase = cfg.shadow_offset_x != null ? cfg.shadow_offset_x : 2
const fontParts: string[] = [] const shadowOffsetYBase = cfg.shadow_offset_y != null ? cfg.shadow_offset_y : 2
if (italic) fontParts.push("italic") const shadowBlurBase = cfg.shadow_blur != null ? cfg.shadow_blur : 4
if (bold) fontParts.push("bold") const shadowColor = cfg.shadow_color || "rgba(0,0,0,0.8)"
fontParts.push(`${size}px`, fontFamily) const lineHeightScale = cfg.line_height != null ? cfg.line_height : 1.2
ctx.font = fontParts.join(" ") const marginTopBase = cfg.margin_top != null ? cfg.margin_top : 24
const bgEnabled = !!cfg.bg_enabled
const bgColor = cfg.bg_color || "rgba(0,0,0,0.5)"
const bgPaddingBase = cfg.bg_padding != null ? cfg.bg_padding : 12
const bgRadiusBase = cfg.bg_radius != null ? cfg.bg_radius : 8
const fontKey = cfg.font || "思源黑体"
const fontFamily = getFontFamily(fontKey)
const setFont = (sz: number, bd: boolean, it: boolean) => {
const parts: string[] = []
if (it) parts.push("italic")
if (bd) parts.push("bold")
parts.push(`${sz}px`, fontFamily)
ctx.font = parts.join(" ")
}
setFont(size, bold, italic)
ctx.fillStyle = color ctx.fillStyle = color
ctx.textAlign = "center" ctx.textAlign = "center"
ctx.textBaseline = "middle" ctx.textBaseline = "middle"
// 阴影(shadow=true 时开启)——按 scale 缩放 const lineGap = size * lineHeightScale
if (shadow) { const totalTextH = lines.length * lineGap - (lineGap - size)
ctx.shadowColor = "rgba(0,0,0,0.8)" let maxLineW = 0
ctx.shadowBlur = r(4) lines.forEach((l: string) => {
ctx.shadowOffsetX = 0 const m = ctx.measureText(l).width
ctx.shadowOffsetY = r(2) if (m > maxLineW) maxLineW = m
} })
// 位置计算:与 PanelLipsyncPreview 的 CSS 对齐(按 scale 缩放 PAD
const PAD = r(16) const PAD = r(16)
let centerX = videoWidth / 2 let centerX = videoWidth / 2
const position = titleConfig.position || "bottom" const position = cfg.position || "bottom"
const lineGap = size * 1.2
const totalTextH = lines.length * lineGap - (lineGap - size) // 所有行的总高度
// 文本块顶部 ytextBaseline=middle 时首行基线)
let firstLineY: number let firstLineY: number
if ( if (position === "custom" && typeof cfg.pos_x === "number" && typeof cfg.pos_y === "number") {
position === "custom" && centerX = (Math.max(0, Math.min(100, cfg.pos_x)) / 100) * videoWidth
typeof titleConfig.pos_x === "number" && const centerY = (Math.max(0, Math.min(100, cfg.pos_y)) / 100) * videoHeight
typeof titleConfig.pos_y === "number"
) {
centerX = (Math.max(0, Math.min(100, titleConfig.pos_x)) / 100) * videoWidth
const centerY = (Math.max(0, Math.min(100, titleConfig.pos_y)) / 100) * videoHeight
firstLineY = centerY - totalTextH / 2 + size / 2 firstLineY = centerY - totalTextH / 2 + size / 2
} else if (position === "top") { } else if (position === "top") {
// 顶部:y = size/2 + PAD firstLineY = size / 2 + PAD + r(marginTopBase)
firstLineY = size / 2 + PAD
} else if (position === "center") { } else if (position === "center") {
firstLineY = videoHeight / 2 - totalTextH / 2 + size / 2 firstLineY = videoHeight / 2 - totalTextH / 2 + size / 2
} else { } else {
// bottom(默认)
firstLineY = videoHeight - totalTextH - PAD + size / 2 firstLineY = videoHeight - totalTextH - PAD + size / 2
} }
// 描边参数:描边 lineWidth 按 scale 缩放(基准 size * 0.06,最小 2px @720p if (shadow) {
const doStroke = stroke ctx.shadowColor = shadowColor
const strokeWidth = Math.max(r(2), Math.round(size * 0.06)) ctx.shadowBlur = r(shadowBlurBase)
// 逐行绘制 ctx.shadowOffsetX = r(shadowOffsetXBase)
lines.forEach((line, idx) => { ctx.shadowOffsetY = r(shadowOffsetYBase)
} else {
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
ctx.shadowOffsetX = 0
ctx.shadowOffsetY = 0
}
const bgPad = r(bgPaddingBase)
const bgR = r(bgRadiusBase)
const bgW = maxLineW + bgPad * 2
const bgH = totalTextH + bgPad * 2
const bgX = centerX - bgW / 2
const bgY = firstLineY - size / 2 - bgPad
if (bgEnabled) {
ctx.save()
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
ctx.shadowOffsetX = 0
ctx.shadowOffsetY = 0
ctx.fillStyle = bgColor
if (
bgR > 0 &&
(
ctx as CanvasRenderingContext2D & {
roundRect?: (x: number, y: number, w: number, h: number, r: number) => void
}
).roundRect
) {
;(
ctx as CanvasRenderingContext2D & {
roundRect?: (x: number, y: number, w: number, h: number, r: number) => void
}
).roundRect(bgX, bgY, bgW, bgH, bgR)
ctx.fill()
} else {
ctx.fillRect(bgX, bgY, bgW, bgH)
}
ctx.restore()
}
const sw = stroke ? Math.max(r(1), r(strokeWidthBase)) : 0
const lineOverrides = cfg.line_overrides || []
lines.forEach((line: string, idx: number) => {
const y = firstLineY + idx * lineGap const y = firstLineY + idx * lineGap
if (doStroke) { const override = lineOverrides.find((lo) => lo.line_index === idx)
const prevShadowColor = ctx.shadowColor const lineSize = override?.size ? r(Math.max(12, Math.round(override.size))) : size
const prevShadowBlur = ctx.shadowBlur const lineColor = override?.color || color
// 描边不要带阴影(避免黑色描边发虚) const lineBold = override?.bold != null ? !!override.bold : bold
const lineItalic = override?.italic != null ? !!override.italic : italic
const lineStroke = override?.stroke != null ? !!override.stroke : stroke
setFont(lineSize, lineBold, lineItalic)
ctx.fillStyle = lineColor
if (shadow) {
ctx.shadowColor = shadowColor
ctx.shadowBlur = r(shadowBlurBase)
ctx.shadowOffsetX = r(shadowOffsetXBase)
ctx.shadowOffsetY = r(shadowOffsetYBase)
} else {
ctx.shadowColor = "rgba(0,0,0,0)" ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0 ctx.shadowBlur = 0
ctx.lineWidth = strokeWidth ctx.shadowOffsetX = 0
ctx.strokeStyle = "#000000" ctx.shadowOffsetY = 0
}
const lineSw = override?.size
? Math.max(r(1), Math.round(lineSize * (strokeWidthBase / baseSize)))
: sw
if (lineStroke && lineSw > 0) {
ctx.save()
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
ctx.shadowOffsetX = 0
ctx.shadowOffsetY = 0
ctx.lineWidth = lineSw
ctx.strokeStyle = strokeColor
ctx.lineJoin = "round" ctx.lineJoin = "round"
ctx.strokeText(line, centerX, y) ctx.strokeText(line, centerX, y)
// 恢复阴影 ctx.restore()
if (shadow) {
ctx.shadowColor = "rgba(0,0,0,0.8)"
ctx.shadowBlur = r(4)
} else {
ctx.shadowColor = prevShadowColor
ctx.shadowBlur = prevShadowBlur
}
} }
ctx.fillText(line, centerX, y) ctx.fillText(line, centerX, y)
if (override?.highlights?.length) {
const fullW = ctx.measureText(line).width
const charW = line.length > 0 ? fullW / line.length : lineSize
override.highlights.forEach((hl) => {
if (!hl.word) return
const pos = line.indexOf(hl.word)
if (pos < 0) return
const hlX = centerX - fullW / 2 + pos * charW + (charW * hl.word.length) / 2
const hlColor = hl.color || "#ffd700"
const hlScale = hl.scale || 1
const hlSize = lineSize * hlScale
const hlBold = hl.bold != null ? !!hl.bold : true
ctx.save()
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
setFont(hlSize, hlBold, lineItalic)
ctx.fillStyle = hlColor
if (lineStroke && lineSw > 0) {
ctx.lineWidth = Math.max(r(1), Math.round(hlSize * (strokeWidthBase / baseSize)))
ctx.strokeStyle = strokeColor
ctx.lineJoin = "round"
ctx.strokeText(hl.word, hlX, y)
}
ctx.fillText(hl.word, hlX, y)
ctx.restore()
})
}
}) })
try { try {
@@ -138,9 +276,6 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
} }
} }
/**
* 获取视频真实分辨率(HTMLVideoElement + loadedmetadata,超时 3 秒兜底 720×1280)。
*/
export function getVideoResolution( export function getVideoResolution(
videoUrl: string, videoUrl: string,
timeoutMs = 3000, timeoutMs = 3000,
+14 -1
View File
@@ -77,6 +77,8 @@ const GeneratePage: React.FC = () => {
setTtsVoiceId, setTtsVoiceId,
ttsVoiceSource, ttsVoiceSource,
setTtsVoiceSource, setTtsVoiceSource,
ttsStyle,
setTtsStyle,
ttsVoiceAssetId, ttsVoiceAssetId,
setTtsVoiceAssetId, setTtsVoiceAssetId,
dedupEnabled, dedupEnabled,
@@ -329,6 +331,7 @@ const GeneratePage: React.FC = () => {
selectedScript, selectedScript,
ttsVoiceId, ttsVoiceId,
ttsVoiceSource, ttsVoiceSource,
ttsStyle,
ttsVoiceAssetId, ttsVoiceAssetId,
dedupEnabled, dedupEnabled,
style, style,
@@ -410,9 +413,15 @@ const GeneratePage: React.FC = () => {
) )
const handleTtsSynthesized = useCallback( const handleTtsSynthesized = useCallback(
(payload: { voiceAssetId: string; ttsVoiceId: string; ttsVoiceSource: "preset" | "clone" }) => { (payload: {
voiceAssetId: string
ttsVoiceId: string
ttsVoiceSource: "preset" | "clone"
ttsStyle?: string
}) => {
setTtsVoiceId(payload.ttsVoiceId) setTtsVoiceId(payload.ttsVoiceId)
setTtsVoiceSource(payload.ttsVoiceSource) setTtsVoiceSource(payload.ttsVoiceSource)
if (payload.ttsStyle) setTtsStyle(payload.ttsStyle)
setTtsVoiceAssetId(payload.voiceAssetId) setTtsVoiceAssetId(payload.voiceAssetId)
if (payload.ttsVoiceSource === "clone") { if (payload.ttsVoiceSource === "clone") {
setSelectedClonedVoice(payload.ttsVoiceId) setSelectedClonedVoice(payload.ttsVoiceId)
@@ -428,6 +437,7 @@ const GeneratePage: React.FC = () => {
[ [
setTtsVoiceId, setTtsVoiceId,
setTtsVoiceSource, setTtsVoiceSource,
setTtsStyle,
setTtsVoiceAssetId, setTtsVoiceAssetId,
setSelectedVoice, setSelectedVoice,
setSelectedClonedVoice, setSelectedClonedVoice,
@@ -630,6 +640,7 @@ const GeneratePage: React.FC = () => {
onToggleStroke={styleUpdaters.toggleStroke} onToggleStroke={styleUpdaters.toggleStroke}
onToggleShadow={styleUpdaters.toggleShadow} onToggleShadow={styleUpdaters.toggleShadow}
onApplyPreset={styleUpdaters.applyPreset} onApplyPreset={styleUpdaters.applyPreset}
onUpdateStyle={styleUpdaters.updateStyle}
activePreset={styleUpdaters.activePreset} activePreset={styleUpdaters.activePreset}
titlePresets={styleUpdaters.titlePresets} titlePresets={styleUpdaters.titlePresets}
bgm={bgm} bgm={bgm}
@@ -791,6 +802,8 @@ const GeneratePage: React.FC = () => {
open={ttsModalOpen} open={ttsModalOpen}
scriptText={selectedScript?.content ?? ""} scriptText={selectedScript?.content ?? ""}
scriptTitle={selectedScript?.title ?? ""} scriptTitle={selectedScript?.title ?? ""}
style={ttsStyle}
onStyleChange={setTtsStyle}
onCancel={() => setTtsModalOpen(false)} onCancel={() => setTtsModalOpen(false)}
onSynthesized={handleTtsSynthesized} onSynthesized={handleTtsSynthesized}
/> />
@@ -51,8 +51,14 @@ export interface GenerateStepContentProps {
onToggleStroke: () => void onToggleStroke: () => void
onToggleShadow: () => void onToggleShadow: () => void
onApplyPreset: (presetKey: string) => void onApplyPreset: (presetKey: string) => void
onUpdateStyle?: (patch: Partial<TitleSettings>) => void
activePreset: string | null activePreset: string | null
titlePresets: { key: string; label: string; previewStyle: React.CSSProperties }[] titlePresets: Array<{
key: string
label: string
emoji?: string
style: Record<string, unknown>
}>
/* ── 封面 ── */ /* ── 封面 ── */
coverSettings: CoverConfig coverSettings: CoverConfig
onCoverSettingsChange: (settings: CoverConfig) => void onCoverSettingsChange: (settings: CoverConfig) => void
@@ -119,6 +125,7 @@ export const GenerateStepContent: React.FC<GenerateStepContentProps> = (props) =
onToggleStroke, onToggleStroke,
onToggleShadow, onToggleShadow,
onApplyPreset, onApplyPreset,
onUpdateStyle,
activePreset, activePreset,
titlePresets, titlePresets,
coverSettings, coverSettings,
@@ -191,6 +198,7 @@ export const GenerateStepContent: React.FC<GenerateStepContentProps> = (props) =
onToggleStroke={onToggleStroke} onToggleStroke={onToggleStroke}
onToggleShadow={onToggleShadow} onToggleShadow={onToggleShadow}
onApplyPreset={onApplyPreset} onApplyPreset={onApplyPreset}
onUpdateStyle={onUpdateStyle}
activePreset={activePreset} activePreset={activePreset}
titlePresets={titlePresets} titlePresets={titlePresets}
previewCount={previewCount} previewCount={previewCount}
@@ -12,7 +12,8 @@ import React, { useMemo, useState } from "react"
import { Input, message } from "antd" import { Input, message } from "antd"
import { LoadingOutlined } from "@ant-design/icons" import { LoadingOutlined } from "@ant-design/icons"
import type { TitleSettings } from "../types" import type { TitleSettings } from "../types"
import { POSITION_OPTIONS, FONT_OPTIONS } from "../constants" import { POSITION_OPTIONS } from "../constants"
import { FONT_OPTIONS } from "@/components/title/constants"
import { useStep4Title } from "../hooks/useStep4Title" import { useStep4Title } from "../hooks/useStep4Title"
import AiTitleGenerator from "./title/AiTitleGenerator" import AiTitleGenerator from "./title/AiTitleGenerator"
import TitleLibraryAutoComplete from "./title/TitleLibraryAutoComplete" import TitleLibraryAutoComplete from "./title/TitleLibraryAutoComplete"
@@ -33,8 +34,15 @@ interface Step4TitleSettingsProps {
onToggleStroke: () => void onToggleStroke: () => void
onToggleShadow: () => void onToggleShadow: () => void
onApplyPreset: (presetKey: string) => void onApplyPreset: (presetKey: string) => void
onUpdateStyle?: (patch: Partial<TitleSettings>) => void
activePreset: string | null activePreset: string | null
titlePresets: { key: string; label: string; previewStyle: React.CSSProperties }[] titlePresets: Array<{
key: string
label: string
emoji?: string
style?: Record<string, unknown>
previewStyle?: React.CSSProperties
}>
/* ── 批量生成(#1677)── */ /* ── 批量生成(#1677)── */
/** 生成数量 */ /** 生成数量 */
previewCount?: number previewCount?: number
@@ -84,6 +92,7 @@ const Step4TitleSettings: React.FC<Step4TitleSettingsProps> = (props) => {
onToggleStroke, onToggleStroke,
onToggleShadow, onToggleShadow,
onApplyPreset, onApplyPreset,
onUpdateStyle,
activePreset, activePreset,
titlePresets, titlePresets,
previewCount = 1, previewCount = 1,
@@ -285,6 +294,8 @@ const Step4TitleSettings: React.FC<Step4TitleSettingsProps> = (props) => {
onToggleStroke={onToggleStroke} onToggleStroke={onToggleStroke}
onToggleShadow={onToggleShadow} onToggleShadow={onToggleShadow}
onApplyPreset={onApplyPreset} onApplyPreset={onApplyPreset}
onUpdateStyle={onUpdateStyle}
showCoverToggle
activePreset={activePreset} activePreset={activePreset}
titlePresets={titlePresets} titlePresets={titlePresets}
POSITION_OPTIONS={POSITION_OPTIONS} POSITION_OPTIONS={POSITION_OPTIONS}
@@ -19,6 +19,8 @@ import { synthesizeSpeech, getTTSJobStatus, saveTtsToLibrary } from "@/api/tts"
import type { PresetVoiceItem } from "@/api/voices" import type { PresetVoiceItem } from "@/api/voices"
import type { VoiceClone } from "@/api/voice-clone" import type { VoiceClone } from "@/api/voice-clone"
import { VOICE_GENDER_ICON } from "../constants" import { VOICE_GENDER_ICON } from "../constants"
import TtsStyleSelector from "@/components/voice/TtsStyleSelector"
import { DEFAULT_TTS_STYLE, type TtsStyle } from "@/api/tts/styles"
interface TtsVoiceModalProps { interface TtsVoiceModalProps {
open: boolean open: boolean
@@ -31,7 +33,11 @@ interface TtsVoiceModalProps {
voiceAssetId: string voiceAssetId: string
ttsVoiceId: string ttsVoiceId: string
ttsVoiceSource: "preset" | "clone" ttsVoiceSource: "preset" | "clone"
ttsStyle: TtsStyle
}) => void }) => void
/** 当前风格 */
style?: TtsStyle
onStyleChange?: (s: TtsStyle) => void
} }
type TtsSynthStatus = "idle" | "synthesizing" | "saving" | "done" | "error" type TtsSynthStatus = "idle" | "synthesizing" | "saving" | "done" | "error"
@@ -42,7 +48,15 @@ const TtsVoiceModal: React.FC<TtsVoiceModalProps> = ({
scriptTitle, scriptTitle,
onCancel, onCancel,
onSynthesized, onSynthesized,
style: externalStyle,
onStyleChange,
}) => { }) => {
const [internalStyle, setInternalStyle] = useState<TtsStyle>(DEFAULT_TTS_STYLE)
const currentStyle: TtsStyle = externalStyle ?? internalStyle
const handleStyleChange = (s: TtsStyle) => {
setInternalStyle(s)
onStyleChange?.(s)
}
const [activeTab, setActiveTab] = useState<"preset" | "clone">("preset") const [activeTab, setActiveTab] = useState<"preset" | "clone">("preset")
const [selectedVoiceId, setSelectedVoiceId] = useState<string>("") const [selectedVoiceId, setSelectedVoiceId] = useState<string>("")
const [status, setStatus] = useState<TtsSynthStatus>("idle") const [status, setStatus] = useState<TtsSynthStatus>("idle")
@@ -77,6 +91,7 @@ const TtsVoiceModal: React.FC<TtsVoiceModalProps> = ({
setStatus("idle") setStatus("idle")
setError(null) setError(null)
setActiveTab("preset") setActiveTab("preset")
setInternalStyle(externalStyle ?? DEFAULT_TTS_STYLE)
} else { } else {
if (timerRef.current) { if (timerRef.current) {
clearInterval(timerRef.current) clearInterval(timerRef.current)
@@ -91,6 +106,7 @@ const TtsVoiceModal: React.FC<TtsVoiceModalProps> = ({
return () => { return () => {
if (timerRef.current) clearInterval(timerRef.current) if (timerRef.current) clearInterval(timerRef.current)
} }
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [open]) }, [open])
const handlePreview = useCallback( const handlePreview = useCallback(
@@ -143,6 +159,7 @@ const TtsVoiceModal: React.FC<TtsVoiceModalProps> = ({
text: textToSynth, text: textToSynth,
speed: 1.0, speed: 1.0,
language: "zh-CN", language: "zh-CN",
style: currentStyle,
} }
if (isClone) { if (isClone) {
payload.voice_clone_profile_id = selectedVoiceId payload.voice_clone_profile_id = selectedVoiceId
@@ -187,13 +204,14 @@ const TtsVoiceModal: React.FC<TtsVoiceModalProps> = ({
voiceAssetId: jobId, voiceAssetId: jobId,
ttsVoiceId: selectedVoiceId, ttsVoiceId: selectedVoiceId,
ttsVoiceSource: isClone ? "clone" : "preset", ttsVoiceSource: isClone ? "clone" : "preset",
ttsStyle: currentStyle,
}) })
} catch (err: unknown) { } catch (err: unknown) {
setStatus("error") setStatus("error")
const msg = err instanceof Error ? err.message : "合成失败,请稍后重试" const msg = err instanceof Error ? err.message : "合成失败,请稍后重试"
setError(msg) setError(msg)
} }
}, [selectedVoiceId, textToSynth, activeTab, scriptTitle, onSynthesized]) }, [selectedVoiceId, textToSynth, activeTab, scriptTitle, onSynthesized, currentStyle])
const renderVoiceCard = (v: { const renderVoiceCard = (v: {
id: string id: string
@@ -393,6 +411,10 @@ const TtsVoiceModal: React.FC<TtsVoiceModalProps> = ({
{textToSynth.length} {textToSynth.length}
</div> </div>
<div style={{ marginBottom: 12 }}>
<TtsStyleSelector value={currentStyle} onChange={handleStyleChange} compact />
</div>
<Tabs <Tabs
activeKey={activeTab} activeKey={activeTab}
onChange={(k) => { onChange={(k) => {
@@ -0,0 +1,221 @@
/**
* 标题迷你 Canvas 预览(#2001
*
* 渲染一张指定宽度的小 Canvas 预览标题效果,用于:
* - 预设卡片缩略图
* - 样式面板顶部的实时预览
*
* 与 titleCanvas.ts 渲染逻辑保持一致,但:
* - 固定分辨率(width × 宽高比约 2:1)
* - 不调用 ffmpeg,只做视觉预览
* - 支持背景色块、描边宽度/颜色、阴影参数化、行距、自动换行
*/
import React, { useEffect, useRef } from "react"
import type { TitleSettings } from "../../types"
import { getFontFamily } from "../../constants"
interface Props {
settings: TitleSettings
width?: number
sampleText?: string
/** 背景(预览用,默认深色渐变模拟视频底) */
background?: string
/** 高度(可选,默认 width/2 */
height?: number
}
/** 按 maxCharsPerLine 自动换行 */
function wrapLines(text: string, maxChars: number): string[] {
const manual = text
.split(/[/\n]/)
.map((l) => l.trim())
.filter(Boolean)
if (!maxChars || maxChars <= 0) return manual
const out: string[] = []
for (const line of manual) {
if (line.length <= maxChars) {
out.push(line)
continue
}
let cur = ""
for (const ch of line) {
cur += ch
if (cur.length >= maxChars) {
out.push(cur)
cur = ""
}
}
if (cur) out.push(cur)
}
return out
}
const TitleMiniPreview: React.FC<Props> = ({
settings,
width = 200,
sampleText,
background = "linear-gradient(135deg,#1f2937,#111827)",
height,
}) => {
const canvasRef = useRef<HTMLCanvasElement>(null)
const h = height ?? Math.round(width / 1.8)
const text = (sampleText || settings.title || "预览标题").trim() || "预览标题"
useEffect(() => {
const cvs = canvasRef.current
if (!cvs) return
const dpr = window.devicePixelRatio || 1
cvs.width = width * dpr
cvs.height = h * dpr
cvs.style.width = `${width}px`
cvs.style.height = `${h}px`
const ctx = cvs.getContext("2d")
if (!ctx) return
ctx.scale(dpr, dpr)
ctx.clearRect(0, 0, width, h)
// 背景
ctx.fillStyle = "#111827"
ctx.fillRect(0, 0, width, h)
// 分辨率缩放:以 360 宽为基准(对应 720p 的一半)
const scale = width / 360
const r = (v: number) => Math.round(v * scale)
// 字体
const size = r(settings.size)
const ff = getFontFamily(settings.font)
const parts: string[] = []
if (settings.italic) parts.push("italic")
if (settings.bold) parts.push("bold")
parts.push(`${size}px`, ff)
ctx.font = parts.join(" ")
ctx.textAlign = "center"
ctx.textBaseline = "middle"
ctx.fillStyle = settings.color
ctx.lineJoin = "round"
// 阴影
const shadowEnabled = !!settings.shadow
const prevShadow = {
c: ctx.shadowColor,
b: ctx.shadowBlur,
ox: ctx.shadowOffsetX,
oy: ctx.shadowOffsetY,
}
if (shadowEnabled) {
ctx.shadowColor = settings.shadowColor ?? "rgba(0,0,0,0.8)"
ctx.shadowBlur = r(settings.shadowBlur ?? 4)
ctx.shadowOffsetX = r(settings.shadowOffsetX ?? 2)
ctx.shadowOffsetY = r(settings.shadowOffsetY ?? 2)
}
// 换行
const lines = wrapLines(text, settings.maxCharsPerLine ?? 0)
const lineH = size * (settings.lineHeight ?? 1.2)
const totalH = lines.length * lineH
let startY: number
if (settings.position === "top") {
startY = size / 2 + r(settings.marginTop ?? 24)
} else if (settings.position === "center") {
startY = h / 2 - totalH / 2 + size / 2
} else {
// bottom
startY = h - totalH - r(16) + size / 2
}
let centerX = width / 2
if (settings.position === "custom" && settings.posX != null) {
centerX = (settings.posX / 100) * width
}
// 背景块
if (settings.bgEnabled) {
const pad = r(settings.bgPadding ?? 12)
const rad = r(settings.bgRadius ?? 8)
let maxLineW = 0
for (const l of lines) {
const m = ctx.measureText(l)
if (m.width > maxLineW) maxLineW = m.width
}
const bw = maxLineW + pad * 2
const bh = totalH + pad * 2
const bx = centerX - bw / 2
const by = startY - size / 2 - pad + (size - lineH) / 2
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
ctx.fillStyle = settings.bgColor ?? "rgba(0,0,0,0.5)"
roundRect(ctx, bx, by, bw, bh, rad)
ctx.fill()
// 恢复阴影
if (shadowEnabled) {
ctx.shadowColor = settings.shadowColor ?? "rgba(0,0,0,0.8)"
ctx.shadowBlur = r(settings.shadowBlur ?? 4)
ctx.shadowOffsetX = r(settings.shadowOffsetX ?? 2)
ctx.shadowOffsetY = r(settings.shadowOffsetY ?? 2)
}
}
// 描边(先画,再画填充)
const strokeEnabled = !!settings.stroke && (settings.strokeWidth ?? 0) > 0
lines.forEach((line, i) => {
const y = startY + i * lineH
if (strokeEnabled) {
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
ctx.lineWidth = r(settings.strokeWidth ?? 4)
ctx.strokeStyle = settings.strokeColor ?? "#000000"
ctx.strokeText(line, centerX, y)
// 恢复阴影
if (shadowEnabled) {
ctx.shadowColor = settings.shadowColor ?? "rgba(0,0,0,0.8)"
ctx.shadowBlur = r(settings.shadowBlur ?? 4)
ctx.shadowOffsetX = r(settings.shadowOffsetX ?? 2)
ctx.shadowOffsetY = r(settings.shadowOffsetY ?? 2)
}
}
ctx.fillText(line, centerX, y)
})
// 恢复
ctx.shadowColor = prevShadow.c
ctx.shadowBlur = prevShadow.b
ctx.shadowOffsetX = prevShadow.ox
ctx.shadowOffsetY = prevShadow.oy
}, [settings, width, h, text])
return (
<canvas
ref={canvasRef}
style={{
borderRadius: 6,
display: "block",
maxWidth: "100%",
background,
}}
/>
)
}
function roundRect(
ctx: CanvasRenderingContext2D,
x: number,
y: number,
w: number,
h: number,
r: number,
) {
const rr = Math.min(r, w / 2, h / 2)
ctx.beginPath()
ctx.moveTo(x + rr, y)
ctx.lineTo(x + w - rr, y)
ctx.quadraticCurveTo(x + w, y, x + w, y + rr)
ctx.lineTo(x + w, y + h - rr)
ctx.quadraticCurveTo(x + w, y + h, x + w - rr, y + h)
ctx.lineTo(x + rr, y + h)
ctx.quadraticCurveTo(x, y + h, x, y + h - rr)
ctx.lineTo(x, y + rr)
ctx.quadraticCurveTo(x, y, x + rr, y)
ctx.closePath()
}
export default TitleMiniPreview
@@ -190,3 +190,255 @@
border-color: var(--primary-color); border-color: var(--primary-color);
color: #fff; color: #fff;
} }
/* ============================================================
#2001 爆款标题样式面板升级 — 新增样式(ts- 前缀)
============================================================ */
.ts-panel {
position: relative;
}
/* 预览 */
.ts-preview-wrap {
margin-bottom: 14px;
display: flex;
justify-content: center;
padding: 10px;
background: #0f172a;
border-radius: 8px;
}
/* 表单字段 */
.ts-form-field {
margin-bottom: 12px;
}
.ts-form-field label {
display: block;
font-weight: 600;
margin-bottom: 6px;
font-size: 12px;
color: var(--text-primary, #1f2937);
}
.ts-field-label-row {
display: flex;
align-items: center;
justify-content: space-between;
margin-bottom: 6px;
}
.ts-field-value {
font-size: 12px;
font-weight: 600;
color: var(--primary-color, #7c3aed);
}
.ts-row-2 {
display: grid;
grid-template-columns: 1fr 1fr;
gap: 10px;
}
.ts-half {
margin-bottom: 0;
}
.ts-select {
width: 100%;
height: 34px;
border: 1px solid var(--border-color, #e5e7eb);
border-radius: 6px;
background: var(--bg-primary, #fff);
padding: 0 10px;
font-size: 13px;
outline: 0;
color: var(--text-primary, #1f2937);
}
.ts-select:focus {
border-color: var(--primary-color, #7c3aed);
box-shadow: 0 0 0 2px rgba(124, 58, 237, 0.1);
}
.ts-input {
width: 100%;
height: 34px;
border: 1px solid var(--border-color, #e5e7eb);
border-radius: 6px;
padding: 0 10px;
font-size: 13px;
outline: 0;
}
.ts-slider {
width: 100%;
height: 4px;
-webkit-appearance: none;
appearance: none;
background: #e5e7eb;
border-radius: 2px;
outline: none;
}
.ts-slider::-webkit-slider-thumb {
-webkit-appearance: none;
appearance: none;
width: 16px;
height: 16px;
border-radius: 50%;
background: #7c3aed;
cursor: pointer;
border: 2px solid #fff;
box-shadow: 0 1px 3px rgba(0, 0, 0, 0.2);
}
.ts-slider::-moz-range-thumb {
width: 16px;
height: 16px;
border-radius: 50%;
background: #7c3aed;
cursor: pointer;
border: 2px solid #fff;
}
/* 样式按钮 B/I/S/☁ */
.ts-style-btns {
display: flex;
gap: 6px;
}
.ts-style-btn {
width: 34px;
height: 34px;
border-radius: 6px;
border: 1px solid #e5e7eb;
background: #fff;
cursor: pointer;
font-size: 14px;
transition: 0.15s;
color: #374151;
display: inline-flex;
align-items: center;
justify-content: center;
}
.ts-style-btn:hover {
border-color: #7c3aed;
color: #7c3aed;
}
.ts-style-btn.active {
background: #faf5ff;
color: #6d28d9;
border-color: #7c3aed;
font-weight: 700;
}
/* 色板 */
.ts-color-row {
display: flex;
flex-wrap: wrap;
gap: 6px;
align-items: center;
}
.ts-color-swatch {
width: 24px;
height: 24px;
border-radius: 4px;
border: 2px solid #fff;
box-shadow: 0 0 0 1px #e5e7eb;
cursor: pointer;
padding: 0;
transition: 0.15s;
}
.ts-color-swatch:hover {
transform: scale(1.1);
}
.ts-color-swatch.active {
box-shadow: 0 0 0 2px #7c3aed;
transform: scale(1.1);
}
.ts-color-custom {
background: repeating-conic-gradient(#ccc 0% 25%, #fff 0% 50%) 50%/8px 8px;
color: #666;
font-size: 14px;
line-height: 20px;
}
.ts-color-native {
width: 0;
height: 0;
border: 0;
padding: 0;
}
/* 预设网格 10个 - 5列 */
.ts-presets-grid {
display: grid;
grid-template-columns: repeat(5, 1fr);
gap: 6px;
}
.ts-preset-card {
border: 1px solid #e5e7eb;
border-radius: 6px;
background: #fff;
padding: 4px;
cursor: pointer;
transition: 0.15s;
display: flex;
flex-direction: column;
gap: 4px;
}
.ts-preset-card:hover {
border-color: #7c3aed;
}
.ts-preset-card.active {
border-color: #7c3aed;
background: #faf5ff;
box-shadow: 0 0 0 1px #7c3aed;
}
.ts-preset-preview {
height: 34px;
display: flex;
align-items: center;
justify-content: center;
overflow: hidden;
border-radius: 4px;
background: #0f172a;
}
.ts-preset-preview canvas {
max-width: 100%;
max-height: 100%;
}
.ts-preset-meta {
display: flex;
align-items: center;
gap: 2px;
font-size: 10px;
color: #4b5563;
justify-content: center;
white-space: nowrap;
overflow: hidden;
text-overflow: ellipsis;
padding: 0 2px 2px;
}
.ts-preset-emoji {
font-size: 11px;
}
.ts-preset-label {
overflow: hidden;
text-overflow: ellipsis;
}
.ts-toggle-row label {
display: inline-flex;
align-items: center;
gap: 6px;
font-size: 13px;
font-weight: 500;
cursor: pointer;
margin-bottom: 10px;
}
.ts-toggle-row input[type="checkbox"] {
width: 16px;
height: 16px;
accent-color: #7c3aed;
}
/* Tabs 紧凑样式 */
.xx-title-style-section .ant-tabs-nav {
margin-bottom: 10px;
}
.xx-title-style-section .ant-tabs-tab {
font-size: 12px !important;
padding: 6px 8px !important;
}
@@ -1,12 +1,26 @@
/** /**
* 标题样式设置 * 标题样式设置面板(#2001 升级)
* 位置/字体/字号/样式按钮/预设 *
* P0:描边宽度滑块 / 描边颜色选择器 / 每行最大字符数 / 行距+顶部边距 /
* 4款爆款字体 / 抖音爆款黄预设
* P1:阴影参数化 / 背景色块 / Canvas 实时迷你预览 /
* 封面独立标题配置入口
*
* 向后兼容:旧的 onToggleBold/Italic/Stroke/Shadow/onUpdatePosition/onUpdateFont/
* onUpdateSize/onApplyPreset props 全部保留;新增字段通过 onUpdateStyle 统一回写。
*/ */
import React from "react" import React, { useState } from "react"
import { Tabs } from "antd"
import type { TitleSettings } from "../../types" import type { TitleSettings } from "../../types"
import TitlePresetsGrid from "./TitlePresetsGrid" import {
// 标题样式面板共用样式(#1809 ⑦):智能剪辑与 AI数字人复用同一组件, FONT_OPTIONS as NEW_FONT_OPTIONS,
// 由组件自带样式,避免 AI数字人页面重复引入整个 generate.css TITLE_PRESETS,
TITLE_COLOR_PALETTE,
STROKE_COLOR_PALETTE,
BG_COLOR_PALETTE,
} from "@/components/title/constants"
import TitleMiniPreview from "./TitleMiniPreview"
import "./TitleStylePanel.css" import "./TitleStylePanel.css"
interface PositionOption { interface PositionOption {
@@ -14,14 +28,17 @@ interface PositionOption {
label: string label: string
} }
interface TitlePresetItem { interface LegacyPreset {
key: string key: string
label: string label: string
previewStyle: React.CSSProperties emoji?: string
style?: Record<string, unknown>
previewStyle?: React.CSSProperties
} }
interface TitleStylePanelProps { interface TitleStylePanelProps {
settings: TitleSettings settings: TitleSettings
/* 旧 props(兼容) */
onUpdatePosition: (position: string) => void onUpdatePosition: (position: string) => void
onUpdateFont: (font: string) => void onUpdateFont: (font: string) => void
onUpdateSize: (size: number) => void onUpdateSize: (size: number) => void
@@ -31,9 +48,130 @@ interface TitleStylePanelProps {
onToggleShadow: () => void onToggleShadow: () => void
onApplyPreset: (presetKey: string) => void onApplyPreset: (presetKey: string) => void
activePreset: string | null activePreset: string | null
titlePresets: TitlePresetItem[] titlePresets: LegacyPreset[]
POSITION_OPTIONS: PositionOption[] POSITION_OPTIONS: PositionOption[]
FONT_OPTIONS: string[] FONT_OPTIONS?: Array<{ value: string; label: string; family?: string; tag?: string }>
/* 新增:统一字段更新 */
onUpdateStyle?: (patch: Partial<TitleSettings>) => void
/* 是否显示封面独立标题切换 */
showCoverToggle?: boolean
/** 画布预览宽度(默认 200) */
previewWidth?: number
}
/* ── 通用 Slider + Label 行 ── */
const SliderRow: React.FC<{
label: string
value: number
min: number
max: number
step?: number
unit?: string
onChange: (v: number) => void
}> = ({ label, value, min, max, step = 1, unit = "px", onChange }) => (
<div className="ts-form-field">
<div className="ts-field-label-row">
<label>{label}</label>
<span className="ts-field-value">
{value}
{unit}
</span>
</div>
<input
type="range"
className="ts-slider"
min={min}
max={max}
step={step}
value={value}
onChange={(e) => onChange(Number(e.target.value))}
/>
</div>
)
/* ── 色板 + 自定义颜色选择 ── */
const ColorPicker: React.FC<{
label?: string
value: string
palette: string[]
onChange: (c: string) => void
}> = ({ label, value, palette, onChange }) => {
const [customOpen, setCustomOpen] = useState(false)
return (
<div className="ts-form-field">
{label && <label>{label}</label>}
<div className="ts-color-row">
{palette.map((c) => (
<button
key={c}
type="button"
className={`ts-color-swatch${value.toLowerCase() === c.toLowerCase() ? " active" : ""}`}
style={{ background: c }}
onClick={() => onChange(c)}
title={c}
/>
))}
<button
type="button"
className="ts-color-swatch ts-color-custom"
onClick={() => setCustomOpen((v) => !v)}
title="自定义颜色"
>
+
</button>
<input
type="color"
className="ts-color-native"
value={value.startsWith("rgba") ? "#000000" : value}
onChange={(e) => onChange(e.target.value)}
style={{
opacity: customOpen ? 1 : 0,
position: customOpen ? "static" : "absolute",
pointerEvents: customOpen ? "auto" : "none",
width: 0,
height: 0,
}}
/>
</div>
<div style={{ fontSize: 11, color: "#9ca3af", marginTop: 2 }}>
<code style={{ fontSize: 11 }}>{value}</code>
</div>
</div>
)
}
/* ── 预设网格(含爆款黄,10 个 + 迷你 Canvas 缩略) ── */
const PresetGrid: React.FC<{
activePreset: string | null
onApply: (key: string) => void
settings: TitleSettings
}> = ({ activePreset, onApply, settings }) => {
return (
<div className="ts-presets-grid">
{TITLE_PRESETS.map((p) => {
const isActive = activePreset === p.key
// 合并当前 style 与 preset.style 用于预览(仅预览时覆盖)
const previewStyle: TitleSettings = { ...settings, ...(p.style as Partial<TitleSettings>) }
return (
<button
key={p.key}
type="button"
className={`ts-preset-card${isActive ? " active" : ""}`}
onClick={() => onApply(p.key)}
title={p.label}
>
<div className="ts-preset-preview">
<TitleMiniPreview settings={previewStyle} width={100} sampleText="标题" />
</div>
<div className="ts-preset-meta">
<span className="ts-preset-emoji">{p.emoji}</span>
<span className="ts-preset-label">{p.label}</span>
</div>
</button>
)
})}
</div>
)
} }
const TitleStylePanel: React.FC<TitleStylePanelProps> = ({ const TitleStylePanel: React.FC<TitleStylePanelProps> = ({
@@ -47,107 +185,352 @@ const TitleStylePanel: React.FC<TitleStylePanelProps> = ({
onToggleShadow, onToggleShadow,
onApplyPreset, onApplyPreset,
activePreset, activePreset,
titlePresets, titlePresets: _titlePresets,
POSITION_OPTIONS, POSITION_OPTIONS,
FONT_OPTIONS, showCoverToggle = false,
previewWidth = 220,
onUpdateStyle,
}) => { }) => {
const upd = (patch: Partial<TitleSettings>) => {
onUpdateStyle?.(patch)
}
/* 封面独立标题切换 */
const [coverOpen, setCoverOpen] = useState(!!settings.coverTitle)
return ( return (
<div className="xx-title-style-section"> <div className="xx-title-style-section ts-panel">
<h4 className="xx-section-subtitle"></h4> {/* 实时迷你预览 */}
<div className="ts-preview-wrap">
{/* 位置 + 字体 一行 */} <TitleMiniPreview
<div className="xx-title-style-row"> settings={settings}
<div className="xx-form-field xx-half-field"> width={previewWidth}
<label></label> sampleText={settings.title || "预览标题文字"}
<select
className="xx-form-select"
value={settings.position}
onChange={(e) => onUpdatePosition(e.target.value)}
>
{POSITION_OPTIONS.map((opt) => (
<option key={opt.value} value={opt.value}>
{opt.label}
</option>
))}
</select>
</div>
<div className="xx-form-field xx-half-field">
<label></label>
<select
className="xx-form-select"
value={settings.font}
onChange={(e) => onUpdateFont(e.target.value)}
>
{FONT_OPTIONS.map((f) => (
<option key={f} value={f}>
{f}
</option>
))}
</select>
</div>
</div>
{/* 字号滑块 */}
<div className="xx-form-field">
<div className="xx-field-label-row">
<label></label>
<span className="xx-field-value">{settings.size}px</span>
</div>
<input
className="xx-slider"
type="range"
min={12}
max={128}
value={settings.size}
onChange={(e) => onUpdateSize(Number(e.target.value))}
/> />
</div> </div>
{/* 预设样式 */} {/* 预设样式10个,含抖音爆款黄) */}
<div className="xx-form-field"> <div className="ts-form-field">
<label></label> <label></label>
<TitlePresetsGrid <PresetGrid activePreset={activePreset} onApply={onApplyPreset} settings={settings} />
presets={titlePresets}
activePreset={activePreset}
onApply={onApplyPreset}
fontFamily={settings.font}
/>
</div> </div>
{/* 样式按钮:粗体/斜体/描边/阴影 */} <Tabs
<div className="xx-form-field"> size="small"
<label></label> defaultActiveKey="basic"
<div className="xx-style-btns"> items={[
<button {
className={`xx-style-btn ${settings.bold ? "active" : ""}`} key: "basic",
onClick={onToggleBold} label: "基础",
title="粗体" children: (
> <>
<b>B</b> {/* 位置 + 字体 */}
</button> <div className="ts-row-2">
<button <div className="ts-form-field ts-half">
className={`xx-style-btn ${settings.italic ? "active" : ""}`} <label></label>
onClick={onToggleItalic} <select
title="斜体" className="ts-select"
> value={settings.position}
<i>I</i> onChange={(e) => onUpdatePosition(e.target.value)}
</button> >
<button {POSITION_OPTIONS.map((o) => (
className={`xx-style-btn ${settings.stroke ? "active" : ""}`} <option key={o.value} value={o.value}>
onClick={onToggleStroke} {o.label}
title="描边" </option>
> ))}
S </select>
</button> </div>
<button <div className="ts-form-field ts-half">
className={`xx-style-btn ${settings.shadow ? "active" : ""}`} <label></label>
onClick={onToggleShadow} <select
title="阴影" className="ts-select"
> value={settings.font}
onChange={(e) => onUpdateFont(e.target.value)}
</button> >
</div> {NEW_FONT_OPTIONS.map((f) => (
</div> <option key={f.value} value={f.value}>
{f.tag === "hot" ? "🔥 " : f.tag === "new" ? "🆕 " : ""}
{f.label}
</option>
))}
</select>
</div>
</div>
<SliderRow
label="字号"
value={settings.size}
min={16}
max={120}
onChange={onUpdateSize}
/>
{/* 样式按钮 */}
<div className="ts-form-field">
<label></label>
<div className="ts-style-btns">
<button
type="button"
className={`ts-style-btn${settings.bold ? " active" : ""}`}
onClick={onToggleBold}
>
<b>B</b>
</button>
<button
type="button"
className={`ts-style-btn${settings.italic ? " active" : ""}`}
onClick={onToggleItalic}
>
<i>I</i>
</button>
<button
type="button"
className={`ts-style-btn${settings.stroke ? " active" : ""}`}
onClick={() => {
onToggleStroke()
// 如果之前 strokeWidth 为 0,启用时给个默认值
if (!settings.stroke && (settings.strokeWidth ?? 0) < 2) {
upd({ strokeWidth: 4 })
}
}}
title="描边"
>
S
</button>
<button
type="button"
className={`ts-style-btn${settings.shadow ? " active" : ""}`}
onClick={() => {
onToggleShadow()
if (!settings.shadow) {
upd({
shadowOffsetX: 2,
shadowOffsetY: 2,
shadowBlur: 4,
shadowColor: "rgba(0,0,0,0.8)",
})
}
}}
title="阴影"
>
</button>
</div>
</div>
{/* 字色 */}
<ColorPicker
label="字色"
value={settings.color}
palette={TITLE_COLOR_PALETTE}
onChange={(c) => upd({ color: c })}
/>
</>
),
},
{
key: "stroke",
label: "描边",
children: (
<>
<div className="ts-toggle-row">
<label>
<input type="checkbox" checked={settings.stroke} onChange={onToggleStroke} />
</label>
</div>
{settings.stroke && (
<>
<SliderRow
label="描边宽度"
value={settings.strokeWidth ?? 4}
min={0}
max={20}
onChange={(v) => upd({ strokeWidth: v })}
/>
<ColorPicker
label="描边颜色"
value={settings.strokeColor ?? "#000000"}
palette={STROKE_COLOR_PALETTE}
onChange={(c) => upd({ strokeColor: c })}
/>
</>
)}
</>
),
},
{
key: "shadow",
label: "阴影",
children: (
<>
<div className="ts-toggle-row">
<label>
<input type="checkbox" checked={settings.shadow} onChange={onToggleShadow} />
</label>
</div>
{settings.shadow && (
<>
<SliderRow
label="X偏移"
value={settings.shadowOffsetX ?? 2}
min={-20}
max={20}
onChange={(v) => upd({ shadowOffsetX: v })}
/>
<SliderRow
label="Y偏移"
value={settings.shadowOffsetY ?? 2}
min={-20}
max={20}
onChange={(v) => upd({ shadowOffsetY: v })}
/>
<SliderRow
label="模糊半径"
value={settings.shadowBlur ?? 4}
min={0}
max={30}
onChange={(v) => upd({ shadowBlur: v })}
/>
<div className="ts-form-field">
<label></label>
<input
type="text"
className="ts-input"
value={settings.shadowColor ?? "rgba(0,0,0,0.8)"}
onChange={(e) => upd({ shadowColor: e.target.value })}
placeholder="rgba(0,0,0,0.8)"
/>
</div>
</>
)}
</>
),
},
{
key: "bg",
label: "背景",
children: (
<>
<div className="ts-toggle-row">
<label>
<input
type="checkbox"
checked={settings.bgEnabled}
onChange={() => upd({ bgEnabled: !settings.bgEnabled })}
/>
</label>
</div>
{settings.bgEnabled && (
<>
<ColorPicker
label="背景颜色(含透明度)"
value={settings.bgColor}
palette={BG_COLOR_PALETTE}
onChange={(c) => upd({ bgColor: c })}
/>
<SliderRow
label="内边距"
value={settings.bgPadding}
min={0}
max={40}
onChange={(v) => upd({ bgPadding: v })}
/>
<SliderRow
label="圆角"
value={settings.bgRadius}
min={0}
max={30}
onChange={(v) => upd({ bgRadius: v })}
/>
</>
)}
</>
),
},
{
key: "layout",
label: "排版",
children: (
<>
<SliderRow
label="每行最大字符数"
value={settings.maxCharsPerLine ?? 0}
min={0}
max={20}
unit=""
onChange={(v) => upd({ maxCharsPerLine: v })}
/>
<div
className="ts-form-field"
style={{ fontSize: 11, color: "#9ca3af", marginTop: -4 }}
>
0 = /
</div>
<SliderRow
label="行距倍数"
value={Math.round((settings.lineHeight ?? 1.2) * 100) / 100}
min={1}
max={2}
step={0.05}
unit=""
onChange={(v) => upd({ lineHeight: Number(v.toFixed(2)) })}
/>
<SliderRow
label="顶部边距"
value={settings.marginTop ?? 24}
min={0}
max={200}
onChange={(v) => upd({ marginTop: v })}
/>
</>
),
},
...(showCoverToggle
? [
{
key: "cover",
label: "封面",
children: (
<>
<div className="ts-toggle-row">
<label>
<input
type="checkbox"
checked={coverOpen}
onChange={(e) => {
setCoverOpen(e.target.checked)
if (!e.target.checked) {
upd({ coverTitle: null })
} else {
upd({
coverTitle: {
font: settings.font,
size: Math.round(settings.size * 0.9),
color: settings.color,
bold: settings.bold,
},
})
}
}}
/>
使
</label>
</div>
{coverOpen && settings.coverTitle && (
<div style={{ fontSize: 12, color: "#6b7280", lineHeight: 1.6 }}>
//
</div>
)}
</>
),
},
]
: []),
]}
/>
</div> </div>
) )
} }
+19 -2
View File
@@ -54,11 +54,28 @@ export const POSITION_OPTIONS = [
{ value: "custom", label: "自定义" }, { value: "custom", label: "自定义" },
] ]
/* ── 标题字体选项 ── */ /* ── 标题字体选项#2001:新增 4 款爆款字体) ── */
export const FONT_OPTIONS = ["思源黑体", "思源宋体", "苹方", "微软雅黑", "楷体"] export const FONT_OPTIONS = [
"优设标题黑",
"阿里普惠体Bold",
"抖音美好体",
"思源黑体Heavy",
"思源黑体",
"思源宋体",
"苹方",
"微软雅黑",
"楷体",
]
/* ── 标题字体 CSS font-family 映射(中文显示名 → 浏览器可识别的字体栈) ── */ /* ── 标题字体 CSS font-family 映射(中文显示名 → 浏览器可识别的字体栈) ── */
export const FONT_FAMILY_MAP: Record<string, string> = { export const FONT_FAMILY_MAP: Record<string, string> = {
:
'"YouSheBiaoTiHei","YouShe Title Black","Source Han Sans SC Heavy","Noto Sans SC","PingFang SC",sans-serif',
Bold:
'"Alibaba PuHuiTi Bold","Alibaba PuHuiTi","Source Han Sans SC","PingFang SC",sans-serif',
: '"Douyin Sans","DouyinSansBold","Source Han Sans SC Heavy","PingFang SC",sans-serif',
Heavy:
'"Source Han Sans SC Heavy","Noto Sans SC Heavy","Source Han Sans CN Heavy","PingFang SC",sans-serif',
: '"Source Han Sans SC", "Noto Sans SC", "PingFang SC", "Microsoft YaHei", sans-serif', : '"Source Han Sans SC", "Noto Sans SC", "PingFang SC", "Microsoft YaHei", sans-serif',
: '"Source Han Serif SC", "Noto Serif SC", "Songti SC", "SimSun", serif', : '"Source Han Serif SC", "Noto Serif SC", "Songti SC", "SimSun", serif',
: '"PingFang SC", -apple-system, "Helvetica Neue", sans-serif', : '"PingFang SC", -apple-system, "Helvetica Neue", sans-serif',
@@ -22,6 +22,8 @@ export interface UseGenerateVideoProps {
ttsVoiceId?: string ttsVoiceId?: string
/** TTS 音色来源 */ /** TTS 音色来源 */
ttsVoiceSource?: "preset" | "clone" ttsVoiceSource?: "preset" | "clone"
/** TTS 配音风格 */
ttsStyle?: string
/** 合成后保存到配音库的 asset id / job id(叙事模式) */ /** 合成后保存到配音库的 asset id / job id(叙事模式) */
ttsVoiceAssetId?: string ttsVoiceAssetId?: string
/** 智能降重开关(默认 true) */ /** 智能降重开关(默认 true) */
@@ -30,8 +30,23 @@ interface UseBatchCoversOptions {
color: string color: string
position: string position: string
bold: boolean bold: boolean
italic?: boolean
stroke: boolean stroke: boolean
strokeWidth?: number
strokeColor?: string
shadow: boolean shadow: boolean
shadowOffsetX?: number
shadowOffsetY?: number
shadowBlur?: number
shadowColor?: string
lineHeight?: number
marginTop?: number
maxCharsPerLine?: number
bgEnabled?: boolean
bgColor?: string
bgPadding?: number
bgRadius?: number
lineOverrides?: unknown[]
} }
covers: string[] covers: string[]
onCoversChange: CoversChangeFn onCoversChange: CoversChangeFn
@@ -100,8 +115,37 @@ export function useBatchCovers({
font_color: titleStyle.color, font_color: titleStyle.color,
position: titleStyle.position, position: titleStyle.position,
bold: titleStyle.bold, bold: titleStyle.bold,
stroke: titleStyle.stroke, italic: titleStyle.italic,
shadow: titleStyle.shadow, stroke: titleStyle.stroke
? {
enabled: true,
width: titleStyle.strokeWidth ?? 4,
color: titleStyle.strokeColor ?? "#000000",
}
: { enabled: false },
shadow: titleStyle.shadow
? {
enabled: true,
offset_x: titleStyle.shadowOffsetX ?? 2,
offset_y: titleStyle.shadowOffsetY ?? 2,
blur: titleStyle.shadowBlur ?? 4,
color: titleStyle.shadowColor ?? "rgba(0,0,0,0.8)",
}
: { enabled: false },
line_height: titleStyle.lineHeight ?? 1.2,
margin_top: titleStyle.marginTop ?? 24,
max_chars_per_line: titleStyle.maxCharsPerLine ?? 0,
background: titleStyle.bgEnabled
? {
enabled: true,
color: titleStyle.bgColor,
padding: titleStyle.bgPadding,
radius: titleStyle.bgRadius,
}
: { enabled: false },
line_overrides: (titleStyle.lineOverrides ?? []) as Array<
Record<string, unknown>
>,
}, },
} }
: {}), : {}),
@@ -13,6 +13,7 @@ import type { EditPlanClip } from "@/api/template-editor"
import type { CoverConfig } from "../../types/cover" import type { CoverConfig } from "../../types/cover"
import type { PresetVoiceItem } from "@/api/voices" import type { PresetVoiceItem } from "@/api/voices"
import type { ScriptItem } from "@/api/scripts" import type { ScriptItem } from "@/api/scripts"
import { DEFAULT_TTS_STYLE, type TtsStyle } from "@/api/tts/styles"
import { DEFAULT_COVER_SETTINGS, DEFAULT_CLIP_COUNT } from "../../constants" import { DEFAULT_COVER_SETTINGS, DEFAULT_CLIP_COUNT } from "../../constants"
import type { TitleSettings } from "../../types" import type { TitleSettings } from "../../types"
import { usePlanConfigLoader } from "./usePlanConfigLoader" import { usePlanConfigLoader } from "./usePlanConfigLoader"
@@ -33,6 +34,21 @@ const DEFAULT_TITLE_SETTINGS: TitleSettings = {
color: "#ffffff", color: "#ffffff",
posX: null, posX: null,
posY: null, posY: null,
lineHeight: 1.2,
marginTop: 24,
maxCharsPerLine: 0,
strokeWidth: 4,
strokeColor: "#000000",
shadowOffsetX: 2,
shadowOffsetY: 2,
shadowBlur: 4,
shadowColor: "rgba(0,0,0,0.8)",
bgEnabled: false,
bgColor: "rgba(0,0,0,0.5)",
bgPadding: 12,
bgRadius: 8,
lineOverrides: [],
coverTitle: null,
} }
export interface GenerateFormState { export interface GenerateFormState {
@@ -95,6 +111,9 @@ export interface GenerateFormState {
/** TTS 音色来源:preset 系统 / clone 克隆 */ /** TTS 音色来源:preset 系统 / clone 克隆 */
ttsVoiceSource: "preset" | "clone" ttsVoiceSource: "preset" | "clone"
setTtsVoiceSource: (src: "preset" | "clone") => void setTtsVoiceSource: (src: "preset" | "clone") => void
/** TTS 配音风格 */
ttsStyle: TtsStyle
setTtsStyle: (s: TtsStyle) => void
/** 合成后配音库 asset id(叙事模式保存到库后获得;随机模式 = selectedVoice */ /** 合成后配音库 asset id(叙事模式保存到库后获得;随机模式 = selectedVoice */
ttsVoiceAssetId: string ttsVoiceAssetId: string
setTtsVoiceAssetId: (id: string) => void setTtsVoiceAssetId: (id: string) => void
@@ -234,6 +253,7 @@ export const useGenerateFormState = (): GenerateFormState => {
const [selectedScript, setSelectedScript] = useState<ScriptItem | null>(null) const [selectedScript, setSelectedScript] = useState<ScriptItem | null>(null)
const [ttsVoiceId, setTtsVoiceId] = useState<string>("") const [ttsVoiceId, setTtsVoiceId] = useState<string>("")
const [ttsVoiceSource, setTtsVoiceSource] = useState<"preset" | "clone">("preset") const [ttsVoiceSource, setTtsVoiceSource] = useState<"preset" | "clone">("preset")
const [ttsStyle, setTtsStyle] = useState<TtsStyle>(DEFAULT_TTS_STYLE)
const [ttsVoiceAssetId, setTtsVoiceAssetId] = useState<string>("") const [ttsVoiceAssetId, setTtsVoiceAssetId] = useState<string>("")
const [dedupEnabled, setDedupEnabled] = useState<boolean>(true) const [dedupEnabled, setDedupEnabled] = useState<boolean>(true)
@@ -311,6 +331,8 @@ export const useGenerateFormState = (): GenerateFormState => {
setTtsVoiceId, setTtsVoiceId,
ttsVoiceSource, ttsVoiceSource,
setTtsVoiceSource, setTtsVoiceSource,
ttsStyle,
setTtsStyle,
ttsVoiceAssetId, ttsVoiceAssetId,
setTtsVoiceAssetId, setTtsVoiceAssetId,
dedupEnabled, dedupEnabled,
@@ -1,6 +1,7 @@
import { useEffect } from "react" import { useEffect } from "react"
import type { CoverConfig } from "../../types/cover" import type { CoverConfig } from "../../types/cover"
import type { TitleSettings } from "../../types" import type { TitleSettings } from "../../types"
import type { TitleLineOverride } from "@/components/title/types"
import type { TitleConfig } from "@/api/template-editor" import type { TitleConfig } from "@/api/template-editor"
import { getEditPlan } from "@/api/template-editor" import { getEditPlan } from "@/api/template-editor"
@@ -12,6 +13,156 @@ interface UsePlanConfigLoaderOptions {
setSelectedMaterials: (ids: string[]) => void setSelectedMaterials: (ids: string[]) => void
} }
/** #2001:统一归一化 title_config snake_case -> camelCase TitleSettings */
function mapTitleCfgToSettings(
prev: TitleSettings,
tc: TitleConfig & Record<string, unknown>,
): TitleSettings {
const stroke = tc.stroke as
boolean | { enabled?: boolean; width?: number; color?: string } | undefined
const strokeEnabled: boolean | undefined =
typeof stroke === "object" && stroke ? stroke.enabled !== false : !!stroke || undefined
const strokeW: number | undefined =
typeof stroke === "object" && stroke
? (stroke.width ?? (tc.stroke_width as number | undefined))
: (tc.stroke_width as number | undefined)
const strokeC: string | undefined =
typeof stroke === "object" && stroke
? (stroke.color ?? (tc.stroke_color as string | undefined))
: (tc.stroke_color as string | undefined)
const shadow = tc.shadow as
| boolean
| { enabled?: boolean; offset_x?: number; offset_y?: number; blur?: number; color?: string }
| undefined
const shadowEnabled: boolean | undefined =
typeof shadow === "object" && shadow ? shadow.enabled !== false : !!shadow || undefined
const shOffX: number | undefined =
typeof shadow === "object" && shadow
? (shadow.offset_x ?? (tc.shadow_offset_x as number | undefined))
: (tc.shadow_offset_x as number | undefined)
const shOffY: number | undefined =
typeof shadow === "object" && shadow
? (shadow.offset_y ?? (tc.shadow_offset_y as number | undefined))
: (tc.shadow_offset_y as number | undefined)
const shBlur: number | undefined =
typeof shadow === "object" && shadow
? (shadow.blur ?? (tc.shadow_blur as number | undefined))
: (tc.shadow_blur as number | undefined)
const shColor: string | undefined =
typeof shadow === "object" && shadow
? (shadow.color ?? (tc.shadow_color as string | undefined))
: (tc.shadow_color as string | undefined)
const bg = tc.background as
{ enabled?: boolean; color?: string; padding?: number; radius?: number } | undefined
const bgEnabled: boolean | undefined =
(bg && typeof bg === "object" ? bg.enabled : undefined) ??
(tc.bg_enabled as boolean | undefined)
const bgColor: string | undefined =
(bg && typeof bg === "object" ? bg.color : undefined) ?? (tc.bg_color as string | undefined)
const bgPadding: number | undefined =
(bg && typeof bg === "object" ? bg.padding : undefined) ?? (tc.bg_padding as number | undefined)
const bgRadius: number | undefined =
(bg && typeof bg === "object" ? bg.radius : undefined) ?? (tc.bg_radius as number | undefined)
const ct = (tc.cover_title_config ?? null) as null | Record<string, unknown>
let coverTitle: TitleSettings["coverTitle"] = prev.coverTitle
if (ct) {
const ctStroke = ct.stroke as
boolean | { enabled?: boolean; width?: number; color?: string } | undefined
const ctShadow = ct.shadow as
| boolean
| { enabled?: boolean; offset_x?: number; offset_y?: number; blur?: number; color?: string }
| undefined
const ctBg = ct.background as
{ enabled?: boolean; color?: string; padding?: number; radius?: number } | undefined
coverTitle = {
title: (ct.title as string | undefined) ?? prev.coverTitle?.title ?? "",
font: (ct.font as string | undefined) ?? prev.coverTitle?.font,
size:
(ct.font_size as number | undefined) ??
(ct.size as number | undefined) ??
prev.coverTitle?.size,
color:
(ct.font_color as string | undefined) ??
(ct.color as string | undefined) ??
prev.coverTitle?.color,
bold: (ct.bold as boolean | undefined) ?? prev.coverTitle?.bold,
italic: (ct.italic as boolean | undefined) ?? prev.coverTitle?.italic,
position: (ct.position as string | undefined) ?? prev.coverTitle?.position,
stroke:
typeof ctStroke === "object" && ctStroke
? ctStroke.enabled !== false
: ((ctStroke as boolean | undefined) ?? prev.coverTitle?.stroke),
strokeWidth:
(typeof ctStroke === "object" && ctStroke ? ctStroke.width : undefined) ??
(ct.stroke_width as number | undefined) ??
prev.coverTitle?.strokeWidth,
strokeColor:
(typeof ctStroke === "object" && ctStroke ? ctStroke.color : undefined) ??
(ct.stroke_color as string | undefined) ??
prev.coverTitle?.strokeColor,
shadow:
typeof ctShadow === "object" && ctShadow
? ctShadow.enabled !== false
: ((ctShadow as boolean | undefined) ?? prev.coverTitle?.shadow),
shadowOffsetX:
(typeof ctShadow === "object" && ctShadow ? ctShadow.offset_x : undefined) ??
(ct.shadow_offset_x as number | undefined) ??
prev.coverTitle?.shadowOffsetX,
shadowOffsetY:
(typeof ctShadow === "object" && ctShadow ? ctShadow.offset_y : undefined) ??
(ct.shadow_offset_y as number | undefined) ??
prev.coverTitle?.shadowOffsetY,
shadowBlur:
(typeof ctShadow === "object" && ctShadow ? ctShadow.blur : undefined) ??
(ct.shadow_blur as number | undefined) ??
prev.coverTitle?.shadowBlur,
shadowColor:
(typeof ctShadow === "object" && ctShadow ? ctShadow.color : undefined) ??
(ct.shadow_color as string | undefined) ??
prev.coverTitle?.shadowColor,
bgEnabled:
ctBg?.enabled ?? (ct.bg_enabled as boolean | undefined) ?? prev.coverTitle?.bgEnabled,
bgColor: ctBg?.color ?? (ct.bg_color as string | undefined) ?? prev.coverTitle?.bgColor,
bgPadding:
ctBg?.padding ?? (ct.bg_padding as number | undefined) ?? prev.coverTitle?.bgPadding,
bgRadius: ctBg?.radius ?? (ct.bg_radius as number | undefined) ?? prev.coverTitle?.bgRadius,
}
}
const result: TitleSettings = {
...prev,
title: (tc.content as string | undefined) || prev.title,
aiAutoSelect: (tc.ai_auto_select as boolean | undefined) || false,
position: prev.position,
font: (tc.font_preset as string | undefined) || prev.font,
size: (tc.font_size as number | undefined) || prev.size,
color: (tc.font_color as string | undefined) || prev.color,
bold: (tc.bold as boolean | undefined) ?? prev.bold,
italic: (tc.italic as boolean | undefined) ?? prev.italic,
stroke: strokeEnabled ?? prev.stroke,
strokeWidth: strokeW ?? prev.strokeWidth,
strokeColor: strokeC ?? prev.strokeColor,
shadow: shadowEnabled ?? prev.shadow,
shadowOffsetX: shOffX ?? prev.shadowOffsetX,
shadowOffsetY: shOffY ?? prev.shadowOffsetY,
shadowBlur: shBlur ?? prev.shadowBlur,
shadowColor: shColor ?? prev.shadowColor,
lineHeight: (tc.line_height as number | undefined) ?? prev.lineHeight,
marginTop: (tc.margin_top as number | undefined) ?? prev.marginTop,
maxCharsPerLine: (tc.max_chars_per_line as number | undefined) ?? prev.maxCharsPerLine,
bgEnabled: bgEnabled ?? prev.bgEnabled,
bgColor: bgColor ?? prev.bgColor,
bgPadding: bgPadding ?? prev.bgPadding,
bgRadius: bgRadius ?? prev.bgRadius,
lineOverrides: ((tc.line_overrides as unknown[] | undefined) ?? []) as TitleLineOverride[],
coverTitle,
}
return result
}
/** /**
* 从 URL 参数或编辑计划 ID 加载表单配置 * 从 URL 参数或编辑计划 ID 加载表单配置
*/ */
@@ -27,14 +178,7 @@ export function usePlanConfigLoader({
if (!planConfigStr) return if (!planConfigStr) return
try { try {
const config = JSON.parse(planConfigStr) as { const config = JSON.parse(planConfigStr) as {
title_config?: { title_config?: Record<string, unknown>
content?: string
ai_auto_select?: boolean
position?: string
font_preset?: string
font_size?: number
font_color?: string
}
subtitle_config?: { enabled?: boolean } subtitle_config?: { enabled?: boolean }
bgm_config?: { enabled?: boolean; music_id?: string } bgm_config?: { enabled?: boolean; music_id?: string }
mode?: string mode?: string
@@ -43,16 +187,8 @@ export function usePlanConfigLoader({
} }
if (config.title_config) { if (config.title_config) {
const tc = config.title_config as TitleConfig const tc = config.title_config as TitleConfig & Record<string, unknown>
setTitleSettings((prev: TitleSettings) => ({ setTitleSettings((prev: TitleSettings) => mapTitleCfgToSettings(prev, tc))
...prev,
title: tc.content || "",
aiAutoSelect: tc.ai_auto_select || false,
position: prev.position, // 强制保留默认/用户选择,不从草稿配置同步位置
font: tc.font_preset || prev.font,
size: tc.font_size || prev.size,
color: tc.font_color || prev.color,
}))
} }
if (config.segments && config.segments.length > 0) { if (config.segments && config.segments.length > 0) {
const assetIds = config.segments const assetIds = config.segments
@@ -76,15 +212,8 @@ export function usePlanConfigLoader({
if (plan.name) setTitleSettings((prev: TitleSettings) => ({ ...prev, title: plan.name })) if (plan.name) setTitleSettings((prev: TitleSettings) => ({ ...prev, title: plan.name }))
const cfg = plan.config const cfg = plan.config
if (cfg?.title_config) { if (cfg?.title_config) {
setTitleSettings((prev: TitleSettings) => ({ const tc2 = cfg.title_config as unknown as TitleConfig & Record<string, unknown>
...prev, setTitleSettings((prev: TitleSettings) => mapTitleCfgToSettings(prev, tc2))
aiAutoSelect: cfg.title_config!.ai_auto_select,
title: cfg.title_config!.content || prev.title,
position: prev.position, // 强制保留默认/用户选择,不从远程草稿同步位置
font: cfg.title_config!.font_preset || prev.font,
size: cfg.title_config!.font_size || prev.size,
color: cfg.title_config!.font_color || prev.color,
}))
} }
if (cfg?.cover_config) { if (cfg?.cover_config) {
const cc = cfg.cover_config as CoverConfig const cc = cfg.cover_config as CoverConfig
@@ -208,6 +208,7 @@ export function useGenerateVideo(props: UseGenerateVideoProps) {
script_id: props.selectedScript.id, script_id: props.selectedScript.id,
tts_voice_id: props.ttsVoiceId || undefined, tts_voice_id: props.ttsVoiceId || undefined,
tts_voice_source: props.ttsVoiceSource || undefined, tts_voice_source: props.ttsVoiceSource || undefined,
tts_style: props.ttsStyle || undefined,
} }
: {}), : {}),
dedup_enabled: dedupEnabled, dedup_enabled: dedupEnabled,
@@ -240,8 +241,91 @@ export function useGenerateVideo(props: UseGenerateVideoProps) {
} }
: {}), : {}),
bold: props.titleSettings.bold, bold: props.titleSettings.bold,
stroke: props.titleSettings.stroke, italic: props.titleSettings.italic,
shadow: props.titleSettings.shadow, stroke: props.titleSettings.stroke
? {
enabled: true,
width: props.titleSettings.strokeWidth ?? 4,
color: props.titleSettings.strokeColor ?? "#000000",
}
: { enabled: false },
shadow: props.titleSettings.shadow
? {
enabled: true,
offset_x: props.titleSettings.shadowOffsetX ?? 2,
offset_y: props.titleSettings.shadowOffsetY ?? 2,
blur: props.titleSettings.shadowBlur ?? 4,
color: props.titleSettings.shadowColor ?? "rgba(0,0,0,0.8)",
}
: { enabled: false },
line_height: props.titleSettings.lineHeight ?? 1.2,
margin_top: props.titleSettings.marginTop ?? 24,
max_chars_per_line: props.titleSettings.maxCharsPerLine ?? 0,
...(props.titleSettings.bgEnabled
? {
background: {
enabled: true,
color: props.titleSettings.bgColor,
padding: props.titleSettings.bgPadding,
radius: props.titleSettings.bgRadius,
},
}
: { background: { enabled: false } }),
line_overrides: (props.titleSettings.lineOverrides ?? []).map((lo) => ({
line_index: lo.line_index,
text: lo.text,
size: lo.size,
color: lo.color,
bold: lo.bold,
italic: lo.italic,
stroke: lo.stroke,
highlights: lo.highlights?.map((h) => ({
word: h.word,
color: h.color,
bold: h.bold,
scale: h.scale,
})),
})),
...(props.titleSettings.coverTitle
? {
cover_title_config: {
title: props.titleSettings.coverTitle.title,
font: props.titleSettings.coverTitle.font,
font_size: props.titleSettings.coverTitle.size,
font_color: props.titleSettings.coverTitle.color,
bold: props.titleSettings.coverTitle.bold,
italic: props.titleSettings.coverTitle.italic,
position: props.titleSettings.coverTitle.position,
stroke: props.titleSettings.coverTitle.stroke
? {
enabled: true,
width: props.titleSettings.coverTitle.strokeWidth ?? 4,
color: props.titleSettings.coverTitle.strokeColor ?? "#000000",
}
: { enabled: false },
shadow: props.titleSettings.coverTitle.shadow
? {
enabled: true,
offset_x: props.titleSettings.coverTitle.shadowOffsetX ?? 2,
offset_y: props.titleSettings.coverTitle.shadowOffsetY ?? 2,
blur: props.titleSettings.coverTitle.shadowBlur ?? 4,
color:
props.titleSettings.coverTitle.shadowColor ?? "rgba(0,0,0,0.8)",
}
: { enabled: false },
...(props.titleSettings.coverTitle.bgEnabled
? {
background: {
enabled: true,
color: props.titleSettings.coverTitle.bgColor,
padding: props.titleSettings.coverTitle.bgPadding,
radius: props.titleSettings.coverTitle.bgRadius,
},
}
: { background: { enabled: false } }),
},
}
: {}),
}, },
} }
: {}), : {}),
@@ -1,5 +1,6 @@
import { useCallback, useMemo } from "react" import { useCallback, useMemo } from "react"
import { TITLE_PRESETS } from "../../constants" import { TITLE_PRESETS } from "../../constants"
import { TITLE_PRESETS as NEW_TITLE_PRESETS } from "@/components/title/constants"
import type { TitleSettings } from "../../types" import type { TitleSettings } from "../../types"
interface UseTitleStyleUpdatersOptions { interface UseTitleStyleUpdatersOptions {
@@ -97,26 +98,45 @@ export function useTitleStyleUpdaters({
onTitleSettingsChange({ ...titleSettings, shadow: !titleSettings.shadow }) onTitleSettingsChange({ ...titleSettings, shadow: !titleSettings.shadow })
}, [titleSettings, onTitleSettingsChange]) }, [titleSettings, onTitleSettingsChange])
/** 应用预设:只覆盖 color/bold/italic/stroke/shadow,不改变字号 */ /** 应用预设(支持新预设细粒度字段) */
const applyPreset = useCallback( const applyPreset = useCallback(
(presetKey: string) => { (presetKey: string) => {
const preset = TITLE_PRESETS.find((p) => p.key === presetKey) // 优先匹配新预设(10个爆款预设),fallback 旧预设
if (!preset) return const newPreset = NEW_TITLE_PRESETS.find((p) => p.key === presetKey)
const oldPreset = TITLE_PRESETS.find((p) => p.key === presetKey)
if (newPreset) {
onTitleSettingsChange({
...titleSettings,
...(newPreset.style as Partial<TitleSettings>),
// 清除逐行覆盖
lineOverrides: [],
})
return
}
if (!oldPreset) return
onTitleSettingsChange({ onTitleSettingsChange({
...titleSettings, ...titleSettings,
color: preset.style.color, color: oldPreset.style.color,
bold: preset.style.bold, bold: oldPreset.style.bold,
italic: preset.style.italic, italic: oldPreset.style.italic,
stroke: preset.style.stroke, stroke: oldPreset.style.stroke,
shadow: preset.style.shadow, shadow: oldPreset.style.shadow,
}) })
}, },
[titleSettings, onTitleSettingsChange], [titleSettings, onTitleSettingsChange],
) )
/** 通用字段更新(patch */
const updateStyle = useCallback(
(patch: Partial<TitleSettings>) => {
onTitleSettingsChange({ ...titleSettings, ...patch })
},
[titleSettings, onTitleSettingsChange],
)
return { return {
activePreset, activePreset,
titlePresets: TITLE_PRESETS, titlePresets: NEW_TITLE_PRESETS,
updateTitle, updateTitle,
toggleAiAutoSelect, toggleAiAutoSelect,
updatePosition, updatePosition,
@@ -129,5 +149,6 @@ export function useTitleStyleUpdaters({
toggleStroke, toggleStroke,
toggleShadow, toggleShadow,
applyPreset, applyPreset,
updateStyle,
} }
} }
+83 -25
View File
@@ -3,8 +3,9 @@
*/ */
import type { AssetItem } from "@/api/assets" import type { AssetItem } from "@/api/assets"
import type { TitleLineOverride } from "@/components/title/types"
/* ── 标题设置 ── */ /* ── 标题设置(#2001 升级:新增描边/阴影/背景/逐行/封面独立样式/排版字段) ── */
export interface TitleSettings { export interface TitleSettings {
aiAutoSelect: boolean aiAutoSelect: boolean
title: string title: string
@@ -19,6 +20,58 @@ export interface TitleSettings {
/** 自由位置坐标(PlayRes 像素),仅当 position="custom" 时有效 */ /** 自由位置坐标(PlayRes 像素),仅当 position="custom" 时有效 */
posX: number | null posX: number | null
posY: number | null posY: number | null
/* ── 排版(P0 ── */
/** 行距倍数,默认 1.2 */
lineHeight: number
/** 顶部边距(position=toppx @720p */
marginTop: number
/** 每行最大字符数(4-20),0=不自动换行 */
maxCharsPerLine: number
/* ── 描边参数化(P0) ── */
strokeWidth: number
strokeColor: string
/* ── 阴影参数化(P1) ── */
shadowOffsetX: number
shadowOffsetY: number
shadowBlur: number
shadowColor: string
/* ── 背景色块(P1) ── */
bgEnabled: boolean
bgColor: string
bgPadding: number
bgRadius: number
/* ── 逐行独立样式(P1) ── */
lineOverrides: TitleLineOverride[]
/* ── 封面独立标题(P1):null=沿用主标题 ── */
coverTitle: null | {
title?: string
font?: string
size?: number
color?: string
bold?: boolean
italic?: boolean
position?: string
stroke?: boolean
strokeWidth?: number
strokeColor?: string
shadow?: boolean
shadowOffsetX?: number
shadowOffsetY?: number
shadowBlur?: number
shadowColor?: string
bgEnabled?: boolean
bgColor?: string
bgPadding?: number
bgRadius?: number
lineHeight?: number
maxCharsPerLine?: number
}
} }
/* ── 智能匹配结果 ── */ /* ── 智能匹配结果 ── */
@@ -49,28 +102,33 @@ export interface StepDef {
label: string label: string
} }
/* ── 标题预设样式 ── */ /** 旧版 TitleSettings 的默认值字段(P0/P1 新字段补齐默认值) */
export interface TitlePresetStyle { export const DEFAULT_TITLE_SETTINGS_FULL: TitleSettings = {
size: number aiAutoSelect: false,
color: string title: "",
bold: boolean position: "top",
italic: boolean font: "思源黑体",
stroke: boolean size: 28,
shadow: boolean bold: true,
} italic: false,
stroke: true,
export interface TitlePreset { shadow: false,
key: string color: "#ffffff",
label: string posX: null,
style: TitlePresetStyle posY: null,
previewStyle: Record<string, string | number> lineHeight: 1.2,
} marginTop: 24,
maxCharsPerLine: 0,
/* ── 生成结果视频 ── */ strokeWidth: 4,
export interface GeneratedVideoResult { strokeColor: "#000000",
id: string shadowOffsetX: 2,
url: string shadowOffsetY: 2,
thumbnail: string shadowBlur: 4,
duration: number shadowColor: "rgba(0,0,0,0.8)",
title: string bgEnabled: false,
bgColor: "rgba(0,0,0,0.5)",
bgPadding: 12,
bgRadius: 8,
lineOverrides: [],
coverTitle: null,
} }
@@ -98,6 +98,7 @@ const VoiceMaterialLibrary: React.FC = () => {
ttsText, ttsText,
ttsVoiceId, ttsVoiceId,
ttsSpeed, ttsSpeed,
ttsStyle,
ttsStatus, ttsStatus,
ttsAudioUrl, ttsAudioUrl,
ttsError, ttsError,
@@ -107,6 +108,7 @@ const VoiceMaterialLibrary: React.FC = () => {
setTtsText, setTtsText,
setTtsVoiceId, setTtsVoiceId,
setTtsSpeed, setTtsSpeed,
setTtsStyle,
handleTtsSynthesize, handleTtsSynthesize,
handleTtsSave, handleTtsSave,
handleTtsClose, handleTtsClose,
@@ -315,6 +317,7 @@ const VoiceMaterialLibrary: React.FC = () => {
text={ttsText} text={ttsText}
voiceId={ttsVoiceId} voiceId={ttsVoiceId}
speed={ttsSpeed} speed={ttsSpeed}
style={ttsStyle}
status={ttsStatus} status={ttsStatus}
audioUrl={ttsAudioUrl ?? ""} audioUrl={ttsAudioUrl ?? ""}
error={ttsError ?? ""} error={ttsError ?? ""}
@@ -324,6 +327,7 @@ const VoiceMaterialLibrary: React.FC = () => {
onTextChange={setTtsText} onTextChange={setTtsText}
onVoiceChange={setTtsVoiceId} onVoiceChange={setTtsVoiceId}
onSpeedChange={setTtsSpeed} onSpeedChange={setTtsSpeed}
onStyleChange={setTtsStyle}
onSynthesize={handleTtsSynthesize} onSynthesize={handleTtsSynthesize}
onSave={handleTtsSave} onSave={handleTtsSave}
/> />
@@ -1,6 +1,8 @@
import React from "react" import React from "react"
import { RobotOutlined, LoadingOutlined, PlusOutlined } from "@ant-design/icons" import { RobotOutlined, LoadingOutlined, PlusOutlined } from "@ant-design/icons"
import { Button } from "@/components/ui" import { Button } from "@/components/ui"
import TtsStyleSelector from "@/components/voice/TtsStyleSelector"
import type { TtsStyle } from "@/api/tts/styles"
export type TtsStatus = "idle" | "synthesizing" | "done" | "error" export type TtsStatus = "idle" | "synthesizing" | "done" | "error"
@@ -20,6 +22,8 @@ interface TtsModalProps {
text: string text: string
voiceId: string voiceId: string
speed: number speed: number
style: TtsStyle
onStyleChange: (style: TtsStyle) => void
status: TtsStatus status: TtsStatus
audioUrl: string audioUrl: string
error: string error: string
@@ -39,6 +43,8 @@ const TtsModal: React.FC<TtsModalProps> = ({
text, text,
voiceId, voiceId,
speed, speed,
style,
onStyleChange,
status, status,
audioUrl, audioUrl,
error, error,
@@ -143,6 +149,9 @@ const TtsModal: React.FC<TtsModalProps> = ({
/> />
</div> </div>
{/* 配音风格 */}
<TtsStyleSelector value={style} onChange={onStyleChange} compact />
{/* 合成按钮 */} {/* 合成按钮 */}
<Button <Button
buttonType="primary" buttonType="primary"
@@ -2,6 +2,7 @@ import { useState, useRef, useCallback, useEffect } from "react"
import { useQuery, useQueryClient } from "@tanstack/react-query" import { useQuery, useQueryClient } from "@tanstack/react-query"
import { message } from "antd" import { message } from "antd"
import { synthesizeSpeech, getTTSJobStatus, saveTtsToLibrary } from "@/api/tts" import { synthesizeSpeech, getTTSJobStatus, saveTtsToLibrary } from "@/api/tts"
import { DEFAULT_TTS_STYLE, type TtsStyle } from "@/api/tts/styles"
import { fetchPresetVoices, type PresetVoiceItem } from "@/api/voices" import { fetchPresetVoices, type PresetVoiceItem } from "@/api/voices"
import { getVoiceClonesWithTotal, toVoiceClone } from "@/api/voice-clone" import { getVoiceClonesWithTotal, toVoiceClone } from "@/api/voice-clone"
@@ -18,6 +19,7 @@ export function useTtsSynthesize() {
const [ttsText, setTtsText] = useState("") const [ttsText, setTtsText] = useState("")
const [ttsVoiceId, setTtsVoiceId] = useState<string>("") const [ttsVoiceId, setTtsVoiceId] = useState<string>("")
const [ttsSpeed, setTtsSpeed] = useState(1.0) const [ttsSpeed, setTtsSpeed] = useState(1.0)
const [ttsStyle, setTtsStyle] = useState<TtsStyle>(DEFAULT_TTS_STYLE)
const [ttsJobId, setTtsJobId] = useState<string | null>(null) const [ttsJobId, setTtsJobId] = useState<string | null>(null)
const [ttsStatus, setTtsStatus] = useState<TtsStatus>("idle") const [ttsStatus, setTtsStatus] = useState<TtsStatus>("idle")
const [ttsAudioUrl, setTtsAudioUrl] = useState<string | null>(null) const [ttsAudioUrl, setTtsAudioUrl] = useState<string | null>(null)
@@ -59,6 +61,7 @@ export function useTtsSynthesize() {
text: ttsText.trim(), text: ttsText.trim(),
voice_id: ttsVoiceId || undefined, voice_id: ttsVoiceId || undefined,
speed: ttsSpeed, speed: ttsSpeed,
style: ttsStyle,
}) })
setTtsJobId(resp.job_id) setTtsJobId(resp.job_id)
@@ -89,7 +92,7 @@ export function useTtsSynthesize() {
setTtsStatus("error") setTtsStatus("error")
setTtsError(msg) setTtsError(msg)
} }
}, [ttsText, ttsVoiceId, ttsSpeed]) }, [ttsText, ttsVoiceId, ttsSpeed, ttsStyle])
/** 保存 TTS 结果到素材库 */ /** 保存 TTS 结果到素材库 */
const handleTtsSave = useCallback(async () => { const handleTtsSave = useCallback(async () => {
@@ -131,6 +134,7 @@ export function useTtsSynthesize() {
ttsText, ttsText,
ttsVoiceId, ttsVoiceId,
ttsSpeed, ttsSpeed,
ttsStyle,
ttsJobId, ttsJobId,
ttsStatus, ttsStatus,
ttsAudioUrl, ttsAudioUrl,
@@ -141,6 +145,7 @@ export function useTtsSynthesize() {
setTtsText, setTtsText,
setTtsVoiceId, setTtsVoiceId,
setTtsSpeed, setTtsSpeed,
setTtsStyle,
handleTtsSynthesize, handleTtsSynthesize,
handleTtsSave, handleTtsSave,
handleTtsClose, handleTtsClose,
@@ -133,6 +133,7 @@ const VoiceLibrary: React.FC = () => {
ttsVoiceId, ttsVoiceId,
ttsSpeed, ttsSpeed,
ttsEmotion, ttsEmotion,
ttsStyle,
ttsLanguage, ttsLanguage,
ttsStatus, ttsStatus,
ttsAudioUrl, ttsAudioUrl,
@@ -140,6 +141,7 @@ const VoiceLibrary: React.FC = () => {
setTtsText, setTtsText,
setTtsSpeed, setTtsSpeed,
setTtsEmotion, setTtsEmotion,
setTtsStyle,
setTtsLanguage, setTtsLanguage,
setTtsOpen, setTtsOpen,
handleVoiceChange, handleVoiceChange,
@@ -368,6 +370,7 @@ const VoiceLibrary: React.FC = () => {
ttsVoiceId={ttsVoiceId} ttsVoiceId={ttsVoiceId}
ttsSpeed={ttsSpeed} ttsSpeed={ttsSpeed}
ttsEmotion={ttsEmotion} ttsEmotion={ttsEmotion}
ttsStyle={ttsStyle}
ttsLanguage={ttsLanguage} ttsLanguage={ttsLanguage}
ttsStatus={ttsStatus} ttsStatus={ttsStatus}
ttsAudioUrl={ttsAudioUrl} ttsAudioUrl={ttsAudioUrl}
@@ -381,6 +384,7 @@ const VoiceLibrary: React.FC = () => {
onTtsVoiceChange={handleVoiceChange} onTtsVoiceChange={handleVoiceChange}
onTtsSpeedChange={setTtsSpeed} onTtsSpeedChange={setTtsSpeed}
onTtsEmotionChange={setTtsEmotion} onTtsEmotionChange={setTtsEmotion}
onTtsStyleChange={setTtsStyle}
onTtsLanguageChange={setTtsLanguage} onTtsLanguageChange={setTtsLanguage}
onTtsSynthesize={handleTtsSynthesize} onTtsSynthesize={handleTtsSynthesize}
onTtsSave={handleTtsSave} onTtsSave={handleTtsSave}
@@ -9,6 +9,7 @@ import LanguageControl from "./tts-modal/LanguageControl"
import SynthesizeButton from "./tts-modal/SynthesizeButton" import SynthesizeButton from "./tts-modal/SynthesizeButton"
import ErrorAlert from "./tts-modal/ErrorAlert" import ErrorAlert from "./tts-modal/ErrorAlert"
import ResultPanel from "./tts-modal/ResultPanel" import ResultPanel from "./tts-modal/ResultPanel"
import TtsStyleSelector from "@/components/voice/TtsStyleSelector"
import { PRESET_TTS_LANGUAGE_OPTIONS, CLONE_TTS_LANGUAGE_OPTIONS } from "./tts-modal/constants" import { PRESET_TTS_LANGUAGE_OPTIONS, CLONE_TTS_LANGUAGE_OPTIONS } from "./tts-modal/constants"
/** AI 配音弹窗 */ /** AI 配音弹窗 */
@@ -18,6 +19,7 @@ const TtsModal: React.FC<TtsModalProps> = ({
ttsVoiceId, ttsVoiceId,
ttsSpeed, ttsSpeed,
ttsEmotion, ttsEmotion,
ttsStyle,
ttsLanguage, ttsLanguage,
ttsStatus, ttsStatus,
ttsAudioUrl, ttsAudioUrl,
@@ -29,6 +31,7 @@ const TtsModal: React.FC<TtsModalProps> = ({
onVoiceChange, onVoiceChange,
onSpeedChange, onSpeedChange,
onEmotionChange, onEmotionChange,
onStyleChange,
onLanguageChange, onLanguageChange,
onSynthesize, onSynthesize,
onSave, onSave,
@@ -72,6 +75,7 @@ const TtsModal: React.FC<TtsModalProps> = ({
/> />
</div> </div>
<SpeedControl speed={ttsSpeed} onChange={onSpeedChange} /> <SpeedControl speed={ttsSpeed} onChange={onSpeedChange} />
<TtsStyleSelector value={ttsStyle} onChange={onStyleChange} compact />
<SynthesizeButton status={ttsStatus} text={ttsText} onClick={onSynthesize} /> <SynthesizeButton status={ttsStatus} text={ttsText} onClick={onSynthesize} />
{ttsError && <ErrorAlert error={ttsError} />} {ttsError && <ErrorAlert error={ttsError} />}
{ttsStatus === "done" && ttsAudioUrl && ( {ttsStatus === "done" && ttsAudioUrl && (
@@ -7,6 +7,7 @@ import type { VoiceClone } from "@/api/voice-clone"
import type { TtsStatus } from "./TtsModal" import type { TtsStatus } from "./TtsModal"
import type { TtsClonedVoiceOption } from "./tts-modal/VoiceSelector" import type { TtsClonedVoiceOption } from "./tts-modal/VoiceSelector"
import type { TtsEmotion, TtsLanguage } from "./tts-modal/constants" import type { TtsEmotion, TtsLanguage } from "./tts-modal/constants"
import type { TtsStyle } from "@/api/tts/styles"
import CloneModal from "@/components/voice/CloneModal" import CloneModal from "@/components/voice/CloneModal"
import CloneDetailModal from "./CloneDetailModal" import CloneDetailModal from "./CloneDetailModal"
import UploadVoiceModal from "./UploadVoiceModal" import UploadVoiceModal from "./UploadVoiceModal"
@@ -44,6 +45,7 @@ export interface VoiceModalsProps {
ttsVoiceId: string ttsVoiceId: string
ttsSpeed: number ttsSpeed: number
ttsEmotion: TtsEmotion ttsEmotion: TtsEmotion
ttsStyle: TtsStyle
ttsLanguage: TtsLanguage ttsLanguage: TtsLanguage
ttsStatus: TtsStatus ttsStatus: TtsStatus
ttsAudioUrl: string | null ttsAudioUrl: string | null
@@ -56,6 +58,7 @@ export interface VoiceModalsProps {
onTtsVoiceChange: (id: string) => void onTtsVoiceChange: (id: string) => void
onTtsSpeedChange: (speed: number) => void onTtsSpeedChange: (speed: number) => void
onTtsEmotionChange: (emotion: TtsEmotion) => void onTtsEmotionChange: (emotion: TtsEmotion) => void
onTtsStyleChange: (style: TtsStyle) => void
onTtsLanguageChange: (language: TtsLanguage) => void onTtsLanguageChange: (language: TtsLanguage) => void
onTtsSynthesize: () => void onTtsSynthesize: () => void
onTtsSave: () => void onTtsSave: () => void
@@ -86,6 +89,7 @@ export const VoiceModals: React.FC<VoiceModalsProps> = ({
ttsVoiceId, ttsVoiceId,
ttsSpeed, ttsSpeed,
ttsEmotion, ttsEmotion,
ttsStyle,
ttsLanguage, ttsLanguage,
ttsStatus, ttsStatus,
ttsAudioUrl, ttsAudioUrl,
@@ -97,6 +101,7 @@ export const VoiceModals: React.FC<VoiceModalsProps> = ({
onTtsVoiceChange, onTtsVoiceChange,
onTtsSpeedChange, onTtsSpeedChange,
onTtsEmotionChange, onTtsEmotionChange,
onTtsStyleChange,
onTtsLanguageChange, onTtsLanguageChange,
onTtsSynthesize, onTtsSynthesize,
onTtsSave, onTtsSave,
@@ -139,6 +144,7 @@ export const VoiceModals: React.FC<VoiceModalsProps> = ({
ttsVoiceId={ttsVoiceId} ttsVoiceId={ttsVoiceId}
ttsSpeed={ttsSpeed} ttsSpeed={ttsSpeed}
ttsEmotion={ttsEmotion} ttsEmotion={ttsEmotion}
ttsStyle={ttsStyle}
ttsLanguage={ttsLanguage} ttsLanguage={ttsLanguage}
ttsStatus={ttsStatus} ttsStatus={ttsStatus}
ttsAudioUrl={ttsAudioUrl} ttsAudioUrl={ttsAudioUrl}
@@ -150,6 +156,7 @@ export const VoiceModals: React.FC<VoiceModalsProps> = ({
onVoiceChange={onTtsVoiceChange} onVoiceChange={onTtsVoiceChange}
onSpeedChange={onTtsSpeedChange} onSpeedChange={onTtsSpeedChange}
onEmotionChange={onTtsEmotionChange} onEmotionChange={onTtsEmotionChange}
onStyleChange={onTtsStyleChange}
onLanguageChange={onTtsLanguageChange} onLanguageChange={onTtsLanguageChange}
onSynthesize={onTtsSynthesize} onSynthesize={onTtsSynthesize}
onSave={onTtsSave} onSave={onTtsSave}
@@ -1,6 +1,7 @@
import { type PresetVoiceDisplay } from "@/pages/voices/types" import { type PresetVoiceDisplay } from "@/pages/voices/types"
import type { TtsClonedVoiceOption } from "./VoiceSelector" import type { TtsClonedVoiceOption } from "./VoiceSelector"
import type { TtsEmotion, TtsLanguage } from "./constants" import type { TtsEmotion, TtsLanguage } from "./constants"
import type { TtsStyle } from "@/api/tts/styles"
export type TtsStatus = "idle" | "synthesizing" | "done" | "error" export type TtsStatus = "idle" | "synthesizing" | "done" | "error"
@@ -10,6 +11,7 @@ export interface TtsModalProps {
ttsVoiceId: string ttsVoiceId: string
ttsSpeed: number ttsSpeed: number
ttsEmotion: TtsEmotion ttsEmotion: TtsEmotion
ttsStyle: TtsStyle
ttsLanguage: TtsLanguage ttsLanguage: TtsLanguage
ttsStatus: TtsStatus ttsStatus: TtsStatus
ttsAudioUrl: string | null ttsAudioUrl: string | null
@@ -22,6 +24,7 @@ export interface TtsModalProps {
onVoiceChange: (voiceId: string) => void onVoiceChange: (voiceId: string) => void
onSpeedChange: (speed: number) => void onSpeedChange: (speed: number) => void
onEmotionChange: (emotion: TtsEmotion) => void onEmotionChange: (emotion: TtsEmotion) => void
onStyleChange: (style: TtsStyle) => void
onLanguageChange: (language: TtsLanguage) => void onLanguageChange: (language: TtsLanguage) => void
onSynthesize: () => void onSynthesize: () => void
onSave: () => void onSave: () => void
@@ -11,6 +11,7 @@ import {
type TtsEmotion, type TtsEmotion,
type TtsLanguage, type TtsLanguage,
} from "../components/tts-modal/constants" } from "../components/tts-modal/constants"
import { DEFAULT_TTS_STYLE, type TtsStyle } from "@/api/tts/styles"
export type TtsStatus = "idle" | "synthesizing" | "done" | "error" export type TtsStatus = "idle" | "synthesizing" | "done" | "error"
@@ -42,6 +43,7 @@ export function useTtsSynthesize({
const [ttsVoiceId, setTtsVoiceId] = useState<string>("") const [ttsVoiceId, setTtsVoiceId] = useState<string>("")
const [ttsSpeed, setTtsSpeed] = useState(1.0) const [ttsSpeed, setTtsSpeed] = useState(1.0)
const [ttsEmotion, setTtsEmotion] = useState<TtsEmotion>(DEFAULT_TTS_EMOTION) const [ttsEmotion, setTtsEmotion] = useState<TtsEmotion>(DEFAULT_TTS_EMOTION)
const [ttsStyle, setTtsStyle] = useState<TtsStyle>(DEFAULT_TTS_STYLE)
const [ttsLanguage, setTtsLanguage] = useState<TtsLanguage>(DEFAULT_TTS_LANGUAGE) const [ttsLanguage, setTtsLanguage] = useState<TtsLanguage>(DEFAULT_TTS_LANGUAGE)
const [ttsJobId, setTtsJobId] = useState<string | null>(null) const [ttsJobId, setTtsJobId] = useState<string | null>(null)
const [ttsStatus, setTtsStatus] = useState<TtsStatus>("idle") const [ttsStatus, setTtsStatus] = useState<TtsStatus>("idle")
@@ -83,6 +85,7 @@ export function useTtsSynthesize({
voice_id: ttsVoiceId || undefined, voice_id: ttsVoiceId || undefined,
speed: ttsSpeed, speed: ttsSpeed,
emotion: ttsEmotion, emotion: ttsEmotion,
style: ttsStyle,
language: effectiveLang, language: effectiveLang,
}) })
setTtsJobId(resp.job_id) setTtsJobId(resp.job_id)
@@ -114,7 +117,7 @@ export function useTtsSynthesize({
setTtsStatus("error") setTtsStatus("error")
setTtsError(msg) setTtsError(msg)
} }
}, [ttsText, ttsVoiceId, ttsSpeed, ttsEmotion, ttsLanguage, clonedVoices]) }, [ttsText, ttsVoiceId, ttsSpeed, ttsEmotion, ttsStyle, ttsLanguage, clonedVoices])
/** 保存 TTS 结果到素材库 */ /** 保存 TTS 结果到素材库 */
const handleTtsSave = useCallback(async () => { const handleTtsSave = useCallback(async () => {
@@ -137,6 +140,7 @@ export function useTtsSynthesize({
setTtsVoiceId("") setTtsVoiceId("")
setTtsSpeed(1.0) setTtsSpeed(1.0)
setTtsEmotion(DEFAULT_TTS_EMOTION) setTtsEmotion(DEFAULT_TTS_EMOTION)
setTtsStyle(DEFAULT_TTS_STYLE)
setTtsLanguage(DEFAULT_TTS_LANGUAGE) setTtsLanguage(DEFAULT_TTS_LANGUAGE)
setTtsStatus("idle") setTtsStatus("idle")
setTtsAudioUrl(null) setTtsAudioUrl(null)
@@ -168,6 +172,7 @@ export function useTtsSynthesize({
ttsVoiceId, ttsVoiceId,
ttsSpeed, ttsSpeed,
ttsEmotion, ttsEmotion,
ttsStyle,
ttsLanguage, ttsLanguage,
ttsJobId, ttsJobId,
ttsStatus, ttsStatus,
@@ -181,6 +186,7 @@ export function useTtsSynthesize({
setTtsVoiceId, setTtsVoiceId,
setTtsSpeed, setTtsSpeed,
setTtsEmotion, setTtsEmotion,
setTtsStyle,
setTtsLanguage, setTtsLanguage,
setTtsOpen, setTtsOpen,
// 覆写 onVoiceChange(带语言回退) // 覆写 onVoiceChange(带语言回退)
+3
View File
@@ -48,6 +48,9 @@ celery_app.conf.imports = (
# PYTHONPATH=/app/apps/api 下,app.tasks.lipsync_tts 可直接导入且不触发 apps/api/__init__.py # PYTHONPATH=/app/apps/api 下,app.tasks.lipsync_tts 可直接导入且不触发 apps/api/__init__.py
# apps/api/__init__.py 会 from .main import app,级联加载整个 FastAPI 栈,Worker 中不需要且会导致注册失败) # apps/api/__init__.py 会 from .main import app,级联加载整个 FastAPI 栈,Worker 中不需要且会导致注册失败)
"app.tasks.lipsync_tts", "app.tasks.lipsync_tts",
# #1998 GPU MuseTalk 异步推理:wait_for_result→签名 URL→回写 lipsync_jobs
# 必须在 Worker 侧注册,否则 apply_async 消息无人消费,job 永远卡在 processing
"app.tasks.lipsync_gpu",
) )
# Celery Beat 定时任务调度 # Celery Beat 定时任务调度
+7 -2
View File
@@ -115,8 +115,13 @@ def _register(task_id: Optional[str] = None) -> bool:
last_heartbeat_at,防止长推理被误判超时回收。 last_heartbeat_at,防止长推理被误判超时回收。
""" """
ok, info = _check_musetalk_health() ok, info = _check_musetalk_health()
free_vram = int(info.get("free_vram_mb", 0) or 0) if isinstance(info, dict) else 0 if isinstance(info, dict):
gpu_name = info.get("gpu_name", "") if isinstance(info, dict) else "" gpu_info = info.get("gpu", info)
free_vram = int(gpu_info.get("free_vram_mb", gpu_info.get("memory_free_mb", 0)) or 0)
gpu_name = gpu_info.get("gpu_name", info.get("gpu_name", ""))
else:
free_vram = 0
gpu_name = ""
if not gpu_name: if not gpu_name:
# 尝试在 Windows 上读 nvidia-smi # 尝试在 Windows 上读 nvidia-smi
gpu_name = _probe_gpu_name() gpu_name = _probe_gpu_name()
+565 -113
View File
@@ -1,13 +1,18 @@
"""MuseTalk Flask HTTP 服务 — 反向轮询架构的服务端部分. """MuseTalk Flask HTTP 服务 — 反向轮询架构的服务端部分.
部署在 RTX2060 本地,接收 gpu_worker.py 的推理请求,调用 MuseTalk 生成口型同步视频。 部署在 RTX2060 本地,接收 gpu_worker.py 的推理请求,调用 MuseTalk 生成口型同步视频。
本文件修复了原 worker.py 的 8 个工程 bug,并新增 /cancel 端点。
#2000 关键修复:
- 集成真实 MuseTalk 推理(替换原有 stub 代码)
- 音频预处理:22050Hz MP3 → 16kHz mono 16bit WAVMuseTalk 要求)
- 模型懒加载:首次推理时加载,后续复用,避免重复加载
- 视频帧循环使用 mirror indexing(乒乓模式),消除循环边界跳变
- bbox_shift 可通过请求参数配置
#1978 性能修复(v2 架构): #1978 性能修复(v2 架构):
MuseTalk 原生支持长音频输入(内部循环视频帧),不需要我们先 loop 视频。 MuseTalk 原生支持长音频输入(内部循环视频帧),不需要我们先 loop 视频。
正确流程:原视频 + 全量音频 → MuseTalk 推理 → 输出时长=音频时长的无声画面 正确流程:原视频 + 全量音频 → MuseTalk 推理 → 输出时长=音频时长的无声画面
→ ffmpeg 快速 -c:v copy 替换音轨。推理时间不变(~14s),后处理几秒。 → ffmpeg 快速 -c:v copy 替换音轨。推理时间不变(~14s),后处理几秒。
禁止在推理前用 ffmpeg 循环视频(会导致 MuseTalk 处理 2x+ 帧数,慢 16 倍)。
环境变量: 环境变量:
MUSE_PORT 监听端口,默认 7861 MUSE_PORT 监听端口,默认 7861
@@ -18,21 +23,27 @@
MUSE_DEFAULT_FPS 视频 fps 兜底值,默认 25.0 MUSE_DEFAULT_FPS 视频 fps 兜底值,默认 25.0
MUSE_TEMP_DIR 临时文件目录,默认 /tmp/musetalk_$$ MUSE_TEMP_DIR 临时文件目录,默认 /tmp/musetalk_$$
MUSE_VIDEO_ENCODER 循环视频时的编码器(仅兜底):auto(默认)/h264_nvenc/libx264 MUSE_VIDEO_ENCODER 循环视频时的编码器(仅兜底):auto(默认)/h264_nvenc/libx264
MUSE_DIR MuseTalk 仓库路径,默认 /home/ying/projects/MuseTalk
MUSE_MODEL_DIR 模型目录(相对 MUSE_DIR),默认 models/musetalk
MUSE_USE_FLOAT16 使用 FP16 推理,默认 1(开启)
MUSE_BATCH_SIZE 推理批次大小,默认 8
接口: 接口:
GET /health 健康检查 + GPU 显存信息 GET /health 健康检查 + GPU 显存信息
POST /inference 推理请求(multipart: video + audio POST /inference 推理请求(multipart: video + audio, form: bbox_shift
POST /cancel 终止当前推理任务 POST /cancel 终止当前推理任务
""" """
from __future__ import annotations from __future__ import annotations
import atexit import atexit
import copy
import logging import logging
import os import os
import shutil import shutil
import signal import signal
import subprocess import subprocess
import sys
import threading import threading
import time import time
from pathlib import Path from pathlib import Path
@@ -68,6 +79,16 @@ class Config:
video_encoder: str = _env("MUSE_VIDEO_ENCODER", "auto") or "auto" video_encoder: str = _env("MUSE_VIDEO_ENCODER", "auto") or "auto"
# 判定音视频时长差异的容差(秒) # 判定音视频时长差异的容差(秒)
duration_epsilon: float = 0.25 duration_epsilon: float = 0.25
# MuseTalk 仓库路径
muse_dir: str = _env("MUSE_DIR", "/home/ying/projects/MuseTalk")
# 模型目录(相对 MUSE_DIR
muse_model_dir: str = _env("MUSE_MODEL_DIR", "models/musetalk")
# 是否使用 FP16(节省显存,RTX2060 建议开启)
use_float16: bool = _env("MUSE_USE_FLOAT16", "1") == "1"
# 推理批次大小(RTX2060 6G 显存建议 4-8
batch_size: int = int(_env("MUSE_BATCH_SIZE", "8"))
# GFPGAN 人脸超分增强(提升生成人脸清晰度,+~170MB VRAM, +60ms/帧)
use_gfpgan: bool = _env("MUSE_USE_GFPGAN", "1") == "1"
# ── 全局状态 ────────────────────────────────────────────────────────── # ── 全局状态 ──────────────────────────────────────────────────────────
@@ -75,6 +96,13 @@ inference_lock = threading.Lock()
current_task: dict = {"task_id": None, "process": None, "start_time": 0.0} current_task: dict = {"task_id": None, "process": None, "start_time": 0.0}
shutdown_event = threading.Event() shutdown_event = threading.Event()
# ── MuseTalk 模型懒加载 ─────────────────────────────────────────────
_muse_models = None
_muse_models_lock = threading.Lock()
_muse_models_loaded = False
_muse_load_error = None
# ── Flask App ───────────────────────────────────────────────────────── # ── Flask App ─────────────────────────────────────────────────────────
app = Flask(__name__) app = Flask(__name__)
@@ -215,6 +243,29 @@ def _pick_video_encoder() -> str:
return "libx264" return "libx264"
def _preprocess_audio(input_path: Path, output_path: Path, target_sr: int = 16000) -> None:
"""将输入音频转换为 MuseTalk 要求的格式:16kHz mono 16bit WAV.
MuseTalk 的 whisper audio2feature 要求 16kHz 采样率的单声道音频。
当前 TTS 输出为 22050Hz MP3,不转换会导致 mel 频谱错位、
音素特征提取错误,口型只跟能量不跟音素。
"""
cmd = [
"ffmpeg", "-y", "-v", "warning",
"-i", str(input_path),
"-ar", str(target_sr), # 重采样到 16kHz
"-ac", "1", # 单声道
"-sample_fmt", "s16", # 16bit PCM
str(output_path),
]
_run_ffmpeg(cmd, timeout=60)
if not output_path.exists() or output_path.stat().st_size < 100:
raise RuntimeError(f"音频预处理失败: {output_path}")
logger.info("音频预处理完成: %s → 16kHz mono WAV", input_path.name)
def _mux_video_with_audio( def _mux_video_with_audio(
video_path: Path, video_path: Path,
audio_path: Path, audio_path: Path,
@@ -340,129 +391,518 @@ def _run_ffmpeg(cmd: list, timeout: float = 120) -> subprocess.CompletedProcess:
raise RuntimeError(f"ffmpeg 超时(>{timeout}s") from exc raise RuntimeError(f"ffmpeg 超时(>{timeout}s") from exc
# ── MuseTalk 模型加载 ────────────────────────────────────────────────
def _load_musetalk_models():
"""懒加载 MuseTalk 模型(全局单例,首次调用时加载).
加载 VAE、UNet、PositionalEncoder 三个核心组件。
加载到 GPU 后转为 FP16(如果配置开启)以节省显存。
RTX2060 6G 显存,FP16 大约需要 3-4GB。
"""
global _muse_models, _muse_models_loaded, _muse_load_error
if _muse_models_loaded:
return _muse_models
if _muse_load_error is not None:
raise _muse_load_error
with _muse_models_lock:
if _muse_models_loaded:
return _muse_models
try:
muse_dir = Path(Config.muse_dir)
if not muse_dir.exists():
raise FileNotFoundError(
f"MuseTalk 目录不存在: {muse_dir}\n"
f"请设置 MUSE_DIR 环境变量指向 MuseTalk 仓库路径"
)
# 将 MuseTalk 加入 sys.path(只在首次加载时)
muse_str = str(muse_dir)
if muse_str not in sys.path:
sys.path.insert(0, muse_str)
import torch
from musetalk.utils.utils import load_all_model
device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu")
logger.info("MuseTalk 使用设备: %s", device)
# 自动检测模型路径
# 优先检测 v1.5 模型,然后回退到 v1
v15_unet = muse_dir / "models" / "musetalkV15" / "unet.pth"
v1_unet = muse_dir / "models" / "musetalk" / "pytorch_model.bin"
if v15_unet.exists():
unet_model_path = str(v15_unet)
unet_config = str(muse_dir / "models" / "musetalkV15" / "musetalk.json")
model_version = "v15"
elif v1_unet.exists():
unet_model_path = str(v1_unet)
unet_config = str(muse_dir / "models" / "musetalk" / "config.json")
model_version = "v1"
else:
raise FileNotFoundError(
f"未找到 MuseTalk 模型权重。\n"
f"检查路径: {v15_unet}{v1_unet}\n"
f"请确认模型已下载到 MuseTalk 仓库的 models/ 目录下"
)
logger.info("加载 MuseTalk %s 模型: %s", model_version, unet_model_path)
vae, unet, pe = load_all_model(
unet_model_path=unet_model_path,
vae_type="sd-vae",
unet_config=unet_config,
device=device,
)
timesteps = torch.tensor([0], device=device)
# FP16 转换(节省 ~50% 显存)
if Config.use_float16:
pe = pe.half()
vae.vae = vae.vae.half()
unet.model = unet.model.half()
logger.info("已启用 FP16 推理")
pe = pe.to(device)
vae.vae = vae.vae.to(device)
unet.model = unet.model.to(device)
# 加载 AudioProcessor 和 face parsing
from musetalk.utils.audio_processor import AudioProcessor
from musetalk.utils.face_parsing import FaceParsing
audio_processor = AudioProcessor()
face_parsing = FaceParsing()
# 加载 GFPGAN 人脸超分模型(FP16,仅 ~170MB VRAM
gfpgan_model = None
if Config.use_gfpgan:
try:
from gfpgan.archs.gfpganv1_clean_arch import GFPGANv1Clean
gfpgan_path = muse_dir / "models" / "GFPGAN" / "GFPGANv1.4.pth"
if gfpgan_path.exists():
logger.info("加载 GFPGANv1.4 人脸超分模型: %s", gfpgan_path)
gfpgan_ckpt = torch.load(str(gfpgan_path), map_location="cpu")
gfpgan_model = GFPGANv1Clean(
out_size=512, num_style_feat=512, channel_multiplier=2,
decoder_load_path=None, fix_decoder=False, num_mlp=8,
input_is_latent=True, different_w=True, narrow=1, sft_half=True,
)
gfpgan_key = "params_ema" if "params_ema" in gfpgan_ckpt else "params"
gfpgan_model.load_state_dict(gfpgan_ckpt[gfpgan_key], strict=True)
gfpgan_model.eval()
if Config.use_float16:
gfpgan_model = gfpgan_model.half()
gfpgan_model = gfpgan_model.to(device)
del gfpgan_ckpt
logger.info("GFPGAN 加载完成 (FP16=%s)", Config.use_float16)
else:
logger.warning("GFPGAN 模型不存在: %s,跳过人脸增强", gfpgan_path)
except Exception as e:
logger.warning("GFPGAN 加载失败,跳过人脸增强: %s", e)
gfpgan_model = None
else:
logger.info("GFPGAN 已禁用 (MUSE_USE_GFPGAN=0)")
_muse_models = {
"vae": vae,
"unet": unet,
"pe": pe,
"timesteps": timesteps,
"audio_processor": audio_processor,
"face_parsing": face_parsing,
"gfpgan": gfpgan_model,
"device": device,
"model_version": model_version,
}
_muse_models_loaded = True
logger.info("MuseTalk 模型加载完成 (版本=%s, 设备=%s, fp16=%s)",
model_version, device, Config.use_float16)
# 打印显存使用情况
if torch.cuda.is_available():
allocated = torch.cuda.memory_allocated() / 1024**2
reserved = torch.cuda.memory_reserved() / 1024**2
logger.info("GPU 显存: 已分配 %.0fMB, 已预留 %.0fMB", allocated, reserved)
return _muse_models
except Exception as exc:
_muse_load_error = exc
logger.error("MuseTalk 模型加载失败: %s", exc)
raise
# ── MuseTalk 推理核心 ─────────────────────────────────────────────────
def _mirror_index(size: int, index: int) -> int:
"""乒乓式循环索引,避免循环边界硬切跳变.
效果: 0→1→2→...→N→N-1→...→1→0→1→...
比简单的 index % size 在边界处更平滑。
"""
if size == 0:
return 0
turn = index // size
res = index % size
if turn % 2 == 0:
return res
else:
return size - res - 1
def _run_inference( def _run_inference(
video_path: Path, video_path: Path,
audio_path: Path, audio_path: Path,
output_path: Path, output_path: Path,
bbox_shift: int = 0,
) -> None: ) -> None:
"""执行 MuseTalk 推理(v2 架构:全量音频直传,不在推理前 loop 视频). """执行 MuseTalk 真实推理.
#1978 性能修复核心 流程
MuseTalk 原生支持长音频输入,内部会自动循环视频帧。 1. 音频预处理:任意格式 → 16kHz mono 16bit WAV
我们只需把【原视频】和【全量音频】传给 MuseTalk, 2. 加载/复用 MuseTalk 模型(VAE + UNet + PE + Whisper
输出视频时长 = 音频时长(MuseTalk 自行处理帧循环)。 3. 视频预处理:提取帧 → 人脸检测 → 获取 bbox → VAE 编码 latent
禁止在推理前用 ffmpeg 循环视频(会导致慢 16 倍)。 4. 音频特征提取:whisper 提取 audio features (50×384 per chunk)
5. 批量推理:UNet 去噪 → VAE 解码 → 得到口型同步的人脸帧
6. 帧合成:将生成的人脸贴回原帧(使用 face parsing 做边缘融合)
7. 输出无声视频(后续由 _mux_video_with_audio 封装 TTS 音频)
实际部署时替换为 MuseTalk 真实推理逻辑。 Args:
此处为示例实现:提取帧 → 模拟 MuseTalk 产出音频时长的无声画面 → 快速封装。 video_path: 输入视频路径
audio_path: 输入音频路径(任意格式,会被预处理为 16kHz WAV)
output_path: 输出无声视频路径
bbox_shift: 口型区域垂直偏移量,默认 0,范围 [-5, 5]
""" """
import cv2
import numpy as np
import torch
from tqdm import tqdm
from musetalk.utils.preprocessing import get_landmark_and_bbox as _orig_get_landmark_and_bbox
from musetalk.utils.blending import get_image
import tempfile as _tempfile, math as _math, shutil as _shutil
from einops import rearrange as _rearrange
# read_imgs: 读取视频帧(支持视频文件路径),返回 numpy BGR 帧列表
def read_imgs(path):
import cv2 as _cv2
cap = _cv2.VideoCapture(str(path))
frames = []
while True:
ret, frame = cap.read()
if not ret:
break
frames.append(frame)
cap.release()
return frames
# get_landmark_and_bbox 适配:旧版签名(img_list, upperbondrange=0),且 img_list 是文件路径列表
def get_landmark_and_bbox(frames, vid_pts=0, bbox_shift=0):
import cv2 as _cv2
_tmpdir = _tempfile.mkdtemp(prefix="muse_frames_")
frame_paths = []
for _i, _frm in enumerate(frames):
_fp = f"{_tmpdir}/{_i:08d}.png"
_cv2.imwrite(_fp, _frm)
frame_paths.append(_fp)
coords_list, _ = _orig_get_landmark_and_bbox(frame_paths, upperbondrange=bbox_shift)
_shutil.rmtree(_tmpdir, ignore_errors=True)
_sentinel = object()
coords_list = [c if c is not None else _sentinel for c in coords_list]
return coords_list, _sentinel
# 加载模型(首次调用时加载,后续复用)
models = _load_musetalk_models()
vae = models["vae"]
unet = models["unet"]
pe = models["pe"]
timesteps = models["timesteps"]
audio_processor = models["audio_processor"]
device = models["device"]
model_version = models["model_version"]
# 给旧版 AudioProcessor 动态添加 feature2chunks 方法
import types as _types
def _feature2chunks(self, feature_array, fps=25, weight_dtype=None,
batch_size=8, audio_padding_length_left=2,
audio_padding_length_right=2):
import torch
sr = 16000
audio_fps = 50
chunk_len = 2 * (audio_padding_length_left + audio_padding_length_right + 1)
whisper_idx_multiplier = audio_fps / fps
num_frames = int(_math.floor((len(feature_array) / sr) * fps))
actual_length = int(_math.floor((len(feature_array) / sr) * audio_fps))
inputs = self.feature_extractor(
feature_array, return_tensors="pt", sampling_rate=sr
).input_features.to(device)
if weight_dtype is not None:
inputs = inputs.to(dtype=weight_dtype)
global _whisper_enc_model
if "_whisper_enc_model" not in globals() or _whisper_enc_model is None:
from transformers import WhisperModel
_wp = str(Path(Config.muse_dir) / "models" / "whisper")
_whisper_enc_model = WhisperModel.from_pretrained(_wp).to(device)
_whisper_enc_model.eval()
if Config.use_float16:
_whisper_enc_model = _whisper_enc_model.half()
with torch.no_grad():
_af = _whisper_enc_model.encoder(inputs, output_hidden_states=True).hidden_states
_af = torch.stack(_af, dim=2)
_af = _af[0, :actual_length, ...]
_pn = int(_math.ceil(whisper_idx_multiplier))
_af = torch.cat([
torch.zeros_like(_af[:_pn * audio_padding_length_left]),
_af,
torch.zeros_like(_af[:_pn * 3 * audio_padding_length_right]),
], dim=0)
_all = []
for _fi in range(num_frames):
_ai = int(_math.floor(_fi * whisper_idx_multiplier))
_clip = _af[_ai:_ai + chunk_len]
if _clip.shape[0] < chunk_len:
_pad = torch.zeros(chunk_len - _clip.shape[0], *_clip.shape[1:],
device=device, dtype=_clip.dtype)
_clip = torch.cat([_clip, _pad], dim=0)
_all.append(_clip)
_prompts = torch.stack(_all, dim=0)
_prompts = _rearrange(_prompts, "b c h w -> b (c h) w")
return _prompts
audio_processor.feature2chunks = _types.MethodType(_feature2chunks, audio_processor)
fps = _get_video_fps(video_path) fps = _get_video_fps(video_path)
audio_duration = _get_media_duration(audio_path) audio_duration = _get_media_duration(audio_path)
video_duration = _get_media_duration(video_path) video_duration = _get_media_duration(video_path)
logger.info( logger.info(
"推理开始: video=%.2fs, audio=%.2fs, fps=%.2f", "MuseTalk 推理开始: video=%.2fs, audio=%.2fs, fps=%.1f, bbox_shift=%d",
video_duration, video_duration, audio_duration, fps, bbox_shift,
audio_duration,
fps,
) )
frames_dir = video_path.parent / "frames" # ── Step 1: 音频预处理(关键修复:22050Hz MP3 → 16kHz mono WAV)──
frames_dir.mkdir(parents=True, exist_ok=True) audio_wav_path = video_path.parent / "audio_16k_mono.wav"
_preprocess_audio(audio_path, audio_wav_path, target_sr=16000)
# 1. 从原视频提取帧(仅原视频长度,不循环) # ── Step 2: 视频提取 ──
_run_ffmpeg( input_frames = read_imgs(str(video_path))
[ total_frames = len(input_frames)
"ffmpeg", if total_frames == 0:
"-y", raise RuntimeError("未能从视频中提取到任何帧")
"-i", logger.info("提取到 %d 帧视频画面", total_frames)
str(video_path),
"-r", # ── Step 3: 人脸检测 & bbox 计算 ──
str(fps), coord_list, coord_placeholder = get_landmark_and_bbox(
str(frames_dir / "frame_%05d.png"), input_frames, vid_pts=0, bbox_shift=bbox_shift
],
timeout=120,
) )
logger.info("人脸检测完成,有效 bbox: %d/%d", sum(1 for c in coord_list if c is not coord_placeholder), total_frames)
frame_files = sorted(frames_dir.glob("*.png")) # 使用 mirror indexing 循环帧和坐标(避免硬切跳变)
if not frame_files: num_output_frames = int(audio_duration * fps)
raise RuntimeError("未从视频中提取到帧") if num_output_frames <= 0:
num_output_frames = total_frames
# 2. 模拟 MuseTalk 推理:输入原视频帧 + 全量音频,输出音频时长的无声画面。 # ── Step 4: 音频特征提取 ──
# TODO: 替换为 MuseTalk 真实推理逻辑。 # 使用 librosa 加载预处理后的 16kHz 音频
# MuseTalk 真实调用示例(伪代码): import librosa
# from musetalk import MuseTalkModel audio_array, _ = librosa.load(str(audio_wav_path), sr=16000, mono=True)
# model = MuseTalkModel(...) whisper_features = audio_processor.feature2chunks(
# silent_video = model.infer(video_path=video_path, audio_path=audio_path) feature_array=audio_array,
# # MuseTalk 内部会循环视频帧匹配音频长度,输出时长=音频时长 fps=fps,
logger.warning("使用示例推理逻辑,未实际调用 MuseTalk 模型") weight_dtype=(torch.float16 if Config.use_float16 else torch.float32),
batch_size=Config.batch_size,
# 示例:生成音频时长的无声画面(循环原视频帧到音频长度)
# 真实部署时 silent_video_path 应替换为 MuseTalk 输出的无声视频路径
silent_video_path = video_path.parent / "visual_silent.mp4"
if audio_duration > video_duration + Config.duration_epsilon:
# 音频更长:循环视频帧到音频长度(仅用于示例,真实 MuseTalk 内部处理)
encoder = _pick_video_encoder()
preset = "p4" if encoder == "h264_nvenc" else "veryfast"
logger.info(
"示例:循环视频帧到音频长度 %.2fs(真实 MuseTalk 内部处理,无需此步骤)",
audio_duration,
)
cmd = [
"ffmpeg",
"-y",
"-stream_loop",
"-1",
"-i",
str(video_path),
"-an",
"-c:v",
encoder,
"-preset",
preset,
"-t",
f"{audio_duration:.3f}",
str(silent_video_path),
]
try:
_run_ffmpeg(cmd, timeout=300)
except RuntimeError:
if encoder == "h264_nvenc":
cmd[cmd.index(encoder)] = "libx264"
cmd[cmd.index(preset) + 1] = "veryfast"
_run_ffmpeg(cmd, timeout=300)
else:
raise
else:
# 音频不长:直接生成无声视频(原视频长度)
_run_ffmpeg(
[
"ffmpeg",
"-y",
"-i",
str(video_path),
"-an",
"-c:v",
"libx264",
"-preset",
"veryfast",
str(silent_video_path),
],
timeout=300,
)
# 3. 快速封装:-map 取推理画面 + 驱动音频,-c:v copy 无损秒级封装
# MuseTalk 输出已匹配音频长度,此处无需循环,仅替换音轨
_mux_video_with_audio(silent_video_path, audio_path, output_path)
if not output_path.exists() or output_path.stat().st_size < 1024:
raise RuntimeError("推理产物不存在或过小")
logger.info(
"推理完成: output=%.2fs (audio=%.2fs)",
_get_media_duration(output_path),
audio_duration,
) )
if isinstance(whisper_features, torch.Tensor):
whisper_features = whisper_features.detach().cpu()
torch.cuda.empty_cache()
logger.info("音频特征提取完成: %d 个 chunk", len(whisper_features))
# ── Step 5: 逐帧裁剪人脸并编码为 8ch latent (masked+ref) ──
face_parsing = models.get("face_parsing", None)
input_latent_list = []
valid_frame_indices = [] # 记录成功编码的帧索引(跳过无脸帧)
with torch.no_grad():
for idx, (frame, bbox) in enumerate(zip(input_frames, coord_list)):
if bbox is coord_placeholder:
continue
x1, y1, x2, y2 = bbox
# v1.5 额外扩展下边界(下巴区域),与 Step 7 保持一致
extra_y2 = 10 if model_version == "v15" else 0
y2_eff = min(y2 + extra_y2, frame.shape[0])
if y2_eff <= y1 or x2 <= x1:
continue
# 裁剪人脸区域 → resize 256×256
crop = frame[y1:y2_eff, x1:x2]
if crop.size == 0:
continue
crop_rgb = cv2.cvtColor(crop, cv2.COLOR_BGR2RGB)
crop_resized = cv2.resize(crop_rgb, (256, 256), interpolation=cv2.INTER_LANCZOS4)
# 使用 VAE 的 get_latents_for_unet 得到 8 通道输入
# get_latents_for_unet 内部: preprocess(half_mask=True) encode + preprocess(half_mask=False) encode → cat → [1,8,32,32]
latents = vae.get_latents_for_unet(crop_resized).detach().cpu()
input_latent_list.append(latents)
# 保存此帧的实际bbox(含extra_y2)和原帧索引供 Step 7 使用
valid_frame_indices.append((idx, x1, y1, x2, y2_eff))
# 构建循环列表:正序+倒序,实现旧版的平滑首尾帧循环
frame_list_cycle = input_frames + input_frames[::-1]
coord_cycle = []
for _i, _x1, _y1, _x2, _y2 in valid_frame_indices:
coord_cycle.append((_x1, _y1, _x2, _y2))
coord_cycle = coord_cycle + coord_cycle[::-1]
latent_cycle = input_latent_list + input_latent_list[::-1]
valid_cycle = valid_frame_indices + [(i, x1, y1, x2, y2) for (i, x1, y1, x2, y2) in reversed(valid_frame_indices)]
torch.cuda.empty_cache()
logger.info("人脸裁剪+VAE 编码完成: %d 个有效latent", len(input_latent_list))
# ── Step 6: 批量推理(仿旧版 datagen 循环)──
res_frame_list = []
video_num = len(whisper_features)
bs = min(Config.batch_size, 2) # RTX2060 6G 限制batch=2防OOM
total_batches = (video_num + bs - 1) // bs
for bi in tqdm(range(total_batches), desc="MuseTalk 推理"):
whisper_batch = whisper_features[bi*bs:(bi+1)*bs]
if len(whisper_batch) == 0:
break
# 对应 latent 索引(循环取 latent_cycle
latent_batch_parts = []
for j in range(len(whisper_batch)):
global_idx = bi*bs + j
lat_idx = global_idx % len(latent_cycle)
latent_batch_parts.append(latent_cycle[lat_idx])
# whisper_batch 是 [bs,50,384] tensor slice (feature2chunks 已返回 stacked tensor)
if isinstance(whisper_batch, list):
whisper_batch_t = torch.stack(whisper_batch).to(device)
else:
whisper_batch_t = whisper_batch.to(device)
latent_batch_t = torch.cat(latent_batch_parts, dim=0).to(device)
if Config.use_float16:
latent_batch_t = latent_batch_t.to(dtype=unet.model.dtype)
whisper_batch_t = whisper_batch_t.to(dtype=unet.model.dtype)
audio_feature_batch = pe(whisper_batch_t)
with torch.no_grad():
pred_latents = unet.model(
latent_batch_t,
timesteps,
encoder_hidden_states=audio_feature_batch,
).sample
recon_frames = vae.decode_latents(pred_latents)
for rf in recon_frames:
res_frame_list.append(rf)
del pred_latents, recon_frames, latent_batch_t, whisper_batch_t
if "audio_feature_batch" in dir():
try: del audio_feature_batch
except: pass
torch.cuda.empty_cache()
logger.info("推理完成,生成 %d", len(res_frame_list))
# ── Step 7: 合成最终帧 → ffmpeg pipe 编码(零磁盘IO ──
gfpgan_enhancer = models.get("gfpgan")
silent_video_path = video_path.parent / "silent_output.mp4"
frame_h, frame_w = frame_list_cycle[0].shape[:2]
# 启动 ffmpegstdin 接收 raw BGR24 帧,直接编码 H.264(省去PNG落盘+回读)
_ff_cmd = [
"ffmpeg", "-y", "-v", "warning",
"-f", "rawvideo", "-pix_fmt", "bgr24",
"-s", f"{frame_w}x{frame_h}", "-r", str(fps),
"-i", "-",
"-vcodec", "libx264", "-preset", "veryfast",
"-vf", "format=yuv420p", "-crf", "18",
str(silent_video_path),
]
import subprocess as _sp
_ff_proc = _sp.Popen(_ff_cmd, stdin=_sp.PIPE, stdout=_sp.DEVNULL, stderr=_sp.PIPE)
n_out = min(len(res_frame_list), num_output_frames)
try:
for i in tqdm(range(n_out), desc="合成帧"):
cyc_i = i % len(coord_cycle)
x1, y1, x2, y2 = coord_cycle[cyc_i]
ori_frame = copy.deepcopy(frame_list_cycle[cyc_i])
res_frame = res_frame_list[i]
try:
res_frame_resized = cv2.resize(
res_frame.astype(np.uint8), (x2-x1, y2-y1),
interpolation=cv2.INTER_LANCZOS4
)
except Exception:
_ff_proc.stdin.write(ori_frame.tobytes())
continue
# GFPGAN 人脸超分增强
if gfpgan_enhancer is not None:
try:
_fh, _fw = res_frame_resized.shape[:2]
_face_up = cv2.resize(res_frame_resized, (512, 512),
interpolation=cv2.INTER_LANCZOS4)
_face_rgb = cv2.cvtColor(_face_up, cv2.COLOR_BGR2RGB).astype(np.float32) / 255.0
_face_t = torch.from_numpy(_face_rgb.transpose(2,0,1)).unsqueeze(0)
_face_t = ((_face_t - 0.5) / 0.5).to(device)
if Config.use_float16:
_face_t = _face_t.half()
with torch.no_grad():
_out = gfpgan_enhancer(_face_t, return_rgb=False, weight=0.5)[0]
_out = _out.squeeze(0).float().cpu().clamp_(-1,1)
_out = ((_out + 1)/2*255).numpy().transpose(1,2,0)
_out_bgr = cv2.cvtColor(_out.astype(np.uint8), cv2.COLOR_RGB2BGR)
res_frame_resized = cv2.resize(_out_bgr, (_fw, _fh),
interpolation=cv2.INTER_LANCZOS4)
del _face_t, _out, _out_bgr
except Exception:
pass
# face parsing 融合
try:
if face_parsing is not None:
combined = get_image(ori_frame, res_frame_resized,
[x1,y1,x2,y2], fp=face_parsing)
else:
combined = get_image(ori_frame, res_frame_resized, [x1,y1,x2,y2])
except Exception:
combined = ori_frame.copy()
try: combined[y1:y2, x1:x2] = res_frame_resized
except Exception: combined = ori_frame
_ff_proc.stdin.write(combined.tobytes())
_ff_proc.stdin.close()
_ff_ret = _ff_proc.wait(timeout=120)
if _ff_ret != 0:
_ff_err = _ff_proc.stderr.read().decode(errors="ignore") if _ff_proc.stderr else ""
raise RuntimeError(f"ffmpeg编码失败(exit={_ff_ret}): {_ff_err[-300:]}")
except Exception:
try: _ff_proc.kill()
except Exception: pass
raise
shutil.copy2(str(silent_video_path), str(output_path))
try:
torch.cuda.empty_cache()
if audio_wav_path.exists():
audio_wav_path.unlink()
if silent_video_path.exists() and str(silent_video_path) != str(output_path):
silent_video_path.unlink()
except Exception as e:
logger.warning("清理中间文件失败: %s", e)
logger.info("MuseTalk 推理完成: output=%s, duration=%.2fs",
output_path.name, _get_media_duration(output_path))
# ── 路由 ────────────────────────────────────────────────────────────── # ── 路由 ──────────────────────────────────────────────────────────────
@@ -470,7 +910,7 @@ def _run_inference(
@app.route("/health", methods=["GET"]) @app.route("/health", methods=["GET"])
def health(): def health():
"""健康检查 + GPU 显存信息.""" """健康检查 + GPU 显存信息 + MuseTalk 模型状态."""
gpu_info = _get_gpu_info() gpu_info = _get_gpu_info()
task_info = { task_info = {
"task_id": current_task["task_id"], "task_id": current_task["task_id"],
@@ -482,6 +922,8 @@ def health():
"status": "healthy", "status": "healthy",
"gpu": gpu_info, "gpu": gpu_info,
"current_task": task_info, "current_task": task_info,
"musetalk_loaded": _muse_models_loaded,
"musetalk_load_error": str(_muse_load_error) if _muse_load_error else None,
"timestamp": time.time(), "timestamp": time.time(),
} }
) )
@@ -491,7 +933,8 @@ def health():
def inference(): def inference():
"""推理请求:multipart form 包含 video 和 audio 文件. """推理请求:multipart form 包含 video 和 audio 文件.
#1978 v2MuseTalk 直接处理全量音频,输出时长=音频时长,无需预处理循环。 可选 form 参数:
bbox_shift: 口型区域垂直偏移量,默认 0,范围 [-5, 5]
""" """
# 并发控制:检查锁 # 并发控制:检查锁
if not inference_lock.acquire(blocking=False): if not inference_lock.acquire(blocking=False):
@@ -510,6 +953,8 @@ def inference():
video_file = request.files["video"] video_file = request.files["video"]
audio_file = request.files["audio"] audio_file = request.files["audio"]
task_id = request.form.get("task_id", f"task_{int(time.time())}") task_id = request.form.get("task_id", f"task_{int(time.time())}")
bbox_shift = int(request.form.get("bbox_shift", "0"))
bbox_shift = max(-5, min(5, bbox_shift)) # 限制范围
# 文件大小检查 # 文件大小检查
err = _check_file_size(video_file, Config.video_max_mb, "视频") err = _check_file_size(video_file, Config.video_max_mb, "视频")
@@ -523,13 +968,14 @@ def inference():
task_dir = Path(Config.temp_dir) / task_id task_dir = Path(Config.temp_dir) / task_id
task_dir.mkdir(parents=True, exist_ok=True) task_dir.mkdir(parents=True, exist_ok=True)
video_path = task_dir / "input.mp4" video_path = task_dir / "input.mp4"
audio_path = task_dir / "input_audio.wav" audio_path = task_dir / "input_audio.bin"
output_path = task_dir / "output.mp4" output_path = task_dir / "output.mp4"
video_file.save(str(video_path)) video_file.save(str(video_path))
audio_file.save(str(audio_path)) audio_file.save(str(audio_path))
logger.info("开始推理 task_id=%s, video=%s, audio=%s", task_id, video_path.name, audio_path.name) logger.info("开始推理 task_id=%s, video=%s, audio=%s, bbox_shift=%d",
task_id, video_path.name, audio_path.name, bbox_shift)
# 更新当前任务信息 # 更新当前任务信息
current_task["task_id"] = task_id current_task["task_id"] = task_id
@@ -541,8 +987,9 @@ def inference():
def inference_thread(): def inference_thread():
try: try:
_run_inference(video_path, audio_path, output_path) _run_inference(video_path, audio_path, output_path, bbox_shift=bbox_shift)
except Exception as exc: except Exception as exc:
logger.exception("推理异常: %s", exc)
result_container["error"] = str(exc) result_container["error"] = str(exc)
thread = threading.Thread(target=inference_thread) thread = threading.Thread(target=inference_thread)
@@ -621,6 +1068,8 @@ def cancel():
def main(): def main():
import os as _os
_os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "max_split_size_mb:128")
"""启动 Flask 服务.""" """启动 Flask 服务."""
# 创建临时目录 # 创建临时目录
Path(Config.temp_dir).mkdir(parents=True, exist_ok=True) Path(Config.temp_dir).mkdir(parents=True, exist_ok=True)
@@ -633,11 +1082,14 @@ def main():
gpu_info["memory_used_mb"], gpu_info["memory_used_mb"],
gpu_info["memory_total_mb"], gpu_info["memory_total_mb"],
) )
logger.info("MuseTalk 仓库路径: %s", Config.muse_dir)
logger.info( logger.info(
"启动 MuseTalk Server: port=%d, timeout=%.0fs, max_concurrent=%d", "启动 MuseTalk Server: port=%d, timeout=%.0fs, max_concurrent=%d, fp16=%s, batch_size=%d",
Config.port, Config.port,
Config.inference_timeout, Config.inference_timeout,
Config.max_concurrent, Config.max_concurrent,
Config.use_float16,
Config.batch_size,
) )
app.run(host="0.0.0.0", port=Config.port, threaded=True) app.run(host="0.0.0.0", port=Config.port, threaded=True)
+1 -1
View File
@@ -3,7 +3,7 @@
REPO_API="https://git.xiaoxiajianji.com/api/v1/repos/xiaoxia/xiaoxia-saas/commits?sha=develop&path=deploy/gpu_worker&limit=1" REPO_API="https://git.xiaoxiajianji.com/api/v1/repos/xiaoxia/xiaoxia-saas/commits?sha=develop&path=deploy/gpu_worker&limit=1"
STATE_FILE="/home/ying/projects/gpu-webhook/.last_commit" STATE_FILE="/home/ying/projects/gpu-webhook/.last_commit"
UPDATE_SCRIPT="/home/ying/projects/update-gpu-worker.sh" UPDATE_SCRIPT="/home/ying/projects/update-gpu-worker.sh"
LOG_FILE="/tmp/gpu-poll.log" LOG_FILE="$HOME/gpu-poll.log"
log() { log() {
echo "[$(date +"%Y-%m-%d %H:%M:%S")] $*" >> "$LOG_FILE" echo "[$(date +"%Y-%m-%d %H:%M:%S")] $*" >> "$LOG_FILE"
+2 -1
View File
@@ -59,4 +59,5 @@ echo " sudo systemctl status musetalk-worker"
echo " sudo systemctl status xiaoxia-gpu-worker" echo " sudo systemctl status xiaoxia-gpu-worker"
echo " sudo systemctl status gpu-poll.timer" echo " sudo systemctl status gpu-poll.timer"
echo "健康检查:curl http://127.0.0.1:7861/health" echo "健康检查:curl http://127.0.0.1:7861/health"
echo "更新日志:tail -f /tmp/gpu-worker-update.log" echo "更新日志:tail -f ~/gpu-worker-update.log"
echo "轮询日志:tail -f ~/gpu-poll.log"
@@ -4,7 +4,7 @@ set -e
REPO_URL="https://git.xiaoxiajianji.com/xiaoxia/xiaoxia-saas/raw/branch/develop/deploy/gpu_worker" REPO_URL="https://git.xiaoxiajianji.com/xiaoxia/xiaoxia-saas/raw/branch/develop/deploy/gpu_worker"
MUSE_DIR="/home/ying/projects/MuseTalk" MUSE_DIR="/home/ying/projects/MuseTalk"
WORKER_DIR="/opt/xiaoxia-gpu-worker" WORKER_DIR="/opt/xiaoxia-gpu-worker"
LOG_FILE="/tmp/gpu-worker-update.log" LOG_FILE="$HOME/gpu-worker-update.log"
log() { log() {
local NOW local NOW
+3 -3
View File
@@ -1,11 +1,11 @@
[Unit] [Unit]
Description=MuseTalk GPU Worker (xiaoxia-saas 反向轮询) Description=MuseTalk GPU Worker (xiaoxia-saas 反向轮询)
After=network.target musetalk.service After=network.target musetalk-worker.service
# 本地 MuseTalk 服务启动后再启动本 Worker;若 MuseTalk 没有 systemd 服务则删除 musetalk.service # 本地 MuseTalk 服务musetalk-worker.service)启动后再启动本 Worker
[Service] [Service]
Type=simple Type=simple
User=%i User=ying
WorkingDirectory=/opt/xiaoxia-gpu-worker WorkingDirectory=/opt/xiaoxia-gpu-worker
# 读取环境变量(API 地址、Token、轮询间隔等) # 读取环境变量(API 地址、Token、轮询间隔等)
EnvironmentFile=/opt/xiaoxia-gpu-worker/.env EnvironmentFile=/opt/xiaoxia-gpu-worker/.env
+6
View File
@@ -31,6 +31,12 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
COPY infra/fonts/NotoSansSC-VF.ttf /usr/share/fonts/opentype/noto/NotoSansSC-VF.ttf COPY infra/fonts/NotoSansSC-VF.ttf /usr/share/fonts/opentype/noto/NotoSansSC-VF.ttf
COPY infra/fonts/NotoSerifCJKsc-VF.otf /usr/share/fonts/opentype/noto/NotoSerifCJKsc-VF.otf COPY infra/fonts/NotoSerifCJKsc-VF.otf /usr/share/fonts/opentype/noto/NotoSerifCJKsc-VF.otf
COPY infra/fonts/LXGWWenKai-Regular.ttf /usr/share/fonts/truetype/lxgw/LXGWWenKai-Regular.ttf COPY infra/fonts/LXGWWenKai-Regular.ttf /usr/share/fonts/truetype/lxgw/LXGWWenKai-Regular.ttf
# #2001 爆款标题字体:优设标题黑 / 阿里普惠体 Bold / 抖音美好体 / 思源黑体 Heavy
RUN mkdir -p /usr/share/fonts/truetype/xiaoxia
COPY infra/fonts/xiaoxia/YouSheBiaoTiHei.ttf /usr/share/fonts/truetype/xiaoxia/YouSheBiaoTiHei.ttf
COPY infra/fonts/xiaoxia/AlibabaPuHuiTi-Bold.ttf /usr/share/fonts/truetype/xiaoxia/AlibabaPuHuiTi-Bold.ttf
COPY infra/fonts/xiaoxia/DouyinSansBold.otf /usr/share/fonts/truetype/xiaoxia/DouyinSansBold.otf
COPY infra/fonts/xiaoxia/NotoSansSC-Black.otf /usr/share/fonts/truetype/xiaoxia/NotoSansSC-Black.otf
RUN fc-cache -fv RUN fc-cache -fv
# 创建虚拟环境 # 创建虚拟环境
+6
View File
@@ -35,6 +35,12 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
COPY infra/fonts/NotoSansSC-VF.ttf /usr/share/fonts/opentype/noto/NotoSansSC-VF.ttf COPY infra/fonts/NotoSansSC-VF.ttf /usr/share/fonts/opentype/noto/NotoSansSC-VF.ttf
COPY infra/fonts/NotoSerifCJKsc-VF.otf /usr/share/fonts/opentype/noto/NotoSerifCJKsc-VF.otf COPY infra/fonts/NotoSerifCJKsc-VF.otf /usr/share/fonts/opentype/noto/NotoSerifCJKsc-VF.otf
COPY infra/fonts/LXGWWenKai-Regular.ttf /usr/share/fonts/truetype/lxgw/LXGWWenKai-Regular.ttf COPY infra/fonts/LXGWWenKai-Regular.ttf /usr/share/fonts/truetype/lxgw/LXGWWenKai-Regular.ttf
# #2001 爆款标题字体:优设标题黑 / 阿里普惠体 Bold / 抖音美好体 / 思源黑体 Heavy
RUN mkdir -p /usr/share/fonts/truetype/xiaoxia
COPY infra/fonts/xiaoxia/YouSheBiaoTiHei.ttf /usr/share/fonts/truetype/xiaoxia/YouSheBiaoTiHei.ttf
COPY infra/fonts/xiaoxia/AlibabaPuHuiTi-Bold.ttf /usr/share/fonts/truetype/xiaoxia/AlibabaPuHuiTi-Bold.ttf
COPY infra/fonts/xiaoxia/DouyinSansBold.otf /usr/share/fonts/truetype/xiaoxia/DouyinSansBold.otf
COPY infra/fonts/xiaoxia/NotoSansSC-Black.otf /usr/share/fonts/truetype/xiaoxia/NotoSansSC-Black.otf
RUN fc-cache -fv RUN fc-cache -fv
# 创建虚拟环境 # 创建虚拟环境
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -711,6 +711,7 @@ class LipsyncJobModel(Base):
voice_id = Column(String(200), nullable=False, default="") voice_id = Column(String(200), nullable=False, default="")
script_text = Column(Text, nullable=False, default="") script_text = Column(Text, nullable=False, default="")
speed = Column(Float, nullable=False, default=1.0) speed = Column(Float, nullable=False, default=1.0)
style = Column(String(32), nullable=False, default="")
emotion = Column(String(20), nullable=False, default="") emotion = Column(String(20), nullable=False, default="")
# MediaKit 任务状态 # MediaKit 任务状态
@@ -750,6 +751,8 @@ class AiAvatarRenderJob(Base):
# b_roll_segments 格式: [{"script_segment_index": 0, "asset_url": "...", "mode": "fullscreen|pip", "start_time": 5.0, "end_time": 10.0}, ...] # b_roll_segments 格式: [{"script_segment_index": 0, "asset_url": "...", "mode": "fullscreen|pip", "start_time": 5.0, "end_time": 10.0}, ...]
title_config = Column(JSON, nullable=False, default=dict) title_config = Column(JSON, nullable=False, default=dict)
cover_config = Column(JSON, nullable=False, default=dict) cover_config = Column(JSON, nullable=False, default=dict)
# #2001 封面独立标题配置(结构同 title_config;为空时封面不叠标题)
cover_title_config = Column(JSON, nullable=False, default=dict)
# 任务状态 # 任务状态
status = Column(String(20), nullable=False, default="pending", index=True) status = Column(String(20), nullable=False, default="pending", index=True)
+126 -13
View File
@@ -26,12 +26,43 @@ from packages.shared.config import get_shared_settings
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
# CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。 # ── Style(语气风格)→ CosyVoice instruct 自然语言指令 ──
# 官方文档:instruction 格式严格为 "你说话的情感是<情感值>。",结尾中文句号不可省略; # 前端 PR#2002 传 6 种 stylenatural/sweet/excited/professional/news/livestream。
# 情感值必须是 7 种英文枚举之一:neutral/happy/sad/angry/surprised/fearful/disgusted # style 是新的统一参数;emotion 为 deprecated 兼容别名,内部映射为 style
# 参考:https://help.aliyun.com/zh/model-studio/cosyvoice-voice-list # 参考:https://help.aliyun.com/zh/model-studio/cosyvoice-voice-list
# 前端可传英文枚举或中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),统一归一化为英文枚举。 STYLE_INSTRUCTION_MAP: dict[str, str] = {
# 映射表 key(不区分大小写): 英文枚举/旧英文/中文标签 → 7 种标准英文枚举 "natural": "用自然、平和的语气说话。",
"sweet": "用温柔甜美、亲切柔和的语气说话。",
"excited": "用兴奋、激动的语气说话。",
"professional": "用专业、正式的语气说话。",
"news": "用新闻播报的语气说话。",
"livestream": "用直播解说的语气说话。",
}
# 有效 style 值集合(供 schema / 校验使用)
VALID_STYLES: frozenset[str] = frozenset(STYLE_INSTRUCTION_MAP.keys())
# ── 旧 emotion → 新 style 兼容映射(方案 B:统一 styleemotion deprecated)──
_EMOTION_TO_STYLE: dict[str, str] = {
"neutral": "natural",
"happy": "excited",
"sad": "sweet",
"angry": "excited",
"surprised": "excited",
"fearful": "sweet",
"disgusted": "natural",
}
# 严格格式系统音色:style → emotion 回退(用于无法使用自由文本指令的音色)
_STYLE_TO_EMOTION: dict[str, str] = {
"natural": "neutral",
"sweet": "sad",
"excited": "happy",
"professional": "neutral",
# news / livestream 无直接对应 emotion,特殊处理
}
# ── 旧 emotion 映射表(deprecated,保留以兼容历史数据)──
EMOTION_MAP: dict[str, str] = { EMOTION_MAP: dict[str, str] = {
# ── 7 种标准英文枚举(CosyVoice v3 官方支持的情感值)── # ── 7 种标准英文枚举(CosyVoice v3 官方支持的情感值)──
"neutral": "neutral", "neutral": "neutral",
@@ -122,6 +153,70 @@ def build_emotion_instruction(voice_id: str, emotion_enum: str) -> str:
return "" return ""
def resolve_style(style: str = "", emotion: str = "") -> str:
"""统一解析 style 参数(方案 B).
- style 有值且合法:直接使用(style 优先级最高)。
- style 为空但 emotion 有值:将旧 emotion 归一化后映射为 style。
- 两者皆空:返回空串(调用方不传 instruction)。
Args:
style: 新的语气风格(natural/sweet/excited/professional/news/livestream
emotion: 旧的情绪参数(deprecated,内部映射为 style
Returns:
解析后的 style 字符串;无需 instruct 时返回空串
"""
s = (style or "").strip().lower()
if s:
if s in VALID_STYLES:
return s
logger.warning("未知的 style 值 %r,忽略 style 参数", style)
# 回退:emotion → style
norm = normalize_emotion(emotion)
if not norm:
return ""
mapped = _EMOTION_TO_STYLE.get(norm)
if not mapped:
logger.warning("emotion %r 无法映射到 style,跳过 instruct", norm)
return mapped or ""
def build_style_instruction(voice_id: str, style: str) -> str:
"""根据 voice 类型构造 style instruction.
- natural:返回空串(不额外加 instruct,使用 CosyVoice 默认自然语气)。
- 克隆/设计音色:使用中文自然语言指令(DashScope 允许任意自然语言)。
- 系统音色中支持 emotion instruct 的白名单音色:
若 style 可映射到 emotion,用严格格式 "你说话的情感是<emotion>。"
news/livestream 尝试直接用中文 instruct(部分音色支持自由文本)。
- 其他系统音色:返回空串。
Args:
voice_id: CosyVoice voice 参数
style: 已通过 resolve_style() 解析的 style 值
Returns:
拼接好的 instruction 字符串;无需 instruct 时返回空串
"""
if not style or style == "natural":
return ""
# 克隆音色:直接使用中文自然语言指令
if _is_cloned_voice(voice_id):
return STYLE_INSTRUCTION_MAP.get(style, "")
# 系统音色白名单:优先映射到严格 emotion 格式
if voice_id in _SYSTEM_VOICES_WITH_EMOTION_INSTRUCT:
emotion_val = _STYLE_TO_EMOTION.get(style)
if emotion_val:
return f"你说话的情感是{emotion_val}"
# news/livestream 无 emotion 对应,尝试自由中文 instruct
desc = STYLE_INSTRUCTION_MAP.get(style, "")
if desc:
logger.info("音色 %s 使用自由文本 style instruct: %s", voice_id, desc)
return desc
return ""
def normalize_emotion(emotion: str) -> str: def normalize_emotion(emotion: str) -> str:
"""将前端情绪值归一化为 CosyVoice v3 官方英文枚举,用于拼入 instruction. """将前端情绪值归一化为 CosyVoice v3 官方英文枚举,用于拼入 instruction.
@@ -573,6 +668,8 @@ class CosyVoiceService:
volume: int = 50, volume: int = 50,
emotion: str = "", emotion: str = "",
language: str = "zh", language: str = "zh",
style: str = "",
pitch: float = 1.0,
) -> dict: ) -> dict:
"""提交语音合成任务(同步非流式,直接返回结果). """提交语音合成任务(同步非流式,直接返回结果).
@@ -586,9 +683,10 @@ class CosyVoiceService:
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值 format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
speed: 语速(0.5-2.0),1.0 为正常速度 speed: 语速(0.5-2.0),1.0 为正常速度
volume: 音量(0-100),默认 50 volume: 音量(0-100),默认 50
emotion: 情绪,英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted style: 语气风格(natural/sweet/excited/professional/news/livestream),
或前端中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),兼容旧值 新的统一参数;优先级高于 emotion
natural/excited/calm/friendly;空串不传,未知值默认 neutral pitch: 音调(0.5-2.0),1.0 为默认值
emotion: 【deprecated】旧情绪参数,内部通过 resolve_style() 映射为 style
language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints
Returns: Returns:
@@ -617,11 +715,20 @@ class CosyVoiceService:
"rate": speed, "rate": speed,
"volume": volume, "volume": volume,
} }
# 情绪 → instruction(按 voice 类型选择格式) # pitch: CosyVoice API 支持 [0.5, 2.0],非默认值时才传
norm_emotion = normalize_emotion(emotion) if pitch and pitch != 1.0:
emotion_instruction = build_emotion_instruction(voice_id, norm_emotion) input_payload["pitch"] = pitch
if emotion_instruction: # style 优先:显式传 style 时走 build_style_instruction()
input_payload["instruction"] = emotion_instruction # 无 style 时回退到旧的 emotion → build_emotion_instruction() 逻辑(向后兼容)
s = (style or "").strip().lower()
instruction = ""
if s:
instruction = build_style_instruction(voice_id, s)
if not instruction:
norm_emotion = normalize_emotion(emotion)
instruction = build_emotion_instruction(voice_id, norm_emotion)
if instruction:
input_payload["instruction"] = instruction
# 语言 → language_hints 数组(仅取第一个元素生效); # 语言 → language_hints 数组(仅取第一个元素生效);
# 系统音色(非克隆/非 voice_id 中包含下划线以外的短 ID)仅传 zh/en,其他语言不传避免报错 # 系统音色(非克隆/非 voice_id 中包含下划线以外的短 ID)仅传 zh/en,其他语言不传避免报错
norm_lang = normalize_language(language) norm_lang = normalize_language(language)
@@ -682,6 +789,8 @@ class CosyVoiceService:
emotion: str = "", emotion: str = "",
language: str = "zh", language: str = "zh",
timeout: float = 120.0, timeout: float = 120.0,
style: str = "",
pitch: float = 1.0,
) -> SynthesizeResult: ) -> SynthesizeResult:
"""语音合成(同步非流式). """语音合成(同步非流式).
@@ -695,6 +804,8 @@ class CosyVoiceService:
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值 format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
speed: 语速(0.5-2.0),1.0 为正常速度 speed: 语速(0.5-2.0),1.0 为正常速度
volume: 音量(0-100),默认 50 volume: 音量(0-100),默认 50
style: 语气风格(natural/sweet/excited/professional/news/livestream
pitch: 音调(0.5-2.0),1.0 为默认值
timeout: 超时时间(秒),保留参数兼容 timeout: 超时时间(秒),保留参数兼容
Returns: Returns:
@@ -714,6 +825,8 @@ class CosyVoiceService:
volume=volume, volume=volume,
emotion=emotion, emotion=emotion,
language=language, language=language,
style=style,
pitch=pitch,
) )
return SynthesizeResult( return SynthesizeResult(
+20
View File
@@ -146,6 +146,9 @@ class TTSWorkflowService:
_meta = dict(job.metadata) _meta = dict(job.metadata)
_speed = float(_meta.get("speed", 1.0) or 1.0) _speed = float(_meta.get("speed", 1.0) or 1.0)
_emotion = str(_meta.get("emotion", "") or "") _emotion = str(_meta.get("emotion", "") or "")
_style = str(_meta.get("style", "") or "")
_volume = int(_meta.get("volume", 50) or 50)
_pitch = float(_meta.get("pitch", 1.0) or 1.0)
_language = str(_meta.get("language", "zh-CN") or "zh-CN") _language = str(_meta.get("language", "zh-CN") or "zh-CN")
submit_result = self.cosyvoice_service.submit_synthesize_task( submit_result = self.cosyvoice_service.submit_synthesize_task(
text=job.input_text, text=job.input_text,
@@ -153,6 +156,9 @@ class TTSWorkflowService:
sample_rate=job.sample_rate, sample_rate=job.sample_rate,
format=job.format, format=job.format,
speed=_speed, speed=_speed,
style=_style,
volume=_volume,
pitch=_pitch,
emotion=_emotion, emotion=_emotion,
language=_language, language=_language,
) )
@@ -290,6 +296,8 @@ class TTSWorkflowService:
job_metadata = job.metadata or {} job_metadata = job.metadata or {}
speed = float(job_metadata.get("speed", 1.0)) speed = float(job_metadata.get("speed", 1.0))
volume = int(job_metadata.get("volume", 50)) volume = int(job_metadata.get("volume", 50))
style = str(job_metadata.get("style", "") or "")
pitch = float(job_metadata.get("pitch", 1.0))
emotion = str(job_metadata.get("emotion", "") or "") emotion = str(job_metadata.get("emotion", "") or "")
language = str(job_metadata.get("language", "zh-CN") or "zh-CN") language = str(job_metadata.get("language", "zh-CN") or "zh-CN")
@@ -299,7 +307,9 @@ class TTSWorkflowService:
sample_rate=job.sample_rate, sample_rate=job.sample_rate,
format=job.format, format=job.format,
speed=speed, speed=speed,
style=style,
volume=volume, volume=volume,
pitch=pitch,
emotion=emotion, emotion=emotion,
language=language, language=language,
) )
@@ -415,6 +425,9 @@ class TTSWorkflowService:
_seg_meta = job.metadata or {} _seg_meta = job.metadata or {}
_seg_speed = float(_seg_meta.get("speed", 1.0) or 1.0) _seg_speed = float(_seg_meta.get("speed", 1.0) or 1.0)
_seg_emotion = str(_seg_meta.get("emotion", "") or "") _seg_emotion = str(_seg_meta.get("emotion", "") or "")
_seg_style = str(_seg_meta.get("style", "") or "")
_seg_volume = int(_seg_meta.get("volume", 50) or 50)
_seg_pitch = float(_seg_meta.get("pitch", 1.0) or 1.0)
with ThreadPoolExecutor(max_workers=max_workers) as executor: with ThreadPoolExecutor(max_workers=max_workers) as executor:
future_to_idx = {} future_to_idx = {}
@@ -426,6 +439,9 @@ class TTSWorkflowService:
sample_rate=job.sample_rate, sample_rate=job.sample_rate,
format=job.format, format=job.format,
speed=_seg_speed, speed=_seg_speed,
style=_seg_style,
volume=_seg_volume,
pitch=_seg_pitch,
emotion=_seg_emotion, emotion=_seg_emotion,
language=_seg_meta.get("language", "zh-CN") or "zh-CN", language=_seg_meta.get("language", "zh-CN") or "zh-CN",
) )
@@ -517,6 +533,8 @@ class TTSWorkflowService:
job_metadata = job.metadata or {} job_metadata = job.metadata or {}
speed = float(job_metadata.get("speed", 1.0)) speed = float(job_metadata.get("speed", 1.0))
volume = int(job_metadata.get("volume", 50)) volume = int(job_metadata.get("volume", 50))
style = str(job_metadata.get("style", "") or "")
pitch = float(job_metadata.get("pitch", 1.0))
emotion = str(job_metadata.get("emotion", "") or "") emotion = str(job_metadata.get("emotion", "") or "")
# 分段文本(用于缺失段重新合成) # 分段文本(用于缺失段重新合成)
@@ -551,7 +569,9 @@ class TTSWorkflowService:
sample_rate=job.sample_rate, sample_rate=job.sample_rate,
format=job.format, format=job.format,
speed=speed, speed=speed,
style=style,
volume=volume, volume=volume,
pitch=pitch,
emotion=emotion, emotion=emotion,
language=job.metadata.get("language", "zh-CN") if hasattr(job, "metadata") else "zh-CN", language=job.metadata.get("language", "zh-CN") if hasattr(job, "metadata") else "zh-CN",
) )
+26
View File
@@ -30,6 +30,8 @@ TITLE_MARGIN_SIDE = 40
# - 楷体 → LXGW WenKai(霞鹜文楷,#1896 新增 SIL OFL 开源楷体) # - 楷体 → LXGW WenKai(霞鹜文楷,#1896 新增 SIL OFL 开源楷体)
# - 苹方/PingFang/微软雅黑:服务器 Linux 无对应字体,fallback 思源黑体 # - 苹方/PingFang/微软雅黑:服务器 Linux 无对应字体,fallback 思源黑体
# - 华康俪金黑:商业字体有版权风险,前端已移除,后端保留映射 fallback 思源黑体(兼容老数据) # - 华康俪金黑:商业字体有版权风险,前端已移除,后端保留映射 fallback 思源黑体(兼容老数据)
# #2001 爆款标题字体(部署到 /usr/share/fonts/truetype/xiaoxia/):
# - 优设标题黑 / 阿里普惠体 Bold / 抖音美好体 / 思源黑体 Heavy(独立 Black 字重)
FONT_NAME_MAP: dict[str, str] = { FONT_NAME_MAP: dict[str, str] = {
"思源黑体": "Noto Sans SC", "思源黑体": "Noto Sans SC",
"思源宋体": "Noto Serif CJK SC", "思源宋体": "Noto Serif CJK SC",
@@ -40,6 +42,15 @@ FONT_NAME_MAP: dict[str, str] = {
"楷体": "LXGW WenKai", "楷体": "LXGW WenKai",
"霞鹜文楷": "LXGW WenKai", "霞鹜文楷": "LXGW WenKai",
"华康俪金黑": "Noto Sans SC", "华康俪金黑": "Noto Sans SC",
# #2001 爆款标题字体
"优设标题黑": "YouSheBiaoTiHei",
"阿里普惠体": "Alibaba PuHuiTi",
"阿里普惠体 Bold": "Alibaba PuHuiTi",
"阿里巴巴普惠体": "Alibaba PuHuiTi",
"抖音美好体": "Douyin Sans",
"抖音体": "Douyin Sans",
"思源黑体 Heavy": "Noto Sans SC",
"思源黑体 Black": "Noto Sans SC",
} }
# ASS Fontsize 是字体 em-square 高度(含 Latin 升降部留白), # ASS Fontsize 是字体 em-square 高度(含 Latin 升降部留白),
@@ -421,6 +432,14 @@ def build_ass_content(
safe_title_text_raw = escape_ass_text(title_text) safe_title_text_raw = escape_ass_text(title_text)
safe_title_text = _wrap_title_text(safe_title_text_raw, video_width, title_font_size) safe_title_text = _wrap_title_text(safe_title_text_raw, video_width, title_font_size)
# #2001 逐行样式覆盖:按 line_overrides 在每行前注入 ASS inline override 标签
# line_overrides 透传自前端爆款标题面板,SubtitleStyle.from_dict 已做安全过滤
if title_config.get("line_overrides"):
from packages.domain.subtitle_style import SubtitleStyle
_title_style_for_overrides = SubtitleStyle.from_dict(title_config)
safe_title_text = _title_style_for_overrides.apply_line_overrides(safe_title_text)
# 自由位置:在文本前注入 \pos override tag(锚点为文本块中心,配合 \an5) # 自由位置:在文本前注入 \pos override tag(锚点为文本块中心,配合 \an5)
if title_pos is not None: if title_pos is not None:
safe_title_text = f"{{\\pos({title_pos[0]},{title_pos[1]})}}{safe_title_text}" safe_title_text = f"{{\\pos({title_pos[0]},{title_pos[1]})}}{safe_title_text}"
@@ -455,6 +474,13 @@ def build_ass_content(
safe_subtitle_text = escape_ass_text(subtitle_text) safe_subtitle_text = escape_ass_text(subtitle_text)
# #2001 逐行样式覆盖(字幕路径同样支持)
if subtitle_config.get("line_overrides"):
from packages.domain.subtitle_style import SubtitleStyle
_sub_style_for_overrides = SubtitleStyle.from_dict(subtitle_config)
safe_subtitle_text = _sub_style_for_overrides.apply_line_overrides(safe_subtitle_text)
events.append( events.append(
"Dialogue: 0,0:00:00.00," "Dialogue: 0,0:00:00.00,"
f"{format_ass_time(video_duration)}," f"{format_ass_time(video_duration)},"
+140 -1
View File
@@ -6,7 +6,7 @@
from __future__ import annotations from __future__ import annotations
from dataclasses import dataclass from dataclasses import dataclass, field
from typing import Any from typing import Any
# ── 常量 ────────────────────────────────────────────────────────────────────── # ── 常量 ──────────────────────────────────────────────────────────────────────
@@ -163,6 +163,12 @@ class SubtitleStyle:
fade_out: float = 0.0 fade_out: float = 0.0
animation_type: str = "none" animation_type: str = "none"
# 逐行独立样式覆盖(#2001 爆款标题样式面板)
# list[dict],每项可选字段: line_index(0-based,支持负数从末尾倒数),
# color/font/size/bold/italic/stroke_color/stroke_width/shadow_color/shadow_offset_x/shadow_offset_y
# 渲染时按行索引匹配,用 ASS 内联 override 标签包裹该行。缺省字段继承主样式。
line_overrides: list[dict[str, Any]] = field(default_factory=list)
@classmethod @classmethod
def from_dict(cls, config: dict[str, Any] | None) -> "SubtitleStyle": def from_dict(cls, config: dict[str, Any] | None) -> "SubtitleStyle":
"""从字典创建样式配置,带安全类型转换.""" """从字典创建样式配置,带安全类型转换."""
@@ -193,6 +199,14 @@ class SubtitleStyle:
if position not in POSITION_ALIGNMENT: if position not in POSITION_ALIGNMENT:
position = DEFAULT_POSITION position = DEFAULT_POSITION
# 逐行覆盖:仅保留 dict 类型项;非 dict 项过滤掉避免渲染崩溃
raw_overrides = config.get("line_overrides") or []
line_overrides: list[dict[str, Any]] = []
if isinstance(raw_overrides, list):
for item in raw_overrides:
if isinstance(item, dict):
line_overrides.append(dict(item))
return cls( return cls(
font_name=safe_str("font", DEFAULT_FONT), font_name=safe_str("font", DEFAULT_FONT),
font_size=safe_int("size", DEFAULT_FONT_SIZE), font_size=safe_int("size", DEFAULT_FONT_SIZE),
@@ -221,6 +235,7 @@ class SubtitleStyle:
fade_in=max(0.0, safe_float("fade_in", 0.0)), fade_in=max(0.0, safe_float("fade_in", 0.0)),
fade_out=max(0.0, safe_float("fade_out", 0.0)), fade_out=max(0.0, safe_float("fade_out", 0.0)),
animation_type=safe_str("animation_type", "none"), animation_type=safe_str("animation_type", "none"),
line_overrides=line_overrides,
) )
@property @property
@@ -248,6 +263,130 @@ class SubtitleStyle:
color_bgr = hex_to_ass_bgr(self.background_color) color_bgr = hex_to_ass_bgr(self.background_color)
return f"&H{alpha_hex}{color_bgr}" return f"&H{alpha_hex}{color_bgr}"
def build_line_override_tag(self, line_index: int, total_lines: int) -> str:
r"""按 line_overrides 配置为指定行构造 ASS 内联 override 标签 {\c&HBBGGRR&...}.
仅返回大括号包裹的 override 标签串;调用方拼到该行文本前即可。
未配置该覆盖项时返回空串。缺省字段继承主样式,不生成对应 tag。
Args:
line_index: 行号(0-based);支持负数(-1 为最后一行)。
total_lines: 总行数(用于解析负数索引)。
"""
if not self.line_overrides or total_lines <= 0:
return ""
# 解析负数索引
resolved = line_index if line_index >= 0 else total_lines + line_index
if resolved < 0 or resolved >= total_lines:
return ""
override: dict[str, Any] | None = None
for item in self.line_overrides:
if not isinstance(item, dict):
continue
idx = item.get("line_index")
try:
idx_int = int(idx) if idx is not None else None
except (TypeError, ValueError):
continue
if idx_int is None:
continue
if idx_int < 0:
idx_int = total_lines + idx_int
if idx_int == resolved:
override = item
break
if not override:
return ""
tags: list[str] = []
# 主色(字体颜色):\c&HBBGGRR&
color_val = override.get("color")
if isinstance(color_val, str) and color_val:
tags.append(f"\\c{hex_to_ass_color(color_val)}")
# 字号:\fsN
size_val = override.get("size") or override.get("font_size")
try:
size_int = int(size_val) if size_val is not None else None
if size_int and size_int > 0:
tags.append(f"\\fs{size_int}")
except (TypeError, ValueError):
pass
# 字体:\fnFontName
font_val = override.get("font") or override.get("font_name")
if isinstance(font_val, str) and font_val:
tags.append(f"\\fn{font_val}")
# 粗体:\b1 / \b0
bold_val = override.get("bold")
if isinstance(bold_val, bool):
tags.append("\\b1" if bold_val else "\\b0")
# 斜体:\i1 / \i0
italic_val = override.get("italic")
if isinstance(italic_val, bool):
tags.append("\\i1" if italic_val else "\\i0")
# 描边色:\3c&HBBGGRR&
stroke_c = override.get("stroke_color")
if isinstance(stroke_c, str) and stroke_c:
tags.append(f"\\3c{hex_to_ass_color(stroke_c)}")
# 描边宽:\bordN
stroke_w = override.get("stroke_width")
try:
sw = float(stroke_w) if stroke_w is not None else None
if sw is not None and sw >= 0:
tags.append(f"\\bord{sw:g}")
except (TypeError, ValueError):
pass
# 阴影色:\4c&HBBGGRR&
shadow_c = override.get("shadow_color")
if isinstance(shadow_c, str) and shadow_c:
tags.append(f"\\4c{hex_to_ass_color(shadow_c)}")
# 阴影偏移:\shadN(单值,同时设置 x/y;精细控制用 \xshad/\yshad
sx = override.get("shadow_offset_x")
sy = override.get("shadow_offset_y")
try:
sx_i = int(sx) if sx is not None else None
sy_i = int(sy) if sy is not None else None
if sx_i is not None:
tags.append(f"\\xshad{sx_i}")
if sy_i is not None:
tags.append(f"\\yshad{sy_i}")
except (TypeError, ValueError):
pass
if not tags:
return ""
return "{" + "".join(tags) + "}"
def apply_line_overrides(self, text: str) -> str:
"""按 line_overrides 对 ASS 文本(已 escape、换行用 \\N 分隔)逐行套 override 标签.
仅对换行后的每行首加对应 override 标签;无 override 的行保持原样。
"""
if not self.line_overrides or not text:
return text
if "\\N" not in text:
# 单行
tag = self.build_line_override_tag(0, 1)
return tag + text if tag else text
lines = text.split("\\N")
total = len(lines)
out: list[str] = []
for i, ln in enumerate(lines):
tag = self.build_line_override_tag(i, total)
out.append(tag + ln if tag else ln)
return "\\N".join(out)
# ── 字幕片段 ────────────────────────────────────────────────────────────────── # ── 字幕片段 ──────────────────────────────────────────────────────────────────
+15
View File
@@ -388,6 +388,11 @@ DRAWTEXT_FONT_SEARCH_PATHS: list[str] = [
"/usr/share/fonts/opentype/noto/NotoSansSC-VF.ttf", "/usr/share/fonts/opentype/noto/NotoSansSC-VF.ttf",
"/usr/share/fonts/opentype/noto/NotoSerifCJKsc-VF.otf", "/usr/share/fonts/opentype/noto/NotoSerifCJKsc-VF.otf",
"/usr/share/fonts/truetype/lxgw/LXGWWenKai-Regular.ttf", "/usr/share/fonts/truetype/lxgw/LXGWWenKai-Regular.ttf",
# #2001 爆款标题字体(优设标题黑 / 阿里普惠体 Bold / 抖音美好体 / 思源黑体 Heavy)
"/usr/share/fonts/truetype/xiaoxia/YouSheBiaoTiHei.ttf",
"/usr/share/fonts/truetype/xiaoxia/AlibabaPuHuiTi-Bold.ttf",
"/usr/share/fonts/truetype/xiaoxia/DouyinSansBold.otf",
"/usr/share/fonts/truetype/xiaoxia/NotoSansSC-Black.otf",
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc", "/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
"/usr/share/fonts/opentype/noto/NotoSansCJK-Bold.ttc", "/usr/share/fonts/opentype/noto/NotoSansCJK-Bold.ttc",
"/usr/share/fonts/noto-cjk/NotoSansCJK-Regular.ttc", "/usr/share/fonts/noto-cjk/NotoSansCJK-Regular.ttc",
@@ -400,6 +405,7 @@ DRAWTEXT_FONT_SEARCH_PATHS: list[str] = [
# #1896 字体映射修复:每个字体映射到独立的关键字,而非全部回退到 NotoSansSC # #1896 字体映射修复:每个字体映射到独立的关键字,而非全部回退到 NotoSansSC
# - 苹方(macOS/ 微软雅黑(Windows/ PingFang:服务器 Linux 无对应文件,fallback 思源黑体 # - 苹方(macOS/ 微软雅黑(Windows/ PingFang:服务器 Linux 无对应文件,fallback 思源黑体
# - 华康俪金黑:商业字体有版权风险,前端已按 #1896 要求移除,后端保留映射但 fallback 思源黑体(兼容老数据) # - 华康俪金黑:商业字体有版权风险,前端已按 #1896 要求移除,后端保留映射但 fallback 思源黑体(兼容老数据)
# #2001 新增爆款标题字体映射
DRAWTEXT_FONT_MAP: dict[str, str] = { DRAWTEXT_FONT_MAP: dict[str, str] = {
"思源黑体": "NotoSansSC", "思源黑体": "NotoSansSC",
"思源宋体": "NotoSerifCJKsc", "思源宋体": "NotoSerifCJKsc",
@@ -410,6 +416,15 @@ DRAWTEXT_FONT_MAP: dict[str, str] = {
"楷体": "LXGWWenKai", "楷体": "LXGWWenKai",
"霞鹜文楷": "LXGWWenKai", "霞鹜文楷": "LXGWWenKai",
"华康俪金黑": "NotoSansSC", "华康俪金黑": "NotoSansSC",
# #2001 爆款标题字体
"优设标题黑": "YouSheBiaoTiHei",
"阿里普惠体": "AlibabaPuHuiTi-Bold",
"阿里普惠体 Bold": "AlibabaPuHuiTi-Bold",
"阿里巴巴普惠体": "AlibabaPuHuiTi-Bold",
"抖音美好体": "DouyinSansBold",
"抖音体": "DouyinSansBold",
"思源黑体 Heavy": "NotoSansSC-Black",
"思源黑体 Black": "NotoSansSC-Black",
} }
@@ -6,6 +6,7 @@ CI 增量映射:
ai_avatar_cover_service 智能选帧 ai_avatar_cover_service 智能选帧
""" """
import importlib
import os import os
from unittest.mock import MagicMock, patch from unittest.mock import MagicMock, patch
@@ -166,6 +167,116 @@ def test_submit_synthesize_payload_cloned_voice_english_emotion():
assert inp["instruction"] == "Speak in a sad tone." assert inp["instruction"] == "Speak in a sad tone."
# ── style(语气风格,#2002)────────────────────────────────────────────
def test_style_natural_omits_instruction():
"""style=natural 不加 instruct,使用 CosyVoice 默认自然语气。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="natural")
inp = captured["json"]["input"]
assert "instruction" not in inp
def test_style_sweet_cloned_voice_uses_chinese_instruction():
"""克隆音色 + style=sweet → 中文自然语言指令。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="sweet")
inp = captured["json"]["input"]
assert inp["instruction"] == "用温柔甜美、亲切柔和的语气说话。"
@pytest.mark.parametrize(
("style", "expected_fragment"),
[
("excited", "兴奋"),
("professional", "专业"),
("news", "新闻"),
("livestream", "直播"),
],
)
def test_style_values_cloned_voice(style, expected_fragment):
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style=style)
inp = captured["json"]["input"]
assert expected_fragment in inp["instruction"]
def test_style_system_voice_uses_emotion_mapping():
"""系统白名单音色 + style=sweet → 严格中文 emotion 格式(映射到 sad)。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="longanyang", style="sweet")
inp = captured["json"]["input"]
assert inp["instruction"] == "你说话的情感是sad。"
def test_style_takes_priority_over_emotion():
"""同时传 style 和 emotion 时以 style 为准。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="excited", emotion="sad")
inp = captured["json"]["input"]
assert "兴奋" in inp["instruction"]
def test_unknown_style_ignored_falls_back_to_emotion():
"""未知 style 值被忽略,回退到 emotion 逻辑。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", style="nonexistent", emotion="sad")
inp = captured["json"]["input"]
assert inp["instruction"] == "Speak in a sad tone."
def test_pitch_passed_only_when_non_default():
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", pitch=1.5)
inp = captured["json"]["input"]
assert inp["pitch"] == 1.5
def test_pitch_omitted_at_default():
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", pitch=1.0)
inp = captured["json"]["input"]
assert "pitch" not in inp
def test_resolve_style_maps_emotion_to_style():
"""旧 emotion 值通过 resolve_style 映射为 style。"""
mod = importlib.import_module("packages.application.cosyvoice_service")
assert mod.resolve_style(emotion="happy") == "excited"
assert mod.resolve_style(emotion="sad") == "sweet"
assert mod.resolve_style(emotion="neutral") == "natural"
assert mod.resolve_style(style="news") == "news"
# style 优先
assert mod.resolve_style(style="news", emotion="happy") == "news"
assert mod.resolve_style() == ""
# ── 对口型 TTS 直生分支 ───────────────────────────────────────────────── # ── 对口型 TTS 直生分支 ─────────────────────────────────────────────────
@@ -50,6 +50,7 @@ def _make_mock_render_job(
m.b_roll_segments = [] m.b_roll_segments = []
m.title_config = {} m.title_config = {}
m.cover_config = {} m.cover_config = {}
m.cover_title_config = {}
m.status = status m.status = status
m.progress = progress m.progress = progress
m.output_video_url = output_video_url m.output_video_url = output_video_url
@@ -93,6 +94,7 @@ class TestRenderRoutes:
body.b_roll_segments = [] body.b_roll_segments = []
body.title_config = {} body.title_config = {}
body.cover_config = {} body.cover_config = {}
body.cover_title_config = {}
body.project_id = "" body.project_id = ""
result = create_render_job( result = create_render_job(
@@ -118,6 +120,7 @@ class TestRenderRoutes:
body.b_roll_segments = [] body.b_roll_segments = []
body.title_config = {} body.title_config = {}
body.cover_config = {} body.cover_config = {}
body.cover_title_config = {}
body.project_id = "" body.project_id = ""
with pytest.raises(HTTPException) as exc_info: with pytest.raises(HTTPException) as exc_info:
@@ -141,6 +144,7 @@ class TestRenderRoutes:
body.b_roll_segments = [] body.b_roll_segments = []
body.title_config = {} body.title_config = {}
body.cover_config = {} body.cover_config = {}
body.cover_title_config = {}
body.project_id = "" body.project_id = ""
with pytest.raises(HTTPException) as exc_info: with pytest.raises(HTTPException) as exc_info:
+250
View File
@@ -0,0 +1,250 @@
"""Celery 任务 lipsync_gpu_process_async 直接单测 (#1978 异步化).
覆盖 apps/api/app/tasks/lipsync_gpu.py 的全部主路径:
- 成功:wait_for_result 返回 done → 签名 URL → completed
- GPU 超时/失败 → MediaKit 兜底(成功/MediaKitError/其他异常)
- job 不存在 / 状态异常提前返回
- 主流程异常 → job 标 failed
- _sign_media_url 各分支
"""
from __future__ import annotations
from unittest.mock import MagicMock, patch
import app.tasks.lipsync_gpu as task_mod
import pytest
def _make_job(status="processing"):
job = MagicMock()
job.id = "job-1"
job.user_id = "u1"
job.status = status
job.video_url = "videos/v.mp4"
job.audio_url = "audios/a.wav"
job.enable_video_loop = True
return job
def _make_gpu_task(status="done", result_url="gpu-lipsync/results/t1.mp4", result_duration=11.2):
t = MagicMock()
t.status = status
t.result_url = result_url
t.result_duration = result_duration
return t
@pytest.fixture()
def db_patch():
"""patch _get_db_session 返回 MagicMock,并在任务结束后断言 close."""
fake_db = MagicMock()
with patch.object(task_mod, "_get_db_session", return_value=fake_db):
yield fake_db
def _patch_gpu_service(final_task):
fake_svc = MagicMock()
fake_svc.wait_for_result.return_value = final_task
return patch(
"app.services.gpu_lipsync_service.GpuLipsyncService",
return_value=fake_svc,
)
def _run_task():
# @shared_task bind=True:直接调用任务对象会自动注入 self
task_mod.lipsync_gpu_process_async("job-1", "u1", "gpu-task-1")
class TestHappyPath:
def test_gpu_done_marks_completed(self, db_patch):
job = _make_job()
db_patch.query.return_value.filter_by.return_value.first.return_value = job
gpu_task = _make_gpu_task()
storage = MagicMock()
storage.get_download_url.return_value = "https://signed.example.com/r1.mp4?sig=x"
with (
_patch_gpu_service(gpu_task),
patch.object(task_mod, "get_shared_storage_service", return_value=storage),
):
_run_task()
assert job.status == "completed"
assert job.output_video_url == "https://signed.example.com/r1.mp4?sig=x"
assert job.output_duration == 11.2
assert job.completed_at is not None
db_patch.commit.assert_called_once()
db_patch.close.assert_called_once()
def test_gpu_done_empty_signed_url_keeps_original(self, db_patch):
job = _make_job()
db_patch.query.return_value.filter_by.return_value.first.return_value = job
gpu_task = _make_gpu_task(result_url="gpu/r2.mp4")
storage = MagicMock()
storage.get_download_url.return_value = ""
with (
_patch_gpu_service(gpu_task),
patch.object(task_mod, "get_shared_storage_service", return_value=storage),
):
_run_task()
assert job.status == "completed"
assert job.output_video_url == "gpu/r2.mp4"
def test_gpu_done_result_duration_none_defaults_zero(self, db_patch):
job = _make_job()
db_patch.query.return_value.filter_by.return_value.first.return_value = job
gpu_task = _make_gpu_task(result_duration=None)
storage = MagicMock()
with (
_patch_gpu_service(gpu_task),
patch.object(task_mod, "get_shared_storage_service", return_value=storage),
):
_run_task()
assert job.output_duration == 0.0
def test_sign_failure_uses_original_url(self, db_patch):
job = _make_job()
db_patch.query.return_value.filter_by.return_value.first.return_value = job
gpu_task = _make_gpu_task(result_url="gpu/r3.mp4")
with (
_patch_gpu_service(gpu_task),
patch.object(task_mod, "get_shared_storage_service", side_effect=RuntimeError("oss down")),
):
_run_task()
assert job.status == "completed"
assert job.output_video_url == "gpu/r3.mp4"
class TestJobGuards:
def test_job_not_found_returns(self, db_patch):
db_patch.query.return_value.filter_by.return_value.first.return_value = None
_run_task()
db_patch.commit.assert_not_called()
db_patch.close.assert_called_once()
def test_job_wrong_status_skipped(self, db_patch):
job = _make_job(status="completed")
db_patch.query.return_value.filter_by.return_value.first.return_value = job
_run_task()
db_patch.commit.assert_not_called()
class TestGpuFailureFallback:
def test_gpu_timeout_falls_back_mediakit_success(self, db_patch):
job = _make_job()
db_patch.query.return_value.filter_by.return_value.first.return_value = job
with (
_patch_gpu_service(None),
patch.object(task_mod, "_fallback_to_mediakit") as fb,
):
_run_task()
fb.assert_called_once_with(db_patch, job)
def test_gpu_failed_status_falls_back(self, db_patch):
job = _make_job()
db_patch.query.return_value.filter_by.return_value.first.return_value = job
gpu_task = _make_gpu_task(status="failed")
with (
_patch_gpu_service(gpu_task),
patch.object(task_mod, "_fallback_to_mediakit") as fb,
):
_run_task()
fb.assert_called_once_with(db_patch, job)
class TestFallbackToMediaKit:
def test_mediakit_success_marks_submitted(self, db_patch):
job = _make_job()
fake_client = MagicMock()
fake_client.submit_lipsync.return_value = {"task_id": "mk-99"}
with (
patch("app.services.mediakit_client.get_mediakit_client", return_value=fake_client),
patch.object(task_mod, "_sign_media_url", side_effect=lambda u: u + "?s"),
):
task_mod._fallback_to_mediakit(db_patch, job)
fake_client.submit_lipsync.assert_called_once()
kwargs = fake_client.submit_lipsync.call_args.kwargs
assert kwargs["enable_video_loop"] is True
assert kwargs["client_token"] == "job-1"
assert job.status == "submitted"
assert job.mediakit_task_id == "mk-99"
db_patch.commit.assert_called_once()
def test_mediakit_error_marks_failed(self, db_patch):
from app.services.mediakit_client import MediaKitError
job = _make_job()
fake_client = MagicMock()
fake_client.submit_lipsync.side_effect = MediaKitError("api reject", code="MkReject")
with (
patch("app.services.mediakit_client.get_mediakit_client", return_value=fake_client),
patch.object(task_mod, "_sign_media_url", side_effect=lambda u: u),
):
task_mod._fallback_to_mediakit(db_patch, job)
assert job.status == "failed"
assert job.error_code == "MkReject"
db_patch.commit.assert_called_once()
def test_other_exception_marks_failed(self, db_patch):
job = _make_job()
with (
patch("app.services.mediakit_client.get_mediakit_client", side_effect=RuntimeError("boom")),
patch.object(task_mod, "_sign_media_url", side_effect=lambda u: u),
):
task_mod._fallback_to_mediakit(db_patch, job)
assert job.status == "failed"
assert job.error_code == "FallbackFailed"
db_patch.commit.assert_called_once()
class TestTaskException:
def test_unexpected_exception_marks_job_failed(self, db_patch):
job = _make_job()
# query 第一次返回 job,异常路径里再次 query 也返回 job
db_patch.query.return_value.filter_by.return_value.first.return_value = job
with patch(
"app.services.gpu_lipsync_service.GpuLipsyncService",
side_effect=RuntimeError("svc ctor fail"),
):
_run_task()
assert job.status == "failed"
assert job.error_code == "GpuAsyncError"
def test_exception_handler_failure_swallowed(self, db_patch):
# 主流程异常,且异常处理中的 query 也抛异常 → 不应再抛
db_patch.query.side_effect = RuntimeError("db totally broken")
_run_task()
db_patch.close.assert_called_once()
class TestSignMediaUrl:
def test_empty_url_returned_as_is(self):
assert task_mod._sign_media_url("") == ""
def test_non_own_host_returned_as_is(self):
storage = MagicMock()
storage.public_url = "https://own-bucket.oss-cn-beijing.aliyuncs.com"
with patch.object(task_mod, "get_shared_storage_service", return_value=storage):
url = "https://other.example.com/a.wav"
assert task_mod._sign_media_url(url) == url
def test_own_host_signed(self):
storage = MagicMock()
storage.public_url = "https://own-bucket.oss-cn-beijing.aliyuncs.com"
storage.get_download_url.return_value = "https://own-bucket.oss-cn-beijing.aliyuncs.com/a?sig=1"
with patch.object(task_mod, "get_shared_storage_service", return_value=storage):
out = task_mod._sign_media_url("https://own-bucket.oss-cn-beijing.aliyuncs.com/a.wav")
assert out.endswith("?sig=1")
storage.get_download_url.assert_called_once()
def test_missing_public_url_returns_original(self):
storage = MagicMock()
storage.public_url = ""
with patch.object(task_mod, "get_shared_storage_service", return_value=storage):
url = "https://own-bucket.oss-cn-beijing.aliyuncs.com/a.wav"
assert task_mod._sign_media_url(url) == url
def test_exception_returns_original(self):
with patch.object(task_mod, "get_shared_storage_service", side_effect=RuntimeError("x")):
url = "https://own-bucket.oss-cn-beijing.aliyuncs.com/a.wav"
assert task_mod._sign_media_url(url) == url
@@ -141,6 +141,46 @@ class TestGpuFallback:
fake_mediakit.submit_lipsync.assert_called_once() fake_mediakit.submit_lipsync.assert_called_once()
assert job.status == "submitted" assert job.status == "submitted"
def test_gpu_create_returns_none_falls_back_mediakit(self, fake_db, fake_mediakit):
"""_submit_to_gpu_create 返回 Nonecreate_task 失败被内部吞掉)→ rollback + MediaKit."""
svc = _make_svc(fake_db, fake_mediakit, use_gpu=True)
fake_gpu_svc = MagicMock()
fake_gpu_svc.has_available_worker.return_value = True
with (
_patch_storage(),
patch("app.services.gpu_lipsync_service.GpuLipsyncService", return_value=fake_gpu_svc),
patch.object(svc, "_submit_to_gpu_create", return_value=None) as m_create,
):
job = _make_job()
svc._submit_audio_direct(job=job)
m_create.assert_called_once()
fake_db.rollback.assert_called_once()
fake_mediakit.submit_lipsync.assert_called_once()
assert job.status == "submitted"
def test_submit_to_gpu_wait_timeout_returns(self, fake_db, fake_mediakit):
"""降级同步等待:wait_for_result 返回 None → 直接返回,job 保持 processing."""
svc = _make_svc(fake_db, fake_mediakit, use_gpu=True)
fake_gpu_svc = MagicMock()
fake_gpu_svc.wait_for_result.return_value = None
job = _make_job()
job.status = "processing"
svc._submit_to_gpu_wait(job=job, gpu_svc=fake_gpu_svc, gpu_task=MagicMock(id="gpu-task-x"))
fake_gpu_svc.wait_for_result.assert_called_once_with("gpu-task-x")
fake_db.commit.assert_not_called()
assert job.status == "processing"
def test_submit_to_gpu_wait_failed_status_returns(self, fake_db, fake_mediakit):
"""降级同步等待:final_task.status != done → 直接返回."""
svc = _make_svc(fake_db, fake_mediakit, use_gpu=True)
fake_gpu_svc = MagicMock()
fake_gpu_svc.wait_for_result.return_value = MagicMock(status="failed", result_url="")
job = _make_job()
job.status = "processing"
svc._submit_to_gpu_wait(job=job, gpu_svc=fake_gpu_svc, gpu_task=MagicMock(id="gpu-task-y"))
fake_db.commit.assert_not_called()
assert job.status == "processing"
def test_gpu_external_audio_persisted_to_own_oss(self, fake_db, fake_mediakit): def test_gpu_external_audio_persisted_to_own_oss(self, fake_db, fake_mediakit):
"""Bug2 回归:dashscope 临时音频 URL 在创建 GPU 任务前转存自家 OSS.""" """Bug2 回归:dashscope 临时音频 URL 在创建 GPU 任务前转存自家 OSS."""
svc = _make_svc(fake_db, fake_mediakit, use_gpu=True) svc = _make_svc(fake_db, fake_mediakit, use_gpu=True)
@@ -212,6 +252,53 @@ class TestGpuFallback:
assert fake_gpu_svc.create_task.call_args.kwargs["audio_url"] == dashscope_url assert fake_gpu_svc.create_task.call_args.kwargs["audio_url"] == dashscope_url
class TestRefreshGpuStale:
"""refresh_job_status 的 GPU 异步 stale 超时分支."""
def test_stale_gpu_job_marked_failed(self, fake_db):
from datetime import UTC, datetime, timedelta
svc = _make_svc(fake_db, MagicMock(), use_gpu=True)
job = MagicMock()
job.status = "processing"
job.mediakit_task_id = "gpu:gpu-task-stale"
job.updated_at = datetime.now(UTC) - timedelta(minutes=31)
with patch.object(svc, "get_job", return_value=job):
result = svc.refresh_job_status("job-stale", "u1")
assert result is job
assert job.status == "failed"
assert job.error_code == "GpuTimeout"
fake_db.commit.assert_called_once()
def test_fresh_gpu_job_left_processing(self, fake_db):
from datetime import UTC, datetime, timedelta
svc = _make_svc(fake_db, MagicMock(), use_gpu=True)
job = MagicMock()
job.status = "processing"
job.mediakit_task_id = "gpu:gpu-task-fresh"
job.updated_at = datetime.now(UTC) - timedelta(minutes=2)
with patch.object(svc, "get_job", return_value=job):
result = svc.refresh_job_status("job-fresh", "u1")
assert result is job
assert job.status == "processing"
fake_db.commit.assert_not_called()
def test_naive_updated_at_stale_marked_failed(self, fake_db):
"""updated_at 为 naive datetime 时按 UTC 补时区后再判定."""
from datetime import UTC, datetime, timedelta
svc = _make_svc(fake_db, MagicMock(), use_gpu=True)
job = MagicMock()
job.status = "gpu_processing"
job.mediakit_task_id = "gpu:gpu-task-naive"
job.updated_at = datetime.now(UTC).replace(tzinfo=None) - timedelta(minutes=31)
with patch.object(svc, "get_job", return_value=job):
svc.refresh_job_status("job-naive", "u1")
assert job.status == "failed"
assert job.error_code == "GpuTimeout"
class TestGpuServiceHelpers: class TestGpuServiceHelpers:
"""GpuLipsyncService.has_available_worker 测试.""" """GpuLipsyncService.has_available_worker 测试."""
+5 -1
View File
@@ -298,6 +298,7 @@ class TestLipsyncServiceUnit:
mock_db = MagicMock() mock_db = MagicMock()
svc = LipsyncService(mock_db, client=mock_mediakit) svc = LipsyncService(mock_db, client=mock_mediakit)
svc.settings.use_gpu_lipsync = False
with pytest.raises(MediaKitError, match="API 调用失败"): with pytest.raises(MediaKitError, match="API 调用失败"):
svc.create_job( svc.create_job(
@@ -473,6 +474,7 @@ class TestLipsyncServiceUnit:
cosyvoice_service=mock_cosyvoice, cosyvoice_service=mock_cosyvoice,
voice_clone_repo=MagicMock(), voice_clone_repo=MagicMock(),
) )
svc.settings.use_gpu_lipsync = False
job = svc.create_job( job = svc.create_job(
user_id="user-1", user_id="user-1",
video_url="https://example.com/video.mp4", video_url="https://example.com/video.mp4",
@@ -616,12 +618,14 @@ class TestSignMediaUrl403Fix:
def _svc(self, mock_mediakit, mock_cosyvoice): def _svc(self, mock_mediakit, mock_cosyvoice):
from app.services.lipsync_service import LipsyncService from app.services.lipsync_service import LipsyncService
return LipsyncService( svc = LipsyncService(
MagicMock(), MagicMock(),
client=mock_mediakit, client=mock_mediakit,
cosyvoice_service=mock_cosyvoice, cosyvoice_service=mock_cosyvoice,
voice_clone_repo=MagicMock(), voice_clone_repo=MagicMock(),
) )
svc.settings.use_gpu_lipsync = False
return svc
def test_own_oss_unsigned_url_gets_resigned(self, mock_mediakit, mock_cosyvoice): def test_own_oss_unsigned_url_gets_resigned(self, mock_mediakit, mock_cosyvoice):
"""裸 public_url(不带签名,私有桶匿名 403)必须被重签.""" """裸 public_url(不带签名,私有桶匿名 403)必须被重签."""
+223
View File
@@ -399,3 +399,226 @@ class TestSubtitleSegment:
def test_is_valid_zero_duration(self): def test_is_valid_zero_duration(self):
seg = SubtitleSegment(start=1, end=1, text="hello") seg = SubtitleSegment(start=1, end=1, text="hello")
assert seg.is_valid is False assert seg.is_valid is False
# ── line_overrides 逐行样式覆盖测试 ─────────────────────────────────────────
class TestBuildLineOverrideTag:
def test_empty_overrides_returns_empty(self):
style = SubtitleStyle()
assert style.build_line_override_tag(0, 1) == ""
def test_no_matching_line_returns_empty(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 1, "color": "#FF0000"}]})
assert style.build_line_override_tag(0, 2) == ""
def test_out_of_range_index_returns_empty(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 5, "color": "#FF0000"}]})
assert style.build_line_override_tag(0, 2) == ""
def test_negative_index_resolves_from_end(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [{"line_index": -1, "color": "#FF0000", "font_size": 48}],
}
)
tag = style.build_line_override_tag(-1, 3) # last line, 3 lines total
assert tag
assert "\\c&H000000FF" in tag # red
assert "\\fs48" in tag
assert tag.startswith("{") and tag.endswith("}")
def test_color_override(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [{"line_index": 0, "color": "#00FF00"}],
}
)
tag = style.build_line_override_tag(0, 1)
assert "\\c&H0000FF00" in tag # green = 00FF00 → bgr 00FF00
def test_font_size_override(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [{"line_index": 0, "font_size": 60}],
}
)
tag = style.build_line_override_tag(0, 1)
assert "\\fs60" in tag
def test_font_override(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [{"line_index": 0, "font": "优设标题黑"}],
}
)
tag = style.build_line_override_tag(0, 1)
assert "\\fn优设标题黑" in tag
def test_bold_on_off(self):
style_bold = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "bold": True}]})
assert "\\b1" in style_bold.build_line_override_tag(0, 1)
style_nobold = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "bold": False}]})
assert "\\b0" in style_nobold.build_line_override_tag(0, 1)
def test_italic_on_off(self):
style_italic = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "italic": True}]})
assert "\\i1" in style_italic.build_line_override_tag(0, 1)
style_noitalic = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "italic": False}]})
assert "\\i0" in style_noitalic.build_line_override_tag(0, 1)
def test_stroke_override(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [{"line_index": 0, "stroke_color": "#0000FF", "stroke_width": 3.0}],
}
)
tag = style.build_line_override_tag(0, 1)
assert "\\3c&H00FF0000" in tag # blue bgr
assert "\\bord3" in tag
def test_shadow_override(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [
{"line_index": 0, "shadow_color": "#000000", "shadow_offset_x": 2, "shadow_offset_y": 3}
],
}
)
tag = style.build_line_override_tag(0, 1)
assert "\\4c&H00000000" in tag # black
assert "\\xshad2" in tag
assert "\\yshad3" in tag
def test_full_combo(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [
{
"line_index": 0,
"color": "#FF0000",
"font_size": 72,
"font": "抖音美好体",
"bold": True,
"italic": False,
"stroke_color": "#FFFFFF",
"stroke_width": 2,
"shadow_color": "#000000",
"shadow_offset_x": 0,
"shadow_offset_y": 4,
}
],
}
)
tag = style.build_line_override_tag(0, 1)
assert "\\c&H000000FF" in tag
assert "\\fs72" in tag
assert "\\fn抖音美好体" in tag
assert "\\b1" in tag
assert "\\i0" in tag
assert "\\3c&H00FFFFFF" in tag
assert "\\bord2" in tag
assert "\\4c&H00000000" in tag
assert "\\xshad0" in tag
assert "\\yshad4" in tag
def test_empty_override_dict_returns_empty(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0}]})
assert style.build_line_override_tag(0, 1) == ""
def test_non_dict_items_filtered(self):
# 从 dict 解析时已过滤,这里手动构造测试
style = SubtitleStyle(line_overrides=["not a dict", {"line_index": 0, "color": "#FF0000"}])
tag = style.build_line_override_tag(0, 1)
assert "\\c&H000000FF" in tag
def test_total_lines_zero_returns_empty(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "color": "#FF0000"}]})
assert style.build_line_override_tag(0, 0) == ""
def test_line_index_none_skipped(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"color": "#FF0000"}]}) # 无 line_index
assert style.build_line_override_tag(0, 1) == ""
class TestApplyLineOverrides:
def test_empty_text_returns_empty(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "color": "#FF0000"}]})
assert style.apply_line_overrides("") == ""
def test_no_overrides_returns_original(self):
style = SubtitleStyle()
assert style.apply_line_overrides("hello\\Nworld") == "hello\\Nworld"
def test_single_line(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "color": "#FF0000", "font_size": 60}]})
result = style.apply_line_overrides("单行标题")
assert result.startswith("{")
assert "单行标题" in result
assert "\\c&H000000FF" in result
def test_multi_line_only_first_line_tagged(self):
style = SubtitleStyle.from_dict({"line_overrides": [{"line_index": 0, "color": "#FF0000", "font_size": 72}]})
result = style.apply_line_overrides("第一行\\N第二行\\N第三行")
lines = result.split("\\N")
assert len(lines) == 3
assert lines[0].startswith("{\\c&H000000FF")
assert "第一行" in lines[0]
assert lines[1] == "第二行" # 无标签
assert lines[2] == "第三行"
def test_multi_line_middle_and_last(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [
{"line_index": 0, "font_size": 72, "bold": True},
{"line_index": -1, "color": "#00FF00", "font_size": 36},
],
}
)
result = style.apply_line_overrides("主标题\\N副标题\\N脚注")
lines = result.split("\\N")
assert lines[0].startswith("{\\fs72\\b1}")
assert lines[1] == "副标题"
assert "\\c&H0000FF00" in lines[2]
assert "\\fs36" in lines[2]
def test_line_without_override_kept_verbatim(self):
style = SubtitleStyle.from_dict(
{
"line_overrides": [{"line_index": 1, "bold": True}],
}
)
result = style.apply_line_overrides("第一行\\N第二行")
lines = result.split("\\N")
assert lines[0] == "第一行"
assert lines[1].startswith("{\\b1}")
assert "第二行" in lines[1]
class TestFromDictLineOverrides:
def test_from_dict_parses_line_overrides(self):
cfg = {
"line_overrides": [
{"line_index": 0, "color": "#FF0000", "bold": True},
{"line_index": 1, "font_size": 36},
],
}
style = SubtitleStyle.from_dict(cfg)
assert len(style.line_overrides) == 2
assert style.line_overrides[0]["line_index"] == 0
assert style.line_overrides[1]["font_size"] == 36
def test_from_dict_filters_non_dict_items(self):
cfg = {"line_overrides": [{"line_index": 0, "color": "#FF0000"}, "bad", None, 123]}
style = SubtitleStyle.from_dict(cfg)
assert len(style.line_overrides) == 1
def test_from_dict_no_field_defaults_empty(self):
style = SubtitleStyle.from_dict({"font": "微软雅黑"})
assert style.line_overrides == []
def test_default_has_empty_list(self):
style = SubtitleStyle()
assert style.line_overrides == []
+12
View File
@@ -101,6 +101,9 @@ class TestTTSPreviewEndpoint:
text="你好世界", text="你好世界",
voice_id="longxiaochun", voice_id="longxiaochun",
speed=1.0, speed=1.0,
style="",
volume=50,
pitch=1.0,
emotion="", emotion="",
language="zh-CN", language="zh-CN",
) )
@@ -148,6 +151,9 @@ class TestTTSPreviewEndpoint:
text="测试", text="测试",
voice_id="v1", voice_id="v1",
speed=1.5, speed=1.5,
style="",
volume=50,
pitch=1.0,
emotion="", emotion="",
language="zh-CN", language="zh-CN",
) )
@@ -334,6 +340,9 @@ class TestTTSPreviewEndpoint:
text="克隆音色测试", text="克隆音色测试",
voice_id="cosyvoice_actual_voice_123", voice_id="cosyvoice_actual_voice_123",
speed=1.0, speed=1.0,
style="",
volume=50,
pitch=1.0,
emotion="", emotion="",
language="zh-CN", language="zh-CN",
) )
@@ -412,6 +421,9 @@ class TestTTSPreviewEndpoint:
text="预设音色测试", text="预设音色测试",
voice_id="longxiaoxia_v3", voice_id="longxiaoxia_v3",
speed=1.0, speed=1.0,
style="",
volume=50,
pitch=1.0,
emotion="", emotion="",
language="zh-CN", language="zh-CN",
) )