Compare commits

...

21 Commits

Author SHA1 Message Date
LingYing Agent 771c5c9a48 style(worker): black格式化edit_plan_service.py 2026-09-14 01:46:55 +08:00
LingYing Agent 53d6b52232 chore: 重新触发CI 2026-09-14 01:46:24 +08:00
lingying 51f236776a fix(worker): 修复批量变体3个P0 bug——配音时长去重/素材区间未传递/节奏模板时机
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 55s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m28s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m13s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m52s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 3m10s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 3m21s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 3m43s
CI/CD Pipeline / Validate - Style (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Security (pull_request) Has been cancelled
CI/CD Pipeline / Build Production API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (pull_request) Has been cancelled
CI/CD Pipeline / Production Browser E2E (pull_request) Has been cancelled
CI/CD Pipeline / Canary Release to Production (pull_request) Has been cancelled
CI/CD Pipeline / CI Gate (pull_request) Has been cancelled
AI Code Review / AI Code Review (pull_request) Has been cancelled
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 62h7m26s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 62h7m33s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 62h8m43s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 62h7m11s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 62h7m12s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 62h7m16s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 62h8m16s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 62h7m0s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 62h7m6s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 62h7m0s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 62h7m5s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 62h8m18s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 62h8m41s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 62h42m32s
2026-09-14 01:14:32 +08:00
xiaoxia 5b931438e0 Merge pull request 'fix(worker): 修复像素扰动color_balance滤镜名错误导致模板编辑器FFmpeg exit=8' (#1880) from fix/colorbalance-filter-name-0913 into develop
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 3s
CI/CD Pipeline / Build Staging API Image (push) Successful in 22s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 23s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m11s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m6s
CI/CD Pipeline / Integration Tests (push) Successful in 3m2s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 3m26s
CI/CD Pipeline / Validate - Style (push) Successful in 4m0s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m44s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 1m47s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 4m19s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m27s
CI/CD Pipeline / Validate - Security (push) Successful in 7m57s
CI/CD Pipeline / Unit Tests (push) Successful in 8m35s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 0s
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1s
AI Code Review / AI Code Review (pull_request) Successful in 24s
CI/CD Pipeline / Validate - Style (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Security (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been cancelled
CI/CD Pipeline / Unit Tests (pull_request) Has been cancelled
CI/CD Pipeline / Integration Tests (pull_request) Has been cancelled
CI/CD Pipeline / Frontend Lint (pull_request) Has been cancelled
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been cancelled
CI/CD Pipeline / PR Build API Image (pull_request) Has been cancelled
CI/CD Pipeline / PR Build Web Image (pull_request) Has been cancelled
CI/CD Pipeline / PR Build Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been cancelled
CI/CD Pipeline / Build Production API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (pull_request) Has been cancelled
CI/CD Pipeline / Production Browser E2E (pull_request) Has been cancelled
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been cancelled
CI/CD Pipeline / Canary Release to Production (pull_request) Has been cancelled
CI/CD Pipeline / CI Gate (pull_request) Has been cancelled
PR Automation / Auto Approve on CI Green (pull_request) Has been cancelled
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been cancelled
Preview Deploy / Deploy Preview Environment (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Failing after 63h7m26s
CI/CD Pipeline / Build Production Web Image (push) Failing after 63h7m30s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 63h14m49s
CI/CD Pipeline / Frontend Lint (push) Failing after 63h16m6s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 63h15m45s
CI/CD Pipeline / PR Build Web Image (push) Failing after 63h15m39s
CI/CD Pipeline / PR Build API Image (push) Failing after 63h15m39s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 63h14m24s
CI/CD Pipeline / CI Gate (push) Failing after 63h7m4s
CI/CD Pipeline / Canary Release to Production (push) Failing after 63h7m1s
CI/CD Pipeline / Build Production API Image (push) Failing after 63h7m5s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 63h7m4s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 63h14m24s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 63h15m39s
2026-09-14 00:07:35 +08:00
灵应 b9acf8c86f fix(worker): 修复像素扰动color_balance滤镜名错误导致ffmpeg exit=8
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 5s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 21s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 32s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Failing after 13m45s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 6m21s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 6m49s
CI/CD Pipeline / Validate - Style (pull_request) Successful in 7m5s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 8m39s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 15m14s
CI/CD Pipeline / CI Gate (pull_request) Successful in 1s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
ACR Cleanup / ACR Image Cleanup (pull_request_target) Successful in 6s
Preview Cleanup / Cleanup Preview Environment (pull_request) Successful in 1m4s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 63h18m24s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 63h47m15s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 63h47m17s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 63h47m19s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 63h18m0s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 63h46m55s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 63h18m0s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 63h46m51s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 63h46m51s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 63h46m53s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 63h46m54s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 63h46m54s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 63h47m4s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 63h46m49s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 63h46m49s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 63h17m59s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 63h18m0s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 63h33m14s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 63h33m15s
根因:像素扰动功能(bb98620)在ffmpeg滤镜链中使用了color_balance滤镜名,
但ffmpeg的实际滤镜名是colorbalance(无下划线),导致滤镜解析失败,
报错'No such filter: color_balance',ffmpeg退出码8。

影响:任何启用了像素扰动的变体渲染(默认批量变体会随机启用)都会失败,
表现为进度卡在56%,error_message只显示ffmpeg版本banner。

修复:将ffmpeg滤镜名从color_balance=改为colorbalance=。
内部config key保持color_balance不变(避免兼容问题)。
2026-09-13 23:36:02 +08:00
xiaoxia 6f4a95e00f Merge pull request 'fix(worker): 改进FFmpeg错误信息可见性便于exit=8排查' (#1879) from fix/ffmpeg-error-visibility-0913 into develop
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 4s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m31s
CI/CD Pipeline / Build Staging API Image (push) Successful in 1m50s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 2m12s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m17s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 4m39s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m21s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 2m57s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 2m59s
CI/CD Pipeline / Validate - Style (push) Successful in 7m13s
CI/CD Pipeline / Integration Tests (push) Successful in 7m45s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 8m31s
CI/CD Pipeline / Unit Tests (push) Successful in 12m18s
CI/CD Pipeline / Validate - Security (push) Successful in 21m39s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Failing after 63h31m51s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 63h51m15s
CI/CD Pipeline / PR Build API Image (push) Failing after 63h53m29s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 63h31m26s
CI/CD Pipeline / Build Production API Image (push) Failing after 63h31m27s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 63h50m50s
CI/CD Pipeline / Deploy Production (push) Failing after 63h31m22s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 63h50m50s
CI/CD Pipeline / PR Build Web Image (push) Failing after 63h53m4s
CI/CD Pipeline / Canary Release to Production (push) Failing after 63h31m22s
CI/CD Pipeline / Build Production Web Image (push) Failing after 63h31m27s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 63h53m4s
CI/CD Pipeline / Frontend Lint (push) Failing after 63h53m5s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 63h53m10s
2026-09-13 23:30:09 +08:00
灵应 ccef935ca7 fix(worker): 改进ffmpeg错误信息可见性,stderr尾部优先+error_detail不丢弃
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 27s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 28s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m30s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m16s
AI Code Review / AI Code Review (pull_request) Successful in 6m27s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 7m11s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 7m43s
CI/CD Pipeline / Validate - Style (pull_request) Successful in 7m45s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 9m9s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 10m32s
ACR Cleanup / ACR Image Cleanup (pull_request_target) Successful in 6s
Preview Cleanup / Cleanup Preview Environment (pull_request) Successful in 1m0s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 22m33s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 4s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 63h46m9s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 64h8m23s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 64h8m32s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 64h8m38s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 64h8m7s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 64h8m14s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 64h8m14s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 64h8m22s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 63h45m44s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 64h7m58s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 64h8m2s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 64h8m7s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 63h45m40s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 63h45m40s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 63h45m44s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 64h7m58s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 64h8m13s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 64h8m16s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 64h8m16s
1. error_message显示stderr尾部500字符(而非头部ffmpeg banner)
2. error_detail保留stderr尾部5000字符
3. generation任务RuntimeError保留error_detail
4. 缩略图ffmpeg错误同理
2026-09-13 23:11:53 +08:00
xiaoxia d7f2e68707 Merge pull request 'fix(lipsync): video_url扩展名校验放宽,支持MOV/M4V/WebM等容器' (#1878) from fix/lipsync-video-ext-0913 into develop
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 14s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m18s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 2m18s
CI/CD Pipeline / Build Staging API Image (push) Successful in 2m23s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m16s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 4m53s
CI/CD Pipeline / Integration Tests (push) Successful in 5m52s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 2m1s
CI/CD Pipeline / Validate - Style (push) Successful in 6m45s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 2m47s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 7m59s
CI/CD Pipeline / Unit Tests (push) Successful in 11m20s
CI/CD Pipeline / Validate - Security (push) Successful in 22m57s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Staging E2E Tests (push) Failing after 21m15s
CI/CD Pipeline / Deploy Production (push) Failing after 64h9m30s
CI/CD Pipeline / Build Production Web Image (push) Failing after 64h9m35s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 64h29m51s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 64h32m36s
CI/CD Pipeline / PR Build Web Image (push) Failing after 64h32m12s
CI/CD Pipeline / PR Build API Image (push) Failing after 64h32m12s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 64h29m27s
CI/CD Pipeline / CI Gate (push) Failing after 64h9m11s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 64h32m13s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 64h29m26s
CI/CD Pipeline / Canary Release to Production (push) Failing after 64h9m6s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 64h9m11s
CI/CD Pipeline / Build Production API Image (push) Failing after 64h9m11s
CI/CD Pipeline / Frontend Lint (push) Failing after 64h32m9s
2026-09-13 22:51:08 +08:00
灵应 78a618f7d7 fix(lint): B904 raise from exc for finalize 500 error handler
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 2s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 32s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 31s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m34s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m39s
AI Code Review / AI Code Review (pull_request) Successful in 6m35s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 6m34s
CI/CD Pipeline / Validate - Style (pull_request) Successful in 7m20s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 8m39s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 12m29s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 10m39s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 22m44s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 4s
ACR Cleanup / ACR Image Cleanup (pull_request_target) Successful in 1m24s
Preview Cleanup / Cleanup Preview Environment (pull_request) Successful in 1m52s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 64h35m51s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 64h58m27s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 64h58m30s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 64h58m33s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 64h58m38s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 64h58m12s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 64h58m13s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 64h58m14s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 64h58m13s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 64h35m27s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 64h35m24s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 64h35m29s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 64h58m16s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 64h35m24s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 64h58m4s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 64h58m7s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 64h58m4s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 64h58m9s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 64h58m14s
2026-09-13 22:25:04 +08:00
灵应 5ab8a66e1b fix(test): 更新smart cover轮询参数断言以匹配新值2s/30次
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 29s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 1m37s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m57s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m47s
AI Code Review / AI Code Review (pull_request) Successful in 6m38s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 7m34s
CI/CD Pipeline / Validate - Style (pull_request) Failing after 8m17s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 8m47s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 13m10s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 10m36s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 25m38s
CI/CD Pipeline / CI Gate (pull_request) Failing after 1s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 64h59m44s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 65h25m16s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 65h25m21s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 65h25m22s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 65h24m49s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 65h24m51s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 65h24m54s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 65h24m57s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 65h24m58s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 65h24m46s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 64h59m15s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 64h59m20s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 65h24m58s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 65h25m4s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 64h59m15s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 64h59m20s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 65h24m46s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 65h24m59s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 66h0m5s
2026-09-13 21:58:12 +08:00
CI Bot a7fc67b822 style: auto-format with black + isort + ruff + prettier [skip ci-format-check]
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 2s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m21s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m44s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 1m46s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m43s
AI Code Review / AI Code Review (pull_request) Successful in 6m35s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 6m58s
CI/CD Pipeline / Validate - Style (pull_request) Failing after 7m10s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 8m28s
CI/CD Pipeline / Unit Tests (pull_request) Failing after 11m58s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 11m6s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 23m43s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Failing after 1s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 65h28m40s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 65h52m28s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 65h52m24s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 65h28m21s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 65h52m1s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 65h52m2s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 65h52m3s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 65h52m8s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 65h52m8s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 65h52m11s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 65h28m20s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 65h28m20s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 65h52m1s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 65h28m16s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 65h52m0s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 65h52m4s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 65h52m8s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 65h52m4s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 66h27m20s
2026-09-13 13:31:12 +00:00
灵应 fc925ef3b3 fix(lipsync): video_url扩展名校验放宽,支持MOV/M4V/WebM/AVI/MKV/3GP
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 2m19s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 2m18s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 2m31s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m56s
AI Code Review / AI Code Review (pull_request) Successful in 6m45s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 6m44s
CI/CD Pipeline / Validate - Style (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Security (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been cancelled
CI/CD Pipeline / Unit Tests (pull_request) Has been cancelled
CI/CD Pipeline / Build Production API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (pull_request) Has been cancelled
CI/CD Pipeline / Production Browser E2E (pull_request) Has been cancelled
CI/CD Pipeline / Canary Release to Production (pull_request) Has been cancelled
CI/CD Pipeline / CI Gate (pull_request) Has been cancelled
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 66h0m2s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 66h0m14s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 66h0m14s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 66h0m18s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 65h59m50s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 65h59m44s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 65h59m49s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 65h59m50s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 65h59m54s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 65h59m54s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 65h59m39s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 65h59m35s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 65h59m50s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 65h59m56s
iPhone等设备拍摄的视频转h264后容器保留.MOV扩展名,
原校验仅允许.mp4导致对口型接口返回422。
同步更新单测。
2026-09-13 21:23:19 +08:00
CI Bot b69595bd28 style: auto-format with black + isort + ruff + prettier [skip ci-format-check]
CI/CD Pipeline / Check push changed paths (push) Successful in 37s
CI/CD Pipeline / Build Staging API Image (push) Successful in 13s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 18s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 30s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 55s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m23s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 2m43s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m24s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Failing after 12m7s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m53s
CI/CD Pipeline / Integration Tests (push) Successful in 5m57s
CI/CD Pipeline / Validate - Style (push) Failing after 7m28s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 8m1s
CI/CD Pipeline / Unit Tests (push) Failing after 11m3s
CI/CD Pipeline / Validate - Security (push) Successful in 23m12s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Failing after 67h10m28s
CI/CD Pipeline / Build Production API Image (push) Failing after 67h10m28s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 67h44m41s
CI/CD Pipeline / PR Build Web Image (push) Failing after 67h45m17s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 67h10m4s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 67h44m53s
CI/CD Pipeline / PR Build API Image (push) Failing after 67h44m53s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 67h44m17s
CI/CD Pipeline / Deploy Production (push) Failing after 67h9m59s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 67h44m17s
CI/CD Pipeline / Canary Release to Production (push) Failing after 67h9m59s
CI/CD Pipeline / Build Production Web Image (push) Failing after 67h10m4s
CI/CD Pipeline / Frontend Lint (push) Failing after 67h33m18s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 67h45m28s
2026-09-13 11:37:46 +00:00
xiaoxia 5c91bb21a7 Merge pull request 'fix(ai-avatar): 智能封面抽帧超时修复 + 封面流程改造(选封面后点完成才入库)' (#1877) from fix/ai-avatar-cover-flow-0913 into develop
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check push changed paths (push) Successful in 14s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m35s
CI/CD Pipeline / Build Staging API Image (push) Successful in 1m52s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 2m15s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m23s
CI/CD Pipeline / Integration Tests (push) Successful in 4m30s
CI/CD Pipeline / Validate - Style (push) Has been cancelled
CI/CD Pipeline / Validate - Security (push) Has been cancelled
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Unit Tests (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build Production API Image (push) Has been cancelled
CI/CD Pipeline / Build Production Web Image (push) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
CI/CD Pipeline / ACR Image Cleanup (push) Has been cancelled
CI/CD Pipeline / Canary Release to Production (push) Has been cancelled
CI/CD Pipeline / CI Gate (push) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 67h48m28s
CI/CD Pipeline / PR Build Web Image (push) Failing after 67h50m58s
CI/CD Pipeline / PR Build API Image (push) Failing after 67h50m35s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 67h48m5s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 67h50m34s
CI/CD Pipeline / Frontend Lint (push) Failing after 67h50m35s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 67h50m38s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 67h48m4s
2026-09-13 19:32:43 +08:00
CI Bot 135e422d45 style: auto-format with black + isort + ruff + prettier [skip ci-format-check]
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 4s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 51s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 51s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 51s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 1m39s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m53s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Successful in 1m50s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 2m1s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 2m11s
CI/CD Pipeline / Validate - Style (pull_request) Failing after 2m14s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m13s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 5m2s
AI Code Review / AI Code Review (pull_request) Successful in 6m21s
CI/CD Pipeline / Unit Tests (pull_request) Failing after 8m51s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Failing after 4s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 5m47s
ACR Cleanup / ACR Image Cleanup (pull_request_target) Successful in 7s
Preview Cleanup / Cleanup Preview Environment (pull_request) Successful in 36s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 67h58m7s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 68h5m29s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 68h5m41s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 68h5m17s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 68h5m43s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 68h5m44s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 68h5m45s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 67h57m43s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 67h57m40s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 67h57m44s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 67h57m40s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 68h5m5s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 68h5m7s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 68h5m18s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 68h6m42s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 68h40m30s
2026-09-13 11:16:39 +00:00
灵应 60d19704bb fix(ai-avatar): 抽帧超时延长至60s + 封面选完才入库流程
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 2s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 1m25s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 1m29s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 2m11s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 2m41s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 2m49s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Successful in 3m9s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m9s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 4m0s
CI/CD Pipeline / Validate - Style (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Security (pull_request) Has been cancelled
CI/CD Pipeline / Unit Tests (pull_request) Has been cancelled
CI/CD Pipeline / Integration Tests (pull_request) Has been cancelled
CI/CD Pipeline / Build Production API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (pull_request) Has been cancelled
CI/CD Pipeline / Production Browser E2E (pull_request) Has been cancelled
CI/CD Pipeline / Canary Release to Production (pull_request) Has been cancelled
CI/CD Pipeline / CI Gate (pull_request) Has been cancelled
AI Code Review / AI Code Review (pull_request) Has been cancelled
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 68h11m50s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 68h11m27s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 68h11m27s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 68h11m20s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 68h11m26s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 68h11m3s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 68h11m9s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 68h11m15s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 68h11m21s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 68h11m34s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 68h46m39s
- ai_avatar_cover_service: COVER_POLL_INTERVAL=2s × COVER_MAX_POLL_ATTEMPTS=30 (最长 60s),适配合成视频(叠加B-roll/标题)较大的下载+抽帧耗时
- ai_avatar_render_service: execute_render 完成后停留在 completed 状态不再自动入库成片库;新增 _persist_to_library(cover_url 可选覆盖) 与 finalize_job(job_id, user_id, cover_url?) 方法,供用户点「完成」时显式入库,幂等(通过 generation_task_id=render_job_id 判重)
- 路由: 新增 POST /{job_id}/finalize 接口,含 404/400/幂等/异常分支
- Schema: 新增 FinalizeRenderResponse
- 前端:
  - aiAvatar.ts 新增 finalizeRenderJob API
  - PanelCoverAndGenerate 拆分 variant=setup(主页面配置+生成按钮,不含封面区) 与 select-cover(弹窗内封面选择);新增 onClose/onCoverSelected props,智能/上传成功后回调通知父组件
  - 新建 ModalCoverSelect 弹窗组件
  - AiAvatarPage: render 完成后不再自动跳转/generated,而是停留在主页面显示封面预览+「🎬选择封面」「完成」按钮;弹窗内选封面、点「完成」才调 finalize 入库并跳转到成片库
- 单元测试: 更新原有 execute_render 自动入库测试为不入库断言,新增 finalize_job 入库/幂等/状态校验 3 个测试
2026-09-13 19:09:39 +08:00
xiaoxia c613f35662 Merge pull request 'fix(ai-avatar): 修复三个bug——封面重影/标题字号缩放/对口型音频截断' (#1876) from fix/ai-avatar-three-bugs-0913 into develop
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 10s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 17s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 19s
CI/CD Pipeline / Build Staging API Image (push) Successful in 1m0s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 1m53s
CI/CD Pipeline / Integration Tests (push) Successful in 1m59s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m0s
CI/CD Pipeline / Validate - Style (push) Successful in 2m39s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m35s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 1m35s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 5m21s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m15s
CI/CD Pipeline / Validate - Security (push) Successful in 5m37s
CI/CD Pipeline / Unit Tests (push) Successful in 8m32s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Failing after 72h59m35s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 73h8m10s
CI/CD Pipeline / Frontend Lint (push) Failing after 73h8m11s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 73h7m54s
CI/CD Pipeline / PR Build Web Image (push) Failing after 73h7m48s
CI/CD Pipeline / PR Build API Image (push) Failing after 73h7m48s
CI/CD Pipeline / Deploy Production (push) Failing after 72h59m8s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 72h59m12s
CI/CD Pipeline / Build Production Web Image (push) Failing after 72h59m12s
CI/CD Pipeline / Build Production API Image (push) Failing after 72h59m12s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 73h6m41s
CI/CD Pipeline / Canary Release to Production (push) Failing after 72h59m8s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 73h6m41s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 73h41m58s
2026-09-13 14:15:31 +08:00
CI Bot 9be89484e6 style: auto-format with black + isort + ruff + prettier [skip ci-format-check]
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 5s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 26s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 32s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 47s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 49s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m50s
AI Code Review / AI Code Review (pull_request) Successful in 6m16s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Failing after 12m8s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Successful in 1m2s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 10m34s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 1m36s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 1m43s
CI/CD Pipeline / Validate - Style (pull_request) Successful in 2m4s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 5m28s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 9m34s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / CI Gate (pull_request) Successful in 4s
ACR Cleanup / ACR Image Cleanup (pull_request_target) Successful in 10s
Preview Cleanup / Cleanup Preview Environment (pull_request) Successful in 17s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 73h50m43s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 73h29m8s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 73h50m44s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 73h50m44s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 73h50m21s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 73h50m21s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 73h28m45s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 73h28m42s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 73h50m20s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 73h50m20s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 73h50m21s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 73h50m26s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 73h28m42s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 73h28m45s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 73h38m21s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 73h50m20s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 73h50m21s
2026-09-13 05:32:51 +00:00
灵应 5f128d175d fix(ai-avatar): 修复三个bug——封面重影/标题字号缩放/对口型音频截断
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 2s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m12s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 1m28s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 1m33s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m49s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 1m48s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 2m16s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Successful in 2m28s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 2m54s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m8s
CI/CD Pipeline / Validate - Style (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Security (pull_request) Has been cancelled
CI/CD Pipeline / Unit Tests (pull_request) Has been cancelled
CI/CD Pipeline / Build Production API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (pull_request) Has been cancelled
CI/CD Pipeline / Production Browser E2E (pull_request) Has been cancelled
CI/CD Pipeline / Canary Release to Production (pull_request) Has been cancelled
CI/CD Pipeline / CI Gate (pull_request) Has been cancelled
AI Code Review / AI Code Review (pull_request) Has been cancelled
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 73h54m41s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 73h54m24s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 73h54m13s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 73h54m18s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 73h54m0s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 73h54m0s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 73h54m7s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 73h54m12s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 73h54m12s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 73h54m18s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 73h54m19s
1. 删除封面展示区重复叠加的标题div(成片帧已含标题,不再CSS叠字)
2. 标题字号按视频分辨率等比缩放,预览容器动态测量宽度计算scale
3. enable_video_loop默认改为true,防止TTS音频长于出镜视频时被截断
2026-09-13 13:28:51 +08:00
xiaoxia 053b00634a feat(ai-avatar): 配音前置 (#1875)
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 3s
CI/CD Pipeline / Check push changed paths (push) Successful in 4s
CI/CD Pipeline / Build Staging API Image (push) Successful in 23s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 23s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 26s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 49s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m32s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m38s
CI/CD Pipeline / Integration Tests (push) Successful in 3m50s
CI/CD Pipeline / Validate - Style (push) Successful in 4m2s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m9s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m43s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m26s
CI/CD Pipeline / Unit Tests (push) Successful in 8m55s
CI/CD Pipeline / Validate - Security (push) Successful in 9m7s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Failing after 82h52m10s
CI/CD Pipeline / Deploy Production (push) Failing after 82h52m6s
CI/CD Pipeline / Frontend Lint (push) Failing after 83h1m19s
CI/CD Pipeline / PR Build Web Image (push) Failing after 83h0m59s
CI/CD Pipeline / PR Build API Image (push) Failing after 83h0m59s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 83h0m28s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 83h0m28s
CI/CD Pipeline / CI Gate (push) Failing after 82h51m46s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 83h1m1s
CI/CD Pipeline / Canary Release to Production (push) Failing after 82h51m42s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 82h51m46s
CI/CD Pipeline / Build Production API Image (push) Failing after 82h51m46s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 83h0m59s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 83h0m28s
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-09-13 04:22:21 +08:00
xiaoxia 3858acf377 Merge pull request 'fix(ai-avatar): B-roll时间戳根因——逗号分句+优先后端时间戳' (#1874) from fix/ai-avatar-broll-comma-split into develop
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check push changed paths (push) Successful in 15s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m37s
CI/CD Pipeline / Build Staging API Image (push) Successful in 2m6s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 2m32s
CI/CD Pipeline / Integration Tests (push) Successful in 3m25s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 59s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m22s
CI/CD Pipeline / Validate - Style (push) Successful in 4m25s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 2m1s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 6m28s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m25s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m7s
CI/CD Pipeline / Validate - Security (push) Successful in 9m30s
CI/CD Pipeline / Unit Tests (push) Successful in 9m39s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Failing after 84h24m57s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 84h31m50s
CI/CD Pipeline / PR Build Web Image (push) Failing after 84h34m38s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 84h31m27s
CI/CD Pipeline / CI Gate (push) Failing after 84h24m38s
CI/CD Pipeline / Build Production API Image (push) Failing after 84h24m39s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 84h31m26s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 84h34m15s
CI/CD Pipeline / PR Build API Image (push) Failing after 84h34m16s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 84h34m21s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 84h24m38s
CI/CD Pipeline / Canary Release to Production (push) Failing after 84h24m33s
CI/CD Pipeline / Frontend Lint (push) Failing after 84h34m16s
CI/CD Pipeline / Build Production Web Image (push) Failing after 84h59m56s
fix(ai-avatar): B-roll逗号分句+标题PNG图层+TTS竞态修复+死代码清理

- B-roll分句正则加入中文逗号,放宽后端时间戳校验直接使用
- 标题改为Canvas渲染透明PNG,FFmpeg overlay图片图层替代drawtext,所见即所得
- Celery事务竞态修复:先commit后发任务+worker侧retry防御
- TTS任务超时防护(180s soft/200s hard)+入口日志
- 删除封面死代码(POST /smart-cover裸视频抽帧,所有封面从成片获取)
- 删除DRAWTEXT_BOLD_FONT_SEARCH_PATHS冗余代码
2026-09-13 02:49:04 +08:00
29 changed files with 2130 additions and 741 deletions
@@ -18,6 +18,7 @@ from app.dependencies import get_db_session
from app.schemas.ai_avatar_render import (
AiAvatarRenderJobResponse,
CreateAiAvatarRenderRequest,
FinalizeRenderResponse,
SmartCoverResponse,
)
from app.services.ai_avatar_cover_service import generate_smart_cover
@@ -257,3 +258,67 @@ def generate_render_smart_cover(
cover_url[:120],
)
return SmartCoverResponse(cover_url=cover_url, status="completed")
# ── POST /{job_id}/finalize — 封面选定后正式入库成片库 ────────────────────
@router.post("/{job_id}/finalize", response_model=FinalizeRenderResponse)
def finalize_render_job(
job_id: str,
current_user: AuthenticatedUser = Depends(get_current_user),
db: Session = Depends(get_db_session),
):
"""用户完成封面选择后,将视频正式保存到成片库.
- 必须等渲染任务 completed 后才可调用
- 如果已通过 smart-cover/custom-cover 设置了封面,会自动带上
- 返回成片库视频ID
- 幂等:已 finalize 的任务重复调用会返回 existing 记录
"""
from app.services.ai_avatar_render_service import AiAvatarRenderError, AiAvatarRenderService
svc = AiAvatarRenderService(db)
job = svc.get_render_job(job_id, current_user.user.id)
if job is None:
raise HTTPException(status_code=404, detail="渲染任务不存在")
if job.status != "completed":
raise HTTPException(status_code=400, detail="请先完成视频生成")
# 幂等检查(通过 generation_task_id=job_id 识别,finalize_job 内部也做了一次,这里提前返回简化)
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
existing = (
db.query(GeneratedVideoModel)
.filter(
GeneratedVideoModel.user_id == current_user.user.id,
GeneratedVideoModel.generation_task_id == job_id,
)
.first()
)
if existing is not None:
return FinalizeRenderResponse(
video_id=existing.id,
cover_url=existing.thumbnail_url or "",
status="already_finalized",
)
try:
video = svc.finalize_job(job_id, current_user.user.id)
return FinalizeRenderResponse(
video_id=video.id,
cover_url=video.thumbnail_url or job.output_cover_url or "",
status="success",
)
except AiAvatarRenderError as exc:
status_map = {
"RenderJobNotFound": 404,
"RenderNotCompleted": 400,
"OutputVideoMissing": 400,
}
raise HTTPException(
status_code=status_map.get(exc.code, 400),
detail=str(exc),
) from exc
except Exception as exc:
logger.error("渲染任务finalize失败: job_id=%s err=%s", job_id, exc, exc_info=True)
raise HTTPException(status_code=500, detail=f"保存到成片库失败: {str(exc)}") from exc
+50 -7
View File
@@ -61,24 +61,38 @@ def _query_voice_durations(db: Session, voice_ids: list[str]) -> list[float]:
"""批量查询配音素材时长(秒),#1749 配音时长分配用。
逐项 try/float 硬化:MagicMock/异常/缺失 → 0.0(无配音不分配,不阻断)。
#1855 P0修复:不再对 voice_ids 去重,保持与调用方传入顺序/长度一致,
允许同配音id多次出现时返回相同时长(支持"同配音N变体"的时长对齐)。
"""
ids = [v for v in dict.fromkeys(voice_ids or []) if v]
if not ids:
# 先去重查询(IN 查询性能优化),但最终按原始 voice_ids 顺序返回
raw_ids = list(voice_ids or [])
if not raw_ids:
return []
# 去重且保序,用于 SQL IN 查询;空字符串/None 视为无效id → 0.0
unique_ids: list[str] = []
_seen: set[str] = set()
for v in raw_ids:
if v and v not in _seen:
_seen.add(v)
unique_ids.append(v)
if not unique_ids:
return [0.0 for _ in raw_ids]
try:
from packages.adapters.sqlalchemy_impl.models import AssetModel
rows = db.query(AssetModel.id, AssetModel.duration).filter(AssetModel.id.in_(ids)).all()
rows = db.query(AssetModel.id, AssetModel.duration).filter(AssetModel.id.in_(unique_ids)).all()
dur_map: dict[str, float] = {}
for row in rows:
try:
dur_map[row[0]] = float(row[1] or 0.0)
except (TypeError, ValueError):
dur_map[row[0]] = 0.0
return [dur_map.get(v, 0.0) for v in ids]
# 按原始 voice_ids 顺序返回,保持长度一致;空/None/未查到 → 0.0
return [dur_map.get(v, 0.0) if v else 0.0 for v in raw_ids]
except Exception:
logger.warning("[生成任务] 配音时长查询失败(按无配音处理,不阻断)", exc_info=True)
return [0.0 for _ in ids]
return [0.0 for _ in raw_ids]
def _to_generation_task_response(task) -> GenerationTaskResponse:
@@ -552,7 +566,26 @@ def create_generation_task(
) from clone_err
variant_plan_ids.append(_plan0.id)
# 变体 1..N-1 独立选片
# #1855 P0:批次区间避让表,从变体0实际clips构建初始值
def _collect_segments(pid):
segs = {}
_sk, _pg = 0, 500
while True:
_b = _plan_svc._clip_repo.list_by_plan(pid, skip=_sk, limit=_pg)
if not _b:
break
for _c in _b:
if _c.asset_id and float(_c.duration or 0) > 0:
_st = float(_c.start_time or 0.0)
segs.setdefault(_c.asset_id, []).append((_st, _st + float(_c.duration)))
if len(_b) < _pg:
break
_sk += _pg
return segs
_batch_segments = _collect_segments(_plan0.id)
# 变体 1..N-1 独立选片(传入累积batch_segments做素材区间避让)
for task_index in range(1, count):
variant = None
last_err: Exception | None = None
@@ -564,6 +597,7 @@ def create_generation_task(
created_by_user_id=user_id,
name_suffix=f"批量{task_index + 1}",
voice_duration=voice_durations[task_index] if task_index < len(voice_durations) else 0.0,
batch_segments=_batch_segments,
)
break
except ValueError as ve:
@@ -595,7 +629,16 @@ def create_generation_task(
) from last_err
variant_plan_ids.append(variant.id)
# ③ 配音时长分配(回传 plan / clone 变体0 均需幂等分配;reselect 已在选片时分配)
# #1855 P0:把新变体的clips区间追加到batch_segments,供下一变体避让
try:
_new_segs = _collect_segments(variant.id)
for _aid, _ivs in _new_segs.items():
_batch_segments.setdefault(_aid, []).extend(_ivs)
except Exception:
logger.exception("[生成任务] 变体%d 区间收集失败(不阻断)", task_index)
# ③ 配音时长分配(回传 plan / clone 变体0 均需幂等分配;reselect 已在选片时分配,
# #1855apply_voice_duration_to_plan 已内置幂等判断,重复调用安全)
for _vi, _pid in enumerate(variant_plan_ids):
_vd = voice_durations[_vi] if _vi < len(voice_durations) else 0.0
if _vd > 0:
+64 -15
View File
@@ -1,11 +1,12 @@
"""对口型 API 路由 — #1796 MediaKit 对口型, #1809 参数调整.
"""对口型 API 路由 — #1796 MediaKit 对口型, #1809 参数调整, #1845 配音前置.
接口:
POST /api/v1/lipsync/jobs 提交对口型任务
POST /api/v1/lipsync/jobs 提交对口型任务(支持 TTS/直传/预合成 三种模式)
GET /api/v1/lipsync/jobs 任务列表
GET /api/v1/lipsync/jobs/{id} 任务详情
POST /api/v1/lipsync/jobs/{id}/refresh 刷新任务状态
POST /api/v1/lipsync/jobs/{id}/cancel 取消任务
POST /api/v1/lipsync/tts-preview #1845 步骤1 TTS 预合成(同步 HTTP~2-3s
"""
from __future__ import annotations
@@ -17,7 +18,12 @@ from app.dependencies import (
get_db_session,
get_voice_clone_profile_repository,
)
from app.schemas.lipsync import CreateLipsyncJobRequest, LipsyncJobResponse
from app.schemas.lipsync import (
AiAvatarTtsPreviewRequest,
AiAvatarTtsPreviewResponse,
CreateLipsyncJobRequest,
LipsyncJobResponse,
)
from app.services.lipsync_service import LipsyncService
from app.services.mediakit_client import MediaKitError
from fastapi import APIRouter, BackgroundTasks, Depends, HTTPException, Query
@@ -33,7 +39,6 @@ def _get_service(
voice_clone_repo=Depends(get_voice_clone_profile_repository),
) -> LipsyncService:
# voice_clone_repo 用于克隆音色 profile 解析
# TTS 合成已移至 Celery 异步任务,无需同步注入 cosyvoice_service
return LipsyncService(
db,
voice_clone_repo=voice_clone_repo,
@@ -51,15 +56,20 @@ def create_lipsync_job(
):
"""提交对口型任务.
#1809/#1822: 前端传 {video_url, voice_id, script_text, speed?, emotion?}
后端创建任务记录(状态 tts_processing),dispatch Celery 异步任务执行 TTS 合成 + MediaKit 提交;
也支持直接传 {video_url, audio_url}(同步提交 MediaKit
三种模式:
- TTS 直生(旧版/降级):传 {video_url, voice_id, script_text, speed?, emotion?}
后端 dispatch Celery 异步任务
- 直接音频:传 {video_url, audio_url},后端同步下载+算timings+提交MediaKit。
- 预合成音频(#1845 新主路径):传 {video_url, audio_url, audio_duration, sentence_timings}
后端同步ffprobe+写入timings+直接提交MediaKit~2-3s)。
"""
try:
job = svc.create_job(
user_id=current_user.user.id,
video_url=body.video_url,
audio_url=body.audio_url,
audio_duration=body.audio_duration,
sentence_timings=body.sentence_timings,
voice_id=body.voice_id,
script_text=body.script_text,
speed=body.speed,
@@ -68,10 +78,8 @@ def create_lipsync_job(
project_id=body.project_id,
)
except ValueError as exc:
# 参数无效(如 voice_id 格式不对、文本过长等)
raise HTTPException(status_code=400, detail=str(exc)) from exc
except MediaKitError as exc:
# 音色无权访问 → 403;参数无效 → 400MediaKit 提交失败 → 502
status_code = 502
if exc.code in ("VoiceForbidden",):
status_code = 403
@@ -86,7 +94,6 @@ def create_lipsync_job(
},
) from exc
except Exception as exc:
# 兜底:任何未预期的错误返回 400 而非 500
logger.error("创建对口型任务异常: %s", exc, exc_info=True)
raise HTTPException(
status_code=400,
@@ -96,6 +103,52 @@ def create_lipsync_job(
return job
# ── POST /tts-preview — #1845 步骤1 TTS 预合成 ──────────────────────────
@router.post("/tts-preview", response_model=AiAvatarTtsPreviewResponse)
def preview_tts(
body: AiAvatarTtsPreviewRequest,
current_user: AuthenticatedUser = Depends(get_current_user),
svc: LipsyncService = Depends(_get_service),
):
"""步骤1「生成配音」同步 TTS 预合成.
同步执行 TTS 合成 → 下载音频 → ffprobe 时长 → 句子时间戳计算,
不创建 LipsyncJob、不转存 OSS,直接返回 CosyVoice 临时 URL~24h 有效)。
耗时约 2-3 秒。
"""
try:
result = svc.preview_tts(
user_id=current_user.user.id,
voice_id=body.voice_id,
script_text=body.script_text,
speed=body.speed,
emotion=body.emotion,
)
except MediaKitError as exc:
status_code = 400
if exc.code in ("VoiceForbidden",):
status_code = 403
elif exc.code in ("TTSNoAudio",):
status_code = 502
raise HTTPException(
status_code=status_code,
detail={
"code": exc.code,
"message": str(exc),
},
) from exc
except Exception as exc:
logger.error("TTS 预合成异常: %s", exc, exc_info=True)
raise HTTPException(
status_code=400,
detail=f"TTS 合成失败: {exc}",
) from exc
return result
# ── GET /jobs — 任务列表 ─────────────────────────────────────────────────
@@ -134,11 +187,7 @@ def get_lipsync_job(
current_user: AuthenticatedUser = Depends(get_current_user),
svc: LipsyncService = Depends(_get_service),
):
"""获取对口型任务详情.
非终态任务:先返回 DB 缓存,挂后台刷新(下次轮询拿到新状态),
避免 MediaKit 慢响应阻塞前端轮询。
"""
"""获取对口型任务详情."""
job = svc.get_job(job_id, current_user.user.id)
if job is None:
raise HTTPException(status_code=404, detail="任务不存在")
+8
View File
@@ -116,3 +116,11 @@ class SmartCoverResponse(BaseModel):
cover_url: str = Field("", description="封面图公网 URL(OSS,非临时);失败为空")
status: str = Field("completed", description="completed / fallback_failed")
message: str = Field("", description="失败原因(如有)")
class FinalizeRenderResponse(BaseModel):
"""封面选好后点「完成」,正式入库成片库的响应."""
video_id: str = Field(..., description="成片库视频ID")
cover_url: str = Field("", description="封面URL")
status: str = Field("success", description="success/already_finalized")
+41 -11
View File
@@ -1,9 +1,12 @@
"""对口型 API Schema 定义 — #1796 / #1809 / #1822.
"""对口型 API Schema 定义 — #1796 / #1809 / #1822 / #1845(配音前置).
支持种输入模式(二选一)
1. TTS 直生模式(推荐):传 voice_id + script_text+ speed/emotion),
后端内部先调 CosyVoice 合成音频,再提交 MediaKit 对口型
支持种输入模式:
1. TTS 直生模式(兼容旧版前端):传 voice_id + script_text+ speed/emotion),
后端 Celery 异步做 TTS 合成 + MediaKit 提交
2. 直接音频模式:传 video_url + audio_url(音频已由调用方准备好)。
3. 预合成音频模式(#1845 配音前置新主路径):前端先调 POST /lipsync/tts-preview
拿到 audio_url + sentence_timings,再在 create_job 时传 audio_url + audio_duration
+ sentence_timings,后端跳过 TTS 和时间戳计算,直接 ffprobe 校验后提交 MediaKit。
"""
from __future__ import annotations
@@ -46,15 +49,19 @@ class LipsyncJobResponse(BaseModel):
class CreateLipsyncJobRequest(BaseModel):
"""创建对口型任务请求.
种模式(选一):
- TTS 直生:voice_id + script_text 必填+ 可选 speed/emotionaudio_url 留空。
种模式(选一):
- TTS 直生(旧版/降级)voice_id + script_text 必填;audio_url 留空。
- 直接音频:video_url + audio_url 必填。
- 预合成音频(#1845 新主路径):audio_url 必填 + 可选 audio_duration/sentence_timings
后端同步 ffprobe 校验时长、写入 timings,直接提交 MediaKit。
"""
video_url: str = Field(..., description="人物视频 URL(MP4,≤30min,单人真人)")
# 模式 2:直接音频
# 模式 2/3:直接/预合成音频
audio_url: str = Field("", description="驱动音频 URLmp3/aac/wav/m4a/flac);直生模式留空")
audio_duration: Optional[float] = Field(None, ge=0, description="预合成音频时长(秒),可选;后端会 ffprobe 校验")
sentence_timings: Optional[list] = Field(None, description="预合成接口返回的句子时间戳,可选;若传入则直接写入 job")
# 模式 1TTS 直生
voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID")
@@ -62,7 +69,9 @@ class CreateLipsyncJobRequest(BaseModel):
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
emotion: str = Field("", description="情绪(natural/excited/calm/friendly 或中文 自然/兴奋/沉稳/亲切)")
enable_video_loop: bool = Field(False, description="音频长于视频时是否循环画面")
enable_video_loop: bool = Field(
True, description="音频长于视频时是否循环画面(AI数字人默认开启,防止音频长于视频被截断)"
)
project_id: str = Field("", description="项目 ID(可选)")
@model_validator(mode="after")
@@ -73,15 +82,16 @@ class CreateLipsyncJobRequest(BaseModel):
if not video.startswith(("http://", "https://")):
raise ValueError("video_url 必须是 HTTP/HTTPS URL")
lower = video.lower().split("?")[0]
if not lower.endswith(".mp4"):
raise ValueError("video_url 仅支持 MP4 格式")
allowed_video_exts = (".mp4", ".mov", ".m4v", ".webm", ".avi", ".mkv", ".3gp")
if not any(lower.endswith(ext) for ext in allowed_video_exts):
raise ValueError("video_url 格式不支持,仅支持: " + ", ".join(allowed_video_exts))
has_audio = bool((self.audio_url or "").strip())
has_tts = bool((self.voice_id or "").strip()) and bool((self.script_text or "").strip())
if not has_audio and not has_tts:
raise ValueError(
"必须提供驱动音频:要么传 audio_url(直接音频模式),"
"必须提供驱动音频:要么传 audio_url(直接/预合成音频模式),"
"要么同时传 voice_id + script_textTTS 直生模式)"
)
@@ -99,3 +109,23 @@ class CreateLipsyncJobRequest(BaseModel):
self.audio_url = au
return self
# ── #1845 TTS 预合成接口 ────────────────────────────────────────────────
class AiAvatarTtsPreviewRequest(BaseModel):
"""步骤1「生成配音」预合成请求(同步 HTTP,~2-3s)."""
voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID")
script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
emotion: str = Field("natural", max_length=32, description="情绪")
class AiAvatarTtsPreviewResponse(BaseModel):
"""TTS 预合成响应(临时 URL,24h 内有效,足够当前会话使用)."""
audio_url: str = Field(..., description="CosyVoice 临时音频 URL")
duration: float = Field(..., ge=0, description="音频总时长(秒),ffprobe 测得")
sentence_timings: list[dict] = Field(..., description="句子级精确时间戳")
@@ -22,9 +22,9 @@ from urllib.parse import urlparse
logger = logging.getLogger(__name__)
# MediaKit 抽帧轮询参数:poll_interval=1s × max_poll=15 → 最长 15s,配合前端 120s 超时足够
COVER_POLL_INTERVAL = 1.0
COVER_MAX_POLL_ATTEMPTS = 15
# MediaKit 抽帧轮询参数:poll_interval=2s × max_poll=30 → 最长 60s(与 mediakit_client 默认值/lipsync 轮询保持一致,防止合成视频下载+抽帧超时)
COVER_POLL_INTERVAL = 2.0
COVER_MAX_POLL_ATTEMPTS = 30
# 帧图片下载超时(秒)
FRAME_DOWNLOAD_TIMEOUT = 20
@@ -399,41 +399,9 @@ class AiAvatarRenderService:
self.db.commit()
logger.info("渲染任务完成: %s", job_id)
# 7. 自动保存成片记录到成片库
if job.output_video_url:
try:
from packages.adapters.sqlalchemy_impl.generated_video_repository import (
SQLAlchemyGeneratedVideoRepository,
)
from packages.domain.generated_video import GeneratedVideo
clip_name = f"AI数字人_{job_id[:8]}"
# AI数字人入口是独立页面,前端可能不传 project_id(无项目概念),
# 兜底为 "ai_avatar" 避免 DB 非空约束/查询问题;generation_task_id 同样兜底用 render_job_id
clip_project_id = (job.project_id or "").strip() or "ai_avatar"
clip_generation_task_id = (job.lipsync_job_id or "").strip() or job_id
clip = GeneratedVideo.create(
project_id=clip_project_id,
generation_task_id=clip_generation_task_id,
name=clip_name,
file_url=job.output_video_url,
user_id=job.user_id,
duration=job.output_duration or 0.0,
thumbnail_url=job.output_cover_url or None,
generation_params={
"source": "ai_avatar_render",
"render_job_id": job.id,
},
)
video_repo = SQLAlchemyGeneratedVideoRepository(self.db)
video_repo.create(clip)
logger.info("成片记录已保存到成片库: clip_id=%s, render_job=%s", clip.id, job_id)
except Exception:
logger.error(
"自动保存成片记录失败(不影响渲染任务状态): render_job=%s",
job_id,
exc_info=True,
)
# 7. 渲染完成,停留在「待选封面」状态:不自动入库。
# 用户在前端选好封面、点「完成」后,由 /{job_id}/finalize 接口显式入库。
logger.info("渲染任务完成,等待用户选择封面后入库: job_id=%s", job_id)
except AiAvatarRenderError as exc:
job.status = "failed"
@@ -450,6 +418,87 @@ class AiAvatarRenderService:
logger.exception("渲染任务异常 [%s]", job_id)
raise
def _persist_to_library(self, job: AiAvatarRenderJob, cover_url: Optional[str] = None):
"""将渲染结果写入成片库,返回 GeneratedVideo 领域对象.
Args:
job: 渲染任务(必须 status=completed 且 output_video_url 非空)
cover_url: 可选的封面 URL 覆盖(finalize 时传入即优先使用,否则取 job.output_cover_url
"""
from packages.adapters.sqlalchemy_impl.generated_video_repository import (
SQLAlchemyGeneratedVideoRepository,
)
from packages.domain.generated_video import GeneratedVideo
clip_name = f"AI数字人_{job.id[:8]}"
# AI数字人入口是独立页面,前端可能不传 project_id(无项目概念),
# 兜底为 "ai_avatar" 避免 DB 非空约束/查询问题;generation_task_id 用 render_job_id 便于反查。
clip_project_id = (job.project_id or "").strip() or "ai_avatar"
clip_generation_task_id = job.id
effective_cover = (cover_url or "").strip() if cover_url else (job.output_cover_url or "").strip()
clip = GeneratedVideo.create(
project_id=clip_project_id,
generation_task_id=clip_generation_task_id,
name=clip_name,
file_url=job.output_video_url,
user_id=job.user_id,
duration=job.output_duration or 0.0,
thumbnail_url=effective_cover or None,
generation_params={
"source": "ai_avatar_render",
"render_job_id": job.id,
},
)
video_repo = SQLAlchemyGeneratedVideoRepository(self.db)
video_repo.create(clip)
logger.info("[数字人渲染] 成片已入库: clip_id=%s render_job=%s", clip.id, job.id)
return clip
def finalize_job(self, job_id: str, user_id: str, cover_url: Optional[str] = None):
"""用户在前端点「完成」后调用:将已 completed 的渲染任务正式入库到成片库.
- 必须 status=completed 才可调用
- cover_url 若传入则优先使用并回写 job.output_cover_url;否则使用 job.output_cover_urlsmart-cover/custom-cover 已写入)
- 幂等:已入库则返回已存在的 GeneratedVideo
"""
from packages.adapters.sqlalchemy_impl.generated_video_repository import (
SQLAlchemyGeneratedVideoRepository,
)
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
job = self.get_render_job(job_id, user_id)
if job is None:
raise AiAvatarRenderError("渲染任务不存在", code="RenderJobNotFound")
if job.status != "completed":
raise AiAvatarRenderError(f"渲染任务未完成(当前状态: {job.status}),无法入库", code="RenderNotCompleted")
if not (job.output_video_url or "").strip():
raise AiAvatarRenderError("渲染成片视频 URL 为空,无法入库", code="OutputVideoMissing")
# 幂等检查:已入库直接返回现有记录(通过 generation_task_id=job_id 识别,
# 因为入库时 generation_task_id 被设置为 render_job_id 自身)
existing = (
self.db.query(GeneratedVideoModel)
.filter(
GeneratedVideoModel.user_id == user_id,
GeneratedVideoModel.generation_task_id == job_id,
)
.first()
)
if existing is not None:
logger.info("[数字人渲染] finalize 幂等命中,返回已存在记录: clip_id=%s job_id=%s", existing.id, job_id)
return SQLAlchemyGeneratedVideoRepository(self.db).get(existing.id)
# 传入 cover_url 时回写到 job
if cover_url and cover_url.strip():
job.output_cover_url = cover_url.strip()
# 同步更新 cover_config,保持 smart-cover 路径一致
if isinstance(job.cover_config, dict):
job.cover_config = {**job.cover_config, "mode": "auto_frame", "url": cover_url.strip()}
job.updated_at = datetime.now(timezone.utc)
self.db.commit()
return self._persist_to_library(job, cover_url=cover_url)
def _download_video(self, url: str) -> str:
"""下载视频到临时文件."""
import httpx
+157 -70
View File
@@ -473,6 +473,7 @@ class EditPlanService:
name_suffix: str = "变体",
voice_duration: float = 0.0,
rng=None,
batch_segments: dict[str, list[tuple[float, float]]] | None = None,
) -> EditPlan:
"""为批量变体生成独立 plan:完整重跑单视频选片流程(#1743)。
@@ -489,6 +490,8 @@ class EditPlanService:
created_by_user_id: 新 plan 归属用户。
name_suffix: plan 名后缀。
rng: 可选随机数(测试注入种子)。
batch_segments: 可选,外部传入的批次内已使用素材区间(前序变体避让用)。
传入时作为初始避让对象;未传则保持原逻辑从源 plan clips 自建(向后兼容)。
Raises:
ValueError: 源 plan 不存在/无片段、素材池为空或时长全未知。
@@ -537,16 +540,26 @@ class EditPlanService:
voice = float(voice_duration or 0.0)
except (TypeError, ValueError):
voice = 0.0
rhythm_template_for_reselect = None
if source.config:
rhythm_template_for_reselect = source.config.get("rhythm_template")
if voice > 0 and source_clips_data:
from packages.domain.voice_duration_planner import plan_clip_durations
_effects: list[str | None] = [c.get("transition_effect") for c in source_clips_data]
_tdurs: list[float] = [float(c.get("transition_duration") or 0.0) for c in source_clips_data]
# #1855 P0:先占位durations为空dict,真正查durations在后面pool_ids确定后执行;
# plan_clip_durations 的 asset_durations 参数在该函数中仅作最大段长钳制,
# 这里先不依赖它(durations 还没查),传 None 让planner用默认策略;
# 真正的asset_durations会在后面 clips_data 生成时传入 reselect_clips_for_variant
target_durations = plan_clip_durations(
len(source_clips_data),
voice,
transition_effects=_effects,
transition_durations=_tdurs,
rhythm_template=rhythm_template_for_reselect,
asset_durations=None,
)
if target_durations:
for _c, _d in zip(source_clips_data, target_durations, strict=False):
@@ -582,12 +595,18 @@ class EditPlanService:
created_by_user_id=created_by_user_id or (source.created_by_user_id or ""),
)
# 批次内区间:以源 plan(变体 0)片段为初始避让对象
batch_segments: dict[str, list[tuple[float, float]]] = {}
for c in clips:
if c.asset_id and float(c.duration or 0) > 0:
st = float(c.start_time or 0.0)
batch_segments.setdefault(c.asset_id, []).append((st, st + float(c.duration)))
# 批次内区间:外部传入时使用外部传入(含前序变体已用区间);
# 否则保持原逻辑从源 plan clips 自建(向后兼容)
if batch_segments is not None:
batch_segments_resolved: dict[str, list[tuple[float, float]]] = {
k: list(v) for k, v in batch_segments.items()
}
else:
batch_segments_resolved = {}
for c in clips:
if c.asset_id and float(c.duration or 0) > 0:
st = float(c.start_time or 0.0)
batch_segments_resolved.setdefault(c.asset_id, []).append((st, st + float(c.duration)))
clips_data = reselect_clips_for_variant(
source_clips_data,
@@ -595,7 +614,7 @@ class EditPlanService:
asset_durations=durations,
asset_scene_points=scene_points,
historical_used_segments=historical,
batch_segments=batch_segments,
batch_segments=batch_segments_resolved,
target_durations=target_durations,
rng=rng,
)
@@ -767,6 +786,17 @@ class EditPlanService:
if plan is None:
return None
# #1855 P0:幂等判断——如果已成功分配过且当前 total_duration 已接近 voice_duration,直接返回
try:
existing_mark = None
if plan.config:
existing_mark = plan.config.get("voice_duration_applied")
cur_total = float(plan.total_duration or 0.0)
if existing_mark is not None and abs(existing_mark - voice) < 1e-6 and abs(cur_total - voice) < 0.5:
return plan
except Exception:
pass
clips: List[EditPlanClip] = []
skip, page = 0, 500
while True:
@@ -838,6 +868,10 @@ class EditPlanService:
)
try:
plan.total_duration = net
# #1855 P0:写入幂等标记,避免二次调用时只重分配 duration 不重算 start_time
new_cfg = dict(plan.config or {})
new_cfg["voice_duration_applied"] = voice
plan.config = new_cfg
db = self._clip_repo.session
db.commit()
except Exception:
@@ -878,6 +912,69 @@ class EditPlanService:
rng = rng or _random.Random()
plan_ids: list[str] = []
# #1855 P0:先确定片段数 clip_count(用于节奏模板生成长度匹配)
from packages.domain.bgm_pool import allocate_bgm_pool_for_variants
from packages.domain.variant_plan_selector import (
generate_pixel_perturbation,
generate_visual_perturbation,
)
from packages.domain.voice_duration_planner import RHYTHM_TEMPLATES, adapt_template_length
clip_count = 0
# 从源 plan 获取片段数(分页读,避免关系加载问题)
_sclips: list = []
_sk, _pg = 0, 500
while True:
_b = self._clip_repo.list_by_plan(source_plan_id, skip=_sk, limit=_pg)
if not _b:
break
_sclips.extend(_b)
if len(_b) < _pg:
break
_sk += _pg
clip_count = len(_sclips)
# 预先生成所有 N 个变体的节奏模板/BGM/扰动参数(时机提前到选片前写入config)
rhythm_templates_for_variants: list = []
for _idx in range(count):
if clip_count > 0:
variant_seed = rng.randint(0, 999999)
_tpl = adapt_template_length(RHYTHM_TEMPLATES[variant_seed % len(RHYTHM_TEMPLATES)], clip_count)
rhythm_templates_for_variants.append(_tpl)
else:
rhythm_templates_for_variants.append(None)
source_bgm_config: dict = {}
source_plan = self.get_plan(source_plan_id)
if source_plan and source_plan.config:
source_bgm_config = source_plan.config.get("bgm", {}) or {}
variant_seeds_for_bgm = [rng.randint(0, 999999) for _ in range(count)]
bgm_pool_assignments = allocate_bgm_pool_for_variants(source_bgm_config, variant_seeds_for_bgm)
def _build_variant_config_update(idx: int) -> dict:
"""构建单个变体的 config 更新(节奏模板/BGM/视觉/像素扰动)。"""
upd: dict = {}
try:
perturbation = generate_visual_perturbation(rng)
if idx == 0:
perturbation["hflip"] = False
upd["visual_perturbation"] = perturbation
except Exception:
logger.exception("变体 %d 视觉扰动生成失败(不阻断)", idx)
try:
pixel_pert = generate_pixel_perturbation(rng)
upd["pixel_perturbation"] = pixel_pert
except Exception:
logger.exception("变体 %d 像素扰动生成失败(不阻断)", idx)
rt = rhythm_templates_for_variants[idx] if idx < len(rhythm_templates_for_variants) else None
if rt is not None:
upd["rhythm_template"] = rt
if idx < len(bgm_pool_assignments):
existing_bgm = dict((source_plan.config or {}).get("bgm", {}) or {})
existing_bgm.update(bgm_pool_assignments[idx])
upd["bgm"] = existing_bgm
return upd
# 变体 0:clone(片段结构同源 plan,起点重算),不污染源 plan
plan0 = self.clone_plan_for_variant(
source_plan_id,
@@ -890,6 +987,15 @@ class EditPlanService:
v0_voice = float(voice_durations[0] or 0.0)
except (TypeError, ValueError):
v0_voice = 0.0
# #1855 P0:在配音分配前先写入变体0的节奏模板/扰动/BGM,确保 apply_voice_duration_to_plan 能读到 rhythm_template
try:
_cfg0 = _build_variant_config_update(0)
if _cfg0:
self.update_plan_config(plan0.id, _cfg0)
except Exception:
logger.exception("变体0 配置写入失败(不阻断): plan=%s", plan0.id)
if v0_voice > 0:
try:
self.apply_voice_duration_to_plan(plan0.id, v0_voice)
@@ -897,7 +1003,27 @@ class EditPlanService:
logger.exception("变体0 配音分配失败(不阻断): plan=%s", plan0.id)
plan_ids.append(plan0.id)
# 变体 1..N-1:独立选片
# #1855 P0:批次内素材区间避让表——从变体0实际落库的clips构建初始值
def _collect_plan_segments(pid: str) -> dict[str, list[tuple[float, float]]]:
"""分页读取 plan 所有 clips,构建 {asset_id: [(start, end), ...]} 区间表。"""
segs: dict[str, list[tuple[float, float]]] = {}
_sk2, _pg2 = 0, 500
while True:
_b2 = self._clip_repo.list_by_plan(pid, skip=_sk2, limit=_pg2)
if not _b2:
break
for _c in _b2:
if _c.asset_id and float(_c.duration or 0) > 0:
_st = float(_c.start_time or 0.0)
segs.setdefault(_c.asset_id, []).append((_st, _st + float(_c.duration)))
if len(_b2) < _pg2:
break
_sk2 += _pg2
return segs
batch_segments_acc: dict[str, list[tuple[float, float]]] = _collect_plan_segments(plan0.id)
# 变体 1..N-1:独立选片(传入累积的 batch_segments 做区间避让)
for i in range(1, count):
voice = 0.0
if voice_durations and i < len(voice_durations):
@@ -905,6 +1031,14 @@ class EditPlanService:
voice = float(voice_durations[i] or 0.0)
except (TypeError, ValueError):
voice = 0.0
# #1855 P0:在reselect前先为"变体i"准备配置更新——但reselect内部复制的是source.config
# 所以每个变体独立的节奏模板需要在reselect后单独写入config
# 但 plan_clip_durations 用的是 source.config.rhythm_template(即源plan的节奏模板),
# 为了让每个变体在选片阶段就使用自己的节奏模板分配段长,这里采用:
# - reselect 仍使用源 plan 的 rhythm_template(保持片段骨架一致)
# - 选片完成后立即写入该变体自己的 rhythm_template/扰动/BGM 到config
# 后续不再二次 apply_voice_duration_to_plan(由幂等标记跳过)
variant = self.reselect_plan_for_variant(
source_plan_id,
candidate_asset_ids,
@@ -912,73 +1046,26 @@ class EditPlanService:
name_suffix=f"变体{i + 1}",
voice_duration=voice,
rng=rng,
batch_segments=batch_segments_acc,
)
# 选片完成后写入该变体的独立配置(节奏模板/扰动/BGM)
try:
_cfgi = _build_variant_config_update(i)
if _cfgi:
self.update_plan_config(variant.id, _cfgi)
except Exception:
logger.exception("变体 %d 配置写入失败(不阻断): plan=%s", i, variant.id)
plan_ids.append(variant.id)
# #1764:为每个变体生成独立节奏模板(让批量视频片段时长分布不同)
from packages.domain.voice_duration_planner import RHYTHM_TEMPLATES, adapt_template_length
clip_count = 0
if voice_durations and len(voice_durations) > 0:
# 从源 plan 获取片段数
source_plan = self.get_plan(source_plan_id)
if source_plan and hasattr(source_plan, "clips"):
clip_count = len(list(source_plan.clips)) if source_plan.clips else 0
rhythm_templates_for_variants = []
if clip_count > 0:
for idx in range(len(plan_ids)):
# 每个变体用不同的 seed 选择节奏模板
variant_seed = rng.randint(0, 999999)
template = adapt_template_length(RHYTHM_TEMPLATES[variant_seed % len(RHYTHM_TEMPLATES)], clip_count)
rhythm_templates_for_variants.append(template)
logger.info("变体 %d 节奏模板: plan=%s template=%s", idx, plan_ids[idx], template)
# #1767:BGM 池差异化分配(让批量变体使用不同 BGM / 段落 / 音量)
from packages.domain.bgm_pool import allocate_bgm_pool_for_variants
source_bgm_config = {}
source_plan = self.get_plan(source_plan_id)
if source_plan and source_plan.config:
source_bgm_config = source_plan.config.get("bgm", {}) or {}
variant_seeds_for_bgm = [rng.randint(0, 999999) for _ in plan_ids]
bgm_pool_assignments = allocate_bgm_pool_for_variants(source_bgm_config, variant_seeds_for_bgm)
# 为每个变体生成独立视觉扰动参数(让批量视频画面本身更不同)
from packages.domain.variant_plan_selector import generate_visual_perturbation
for idx, pid in enumerate(plan_ids):
# #1855 P0:把当前新变体的 clips 区间追加到 batch_segments,供下一变体避让
try:
perturbation = generate_visual_perturbation(rng)
# 变体 0 不做 hflip(保持预览 plan 原始画面方向)
if idx == 0:
perturbation["hflip"] = False
config_update = {"visual_perturbation": perturbation}
# #1764:写入节奏模板
if idx < len(rhythm_templates_for_variants):
config_update["rhythm_template"] = rhythm_templates_for_variants[idx]
# #1765:写入像素级扰动滤镜
from packages.domain.variant_plan_selector import generate_pixel_perturbation
pixel_pert = generate_pixel_perturbation(rng)
config_update["pixel_perturbation"] = pixel_pert
# #1767:写入 BGM 池分配(覆盖 bgm 配置中的 preset_id / audio_offset / volume_adjust_db
if idx < len(bgm_pool_assignments):
existing_bgm = dict((source_plan.config or {}).get("bgm", {}) or {})
existing_bgm.update(bgm_pool_assignments[idx])
config_update["bgm"] = existing_bgm
self.update_plan_config(pid, config_update)
logger.info(
"变体 %d 视觉扰动+像素扰动+BGM池: plan=%s vis=%s pix=%s bgm=%s",
idx,
pid,
perturbation,
pixel_pert,
bgm_pool_assignments[idx] if idx < len(bgm_pool_assignments) else None,
)
_new_segs = _collect_plan_segments(variant.id)
for _aid, _ivs in _new_segs.items():
batch_segments_acc.setdefault(_aid, []).extend(_ivs)
except Exception:
logger.exception("变体 %d 视觉扰动生成失败(不阻断): plan=%s", idx, pid)
logger.exception("变体 %d 区间收集失败(不阻断): plan=%s", i, variant.id)
# 标记所有变体 plan 的 clips 为 ready(已分配素材+起点,语义上就是 ready)
for pid in plan_ids:
+228 -67
View File
@@ -1,8 +1,12 @@
"""对口型 Service — #1796 MediaKit 对口型业务逻辑, #1809 参数调整.
"""对口型 Service — #1796 MediaKit 对口型业务逻辑, #1809 参数调整, #1845 配音前置.
职责:
- 创建/查询对口型任务
- 输入模式:TTS 直生(voice_id + script_text,内部先合成音频转存 OSS)或直接音频(audio_url
- 输入模式:
1. TTS 直生(voice_id + script_text)→ 走 Celery 异步(降级路径)
2. 直接音频(audio_url,前端未传 timings)→ 同步下载 + 算 timings + 提交 MediaKit
3. 预合成音频(audio_url + sentence_timings#1845 新主路径)→ 同步 ffprobe 校验时长 +
写入前端传来的 timings → 直接提交 MediaKit~2-3s
- 调用 MediaKit 客户端提交异步任务
- 轮询更新任务状态(中间状态同步 DB,成片转存自家 OSS)
- 用户隔离(每个用户只能操作自己的任务)
@@ -26,19 +30,22 @@ from app.services.mediakit_client import (
get_mediakit_client,
)
# Celery 异步任务:TTS 合成 + MediaKit 提交(#lipsync-speed-optimization
# Celery 异步任务:TTS 合成 + MediaKit 提交(降级路径
from app.tasks.lipsync_tts import tts_synthesize_and_submit
from sqlalchemy.orm import Session
from packages.adapters.sqlalchemy_impl.models import LipsyncJobModel
from packages.application.cosyvoice_service import CosyVoiceError, normalize_emotion
from packages.domain.sentence_timings import (
compute_sentence_timings,
probe_audio_duration,
)
from packages.shared.storage import get_shared_storage_service
from packages.shared.url_security import ALLOWED_AUDIO_MIME_TYPES, safe_download_bytes
logger = logging.getLogger(__name__)
# 传给 MediaKit GPU worker / 回给前端播放的 OSS 预签名有效期:7 天。
# MediaKit 排队 + 拉取可能延迟,私有桶裸 URL 或 1 小时短预签名都会 403,故统一重签长有效期。
MEDIAKIT_URL_TTL_SECONDS = 7 * 24 * 3600
@@ -142,6 +149,100 @@ class LipsyncService:
logger.warning("TTS 音频转存 OSS 失败,回退临时 URL: job_id=%s err=%s", job_id, exc)
return temp_url
def _submit_audio_direct(
self,
*,
job: LipsyncJobModel,
supplied_timings: Optional[list] = None,
supplied_duration: Optional[float] = None,
) -> None:
"""音频直传模式(包含 #1845 预合成路径):同步下载 → ffprobe → timings → 提交 MediaKit.
直接在 HTTP 请求内完成,不走 Celery。job.status 成功后置为 submitted。
失败时把 job 标成 failed 并 commit,然后抛 MediaKitError。
Args:
job: 已 commit 的 LipsyncJobModelaudio_url / video_url 已写入)
supplied_timings: 前端传来的预合成 timings(可选,可信时直接用)
supplied_duration: 前端传来的预合成时长(可选,用于优先避免重复探测)
"""
# 1. 下载音频
audio_data: bytes | None = None
try:
audio_data = safe_download_bytes(
job.audio_url,
purpose="lipsync_direct_audio",
allowed_mime_types=ALLOWED_AUDIO_MIME_TYPES,
timeout=60.0,
)
logger.info(
"[lipsync] 直传音频下载完成: job_id=%s size=%d",
job.id,
len(audio_data) if audio_data else 0,
)
except Exception as exc:
logger.warning("[lipsync] 直传音频下载失败,跳过 timings 计算: job_id=%s err=%s", job.id, exc)
# 2. ffprobe 探测时长(优先用前端传入的预合成时长,但以 ffprobe 为准做兜底校验)
audio_duration = 0.0
if audio_data:
audio_duration = probe_audio_duration(audio_data)
if audio_duration <= 0 and supplied_duration and supplied_duration > 0:
audio_duration = supplied_duration
logger.info(
"[lipsync] ffprobe 失败,使用前端传入的预合成时长: job_id=%s duration=%.2f", job.id, audio_duration
)
# 3. 句子时间戳:优先用前端预合成传入的 timings(后端预合成接口已经算过,可信);
# 否则若音频下载成功则重算;否则不设置(不阻塞主流程)
timings: Optional[list] = None
if supplied_timings:
timings = supplied_timings
logger.info("[lipsync] 使用前端预合成句子时间戳: job_id=%s sentences=%d", job.id, len(timings))
elif audio_data and audio_duration > 0 and job.script_text:
try:
timings = compute_sentence_timings(audio_data, job.script_text, audio_duration)
logger.info(
"[lipsync] 后端重算句子时间戳: job_id=%s sentences=%d duration=%.2f",
job.id,
len(timings) if timings else 0,
audio_duration,
)
except Exception as exc:
logger.warning("[lipsync] 句子时间戳计算失败(不阻塞): job_id=%s err=%s", job.id, exc)
if timings:
job.sentence_timings = timings
# 4. 签名 URL 并提交 MediaKit
video_url = self._sign_media_url(job.video_url)
signed_audio_url = self._sign_media_url(job.audio_url)
job.audio_url = signed_audio_url
try:
result = self.client.submit_lipsync(
video_url=video_url,
audio_url=signed_audio_url,
enable_video_loop=job.enable_video_loop,
client_token=job.id,
)
job.mediakit_task_id = result["task_id"]
job.status = "submitted"
job.submitted_at = datetime.now(timezone.utc)
self.db.commit()
logger.info(
"[lipsync] 直传音频已提交 MediaKit: job_id=%s task_id=%s",
job.id,
result["task_id"],
)
except MediaKitError as exc:
job.status = "failed"
job.error_message = str(exc)
job.error_code = exc.code
logger.error("[lipsync] 直传音频提交 MediaKit 失败: job_id=%s err=%s", job.id, exc)
self.db.commit()
raise
# ── 创建任务 ──────────────────────────────────────────────────────────
def create_job(
@@ -150,27 +251,35 @@ class LipsyncService:
user_id: str,
video_url: str,
audio_url: str = "",
audio_duration: Optional[float] = None,
sentence_timings: Optional[list] = None,
voice_id: str = "",
script_text: str = "",
speed: float = 1.0,
emotion: str = "",
enable_video_loop: bool = False,
enable_video_loop: bool = True,
project_id: str = "",
) -> LipsyncJobModel:
"""创建对口型任务.
种输入模式:
种输入模式:
- TTS 直生:voice_id + script_textaudio_url 留空)
创建 DB 记录(状态 tts_processing),dispatch Celery 异步任务
执行 TTS 合成 + MediaKit 提交。API 响应 <1s。
- 直接音频:提供 audio_url
→ 同步提交 MediaKit,状态直接设为 submitted
→ 创建 DB 记录(状态 tts_processing),dispatch Celery 异步任务(降级路径)。
API 响应 <1s。
- 直接音频:audio_url 非空 + 无 sentence_timings
→ 同步下载音频 + 重算 timings + 提交 MediaKit(几秒完成)
- 预合成音频(#1845 新主路径):audio_url 非空 + 传 sentence_timings
→ 同步 ffprobe 校验时长 + 写入 timings + 提交 MediaKit~2-3s)。
Raises:
MediaKitError: 参数校验失败或 MediaKit 提交失败(仅直接音频模式)
MediaKitError: 参数校验失败或 MediaKit 提交失败
"""
# 0. 输入校验
if not audio_url:
is_pre_synth = bool(audio_url) and bool(sentence_timings)
bool(audio_url) and not is_pre_synth
is_tts_mode = not bool(audio_url)
if is_tts_mode:
if not (voice_id and script_text):
raise MediaKitError(
"必须提供 audio_url 或 voice_id+script_text",
@@ -178,10 +287,13 @@ class LipsyncService:
)
# TTS 模式:在 HTTP 请求中同步校验音色归属,快速失败
self._resolve_voice_id(voice_id, user_id)
elif is_pre_synth:
# 预合成模式:script_text 可空(因为 timings 已自带句子文本),但仍建议传
if not isinstance(sentence_timings, list) or len(sentence_timings) == 0:
raise MediaKitError("预合成模式 sentence_timings 不能为空", code="InvalidInput")
# 1. 创建数据库记录
job_id = str(uuid.uuid4())
is_tts_mode = not bool(audio_url)
job = LipsyncJobModel(
id=job_id,
user_id=user_id,
@@ -192,20 +304,19 @@ class LipsyncService:
voice_id=voice_id or "",
script_text=script_text or "",
speed=speed,
emotion=normalize_emotion(emotion),
emotion=normalize_emotion(emotion) if is_tts_mode else (emotion or ""),
# 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing
status="tts_processing" if is_tts_mode else "pending",
)
self.db.add(job)
self.db.flush()
# ⚠️ 必须先 commit 再发 Celery 任务,避免事务竞态
# worker 是独立进程+独立DB连接,任务被消费(<4ms)时若本事务还未提交,
# worker 查询 job 会返回 None → 静默 return 不重试,job 永远卡在 tts_processing。
# ⚠️ 必须先 commit 再发 Celery 任务 / 后续同步操作,避免事务竞态
self.db.commit()
self.db.refresh(job)
if is_tts_mode:
# 2a. TTS 模式:dispatch Celery 异步任务处理 TTS 合成 + MediaKit 提交
# 2a. TTS 模式:dispatch Celery 异步任务处理 TTS 合成 + MediaKit 提交(降级路径)
try:
tts_synthesize_and_submit.apply_async(
args=(
@@ -218,8 +329,6 @@ class LipsyncService:
)
)
except Exception as exc:
# 投递失败时立即把 job 标成 failed 并写入 error_message
# 前端轮询时能直接看到失败原因,不会无限卡在 tts_processing。
logger.exception(
"Celery 任务提交失败,TTS 任务已创建但未触发执行: job_id=%s err=%s",
job_id,
@@ -229,35 +338,102 @@ class LipsyncService:
job.error_message = f"Celery 任务投递失败: {exc}"
job.error_code = "AsyncDispatchFailed"
job.updated_at = datetime.now(timezone.utc)
self.db.commit() # 投递失败也要落库失败状态
else:
# 2b. 直接音频模式:同步签名并提交 MediaKit
video_url = self._sign_media_url(video_url)
if audio_url:
audio_url = self._sign_media_url(audio_url)
job.audio_url = audio_url
try:
result = self.client.submit_lipsync(
video_url=video_url,
audio_url=audio_url,
enable_video_loop=enable_video_loop,
client_token=job_id,
)
job.mediakit_task_id = result["task_id"]
job.status = "submitted"
job.submitted_at = datetime.now(timezone.utc)
self.db.commit() # submitted 状态落库
except MediaKitError as exc:
job.status = "failed"
job.error_message = str(exc)
job.error_code = exc.code
logger.error("提交对口型任务失败: %s", exc)
self.db.commit()
raise
else:
# 2b/2c. 直接音频 / 预合成音频:同步路径
self._submit_audio_direct(
job=job,
supplied_timings=sentence_timings,
supplied_duration=audio_duration,
)
self.db.refresh(job)
return job
# ── TTS 预合成(#1845 步骤1「生成配音」同步接口使用) ──────────────────
def preview_tts(
self,
*,
user_id: str,
voice_id: str,
script_text: str,
speed: float = 1.0,
emotion: str = "natural",
) -> dict:
"""同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算.
不创建 LipsyncJob、不转存 OSS,直接返回 CosyVoice 临时 URL~24h 有效期)。
耗时约 2-3 秒,由前端在步骤1点「生成配音」时同步等待。
Returns:
{"audio_url": str, "duration": float, "sentence_timings": list[dict]}
Raises:
MediaKitError: TTS 合成失败 / 下载失败 / ffprobe 失败
"""
# 1. 音色解析(校验克隆音色归属)
actual_voice_id = self._resolve_voice_id(voice_id, user_id)
cosyvoice = self._get_cosyvoice()
# 2. TTS 合成(同步,~2-3s
try:
result = cosyvoice.submit_synthesize_task(
text=script_text,
voice_id=actual_voice_id,
speed=speed,
emotion=normalize_emotion(emotion),
)
except CosyVoiceError as exc:
raise MediaKitError(f"TTS 合成失败: {exc}", code="TTSSynthesisFailed") from exc
except ValueError as exc:
raise MediaKitError(f"TTS 参数错误: {exc}", code="TTSInvalidParam") from exc
temp_url = result.get("audio_url", "")
if not temp_url:
raise MediaKitError("TTS 未返回音频 URL", code="TTSNoAudio")
# 3. 下载音频到内存(用于 ffprobe + 静音检测)
try:
audio_data = safe_download_bytes(
temp_url,
purpose="tts_preview_audio",
allowed_mime_types=ALLOWED_AUDIO_MIME_TYPES,
timeout=60.0,
)
except Exception as exc:
logger.warning("[tts-preview] TTS 音频下载失败,仍返回 audio_url: user_id=%s err=%s", user_id, exc)
return {
"audio_url": temp_url,
"duration": 0.0,
"sentence_timings": [],
}
# 4. ffprobe 时长
duration = probe_audio_duration(audio_data)
if duration <= 0:
logger.warning("[tts-preview] ffprobe 未返回有效时长,timings 留空: user_id=%s", user_id)
return {
"audio_url": temp_url,
"duration": 0.0,
"sentence_timings": [],
}
# 5. 句子时间戳
timings = compute_sentence_timings(audio_data, script_text, duration)
logger.info(
"[tts-preview] TTS 预合成完成: user_id=%s duration=%.2f sentences=%d",
user_id,
duration,
len(timings),
)
return {
"audio_url": temp_url,
"duration": round(duration, 2),
"sentence_timings": timings,
}
# ── 查询任务 ──────────────────────────────────────────────────────────
def get_job(self, job_id: str, user_id: str) -> Optional[LipsyncJobModel]:
@@ -291,11 +467,7 @@ class LipsyncService:
# ── 更新任务状态(轮询) ──────────────────────────────────────────────
def refresh_job_status(self, job_id: str, user_id: str) -> Optional[LipsyncJobModel]:
"""从 MediaKit 拉取最新状态并更新本地记录.
Returns:
更新后的 Job,或 None(任务不存在/不属于该用户)
"""
"""从 MediaKit 拉取最新状态并更新本地记录."""
job = self.get_job(job_id, user_id)
if job is None:
return None
@@ -321,13 +493,12 @@ class LipsyncService:
result = status_data.get("result", {})
job.status = STATUS_COMPLETED
temp_url = result.get("video_url", "")
# 先以临时 URL 立即返回前端(前端可立即播放),再异步 Celery 任务转存自家 OSS(步骤⑦)
job.output_video_url = temp_url
job.output_duration = result.get("duration", 0.0)
job.completed_at = datetime.now(timezone.utc)
job.updated_at = datetime.now(timezone.utc)
self.db.commit()
# 异步转存自家 OSS(注意:必须在 commit 之后 dispatch,避免 commit 失败任务已发出)
# 异步转存自家 OSS
try:
from app.tasks.lipsync_tts import persist_output_video_task
@@ -347,7 +518,6 @@ class LipsyncService:
job.error_code = error.get("code", "TaskFailed")
job.completed_at = datetime.now(timezone.utc)
else:
# 中间状态(running/processing/queued 等)同步到 DB,避免前端永远卡在 submitted
if isinstance(mk_status, str) and mk_status:
job.status = mk_status
job.updated_at = datetime.now(timezone.utc)
@@ -356,10 +526,7 @@ class LipsyncService:
return job
def _persist_output_video(self, temp_url: str, job_id: str, user_id: str) -> str:
"""将 MediaKit 输出的临时视频 URL 转存到自家 OSS.
失败时回退返回原始临时 URL,不影响任务完成。
"""
"""将 MediaKit 输出的临时视频 URL 转存到自家 OSS. 失败时回退返回原始临时 URL."""
if not temp_url:
return ""
try:
@@ -379,27 +546,21 @@ class LipsyncService:
return temp_url
def _sign_media_url(self, url: str) -> str:
"""对自家 OSS 私有桶 URL 重签长有效期预签名,供 MediaKit 拉取 / 前端播放。
- 裸 public_urlupload_file 返回,不带签名)→ 私有桶匿名访问 403,重签。
- 已带签名但即将过期的 URL(如前端 1h 预签名)→ 抽 storage_key 后重签。
- 外部 URLCosyVoice/MediaKit 临时链接,非本桶 host)→ 原样透传。
- 任何异常都降级原样返回,不阻断主流程。
"""
"""对自家 OSS 私有桶 URL 重签长有效期预签名."""
if not url:
return url
try:
storage = get_shared_storage_service()
public_base = getattr(storage, "public_url", "")
if not isinstance(public_base, str) or not public_base:
return url # 无法判定归属,保守透传
return url
own_host = urlparse(public_base).netloc.lower()
host = urlparse(url).netloc.lower()
if not own_host or host != own_host:
return url # 非自家 OSS外部临时链接),不处理
return url # 外部临时链接原样透传
signed = storage.get_download_url(url, expires_seconds=MEDIAKIT_URL_TTL_SECONDS)
return signed or url
except Exception as exc: # noqa: BLE001 - 签名失败不阻断,降级原 URL
except Exception as exc:
logger.warning("对口型 URL 重签失败,原样返回: url_prefix=%s err=%s", url[:80], exc)
return url
+2 -3
View File
@@ -75,7 +75,7 @@ class MediaKitClient:
*,
video_url: str,
audio_url: str,
enable_video_loop: bool = False,
enable_video_loop: bool = True,
callback_url: Optional[str] = None,
callback_args: Optional[str] = None,
client_token: Optional[str] = None,
@@ -103,8 +103,7 @@ class MediaKitClient:
"video_url": video_url,
"audio_url": audio_url,
}
if enable_video_loop:
payload["enable_video_loop"] = True
payload["enable_video_loop"] = bool(enable_video_loop)
if callback_url:
payload["callback_url"] = callback_url
if callback_args:
+18 -213
View File
@@ -12,6 +12,10 @@
注意:使用 @shared_task 而非绑定到某个 celery_app 实例,
确保任务能被 Worker 侧 celery_app 正确注册,同时 API 侧 send_task/apply_async 仍可正常调用。
#1845:句子时间戳计算已提取至 packages/domain/sentence_timings.py,本模块保留
_ 开头别名兼容历史导入,但 _compute_sentence_timings/_split_script_into_sentences/
_estimate_sentence_timings_by_chars 等内部函数已复用共享实现,避免重复代码。
"""
import io
@@ -21,6 +25,12 @@ from urllib.parse import urlparse
from celery import shared_task
# 复用共享的句子时间戳工具(#1845 配音前置)
from packages.domain.sentence_timings import compute_sentence_timings as _compute_sentence_timings
from packages.domain.sentence_timings import (
probe_audio_duration,
)
logger = logging.getLogger(__name__)
# MediaKit 预签名 URL 有效期(7天,秒),与 LipsyncService._sign_media_url 保持一致
@@ -54,164 +64,6 @@ def _sign_media_url(url: str) -> str:
return url
def _split_script_into_sentences(script_text: str) -> list[str]:
"""按句号/问号/感叹号/分号/逗号/换行分句(与前端 SENTENCE_SPLIT_RE 一致).
中文短视频文案习惯用「,」断小句(如"卖花的叫花无缺,卖姜的叫姜子牙"),
必须把逗号也纳入分隔符,否则多句文案会被识别成一整句,导致 B-roll 时间戳错位。
"""
import re
text = (script_text or "").strip()
if not text:
return []
parts = re.split(r"[。!?!??!;,\n\r]+", text)
return [p.strip() for p in parts if p.strip()]
def _compute_sentence_timings(audio_data: bytes, script_text: str, total_duration: float) -> list[dict]:
"""基于 TTS 音频的静音检测,精确计算每句文案的起止时间.
使用 ffmpeg silencedetect 检测静音段,将静音点与句子边界对齐。
比字数比例估算准确得多。
Args:
audio_data: TTS 音频二进制数据(MP3
script_text: 文案全文
total_duration: 音频总时长(秒)
Returns:
list[{"index": int, "text": str, "start_time": float, "end_time": float}]
"""
import re
import subprocess
import tempfile
sentences = _split_script_into_sentences(script_text)
if not sentences:
return []
# 写入临时音频文件
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp:
tmp.write(audio_data)
tmp_path = tmp.name
try:
# 用 ffmpeg silencedetect 检测静音段
result = subprocess.run(
[
"ffmpeg",
"-i",
tmp_path,
"-af",
"silencedetect=noise=-25dB:d=0.3",
"-f",
"null",
"-",
],
capture_output=True,
text=True,
timeout=30,
)
stderr = result.stderr or ""
# 解析静音结束时间点(silence_end: X.XXX
silence_ends = []
for match in re.finditer(r"silence_end:\s*([\d.]+)", stderr):
t = float(match.group(1))
if 0 < t < total_duration:
silence_ends.append(t)
# 如果没有检测到足够的静音点,降级为字数比例估算
if len(silence_ends) < len(sentences) - 1:
logger.warning(
"[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算",
len(silence_ends),
len(sentences) - 1,
)
return _estimate_sentence_timings_by_chars(sentences, total_duration)
# 贪心匹配:N-1 个句子边界对应 N-1 个静音点
# 按时间均匀分布期望值,选择最近的静音点
n_boundaries = len(sentences) - 1
boundaries = []
used_indices = set()
for i in range(n_boundaries):
# 期望的边界位置(按句子数量均匀分布)
expected_pos = (i + 1) / len(sentences) * total_duration
# 找最近的未使用静音点
best_idx = None
best_dist = float("inf")
for j, t in enumerate(silence_ends):
if j in used_indices:
continue
dist = abs(t - expected_pos)
if dist < best_dist:
best_dist = dist
best_idx = j
if best_idx is not None:
used_indices.add(best_idx)
boundaries.append(silence_ends[best_idx])
boundaries.sort()
# 构建 sentence_timings
timings = []
prev_end = 0.0
for i, sent in enumerate(sentences):
start = prev_end
end = boundaries[i] if i < len(boundaries) else total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
prev_end = end
return timings
except Exception as exc:
logger.warning("[sentence_timings] 静音检测异常,降级为字数比例估算: %s", exc)
return _estimate_sentence_timings_by_chars(sentences, total_duration)
finally:
import os
try:
os.unlink(tmp_path)
except Exception:
pass
def _estimate_sentence_timings_by_chars(sentences: list[str], total_duration: float) -> list[dict]:
"""降级方案:按字数比例估算句子时间(与原前端逻辑一致)."""
if not sentences or total_duration <= 0:
return []
total_chars = sum(len(s.replace(r"\s", "")) for s in sentences)
if total_chars == 0:
return []
timings = []
acc = 0
for i, sent in enumerate(sentences):
chars = len(sent.replace(r"\s", ""))
start = (acc / total_chars) * total_duration
end = ((acc + chars) / total_chars) * total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
acc += chars
return timings
@shared_task(
bind=True,
name="lipsync_tts.synthesize_and_submit",
@@ -234,7 +86,8 @@ def tts_synthesize_and_submit(
):
"""异步执行 TTS 合成 + OSS 转存 + MediaKit 提交.
在 Celery worker 中运行,不阻塞 HTTP 请求。
在 Celery worker 中运行,不阻塞 HTTP 请求。保留作为降级路径
(预合成失败 / 旧版前端未传 audio_url 时走此路径)。
"""
from app.services.mediakit_client import MediaKitError, get_mediakit_client
from sqlalchemy.orm import Session as DBSession
@@ -267,9 +120,6 @@ def tts_synthesize_and_submit(
if job is None:
# 事务竞态防御:API 在 commit 前投递了任务,worker 消费时事务尚未提交。
# Celery 内置 autoretry_for 不支持"业务条件重试",这里手动 retry 3 次,
# 间隔递增(1s/3s/7s),让 API 事务有时间提交。
# max_retries 由 self.request(retries) 维护;默认 self.max_retries=3 由装饰器 soft_time_limit 下方指定。
retries = getattr(self.request, "retries", 0)
max_retries = 3
if retries < max_retries:
@@ -340,7 +190,6 @@ def tts_synthesize_and_submit(
# 2. 下载 TTS 音频到内存(用于 2.5 静音检测;不转存自家 OSS,直接使用 CosyVoice 临时 URL
audio_data: bytes | None = None
_st_tmp_path: str | None = None
try:
audio_data = safe_download_bytes(
temp_url,
@@ -349,7 +198,7 @@ def tts_synthesize_and_submit(
"audio/mpeg",
"audio/mp3",
"audio/wav",
"audio/x-wav", # CosyVoice 部分接口返回 audio/x-wav,与 audio/wav 等价(RIFF/WAVE
"audio/x-wav", # CosyVoice 部分接口返回 audio/x-wav
"audio/mp4",
"audio/x-m4a",
},
@@ -361,7 +210,6 @@ def tts_synthesize_and_submit(
len(audio_data) if audio_data else 0,
)
except Exception as exc:
# 下载失败:audio_data 保持 None2.5 静音检测会跳过;后续仍用 temp_url 提交 MediaKit
logger.warning(
"[lipsync_tts] TTS 音频下载失败,跳过静音检测,直接使用临时 URL 提交: job_id=%s err=%s",
job_id,
@@ -373,45 +221,16 @@ def tts_synthesize_and_submit(
db.commit()
# 2.5 计算精确句子时间戳(基于 TTS 音频静音检测)
# 直接复用步骤 2 已下载到内存的 audio_data,避免重新下载
import os as _os
# 2.5 计算精确句子时间戳(基于 TTS 音频静音检测)—— 复用共享工具
try:
import subprocess as _sp
import tempfile as _tmpf
if not audio_data:
logger.warning("[lipsync_tts] 无音频数据,跳过句子时间戳计算: job_id=%s", job_id)
else:
# 写入临时文件供 ffprobe/ffmpeg 使用
with _tmpf.NamedTemporaryFile(suffix=".mp3", delete=False) as _atmp:
_atmp.write(audio_data)
_st_tmp_path = _atmp.name
# ffprobe 获取音频时长
_probe_result = _sp.run(
[
"ffprobe",
"-v",
"error",
"-show_entries",
"format=duration",
"-of",
"default=noprint_wrappers=1:nokey=1",
_st_tmp_path,
],
capture_output=True,
text=True,
timeout=10,
)
_audio_duration = float(_probe_result.stdout.strip()) if _probe_result.stdout.strip() else 0.0
_audio_duration = probe_audio_duration(audio_data)
logger.info(
"[lipsync_tts] 音频时长探测: job_id=%s duration=%.2f probe_stdout=%s probe_stderr=%s",
"[lipsync_tts] 音频时长探测: job_id=%s duration=%.2f",
job_id,
_audio_duration,
_probe_result.stdout.strip()[:50],
_probe_result.stderr.strip()[:100] if _probe_result.stderr else "",
)
if _audio_duration > 0:
@@ -428,22 +247,14 @@ def tts_synthesize_and_submit(
logger.warning("[lipsync_tts] 句子时间戳计算返回空结果: job_id=%s", job_id)
else:
logger.warning(
"[lipsync_tts] ffprobe 未获取到有效时长,跳过句子时间戳: job_id=%s stdout=%s stderr=%s",
"[lipsync_tts] ffprobe 未获取到有效时长,跳过句子时间戳: job_id=%s",
job_id,
_probe_result.stdout.strip()[:100],
_probe_result.stderr.strip()[:200] if _probe_result.stderr else "",
)
db.commit()
except Exception as _st_err:
logger.warning(
"[lipsync_tts] 句子时间戳计算失败(不影响主流程): job_id=%s err=%s", job_id, _st_err, exc_info=True
)
finally:
if _st_tmp_path:
try:
_os.unlink(_st_tmp_path)
except Exception:
pass
# 3. 签名 URL 并提交到 MediaKit(复用模块内 _sign_media_url,避免对 LipsyncService 的耦合)
audio_url = _sign_media_url(job.audio_url)
@@ -495,12 +306,7 @@ def tts_synthesize_and_submit(
default_retry_delay=30,
)
def persist_output_video_task(job_id: str, user_id: str, temp_url: str):
"""异步转存对口型输出视频到自家 OSS(步骤⑦ — 将同步阻塞挪到后台,加速前端响应).
- MediaKit 返回 completed 后先以 temp_url 回前端(前端可立即播放临时 URL)
- Celery 后台下载 temp_url 并转存 OSS,成功后更新 job.output_video_url 为永久 URL
- 失败则保留 temp_url,不阻断主流程
"""
"""异步转存对口型输出视频到自家 OSS(步骤⑦ — 将同步阻塞挪到后台,加速前端响应)."""
try:
from worker_app.db import SessionLocal # type: ignore
@@ -532,7 +338,6 @@ def persist_output_video_task(job_id: str, user_id: str, temp_url: str):
storage = get_shared_storage_service()
storage_key = f"lipsync-outputs/{user_id}/{job_id}.mp4"
permanent_url = storage.upload_file(io.BytesIO(data), storage_key, content_type="video/mp4")
# 对自家 OSS URL 重签 7 天有效期预签名,供前端播放
final_url = _sign_media_url(permanent_url) if permanent_url else temp_url
job.output_video_url = final_url
job.updated_at = datetime.now(timezone.utc)
+17
View File
@@ -1265,3 +1265,20 @@
width: auto;
min-width: 300px;
}
/* 渲染完成后的封面确认区 */
.aa-finalize-section {
display: flex;
flex-direction: column;
align-items: center;
padding: 16px 0 8px;
}
.aa-finalize-cover {
width: 100%;
max-width: 240px;
aspect-ratio: 9/16;
border-radius: 12px;
overflow: hidden;
position: relative;
}
+493 -66
View File
@@ -1,7 +1,7 @@
/**
* AI数字人 — 主页面(v3 两步骤版)
* 步骤1:出镜视频 / 配音库 / 文案
* 步骤2:对口型预览(含插入画面)/ 标题配置 / 封面&生成
* AI数字人 — 主页面(v3 两步骤版 + #1845 配音前置
* 步骤1:出镜视频 / 配音库 / 文案 → 点击「🎵 生成配音」做 TTS 预合成(同步,~2-3s)
* 步骤2:对口型预览(音频已就绪、B-roll 句子时间戳立即可用)/ 标题配置 / 封面&生成
*/
import React, { useState, useCallback, useEffect, useRef } from "react"
import { message } from "antd"
@@ -16,17 +16,20 @@ import PanelTitleConfig from "./components/PanelTitleConfig"
import PanelCoverAndGenerate from "./components/PanelCoverAndGenerate"
import { ModalAssetPicker } from "./components/ModalAssetPicker"
import ModalBRollEditor from "./components/ModalBRollEditor"
import ModalCoverSelect from "./components/ModalCoverSelect"
import {
getScripts,
getAssetById,
createLipsyncJob,
getLipsyncJob,
previewTts,
submitRender,
getRenderJob,
generateRenderSmartCover,
finalizeRenderJob,
} from "./api/aiAvatar"
import { getOrCreateDefaultProject } from "@/api/projects"
import type { RenderJob } from "./types"
import type { RenderJob, SentenceTiming } from "./types"
import {
normalizeEmotion,
buildTitleConfigPayload,
@@ -50,6 +53,12 @@ const AiAvatarPage: React.FC = () => {
cover: false,
})
/* ── #1845 TTS 预合成弹窗 ── */
const [showTtsModal, setShowTtsModal] = useState(false)
const [ttsProgress, setTtsProgress] = useState(0)
const [ttsErrorMessage, setTtsErrorMessage] = useState("")
const ttsProgressTimerRef = useRef<ReturnType<typeof setInterval> | null>(null)
/* ── 对口型生成弹窗 ── */
const [showLipsyncModal, setShowLipsyncModal] = useState(false)
const [lipsyncStatus, setLipsyncStatus] = useState<"generating" | "completed" | "failed">(
@@ -63,8 +72,12 @@ const AiAvatarPage: React.FC = () => {
)
const [renderProgress, setRenderProgress] = useState(0)
const [renderErrorMessage, setRenderErrorMessage] = useState("")
/* ── 当前渲染任务对象(轮询更新;用于封面区判断渲染是否完成) ── */
/* ── 当前渲染任务对象 ── */
const [currentRenderJob, setCurrentRenderJob] = useState<RenderJob | null>(null)
/* ── 封面选择弹窗 ── */
const [showCoverModal, setShowCoverModal] = useState(false)
const [selectedCoverUrl, setSelectedCoverUrl] = useState("")
const [finalizeLoading, setFinalizeLoading] = useState(false)
/* ── 对口型轮询 ── */
const lipsyncTimerRef = useRef<ReturnType<typeof setInterval> | null>(null)
@@ -75,8 +88,29 @@ const AiAvatarPage: React.FC = () => {
setCollapsed((prev) => ({ ...prev, [key]: !prev[key] }))
}, [])
/* ── 步骤切换 ── */
const handleNextStep = useCallback(() => {
/* ── #1845 文案/音色/语速变更时重置 TTS 预合成状态,避免音频与文案不一致 ── */
useEffect(() => {
if (state.ttsPreview.status !== "idle") {
state.resetTtsPreview()
}
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [state.scriptText, state.selectedVoice?.voice_id, state.speed, state.emotion])
const _clearTtsProgressTimer = useCallback(() => {
if (ttsProgressTimerRef.current) {
clearInterval(ttsProgressTimerRef.current)
ttsProgressTimerRef.current = null
}
}, [])
useEffect(() => {
return () => {
_clearTtsProgressTimer()
}
}, [_clearTtsProgressTimer])
/* ── #1845 步骤1:点击「🎵 生成配音」→ 同步 TTS 预合成 ── */
const handleGenerateTts = useCallback(async () => {
const missing: string[] = []
if (!state.selectedVideo) missing.push("出镜视频")
if (!state.selectedVoice) missing.push("配音")
@@ -85,45 +119,120 @@ const AiAvatarPage: React.FC = () => {
message.warning(`请先完成${missing.join("、")}`)
return
}
setCurrentStep(2)
}, [state.selectedVideo, state.selectedVoice, state.scriptText])
// 打开弹窗 & 启动模拟进度条
setShowTtsModal(true)
setTtsProgress(0)
setTtsErrorMessage("")
state.setTtsPreview({
audioUrl: null,
duration: 0,
sentenceTimings: [],
status: "generating",
error: null,
})
// 模拟进度:每 300ms +10%,到 90% 停住,真完成后瞬间到 100%
_clearTtsProgressTimer()
let fake = 0
ttsProgressTimerRef.current = setInterval(() => {
fake = Math.min(fake + 10, 90)
setTtsProgress(fake)
if (fake >= 90) {
_clearTtsProgressTimer()
}
}, 300)
try {
const res = await previewTts({
voice_id: state.selectedVoice!.voice_id,
script_text: state.scriptText,
speed: state.speed,
emotion: normalizeEmotion(state.emotion),
})
_clearTtsProgressTimer()
setTtsProgress(100)
state.setTtsPreview({
audioUrl: res.audio_url,
duration: res.duration,
sentenceTimings: res.sentence_timings as SentenceTiming[],
status: "done",
error: null,
})
message.success("配音合成完成")
} catch (err) {
_clearTtsProgressTimer()
const errMsg =
(err as { response?: { data?: { message?: string; detail?: unknown } } })?.response?.data
?.message || (err instanceof Error ? err.message : "配音合成失败,请重试")
setTtsErrorMessage(typeof errMsg === "string" ? errMsg : "配音合成失败,请重试")
state.setTtsPreview({
audioUrl: null,
duration: 0,
sentenceTimings: [],
status: "failed",
error: typeof errMsg === "string" ? errMsg : "配音合成失败",
})
}
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.emotion])
const handleRetryTts = useCallback(() => {
handleGenerateTts()
}, [handleGenerateTts])
const handleTtsNext = useCallback(() => {
setShowTtsModal(false)
setTtsProgress(0)
setCurrentStep(2)
}, [])
const handleCancelTts = useCallback(() => {
_clearTtsProgressTimer()
setShowTtsModal(false)
setTtsProgress(0)
setTtsErrorMessage("")
// 若用户在生成中途关闭,把状态重置回 idle,允许重新点击
if (state.ttsPreview.status === "generating") {
state.resetTtsPreview()
}
}, [_clearTtsProgressTimer, state])
/* ── 上一步(返回步骤1,不会丢失 TTS 预合成结果) ── */
const handlePrevStep = useCallback(() => {
setCurrentStep(1)
}, [])
/* ── 对口型 ── */
const handleGenerateLipsync = useCallback(async () => {
// ② 缺项明确提示(#1809):不再静默 return
const video = state.selectedVideo
const voice = state.selectedVoice
const text = state.scriptText.trim()
const missing: string[] = []
if (!video) missing.push("出镜视频")
if (!voice) missing.push("音色")
if (!text) missing.push("文案")
if (missing.length > 0 || !video || !voice) {
if (missing.length > 0 || !video) {
message.warning(`请先选择${missing.join("、")}`)
return
}
// #1845:预合成模式下必须要有 audioUrl(理论上到了步骤2肯定有,兜底防御)
const isPreSynth = state.ttsPreview.status === "done" && !!state.ttsPreview.audioUrl
if (!isPreSynth && !state.selectedVoice) {
message.warning("请先选择音色或完成配音合成")
return
}
try {
// 显示生成弹窗
setShowLipsyncModal(true)
setLipsyncStatus("generating")
setLipsyncErrorMessage("")
// ① 先按素材 id 拿 file_url(#1809 补充:对齐后端新参数 video_url)
console.log("[对口型] 开始生成:", {
videoId: video.id,
voiceId: voice.voice_id,
voiceType: voice.type,
mode: isPreSynth ? "pre-synth" : "tts-direct",
textLen: state.scriptText.length,
})
const asset = await getAssetById(video.id)
console.log("[对口型] getAssetById 响应:", {
id: asset?.id,
file_url: asset?.file_url?.substring(0, 100),
})
const videoUrl = asset?.file_url
if (!videoUrl) {
console.error("[对口型] file_url 为空,asset:", asset)
@@ -131,19 +240,35 @@ const AiAvatarPage: React.FC = () => {
message.error("获取出镜视频播放地址失败,请重新选择素材")
return
}
// ② 模式A TTS直生:video_url + voice_id + script_text,语速/情绪英文枚举透传(#1822)
const payload = {
voice_id: voice.voice_id,
script_text: state.scriptText,
video_url: videoUrl,
speed: state.speed, // 语速 0.5~2.0
emotion: normalizeEmotion(state.emotion), // natural/excited/calm/friendly
type LipsyncPayload = Parameters<typeof createLipsyncJob>[0]
let payload: LipsyncPayload
if (isPreSynth) {
// 预合成模式:传 audio_url + audio_duration + sentence_timings(后端直接提交 MediaKit~2-3s
payload = {
video_url: videoUrl,
audio_url: state.ttsPreview.audioUrl!,
audio_duration: state.ttsPreview.duration,
sentence_timings: state.ttsPreview.sentenceTimings,
enable_video_loop: true,
}
} else {
// 降级:TTS 直生(旧路径,前端未预合成时)
payload = {
voice_id: state.selectedVoice!.voice_id,
script_text: state.scriptText,
video_url: videoUrl,
speed: state.speed,
emotion: normalizeEmotion(state.emotion),
}
}
console.log("[对口型] createLipsyncJob 请求:", payload)
const job = await createLipsyncJob(payload)
console.log("[对口型] createLipsyncJob 响应:", { id: job.id, status: job.status })
state.setLipsyncJob(job)
// 开始轮询
// 如果是预合成模式,后端会同步把状态置为 submitted(甚至可能已返回 running),
// 但仍需轮询等 completed
if (lipsyncTimerRef.current) clearInterval(lipsyncTimerRef.current)
lipsyncTimerRef.current = setInterval(async () => {
try {
@@ -180,7 +305,14 @@ const AiAvatarPage: React.FC = () => {
message.error(err instanceof Error ? err.message : "对口型任务提交失败,请重试")
}
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.emotion])
}, [
state.selectedVideo,
state.selectedVoice,
state.scriptText,
state.speed,
state.emotion,
state.ttsPreview,
])
// 取消对口型生成
const handleCancelLipsync = useCallback(() => {
@@ -201,6 +333,14 @@ const AiAvatarPage: React.FC = () => {
}
}, [])
/* ── B-roll 弹窗可用的句子时间戳:优先 lipsyncJob.sentence_timings,否则用 ttsPreview.sentenceTimings ── */
const bRollSentenceTimings: SentenceTiming[] | undefined =
(state.lipsyncJob?.sentence_timings as SentenceTiming[] | undefined) ??
(state.ttsPreview.status === "done" ? state.ttsPreview.sentenceTimings : undefined)
/* ── B-roll 可用的总时长:优先 lipsyncJob.output_duration,否则用 ttsPreview.duration ── */
const bRollDuration = state.lipsyncJob?.output_duration || state.ttsPreview.duration || 0
/* ── 生成视频(含实时进度轮询) ── */
const handleGenerate = useCallback(async () => {
if (!state.lipsyncJob || state.lipsyncJob.status !== "completed") {
@@ -209,10 +349,9 @@ const AiAvatarPage: React.FC = () => {
}
state.setIsGenerating(true)
try {
// 确保有 project_id(AI数字人入口独立,不在项目内,自动取默认项目;#1860 P0 bugfix
const defaultProject = await getOrCreateDefaultProject()
// 用 Canvas 预渲染标题为 PNG dataURL(所见即所得,后端用 overlay 直接叠加)
// 用 Canvas 预渲染标题为 PNG dataURL
let titleImageDataUrl: string | null = null
if (state.titleConfig.title?.trim()) {
try {
@@ -242,7 +381,6 @@ const AiAvatarPage: React.FC = () => {
pip_scale: seg.pip_scale,
})) as never,
title_config: buildTitleConfigPayload(state.titleConfig, titleImageDataUrl),
// 封面不阻塞渲染:用户未选定封面时传空 dict,后端不生成封面;渲染完成后再单独抽帧
cover_config:
state.coverConfig.smart_cover_url ||
(state.coverConfig.upload_url && !state.coverConfig.upload_url.startsWith("blob:"))
@@ -250,12 +388,14 @@ const AiAvatarPage: React.FC = () => {
: {},
})
// 打开渲染进度弹窗,启动轮询
setShowRenderModal(true)
setRenderStatus("generating")
setRenderProgress(job.progress ?? 0)
setRenderErrorMessage("")
setCurrentRenderJob(job as RenderJob)
// 每次新渲染重置封面状态
setSelectedCoverUrl("")
setShowCoverModal(false)
if (renderTimerRef.current) clearInterval(renderTimerRef.current)
renderTimerRef.current = setInterval(async () => {
@@ -267,8 +407,8 @@ const AiAvatarPage: React.FC = () => {
if (renderTimerRef.current) clearInterval(renderTimerRef.current)
renderTimerRef.current = null
setRenderStatus("completed")
// 渲染完成后:如果后端已返回封面(用户预上传/预设)则同步到前端;
// 否则不自动设置封面,由用户在封面区点击"智能获取封面"主动抽帧(步骤③④)
// 不再自动入库/自动跳转:渲染完成后停留在主页面,等用户选封面、点「完成」才入库
// 如果后端在透传时已经带了封面(旧逻辑兜底),同步本地状态
if (updated.output_cover_url) {
state.setCoverConfig((prev) => ({
...prev,
@@ -276,8 +416,9 @@ const AiAvatarPage: React.FC = () => {
smart_cover_url: updated.output_cover_url,
thumbnail_url: updated.output_cover_url,
}))
setSelectedCoverUrl(updated.output_cover_url)
}
message.success("视频生成并保存到成片库")
message.success("视频生成完成,请选择封面")
} else if (updated.status === "failed") {
if (renderTimerRef.current) clearInterval(renderTimerRef.current)
renderTimerRef.current = null
@@ -309,7 +450,7 @@ const AiAvatarPage: React.FC = () => {
setRenderErrorMessage("")
}, [])
/* ── 智能封面:从最终渲染成片抽帧(POST /renders/{id}/smart-cover,步骤③④) ── */
/* ── 智能封面 ── */
const handleGenerateRenderSmartCover = useCallback(
async (renderId: string): Promise<{ cover_url: string; message?: string }> => {
try {
@@ -334,11 +475,72 @@ const AiAvatarPage: React.FC = () => {
return { cover_url: "", message: errMsg }
}
},
// state.setCoverConfig 是 zustand action 引用稳定,eslint 不需要检查
// eslint-disable-next-line react-hooks/exhaustive-deps
[],
)
/* ── 封面弹窗回调 ── */
const handleCoverSelected = useCallback((coverUrl: string) => {
setSelectedCoverUrl(coverUrl || "")
}, [])
const handleOpenCoverModal = useCallback(() => {
if (currentRenderJob?.status !== "completed") {
message.warning("请先完成视频生成")
return
}
setShowCoverModal(true)
}, [currentRenderJob])
const handleCloseCoverModal = useCallback(() => {
setShowCoverModal(false)
}, [])
/* ── 自定义上传封面(本地预览,不单独上传;点完成时一起入库) ── */
const handleUploadCover = useCallback(
(file: File) => {
const url = URL.createObjectURL(file)
state.setCoverConfig((prev) => ({
...prev,
mode: "upload",
upload_url: url,
thumbnail_url: url,
}))
setSelectedCoverUrl(url)
},
[state],
)
/* ── 点「完成」:调用 finalize 入库成片库,成功后跳转到成片库 ── */
const handleFinalize = useCallback(async () => {
if (!currentRenderJob?.id) {
message.error("渲染任务不存在")
return
}
if (currentRenderJob.status !== "completed") {
message.warning("请先完成视频生成")
return
}
setFinalizeLoading(true)
try {
const res = await finalizeRenderJob(currentRenderJob.id)
if (res.data?.status === "success" || res.data?.status === "already_finalized") {
message.success("已保存到成片库")
navigate("/app/products")
} else {
message.error("保存失败,请重试")
}
} catch (err) {
console.error("finalize 失败:", err)
const errMsg =
(err as { response?: { data?: { detail?: unknown } } })?.response?.data?.detail ||
(err instanceof Error ? err.message : "保存到成片库失败")
message.error(typeof errMsg === "string" ? errMsg : "保存到成片库失败")
} finally {
setFinalizeLoading(false)
}
}, [currentRenderJob, navigate])
/* ── 配置汇总 ── */
const coverStatus: "not_ready" | "pending" | "selected" = (() => {
if (
@@ -431,19 +633,33 @@ const AiAvatarPage: React.FC = () => {
onOpenScriptModal={() => state.setShowScriptModal(true)}
/>
<div className="aa-step-btn-row">
<button type="button" className="aa-btn aa-btn--primary" onClick={handleNextStep}>
<button
type="button"
className="aa-btn aa-btn--primary"
onClick={handleGenerateTts}
disabled={state.ttsPreview.status === "generating"}
>
{state.ttsPreview.status === "done" ? "🎵 重新生成配音" : "🎵 生成配音"}
</button>
{state.ttsPreview.status === "done" && (
<button
type="button"
className="aa-btn aa-btn--primary"
onClick={() => setCurrentStep(2)}
style={{ marginLeft: 12 }}
>
</button>
)}
</div>
</div>
</div>
</>
)}
{/* ════ 步骤 2:对口型预览(含插入画面)/ 标题配置 / 封面&生成 ════ */}
{/* ════ 步骤 2:对口型预览 / 标题配置 / 封面&生成 ════ */}
{currentStep === 2 && (
<>
{/* 面板:对口型预览 + 插入画面 */}
<div className={`aa-panel aa-panel--s2-wide${collapsed.lipsync ? " collapsed" : ""}`}>
<div className="aa-panel__header" onClick={() => togglePanel("lipsync")}>
<span className="aa-panel__title"></span>
@@ -481,27 +697,98 @@ const AiAvatarPage: React.FC = () => {
</div>
</div>
{/* 面板5:封面 & 生成 */}
{/* 面板5:封面 & 生成
- 渲染未完成:显示分辨率/配置摘要/「开始生成」按钮(PanelCoverAndGenerate setup 变体,无封面区)
- 渲染完成:显示封面预览 + 「🎬 选择封面」/「✅ 完成」按钮,封面选择在弹窗中完成 */}
<div className={`aa-panel aa-panel--s2${collapsed.cover ? " collapsed" : ""}`}>
<div className="aa-panel__header" onClick={() => togglePanel("cover")}>
<span className="aa-panel__title"> & </span>
<span className="aa-panel__title">
{currentRenderJob?.status === "completed" ? "视频已生成" : "封面 & 生成"}
</span>
<span className="aa-panel__toggle"></span>
</div>
<div className="aa-panel__body">
<PanelCoverAndGenerate
coverConfig={state.coverConfig}
onCoverConfigChange={(partial) =>
state.setCoverConfig((prev) => ({ ...prev, ...partial }))
}
titleConfig={state.titleConfig}
renderJob={currentRenderJob}
onGenerateRenderSmartCover={handleGenerateRenderSmartCover}
resolution={state.resolution}
onResolutionChange={state.setResolution}
isGenerating={state.isGenerating}
onGenerate={handleGenerate}
summary={summary}
/>
{currentRenderJob?.status !== "completed" ? (
<PanelCoverAndGenerate
variant="setup"
coverConfig={state.coverConfig}
onCoverConfigChange={(partial) =>
state.setCoverConfig((prev) => ({ ...prev, ...partial }))
}
renderJob={currentRenderJob}
onGenerateRenderSmartCover={handleGenerateRenderSmartCover}
resolution={state.resolution}
onResolutionChange={state.setResolution}
isGenerating={state.isGenerating}
onGenerate={handleGenerate}
summary={summary}
/>
) : (
<div className="aa-finalize-section">
<div
style={{ marginBottom: 8, fontSize: 13, color: "#1a1a2e", fontWeight: 500 }}
>
</div>
<div className="aa-finalize-cover">
{selectedCoverUrl ? (
<img
src={selectedCoverUrl}
alt="封面预览"
draggable={false}
style={{
width: "100%",
height: "100%",
objectFit: "cover",
borderRadius: 8,
}}
/>
) : (
<div
style={{
width: "100%",
height: "100%",
border: "2px dashed #d9d9d9",
borderRadius: 8,
display: "flex",
alignItems: "center",
justifyContent: "center",
color: "#8c8ca1",
fontSize: 12,
flexDirection: "column",
gap: 4,
}}
>
<span style={{ fontSize: 24 }}>🎬</span>
<span></span>
</div>
)}
</div>
<div
style={{
display: "flex",
gap: 10,
marginTop: 12,
}}
>
<button
type="button"
className="aa-btn aa-btn--ghost"
onClick={handleOpenCoverModal}
>
🎬
</button>
<button
type="button"
className="aa-btn aa-btn--primary"
onClick={handleFinalize}
disabled={finalizeLoading}
>
{finalizeLoading ? "⏳ 保存中..." : "✅ 完成"}
</button>
</div>
</div>
)}
</div>
</div>
</>
@@ -527,20 +814,144 @@ const AiAvatarPage: React.FC = () => {
/>
)}
{/* B-roll 编辑器弹窗 */}
{/* B-roll 编辑器弹窗 — #1845:timings 在对口型完成前就可用(来自 TTS 预合成) */}
{state.showBRollModal && (
<ModalBRollEditor
open={state.showBRollModal}
onClose={() => state.setShowBRollModal(false)}
existingSegments={state.bRollSegments}
scriptText={state.lipsyncJob?.script_text || state.scriptText}
outputDuration={state.lipsyncJob?.output_duration ?? 0}
sentenceTimings={state.lipsyncJob?.sentence_timings}
outputDuration={bRollDuration}
sentenceTimings={bRollSentenceTimings}
onConfirm={state.addBRollSegment}
onRemove={state.removeBRollSegment}
/>
)}
{/* #1845 TTS 预合成弹窗 */}
{showTtsModal && (
<div className="aa-modal-overlay">
<div className="aa-modal" onClick={(e) => e.stopPropagation()}>
<div className="aa-modal__header">
<span className="aa-modal__title"></span>
{state.ttsPreview.status !== "generating" && (
<button className="aa-modal__close" onClick={handleCancelTts}>
</button>
)}
</div>
<div
className="aa-modal__body"
style={{
display: "flex",
flexDirection: "column",
alignItems: "center",
padding: "40px 20px",
}}
>
{state.ttsPreview.status === "generating" && (
<>
<div className="aa-lipsync-spinner" />
<div style={{ marginTop: 20, fontSize: 15, color: "#1a1a2e" }}>
</div>
<div
style={{
marginTop: 20,
fontSize: 32,
fontWeight: 700,
color: "#1890ff",
}}
>
{ttsProgress}%
</div>
<div
style={{
marginTop: 12,
width: "80%",
height: 8,
backgroundColor: "#f0f0f0",
borderRadius: 4,
overflow: "hidden",
}}
>
<div
style={{
width: `${ttsProgress}%`,
height: "100%",
backgroundColor: "#1890ff",
borderRadius: 4,
transition: "width 0.3s ease",
}}
/>
</div>
<div style={{ marginTop: 12, fontSize: 13, color: "#8c8ca1" }}>
</div>
</>
)}
{state.ttsPreview.status === "done" && (
<>
<div style={{ fontSize: 48 }}></div>
<div style={{ marginTop: 16, fontSize: 15, color: "#1a1a2e" }}>
</div>
<div style={{ marginTop: 8, fontSize: 13, color: "#8c8ca1" }}>
{state.ttsPreview.duration.toFixed(1)}s{" "}
{state.ttsPreview.sentenceTimings.length}
</div>
</>
)}
{state.ttsPreview.status === "failed" && (
<>
<div style={{ fontSize: 48 }}></div>
<div style={{ marginTop: 16, fontSize: 15, color: "#1a1a2e" }}></div>
{ttsErrorMessage && (
<div
style={{
marginTop: 8,
fontSize: 13,
color: "#ff4d4f",
textAlign: "center",
padding: "0 20px",
}}
>
{ttsErrorMessage}
</div>
)}
</>
)}
</div>
<div className="aa-modal__footer">
{state.ttsPreview.status === "generating" && (
<button className="aa-btn aa-btn--danger" onClick={handleCancelTts}>
</button>
)}
{state.ttsPreview.status === "done" && (
<button className="aa-btn aa-btn--primary" onClick={handleTtsNext}>
</button>
)}
{state.ttsPreview.status === "failed" && (
<>
<button className="aa-btn" onClick={handleCancelTts}>
</button>
<button
className="aa-btn aa-btn--primary"
onClick={handleRetryTts}
style={{ marginLeft: 12 }}
>
</button>
</>
)}
</div>
</div>
</div>
)}
{/* 对口型生成弹窗 */}
{showLipsyncModal && (
<div className="aa-modal-overlay">
@@ -673,17 +1084,21 @@ const AiAvatarPage: React.FC = () => {
<>
<div style={{ fontSize: 48 }}></div>
<div style={{ marginTop: 16, fontSize: 15, color: "#1a1a2e" }}>
</div>
<div
style={{ marginTop: 8, fontSize: 13, color: "#8c8ca1", textAlign: "center" }}
>
</div>
<button
className="aa-btn"
className="aa-btn aa-btn--primary"
style={{ marginTop: 16 }}
onClick={() => {
setShowRenderModal(false)
navigate("/app/products")
}}
>
📁
🎬
</button>
</>
)}
@@ -714,6 +1129,18 @@ const AiAvatarPage: React.FC = () => {
</div>
</div>
)}
{/* 封面选择弹窗 */}
<ModalCoverSelect
open={showCoverModal}
onClose={handleCloseCoverModal}
renderJob={currentRenderJob}
coverConfig={state.coverConfig}
onCoverConfigChange={(partial) => state.setCoverConfig((prev) => ({ ...prev, ...partial }))}
onGenerateRenderSmartCover={handleGenerateRenderSmartCover}
onUploadCover={handleUploadCover}
onCoverSelected={handleCoverSelected}
/>
</div>
)
}
+45 -9
View File
@@ -2,7 +2,7 @@
* AI数字人 — API 封装(#1822 契约对齐)
*/
import apiClient from "@/api/client"
import type { Script, LipsyncJob, RenderJob, BRollSegment } from "../types"
import type { Script, LipsyncJob, RenderJob, BRollSegment, SentenceTiming } from "../types"
/* ── 文案库 ── */
export const getScripts = async (): Promise<Script[]> => {
@@ -34,17 +34,28 @@ export const getAssetById = async (id: string): Promise<{ file_url?: string; id:
return response.data
}
/* ── 对口型(模式A:TTS 直生,后端内部合成音频;不要先调 TTS 拿 audio_url ── */
/* ── 对口型(支持三种模式) ──
* 1. TTS 直生(降级/旧版):传 voice_id + script_text+speed/emotion),后端 Celery 异步合成
* 2. 直接音频:传 video_url + audio_url,后端同步下载+算timings+提交MediaKit
* 3. 预合成音频(#1845 新主路径):先调 previewTts 拿 audio_url+sentence_timings
* 再把 audio_url + audio_duration + sentence_timings 一起传过来,后端直接提交 MediaKit
*/
export const createLipsyncJob = async (data: {
/** 人物视频 URLMP4);由素材 id 经 getAssetById 拿 file_url,禁止传 video_asset_id */
/** 人物视频 URLMP4);由素材 id 经 getAssetById 拿 file_url */
video_url: string
/** 音色 ID(预置音色 或 克隆音色 profile UUID,后端会解析 */
voice_id: string
/** 合成的文案(手动输入或文案库内容) */
script_text: string
/** 语速 0.5~2.0,默认 1.0 */
/** 预合成/直接音频模式:音频 URL(#1845 步骤1 预合成的 CosyVoice 临时 URL,或外部音频 URL */
audio_url?: string
/** 合成音频时长(秒),由 previewTts 返回 */
audio_duration?: number
/** 预合成接口返回的句子时间戳(精确),后端直接写入 job */
sentence_timings?: SentenceTiming[]
/** 音色 IDTTS 直生模式用) */
voice_id?: string
/** 要合成的文案(TTS 直生模式用) */
script_text?: string
/** 语速 0.5~2.0,默认 1.0TTS 直生模式用) */
speed?: number
/** 情绪英文枚举:natural/excited/calm/friendly */
/** 情绪英文枚举:natural/excited/calm/friendlyTTS 直生模式用) */
emotion?: string
enable_video_loop?: boolean
project_id?: string
@@ -53,6 +64,25 @@ export const createLipsyncJob = async (data: {
return response.data
}
/* ── #1845 TTS 预合成(步骤1「生成配音」同步接口,~2-3s) ── */
export const previewTts = async (data: {
voice_id: string
script_text: string
speed?: number
emotion?: string
}): Promise<{
audio_url: string
duration: number
sentence_timings: SentenceTiming[]
}> => {
const response = await apiClient.post<{
audio_url: string
duration: number
sentence_timings: SentenceTiming[]
}>("/lipsync/tts-preview", data, { timeout: 30000 })
return response.data
}
export const getLipsyncJob = async (id: string): Promise<LipsyncJob> => {
const response = await apiClient.get<LipsyncJob>(`/lipsync/jobs/${id}`, { timeout: 60000 })
return response.data
@@ -93,3 +123,9 @@ export const generateRenderSmartCover = async (
)
return response.data
}
/* ── 封面选定后点「完成」正式入库(POST /ai-avatar/render/{job_id}/finalize ── */
export const finalizeRenderJob = (renderId: string) =>
apiClient.post<{ video_id: string; cover_url: string; status: string }>(
`/ai-avatar/render/${renderId}/finalize`,
)
@@ -0,0 +1,59 @@
/**
* AI数字人 — 封面选择弹窗
* 渲染完成后由主页面唤起,内部用 PanelCoverAndGenerateselect-cover 变体)提供
* 智能抽帧 + 自定义上传 + 预览 + 确定按钮。
*/
import React from "react"
import type { AiAvatarCoverConfig, RenderJob } from "../types"
import PanelCoverAndGenerate from "./PanelCoverAndGenerate"
interface ModalCoverSelectProps {
open: boolean
onClose: () => void
renderJob: RenderJob | null
coverConfig: AiAvatarCoverConfig
onCoverConfigChange: (partial: Partial<AiAvatarCoverConfig>) => void
onGenerateRenderSmartCover: (renderId: string) => Promise<{ cover_url: string; message?: string }>
onUploadCover?: (file: File) => void
onCoverSelected: (coverUrl: string) => void
}
const ModalCoverSelect: React.FC<ModalCoverSelectProps> = ({
open,
onClose,
renderJob,
coverConfig,
onCoverConfigChange,
onGenerateRenderSmartCover,
onUploadCover,
onCoverSelected,
}) => {
if (!open) return null
return (
<div className="aa-modal-overlay" onClick={onClose}>
<div className="aa-modal" onClick={(e) => e.stopPropagation()} style={{ maxWidth: 480 }}>
<div className="aa-modal__header">
<span className="aa-modal__title"></span>
<button type="button" className="aa-modal__close" onClick={onClose} aria-label="关闭">
×
</button>
</div>
<div className="aa-modal__body" style={{ padding: 20 }}>
<PanelCoverAndGenerate
variant="select-cover"
coverConfig={coverConfig}
onCoverConfigChange={onCoverConfigChange}
renderJob={renderJob}
onGenerateRenderSmartCover={onGenerateRenderSmartCover}
onUploadCover={onUploadCover}
onClose={onClose}
onCoverSelected={onCoverSelected}
/>
</div>
</div>
</div>
)
}
export default ModalCoverSelect
@@ -1,32 +1,37 @@
/**
* AI数字人 — 面板5:分辨率/配置摘要/生成按钮/封面
* v3 调整(步骤③④):
* - 布局顺序:分辨率 → 配置摘要卡片 → 🔘「开始生成视频」按钮 → (渲染完成后)封面区域
* - 渲染未完成时封面区域显示占位态,按钮 disabled
* - 「智能获取封面」从最终成片抽帧(调用 POST /renders/{id}/smart-cover),不再依赖 lipsync 状态
* - 修复点 2 次 bug:内部维护 smartCoverLoading,不依赖外层异步 state 更新
* AI数字人 — 面板5 / 封面选择弹窗内容:
* - variant="setup"(默认):分辨率 / 配置摘要 / 「开始生成视频」按钮,用于主页面步骤2配置阶段;
* 渲染完成后仍内嵌封面预览与按钮,方便不打开弹窗直接操作。
* - variant="select-cover":只渲染封面选择区(智能获取封面 + 自定义上传 + 预览),
* 用于 ModalCoverSelect 弹窗中;传 onClose 时底部显示「确定」按钮。
*
* 注意:v3 已删除"画面插入模式",本面板不包含该选项
* 封面一律从最终成片(已叠加标题/B-roll)抽帧,本面板不再叠加标题
*/
import React, { useMemo, useRef, useState } from "react"
import type { AiAvatarCoverConfig, AiAvatarTitleConfig, RenderJob } from "../types"
import React, { useRef, useState } from "react"
import type { AiAvatarCoverConfig, RenderJob } from "../types"
type PanelVariant = "setup" | "select-cover"
interface PanelCoverAndGenerateProps {
variant?: PanelVariant
coverConfig: AiAvatarCoverConfig
titleConfig: AiAvatarTitleConfig
onCoverConfigChange: (partial: Partial<AiAvatarCoverConfig>) => void
resolution: string
onResolutionChange: (r: string) => void
isGenerating: boolean
onGenerate: () => void
resolution?: string
onResolutionChange?: (r: string) => void
isGenerating?: boolean
onGenerate?: () => void
/** 当前渲染任务(渲染完成后才有 output_video_url,才能抽封面) */
renderJob: RenderJob | null
/** 从最终成片智能抽帧(参数 renderId),返回 { cover_url } */
onGenerateRenderSmartCover: (renderId: string) => Promise<{ cover_url: string; message?: string }>
/** 自定义上传封面(选择本地文件后由父组件处理实际上传) */
onUploadCover?: (file: File) => void
/** 配置汇总信息 */
summary: {
/** 弹窗关闭回调(传入则表示在弹窗中使用,底部显示「确定」按钮) */
onClose?: () => void
/** 封面选好(智能抽帧/自定义上传成功)后通知父组件,参数为封面 URL */
onCoverSelected?: (coverUrl: string) => void
/** 配置汇总信息(仅 variant="setup" 使用) */
summary?: {
videoName: string | null
voiceName: string | null
scriptLength: number
@@ -52,27 +57,19 @@ const LIPSYNC_STATUS_LABEL: Record<string, { text: string; cls: string }> = {
failed: { text: "失败", cls: "aa-status-badge--failed" },
}
/** 字体名 → CSS font-family 映射(与后端 drawtext 对齐) */
const FONT_FAMILY_MAP: Record<string, string> = {
: "'Noto Sans SC', 'Source Han Sans SC', 'PingFang SC', 'Microsoft YaHei', sans-serif",
: "'Noto Serif SC', 'Source Han Serif SC', 'SimSun', serif",
: "KaiTi, 'STKaiti', serif",
: "'Heiti SC', 'SimHei', 'Microsoft YaHei', sans-serif",
}
const getFontFamily = (font: string): string => FONT_FAMILY_MAP[font] || FONT_FAMILY_MAP["思源黑体"]
const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
variant = "setup",
coverConfig,
titleConfig,
onCoverConfigChange,
resolution,
resolution = "720p",
onResolutionChange,
isGenerating,
isGenerating = false,
onGenerate,
renderJob,
onGenerateRenderSmartCover,
onUploadCover,
onClose,
onCoverSelected,
summary,
}) => {
const uploadInputRef = useRef<HTMLInputElement>(null)
@@ -84,16 +81,31 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
uploadInputRef.current?.click()
}
const _applyCoverUrl = (url: string, mode: "upload" | "auto_frame") => {
const partial: Partial<AiAvatarCoverConfig> = {
mode,
thumbnail_url: url,
}
if (mode === "auto_frame") {
partial.smart_cover_url = url
} else {
partial.upload_url = url
}
onCoverConfigChange(partial)
onCoverSelected?.(url)
}
const handleFileChange = (e: React.ChangeEvent<HTMLInputElement>) => {
const file = e.target.files?.[0]
if (!file) return
if (onUploadCover) {
onUploadCover(file)
} else {
// 本地预览兜底(实际上传由父级处理;blob URL 仅作本地展示)
const url = URL.createObjectURL(file)
onCoverConfigChange({ mode: "upload", upload_url: url, thumbnail_url: url })
e.target.value = ""
return
}
// 本地预览兜底(实际上传由父级处理;blob URL 仅作本地展示)
const url = URL.createObjectURL(file)
_applyCoverUrl(url, "upload")
e.target.value = ""
}
@@ -104,11 +116,7 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
try {
const res = await onGenerateRenderSmartCover(renderJob.id)
if (res.cover_url) {
onCoverConfigChange({
mode: "auto_frame",
smart_cover_url: res.cover_url,
thumbnail_url: res.cover_url,
})
_applyCoverUrl(res.cover_url, "auto_frame")
} else {
// 失败由父组件 message 提示,这里不重复弹窗
console.warn("[智能封面] 返回空 cover_url:", res.message)
@@ -120,8 +128,8 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
}
}
const lipsync = summary.lipsyncStatus ? LIPSYNC_STATUS_LABEL[summary.lipsyncStatus] : null
const canGenerate = summary.lipsyncStatus === "completed" && !isGenerating
const lipsync = summary?.lipsyncStatus ? LIPSYNC_STATUS_LABEL[summary.lipsyncStatus] : null
const canGenerate = summary?.lipsyncStatus === "completed" && !isGenerating
// 渲染已完成 → 封面区可用
const isRenderCompleted = renderJob?.status === "completed"
const canSmartCover = isRenderCompleted && !smartCoverLoading
@@ -131,63 +139,12 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
coverConfig.smart_cover_url || coverConfig.thumbnail_url || coverConfig.upload_url
const hasCoverImage = Boolean(coverUrl)
/** 是否显示标题叠加层:有图、有文字、非加载中 */
const showTitleOverlay =
hasCoverImage && !smartCoverLoading && titleConfig.title.trim().length > 0
/** 计算标题叠加层的 inline 样式 */
const titleOverlayStyle = useMemo<React.CSSProperties>(() => {
const style: React.CSSProperties = {
position: "absolute",
left: "50%",
width: "90%",
transform: "translateX(-50%)",
textAlign: "center",
boxSizing: "border-box",
padding: "0 4px",
wordBreak: "break-word",
whiteSpace: "pre-wrap",
color: titleConfig.color || "#ffffff",
fontSize: `${(titleConfig.size || 48) * 0.35}px`,
fontFamily: getFontFamily(titleConfig.font),
fontWeight: titleConfig.bold ? "bold" : "normal",
fontStyle: titleConfig.italic ? "italic" : "normal",
lineHeight: 1.3,
pointerEvents: "none",
}
const pos = titleConfig.position || "bottom"
if (pos === "top") {
style.top = "40px"
} else if (pos === "center") {
style.top = "50%"
style.transform = "translate(-50%, -50%)"
} else if (pos === "custom" && titleConfig.pos_x != null && titleConfig.pos_y != null) {
style.left = `${titleConfig.pos_x}%`
style.top = `${titleConfig.pos_y}%`
style.transform = "translate(-50%, -50%)"
} else {
style.bottom = "40px"
}
if (titleConfig.stroke) {
const strokeWidth = Math.max(1, Math.round(titleConfig.size / 18))
;(style as React.CSSProperties)["WebkitTextStroke"] = `${strokeWidth}px rgba(0,0,0,0.75)`
style.textShadow = "none"
} else if (titleConfig.shadow) {
style.textShadow = "0 2px 8px rgba(0,0,0,0.7), 0 0 2px rgba(0,0,0,0.5)"
} else {
style.textShadow = "none"
}
return style
}, [titleConfig])
/** 封面区占位文字 */
const coverPlaceholder = isRenderCompleted ? "暂无封面" : "视频生成后可选择封面"
/** 封面摘要状态文本 */
/** 配置摘要中的封面状态标签 */
const coverSummaryNode = (() => {
if (!summary) return null
if (summary.coverStatus === "selected") {
return <span className="aa-config-summary__value"></span>
}
@@ -197,6 +154,69 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
return <span className="aa-config-summary__empty"></span>
})()
// ── 封面选择区(两种 variant 共用) ─────────────────────────────────
const coverSection = (
<div className="aa-cover-section" style={{ marginTop: variant === "select-cover" ? 0 : 16 }}>
<div className="aa-label" style={{ marginBottom: 8 }}>
{variant === "select-cover" ? "选择封面" : "封面"}
</div>
{/* 封面预览(竖屏 9:16)——成片帧已经通过 Canvas PNG overlay 带有标题,直接展示原图即可 */}
<div className="aa-cover-preview" style={{ opacity: isRenderCompleted ? 1 : 0.5 }}>
{hasCoverImage ? (
<img src={coverUrl!} alt="封面预览" draggable={false} />
) : (
<span className="aa-cover-preview__placeholder">{coverPlaceholder}</span>
)}
{smartCoverLoading && <div className="aa-cover-preview__loading"> </div>}
</div>
<div className="aa-cover-actions">
<button
type="button"
className={`aa-btn aa-btn--ghost${coverConfig.mode === "auto_frame" ? " active" : ""}`}
onClick={handleSmartCover}
disabled={!canSmartCover}
title={isRenderCompleted ? "从成片智能选帧" : "请先生成视频"}
>
{smartCoverLoading ? "⏳ 智能选帧中…" : "🎬 智能获取封面"}
</button>
<button
type="button"
className={`aa-btn aa-btn--ghost${coverConfig.mode === "upload" ? " active" : ""}`}
onClick={handleUploadClick}
disabled={!isRenderCompleted || smartCoverLoading}
title={isRenderCompleted ? "自定义上传封面" : "请先生成视频"}
>
📷
</button>
<input
ref={uploadInputRef}
type="file"
accept="image/*"
style={{ display: "none" }}
onChange={handleFileChange}
/>
</div>
</div>
)
// ── select-cover 变体:只渲染封面区 + 弹窗确定按钮 ──
if (variant === "select-cover") {
return (
<div className="aa-cover-generate">
{coverSection}
{onClose && (
<div style={{ marginTop: 16, display: "flex", justifyContent: "flex-end" }}>
<button type="button" className="aa-btn aa-btn--primary" onClick={onClose}>
</button>
</div>
)}
</div>
)
}
// ── setup 变体:分辨率 / 配置摘要 / 生成按钮(渲染完成后内嵌封面区) ──
return (
<div className="aa-cover-generate">
{/* 分辨率选择 */}
@@ -205,7 +225,7 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
<select
className="aa-select"
value={resolution}
onChange={(e) => onResolutionChange(e.target.value)}
onChange={(e) => onResolutionChange?.(e.target.value)}
disabled={isGenerating}
>
{RESOLUTION_OPTIONS.map((opt) => (
@@ -221,7 +241,7 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
<div className="aa-config-summary">
<div className="aa-config-summary__row">
<span></span>
{summary.videoName ? (
{summary?.videoName ? (
<span className="aa-config-summary__value">{summary.videoName}</span>
) : (
<span className="aa-config-summary__empty"></span>
@@ -229,7 +249,7 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
</div>
<div className="aa-config-summary__row">
<span></span>
{summary.voiceName ? (
{summary?.voiceName ? (
<span className="aa-config-summary__value">{summary.voiceName}</span>
) : (
<span className="aa-config-summary__empty"></span>
@@ -237,7 +257,7 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
</div>
<div className="aa-config-summary__row">
<span></span>
{summary.scriptLength > 0 ? (
{summary && summary.scriptLength > 0 ? (
<span className="aa-config-summary__value">{summary.scriptLength} </span>
) : (
<span className="aa-config-summary__empty"></span>
@@ -254,12 +274,12 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
<div className="aa-config-summary__row">
<span>B-roll </span>
<span className="aa-config-summary__value">
{summary.brollCount > 0 ? `${summary.brollCount}` : "无"}
{summary && summary.brollCount > 0 ? `${summary.brollCount}` : "无"}
</span>
</div>
<div className="aa-config-summary__row">
<span></span>
{summary.hasTitle ? (
{summary?.hasTitle ? (
<span className="aa-config-summary__value"></span>
) : (
<span className="aa-config-summary__empty"></span>
@@ -280,7 +300,7 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
>
{isGenerating ? "⏳ 生成中..." : "🚀 开始生成视频"}
</button>
{summary.lipsyncStatus !== "completed" && !isGenerating && (
{summary?.lipsyncStatus !== "completed" && !isGenerating && (
<div style={{ marginTop: 8, fontSize: 11, color: "#8c8ca1", textAlign: "center" }}>
</div>
@@ -291,55 +311,6 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
</div>
)}
</div>
{/* 封面区域(视频生成后才激活;步骤③④要求:按钮在封面上方,完成后再显示封面区) */}
<div className="aa-cover-section" style={{ marginTop: 16 }}>
<div className="aa-label" style={{ marginBottom: 8 }}>
</div>
{/* 封面预览(竖屏 9:16 */}
<div className="aa-cover-preview" style={{ opacity: isRenderCompleted ? 1 : 0.5 }}>
{hasCoverImage ? (
<img src={coverUrl!} alt="封面预览" draggable={false} />
) : (
<span className="aa-cover-preview__placeholder">{coverPlaceholder}</span>
)}
{smartCoverLoading && <div className="aa-cover-preview__loading"> </div>}
{showTitleOverlay && (
<div style={titleOverlayStyle} aria-hidden="true">
{titleConfig.title}
</div>
)}
</div>
<div className="aa-cover-actions">
<button
type="button"
className={`aa-btn aa-btn--ghost${coverConfig.mode === "auto_frame" ? " active" : ""}`}
onClick={handleSmartCover}
disabled={!canSmartCover}
title={isRenderCompleted ? "从成片智能选帧" : "请先生成视频"}
>
{smartCoverLoading ? "⏳ 智能选帧中…" : "🎬 智能获取封面"}
</button>
<button
type="button"
className={`aa-btn aa-btn--ghost${coverConfig.mode === "upload" ? " active" : ""}`}
onClick={handleUploadClick}
disabled={!isRenderCompleted || smartCoverLoading}
title={isRenderCompleted ? "自定义上传封面" : "请先生成视频"}
>
📷
</button>
<input
ref={uploadInputRef}
type="file"
accept="image/*"
style={{ display: "none" }}
onChange={handleFileChange}
/>
</div>
</div>
</div>
)
}
@@ -1,9 +1,9 @@
/**
* AI数字人 — 对口型预览面板(步骤2用)
* B-roll 画面插入 + 对口型视频预览 + 生成/重新生成按钮
* v3.1: 预览容器按 1/2 缩放、标题实时叠加预览
* v3.1: 标题字号按预览容器实际宽度动态计算 previewScale(基准 720p),与成片一致
*/
import React, { useRef } from "react"
import React, { useCallback, useEffect, useRef, useState } from "react"
import type { LipsyncJob, BRollSegment, AiAvatarTitleConfig } from "../types"
interface PanelLipsyncPreviewProps {
@@ -29,6 +29,16 @@ function formatTime(seconds: number): string {
return `${m}:${s.toString().padStart(2, "0")}`
}
/** 字体名 → CSS font-family 映射(与 titleCanvas 字体链对齐) */
const FONT_FAMILY_MAP: Record<string, string> = {
:
"'Noto Sans CJK SC', 'Source Han Sans CN', 'PingFang SC', 'Microsoft YaHei', sans-serif",
: "'Noto Serif SC', 'Source Han Serif SC', 'SimSun', serif",
: "KaiTi, 'STKaiti', serif",
: "'Heiti SC', 'SimHei', 'Microsoft YaHei', sans-serif",
}
const getFontFamily = (font: string): string => FONT_FAMILY_MAP[font] || FONT_FAMILY_MAP["思源黑体"]
export function PanelLipsyncPreview({
lipsyncJob,
onGenerateLipsync,
@@ -41,6 +51,8 @@ export function PanelLipsyncPreview({
const titleDragRef = useRef<HTMLDivElement>(null)
const draggingTitleRef = useRef(false)
const previewContainerRef = useRef<HTMLDivElement>(null)
// 预览容器实际宽度(通过 ResizeObserver 监听),用于动态计算 previewScale
const [containerWidth, setContainerWidth] = useState(0)
const isGenerating = lipsyncJob?.status === "pending" || lipsyncJob?.status === "processing"
const isDone = lipsyncJob?.status === "completed"
const isFailed = lipsyncJob?.status === "failed"
@@ -52,35 +64,84 @@ export function PanelLipsyncPreview({
? "排队中…"
: "对口型生成中…"
/** 标题叠加样式 */
const titleOverlayStyle: React.CSSProperties | null = titleConfig?.title
? {
position: "absolute",
color: titleConfig.color || "#ffffff",
fontFamily: titleConfig.font || "思源黑体",
fontSize: `${(titleConfig.size || 48) * 0.35}px`,
fontWeight: titleConfig.bold ? 700 : 400,
fontStyle: titleConfig.italic ? "italic" : "normal",
textAlign: "center",
width: "90%",
padding: "4px 8px",
textShadow: titleConfig.shadow ? "0 2px 4px rgba(0,0,0,0.8)" : undefined,
WebkitTextStroke: titleConfig.stroke ? "1.5px #000" : undefined,
...(titleConfig.position === "custom" &&
titleConfig.pos_x != null &&
titleConfig.pos_y != null
? {
left: `${titleConfig.pos_x}%`,
top: `${titleConfig.pos_y}%`,
transform: "translateX(-50%) translateY(-50%)",
}
: titleConfig.position === "top"
? { left: "50%", top: 8, transform: "translateX(-50%)" }
: titleConfig.position === "bottom"
? { left: "50%", bottom: 8, transform: "translateX(-50%)" }
: { left: "50%", top: "50%", transform: "translateX(-50%) translateY(-50%)" }),
}
: null
// 监听预览容器尺寸变化,动态测量宽度以计算 previewScale(基准 720p
useEffect(() => {
const el = previewContainerRef.current
if (!el) return
const update = () => setContainerWidth(el.clientWidth || 0)
update()
if (typeof ResizeObserver !== "undefined") {
const ro = new ResizeObserver(update)
ro.observe(el)
return () => ro.disconnect()
}
window.addEventListener("resize", update)
return () => window.removeEventListener("resize", update)
}, [])
// 预览缩放比:预览宽度 / 720(基准宽度)
const previewScale = containerWidth > 0 ? containerWidth / 720 : 0.35
const ps = useCallback((v: number) => Math.round(v * previewScale * 100) / 100, [previewScale])
/** 标题叠加样式(字号/padding/描边/阴影均按 previewScale 缩放,保持与成片视觉一致) */
const titleOverlayStyle: React.CSSProperties | null =
titleConfig?.title && containerWidth > 0
? (() => {
const baseSize = titleConfig.size || 48
const fontSize = ps(baseSize)
// 描边宽度基准 ≈ size * 0.06,最小 1.5px @720p
const strokeW = Math.max(ps(1.5), +(baseSize * 0.06 * previewScale).toFixed(2))
// 阴影按比例缩放
const shadowBlur = ps(4)
const shadowOffsetY = ps(2)
// padding / top 边距按比例(基准 8px 对应预览小窗,成片基准 16px,这里 8px 对应约 0.33 缩放)
const padV = ps(16) * 0.5 // ≈ 8px in ~240px container
const padH = ps(24) * 0.5
const style: React.CSSProperties = {
position: "absolute",
color: titleConfig.color || "#ffffff",
fontFamily: getFontFamily(titleConfig.font || "思源黑体"),
fontSize: `${fontSize}px`,
fontWeight: titleConfig.bold ? 700 : 400,
fontStyle: titleConfig.italic ? "italic" : "normal",
textAlign: "center",
width: "90%",
lineHeight: 1.2,
padding: `${ps(4)}px ${padH}px`,
textShadow: titleConfig.shadow
? `0 ${shadowOffsetY}px ${shadowBlur}px rgba(0,0,0,0.8), 0 0 ${ps(2)}px rgba(0,0,0,0.5)`
: undefined,
WebkitTextStroke: titleConfig.stroke ? `${strokeW}px #000` : undefined,
boxSizing: "border-box",
wordBreak: "break-word",
whiteSpace: "pre-wrap",
}
if (
titleConfig.position === "custom" &&
titleConfig.pos_x != null &&
titleConfig.pos_y != null
) {
style.left = `${titleConfig.pos_x}%`
style.top = `${titleConfig.pos_y}%`
style.transform = "translateX(-50%) translateY(-50%)"
} else if (titleConfig.position === "top") {
style.left = "50%"
style.top = padV
style.transform = "translateX(-50%)"
} else if (titleConfig.position === "bottom") {
style.left = "50%"
style.bottom = padV
style.transform = "translateX(-50%)"
} else {
style.left = "50%"
style.top = "50%"
style.transform = "translateX(-50%) translateY(-50%)"
}
return style
})()
: null
const handleTitlePointerDown = (e: React.PointerEvent<HTMLDivElement>) => {
if (!onTitlePositionChange || !previewContainerRef.current) return
@@ -183,7 +244,7 @@ export function PanelLipsyncPreview({
)}
</div>
{/* ── 对口型预览(v3.1: 缩放1/2 + 标题叠加 ─ */}
{/* ── 对口型预览(标题字号按 previewScale 动态缩放 ─ */}
<div className="aa-lipsync-section">
<div className="aa-lipsync-section__title"></div>
@@ -1,5 +1,5 @@
/**
* AI数字人 — 页面全局状态管理 hook(v3)
* AI数字人 — 页面全局状态管理 hook(v3 + #1845 配音前置
*/
import { useState, useCallback } from "react"
import type { AssetItem } from "@/api/assets"
@@ -13,10 +13,19 @@ import {
type BRollSegment,
type AiAvatarTitleConfig,
type AiAvatarCoverConfig,
type TtsPreviewResult,
DEFAULT_TITLE_CONFIG,
DEFAULT_COVER_CONFIG,
} from "../types"
const DEFAULT_TTS_PREVIEW: TtsPreviewResult = {
audioUrl: null,
duration: 0,
sentenceTimings: [],
status: "idle",
error: null,
}
export function useAiAvatar() {
/* ── 面板1:出镜视频 ── */
const [selectedVideo, setSelectedVideo] = useState<AssetItem | null>(null)
@@ -36,6 +45,9 @@ export function useAiAvatar() {
const [showScriptModal, setShowScriptModal] = useState(false)
const [showBRollModal, setShowBRollModal] = useState(false)
/* ── #1845 TTS 预合成(步骤1「生成配音」) ── */
const [ttsPreview, setTtsPreview] = useState<TtsPreviewResult>(DEFAULT_TTS_PREVIEW)
/* ── 面板3.5B-roll ── */
const [bRollSegments, setBRollSegments] = useState<BRollSegment[]>([])
@@ -81,6 +93,7 @@ export function useAiAvatar() {
setScript(null)
setScriptText("")
setLipsyncJob(null)
setTtsPreview(DEFAULT_TTS_PREVIEW)
setBRollSegments([])
setTitleConfig(DEFAULT_TITLE_CONFIG)
setCoverConfig(DEFAULT_COVER_CONFIG)
@@ -118,6 +131,10 @@ export function useAiAvatar() {
showBRollModal,
setShowBRollModal,
selectScript,
// #1845 TTS 预合成
ttsPreview,
setTtsPreview,
resetTtsPreview: useCallback(() => setTtsPreview(DEFAULT_TTS_PREVIEW), []),
// B-roll
bRollSegments,
addBRollSegment,
+11
View File
@@ -28,6 +28,17 @@ export const VOICE_LANGUAGE_OPTIONS: { value: VoiceLanguage; label: string }[] =
/* ── 对口型任务状态 ── */
export type LipsyncStatus = "idle" | "pending" | "processing" | "completed" | "failed"
/* ── TTS 预合成(#1845 配音前置:步骤1「生成配音」状态) ── */
export type TtsPreviewStatus = "idle" | "generating" | "done" | "failed"
export interface TtsPreviewResult {
audioUrl: string | null
duration: number
sentenceTimings: SentenceTiming[]
status: TtsPreviewStatus
error: string | null
}
/* ── 文案 ── */
export interface Script {
id: string
@@ -4,6 +4,10 @@
* 把标题按前端预览的 HTML/CSS 效果画到透明背景 PNG 上(与视频同分辨率),
* 以 dataURL 形式传给后端,后端用 FFmpeg overlay 直接叠加图层,
* 彻底解决前端 HTML/CSS 预览 ≠ FFmpeg drawtext 成片的 WYSIWYG 问题。
*
* 约定:titleConfig.size 的语义是"720p 基准宽度下的字号(px",
* 按 videoWidth / 720 得到 scale,所有长度类参数乘以 scale,
* 保证 1080p / 4K 成片里标题视觉大小与预览一致。
*/
import type { AiAvatarTitleConfig } from "../types"
@@ -35,13 +39,18 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
.filter((l) => l.length > 0)
if (lines.length === 0) return null
// 分辨率缩放系数:基准 720p,所有长度类参数乘以 scale
const scale = videoWidth / 720
const r = (v: number) => Math.round(v * scale)
const canvas = document.createElement("canvas")
canvas.width = videoWidth
canvas.height = videoHeight
const ctx = canvas.getContext("2d")
if (!ctx) return null
const size = Math.max(12, Math.round(titleConfig.size || 48))
const baseSize = Math.max(12, Math.round(titleConfig.size || 48))
const size = r(baseSize)
const bold = !!titleConfig.bold
const italic = !!titleConfig.italic
const color = titleConfig.color || "#ffffff"
@@ -60,19 +69,16 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
ctx.textAlign = "center"
ctx.textBaseline = "middle"
// 阴影(shadow=true 时开启)
// 阴影(shadow=true 时开启)——按 scale 缩放
if (shadow) {
ctx.shadowColor = "rgba(0,0,0,0.8)"
ctx.shadowBlur = 4
ctx.shadowBlur = r(4)
ctx.shadowOffsetX = 0
ctx.shadowOffsetY = 2
ctx.shadowOffsetY = r(2)
}
// 位置计算:与 PanelLipsyncPreview 的 CSS 对齐
// 预览用 top/bottom 8px padding + transform translateX(-50%) 居中;
// 这里画到整尺寸 canvas,padding 按比例放大到全分辨率(预览缩放 0.35x 时 8px ≈ 23px 全尺寸,
// 为更贴近原 CSS 16px 安全边距,用 16px 作为内边距)。
const PAD = 16
// 位置计算:与 PanelLipsyncPreview 的 CSS 对齐(按 scale 缩放 PAD
const PAD = r(16)
let centerX = videoWidth / 2
const position = titleConfig.position || "bottom"
const lineGap = size * 1.2
@@ -97,10 +103,9 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
firstLineY = videoHeight - totalTextH - PAD + size / 2
}
// 描边参数stroke=true 或 bold 默认细描边模拟粗体时都画;
// 注意:浏览器原生 bold 已经是粗体 glyphCanvas 这里对 stroke=true 才加黑描边,
// 与预览 CSS 的 WebkitTextStroke 保持一致,不对 bold 自动加描边避免双粗)。
// 描边参数:描边 lineWidth 按 scale 缩放(基准 size * 0.06,最小 2px @720p
const doStroke = stroke
const strokeWidth = Math.max(r(2), Math.round(size * 0.06))
// 逐行绘制
lines.forEach((line, idx) => {
const y = firstLineY + idx * lineGap
@@ -110,14 +115,14 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
// 描边不要带阴影(避免黑色描边发虚)
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
ctx.lineWidth = Math.max(2, size * 0.06)
ctx.lineWidth = strokeWidth
ctx.strokeStyle = "#000000"
ctx.lineJoin = "round"
ctx.strokeText(line, centerX, y)
// 恢复阴影
if (shadow) {
ctx.shadowColor = "rgba(0,0,0,0.8)"
ctx.shadowBlur = 4
ctx.shadowBlur = r(4)
} else {
ctx.shadowColor = prevShadowColor
ctx.shadowBlur = prevShadowBlur
@@ -221,17 +221,18 @@ class RenderAdapter:
except subprocess.CalledProcessError as exc:
stderr_text = (exc.stderr or "").strip()
stderr_tail = stderr_text[-5000:] if len(stderr_text) > 5000 else stderr_text
logger.error(
"[render-adapter] ffmpeg渲染失败: plan_id=%s job_id=%s exit_code=%d\nstderr:\n%s",
plan_id,
job_id,
exc.returncode,
stderr_text[-2000:] if len(stderr_text) > 2000 else stderr_text,
stderr_tail,
)
return RenderAdapterResult(
success=False,
error_message=f"FFmpeg渲染失败(exit={exc.returncode}): {stderr_text[:200]}",
error_detail=stderr_text[-2000:] if len(stderr_text) > 2000 else stderr_text,
error_message=f"FFmpeg渲染失败(exit={exc.returncode}): {stderr_text[-500:]}",
error_detail=stderr_tail,
)
except Exception as exc:
logger.exception(
@@ -721,16 +722,17 @@ class RenderAdapter:
except subprocess.CalledProcessError as exc:
stderr_text = (exc.stderr or "").strip()
stderr_tail = stderr_text[-5000:] if len(stderr_text) > 5000 else stderr_text
logger.error(
"[render-adapter] 内存模式渲染失败: plan_id=%s exit_code=%d\nstderr:\n%s",
actual_plan_id,
exc.returncode,
stderr_text[-2000:] if len(stderr_text) > 2000 else stderr_text,
stderr_tail,
)
return RenderAdapterResult(
success=False,
error_message=f"FFmpeg渲染失败(exit={exc.returncode}): {stderr_text[:200]}",
error_detail=stderr_text[-2000:] if len(stderr_text) > 2000 else stderr_text,
error_message=f"FFmpeg渲染失败(exit={exc.returncode}): {stderr_text[-500:]}",
error_detail=stderr_tail,
)
except Exception as exc:
logger.exception(
@@ -2301,13 +2301,13 @@ class UnifiedRenderService:
filters.append(f"eq=contrast={contrast:.3f}")
elif filt == "color_balance":
# RGB 通道偏移:color_balance=rs=...:gs=...:bs=...
# RGB 通道偏移:colorbalance=rs=...:gs=...:bs=...
r = pixel_pert.get("color_r", 0)
g = pixel_pert.get("color_g", 0)
b = pixel_pert.get("color_b", 0)
if r != 0 or g != 0 or b != 0:
# color_balance 参数范围 -1.0 ~ 1.0,这里用 /100 转换
filters.append(f"color_balance=rs={r/100:.3f}:gs={g/100:.3f}:bs={b/100:.3f}")
filters.append(f"colorbalance=rs={r/100:.3f}:gs={g/100:.3f}:bs={b/100:.3f}")
@staticmethod
def _clip_volume(clip: ResolvedClip) -> float:
+2 -1
View File
@@ -705,7 +705,8 @@ def _render_from_edit_plan(
)
if not result.success:
raise RuntimeError(f"渲染失败: {result.error_message}")
detail_suffix = f"\n[detail] {result.error_detail}" if result.error_detail else ""
raise RuntimeError(f"渲染失败: {result.error_message}{detail_suffix}")
render_elapsed = time.monotonic() - render_start
logger.info(
+205
View File
@@ -0,0 +1,205 @@
"""共享的句子时间戳计算工具 — 供 Celery TTS 任务和 /lipsync/tts-preview 同步接口复用.
- `_split_script_into_sentences`: 按标点分句(中英文逗号/句号/问号/感叹号/分号/换行)
- `_estimate_sentence_timings_by_chars`: 按字数比例估算(静音检测失败时降级)
- `_probe_audio_duration`: ffprobe 读取音频时长
- `compute_sentence_timings`: 基于 ffmpeg silencedetect 精确计算每句起止时间
"""
from __future__ import annotations
import logging
import os
import re
import subprocess
import tempfile
from typing import Optional
logger = logging.getLogger(__name__)
def split_script_into_sentences(script_text: str) -> list[str]:
"""按句号/问号/感叹号/分号/逗号/换行分句(与前端 SENTENCE_SPLIT_RE 一致).
中文短视频文案习惯用「,」断小句(如"卖花的叫花无缺,卖姜的叫姜子牙"),
必须把逗号也纳入分隔符,否则多句文案会被识别成一整句,导致 B-roll 时间戳错位。
"""
text = (script_text or "").strip()
if not text:
return []
parts = re.split(r"[。!?!??!;,\n\r]+", text)
return [p.strip() for p in parts if p.strip()]
def estimate_sentence_timings_by_chars(sentences: list[str], total_duration: float) -> list[dict]:
"""降级方案:按字数比例估算句子时间(与原前端逻辑一致)."""
if not sentences or total_duration <= 0:
return []
total_chars = sum(len(s.replace(r"\s", "")) for s in sentences)
if total_chars == 0:
return []
timings = []
acc = 0
for i, sent in enumerate(sentences):
chars = len(sent.replace(r"\s", ""))
start = (acc / total_chars) * total_duration
end = ((acc + chars) / total_chars) * total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
acc += chars
return timings
def probe_audio_duration(audio_data: bytes, timeout: int = 10) -> float:
"""用 ffprobe 读取音频字节流的时长(秒).
Returns:
时长(秒),失败返回 0.0
"""
if not audio_data:
return 0.0
tmp_path: Optional[str] = None
try:
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp:
tmp.write(audio_data)
tmp_path = tmp.name
result = subprocess.run(
[
"ffprobe",
"-v",
"error",
"-show_entries",
"format=duration",
"-of",
"default=noprint_wrappers=1:nokey=1",
tmp_path,
],
capture_output=True,
text=True,
timeout=timeout,
)
stdout = (result.stdout or "").strip()
if not stdout:
logger.warning("[sentence_timings] ffprobe 无输出: stderr=%s", (result.stderr or "")[:200])
return 0.0
return float(stdout)
except Exception as exc:
logger.warning("[sentence_timings] ffprobe 时长探测失败: %s", exc)
return 0.0
finally:
if tmp_path:
try:
os.unlink(tmp_path)
except Exception:
pass
def compute_sentence_timings(audio_data: bytes, script_text: str, total_duration: float) -> list[dict]:
"""基于 TTS 音频的静音检测,精确计算每句文案的起止时间.
使用 ffmpeg silencedetect 检测静音段,将静音点与句子边界对齐。
比字数比例估算准确得多。
Args:
audio_data: TTS 音频二进制数据(MP3
script_text: 文案全文
total_duration: 音频总时长(秒)
Returns:
list[{"index": int, "text": str, "start_time": float, "end_time": float}]
"""
sentences = split_script_into_sentences(script_text)
if not sentences:
return []
tmp_path: Optional[str] = None
try:
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp:
tmp.write(audio_data)
tmp_path = tmp.name
result = subprocess.run(
[
"ffmpeg",
"-i",
tmp_path,
"-af",
"silencedetect=noise=-25dB:d=0.3",
"-f",
"null",
"-",
],
capture_output=True,
text=True,
timeout=30,
)
stderr = result.stderr or ""
silence_ends = []
for match in re.finditer(r"silence_end:\s*([\d.]+)", stderr):
t = float(match.group(1))
if 0 < t < total_duration:
silence_ends.append(t)
if len(silence_ends) < len(sentences) - 1:
logger.warning(
"[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算",
len(silence_ends),
len(sentences) - 1,
)
return estimate_sentence_timings_by_chars(sentences, total_duration)
n_boundaries = len(sentences) - 1
boundaries = []
used_indices = set()
for i in range(n_boundaries):
expected_pos = (i + 1) / len(sentences) * total_duration
best_idx = None
best_dist = float("inf")
for j, t in enumerate(silence_ends):
if j in used_indices:
continue
dist = abs(t - expected_pos)
if dist < best_dist:
best_dist = dist
best_idx = j
if best_idx is not None:
used_indices.add(best_idx)
boundaries.append(silence_ends[best_idx])
boundaries.sort()
timings = []
prev_end = 0.0
for i, sent in enumerate(sentences):
start = prev_end
end = boundaries[i] if i < len(boundaries) else total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
prev_end = end
return timings
except Exception as exc:
logger.warning("[sentence_timings] 静音检测异常,降级为字数比例估算: %s", exc)
return estimate_sentence_timings_by_chars(sentences, total_duration)
finally:
if tmp_path:
try:
os.unlink(tmp_path)
except Exception:
pass
@@ -262,8 +262,8 @@ def test_smart_cover_selects_best_frame_and_persists():
score_patch.assert_called_once()
# 验证使用了增大的轮询参数
call_kwargs = mk.extract_frames.call_args
assert call_kwargs.kwargs.get("poll_interval") == 1.0 or call_kwargs[1].get("poll_interval") == 1.0
assert call_kwargs.kwargs.get("max_poll_attempts") == 15 or call_kwargs[1].get("max_poll_attempts") == 15
assert call_kwargs.kwargs.get("poll_interval") == 2.0 or call_kwargs[1].get("poll_interval") == 2.0
assert call_kwargs.kwargs.get("max_poll_attempts") == 30 or call_kwargs[1].get("max_poll_attempts") == 30
def test_smart_cover_returns_empty_when_mediakit_unavailable():
@@ -334,8 +334,8 @@ def test_extract_frames_uses_extended_poll_params():
cov.select_best_cover_frame("https://other/avatar.mp4", max_frames=3)
call_kwargs = mk.extract_frames.call_args
assert call_kwargs.kwargs.get("poll_interval") == 1.0 or call_kwargs[1].get("poll_interval") == 1.0
assert call_kwargs.kwargs.get("max_poll_attempts") == 15 or call_kwargs[1].get("max_poll_attempts") == 15
assert call_kwargs.kwargs.get("poll_interval") == 2.0 or call_kwargs[1].get("poll_interval") == 2.0
assert call_kwargs.kwargs.get("max_poll_attempts") == 30 or call_kwargs[1].get("max_poll_attempts") == 30
assert call_kwargs.kwargs.get("max_retries") == 1 or call_kwargs[1].get("max_retries") == 1
+125 -23
View File
@@ -519,8 +519,8 @@ class TestAiAvatarRenderService:
# 不应执行渲染逻辑
mock_db.commit.assert_not_called()
def test_execute_render_success_creates_clip_record(self):
"""execute_render 完成后自动创建成片记录到成片库."""
def test_execute_render_completed_does_not_auto_persist(self):
"""execute_render 完成后自动入库成片库(改为用户点「完成」时由 finalize_job 入库)."""
from app.services.ai_avatar_render_service import AiAvatarRenderService
mock_db = _make_mock_db()
@@ -549,36 +549,139 @@ class TestAiAvatarRenderService:
patch.object(svc, "_upload_to_oss", side_effect=lambda path, key: f"https://oss/{key}"),
patch("subprocess.run") as mock_run,
patch("tempfile.TemporaryDirectory") as tmpdir_mock,
patch(
"app.services.ai_avatar_cover_service.generate_smart_cover", return_value="https://oss/smart_cover.jpg"
),
patch("packages.domain.generated_video.GeneratedVideo.create") as gv_create,
):
import subprocess as _sp
mock_run.return_value = _sp.CompletedProcess(args=[], returncode=0, stdout="", stderr="")
tmpdir_mock.return_value.__enter__ = MagicMock(return_value="/tmp/testdir")
tmpdir_mock.return_value.__exit__ = MagicMock(return_value=False)
svc.execute_render("render-ok")
# 状态应为 completed,但没有自动入库
assert mock_job.status == "completed"
gv_create.assert_not_called()
assert mock_job.output_video_url.startswith("https://oss/")
def test_finalize_job_persists_to_library(self):
"""finalize_job 在用户点「完成」后写入成片库,thumbnail_url 使用 job.output_cover_url."""
from app.services.ai_avatar_render_service import AiAvatarRenderService
mock_db = MagicMock()
mock_job = _make_mock_render_job(
job_id="render-finalize",
status="completed",
output_video_url="https://oss/ai-avatar/render-finalize/output.mp4",
output_cover_url="https://oss/cover.jpg",
output_duration=12.0,
)
# get_render_job → db.query(AiAvatarRenderJob).filter().first() 返回 mock_job
# finalize 幂等检查 → db.query(GeneratedVideoModel).filter().first() 返回 None(未入库)
def _query_side_effect(model):
q = MagicMock()
if model.__name__ == "AiAvatarRenderJob":
q.filter.return_value.first.return_value = mock_job
else:
# GeneratedVideoModel
q.filter.return_value.first.return_value = None
return q
mock_db.query.side_effect = _query_side_effect
svc = AiAvatarRenderService(mock_db)
with (
patch("packages.domain.generated_video.GeneratedVideo.create") as gv_create,
patch(
"packages.adapters.sqlalchemy_impl.generated_video_repository.SQLAlchemyGeneratedVideoRepository"
) as repo_cls,
):
import subprocess as _sp
mock_run.return_value = _sp.CompletedProcess(args=[], returncode=0, stdout="", stderr="")
import tempfile as _tf
tmpdir_mock.return_value.__enter__ = MagicMock(return_value="/tmp/testdir")
tmpdir_mock.return_value.__exit__ = MagicMock(return_value=False)
mock_clip = MagicMock()
mock_clip.id = "clip-001"
mock_clip.id = "clip-new"
mock_clip.thumbnail_url = "https://oss/cover.jpg"
gv_create.return_value = mock_clip
mock_repo = MagicMock()
mock_repo.create.return_value = mock_clip
mock_repo.get.return_value = mock_clip
repo_cls.return_value = mock_repo
svc.execute_render("render-ok")
video = svc.finalize_job("render-finalize", "user-1")
assert video.id == "clip-new"
gv_create.assert_called_once()
call_kwargs = gv_create.call_args.kwargs
assert call_kwargs["file_url"].endswith("output.mp4")
assert call_kwargs["thumbnail_url"] == "https://oss/cover.jpg"
assert call_kwargs["generation_task_id"] == "render-finalize"
mock_repo.create.assert_called_once()
assert mock_job.status == "completed"
gv_create.assert_called_once()
call_kwargs = gv_create.call_args
assert "https://oss/" in call_kwargs.kwargs["file_url"]
assert call_kwargs.kwargs["user_id"] == "user-1"
mock_repo.create.assert_called_once_with(mock_clip)
def test_finalize_job_idempotent_when_already_persisted(self):
"""finalize_job 重复调用:幂等检查命中后直接返回已有记录,不再 create."""
from app.services.ai_avatar_render_service import AiAvatarRenderService
mock_db = MagicMock()
mock_job = _make_mock_render_job(
job_id="render-finalize-2",
status="completed",
output_video_url="https://oss/output.mp4",
output_cover_url="https://oss/cover.jpg",
output_duration=12.0,
)
existing_model = MagicMock()
existing_model.id = "clip-existing"
existing_model.thumbnail_url = "https://oss/cover.jpg"
def _query_side_effect(model):
q = MagicMock()
if model.__name__ == "AiAvatarRenderJob":
q.filter.return_value.first.return_value = mock_job
else:
q.filter.return_value.first.return_value = existing_model
return q
mock_db.query.side_effect = _query_side_effect
svc = AiAvatarRenderService(mock_db)
with (
patch("packages.domain.generated_video.GeneratedVideo.create") as gv_create,
patch(
"packages.adapters.sqlalchemy_impl.generated_video_repository.SQLAlchemyGeneratedVideoRepository"
) as repo_cls,
):
mock_existing = MagicMock()
mock_existing.id = "clip-existing"
mock_repo = MagicMock()
mock_repo.get.return_value = mock_existing
repo_cls.return_value = mock_repo
video = svc.finalize_job("render-finalize-2", "user-1")
assert video.id == "clip-existing"
gv_create.assert_not_called()
mock_repo.create.assert_not_called()
def test_finalize_job_requires_completed_status(self):
"""finalize_job 在非 completed 状态下抛异常."""
from app.services.ai_avatar_render_service import AiAvatarRenderError, AiAvatarRenderService
mock_db = MagicMock()
mock_job = _make_mock_render_job(
job_id="render-pending",
status="processing",
output_video_url="",
output_cover_url="",
output_duration=0.0,
)
filter_mock = MagicMock()
filter_mock.first.return_value = mock_job
query_mock = MagicMock()
query_mock.filter.return_value = filter_mock
mock_db.query.return_value = query_mock
svc = AiAvatarRenderService(mock_db)
with pytest.raises(AiAvatarRenderError):
svc.finalize_job("render-pending", "user-1")
def test_execute_render_clip_failure_does_not_affect_render(self):
"""成片创建失败不影响渲染任务标记为成功."""
@@ -610,7 +713,6 @@ class TestAiAvatarRenderService:
patch.object(svc, "_upload_to_oss", side_effect=lambda path, key: f"https://oss/{key}"),
patch("subprocess.run") as mock_run,
patch("tempfile.TemporaryDirectory") as tmpdir_mock,
patch("app.services.ai_avatar_cover_service.generate_smart_cover", side_effect=RuntimeError("DB error")),
):
import subprocess as _sp
@@ -619,7 +721,7 @@ class TestAiAvatarRenderService:
tmpdir_mock.return_value.__exit__ = MagicMock(return_value=False)
svc.execute_render("render-clip-fail")
# 即使成片创建失败,渲染任务仍应标记为 completed
# 渲染任务仍应标记为 completed(不入库不影响渲染成功)
assert mock_job.status == "completed"
def test_error_exception_has_code(self):
+16 -4
View File
@@ -91,16 +91,27 @@ class TestSchemaValidation:
assert req.voice_id == "longxiaochun_v3"
assert req.script_text == "大家好,欢迎来到直播间"
def test_invalid_video_url_not_mp4(self):
def test_invalid_video_url_unsupported_ext(self):
from app.schemas.lipsync import CreateLipsyncJobRequest
with pytest.raises(ValueError, match="MP4"):
# 不支持的扩展名(.txt)应报错
with pytest.raises(ValueError, match="格式不支持"):
CreateLipsyncJobRequest(
video_url="https://example.com/video.mov",
video_url="https://example.com/video.txt",
voice_id="longxiaochun_v3",
script_text="测试文本",
)
def test_mov_video_url_accepted(self):
from app.schemas.lipsync import CreateLipsyncJobRequest
# .MOV 是 iPhone 拍摄的常见容器,h264 编码可直接被 MediaKit 处理
req = CreateLipsyncJobRequest(
video_url="https://example.com/video.mov",
audio_url="https://example.com/audio.mp3",
)
assert req.video_url.endswith(".mov")
def test_invalid_video_url_empty(self):
from app.schemas.lipsync import CreateLipsyncJobRequest
@@ -159,7 +170,8 @@ class TestSchemaValidation:
voice_id="longxiaochun_v3",
script_text="测试文本",
)
assert req.enable_video_loop is False
# AI数字人场景文案长度不可控,默认开启视频循环,防止音频长于视频时被截断
assert req.enable_video_loop is True
def test_video_url_strip_query_params(self):
"""视频 URL 含查询参数时,扩展名检查应忽略 ? 后面的部分."""
+169
View File
@@ -0,0 +1,169 @@
"""AI 数字人 对口型 TTS 预合成接口(#1845)单元测试 — 覆盖 LipsyncService.preview_tts 成功/失败路径.
直接调用 LipsyncService.preview_tts()mock CosyVoiceService / safe_download_bytes / ffprobe
验证返回结构、错误码、与共享 sentence_timings 工具的协作。
"""
import os
from unittest.mock import MagicMock, patch
import pytest
os.environ.setdefault("JWT_SECRET_KEY", "dev-secret-key-for-testing")
def _make_service(
*,
cosyvoice=None,
download_bytes=b"FAKE_MP3_DATA",
download_error=None,
ffprobe_duration=5.0,
timings_result=None,
):
"""构造 LipsyncService 并把 CosyVoiceService/safe_download_bytes/probe/compute 全部 mock 掉。"""
from app.services.lipsync_service import LipsyncService
db = MagicMock()
# 构造唯一的 cosyvoice mock 实例,便于断言
_cosy_inst = MagicMock()
if cosyvoice is None:
_cosy_inst.submit_synthesize_task.return_value = {"audio_url": "https://cosy.example.com/tts.mp3"}
elif isinstance(cosyvoice, Exception):
_cosy_inst.submit_synthesize_task.side_effect = cosyvoice
else:
_cosy_inst.submit_synthesize_task.return_value = cosyvoice
def _fake_get_cosyvoice(self): # noqa: ARG001
return _cosy_inst
def _fake_resolve_voice_id(self, voice_id, user_id): # noqa: ARG001
return voice_id
svc = LipsyncService(db=db, client=MagicMock(), voice_clone_repo=MagicMock())
svc._cosyvoice = _cosy_inst
patch.object(LipsyncService, "_get_cosyvoice", _fake_get_cosyvoice).start()
patch.object(LipsyncService, "_resolve_voice_id", _fake_resolve_voice_id).start()
# mock safe_download_bytes
if download_error is not None:
patch(
"app.services.lipsync_service.safe_download_bytes",
side_effect=download_error,
).start()
else:
patch(
"app.services.lipsync_service.safe_download_bytes",
return_value=download_bytes,
).start()
# mock probe_audio_durationpatch 到 lipsync_service 模块的命名空间)
patch(
"app.services.lipsync_service.probe_audio_duration",
return_value=ffprobe_duration,
).start()
# mock compute_sentence_timings
default_timings = [
{"index": 0, "text": "你好", "start_time": 0.0, "end_time": 1.5},
{"index": 1, "text": "世界", "start_time": 1.5, "end_time": 5.0},
]
patch(
"app.services.lipsync_service.compute_sentence_timings",
return_value=timings_result if timings_result is not None else default_timings,
).start()
svc.__dict__["_test_cosy"] = _cosy_inst
return svc
def test_preview_tts_success():
"""正常路径:TTS 合成成功 → 下载 → ffprobe → 计算 timings,返回完整结构。"""
svc = _make_service(ffprobe_duration=5.0)
try:
result = svc.preview_tts(
user_id="user-1",
voice_id="longxiaochun",
script_text="你好,世界",
speed=1.0,
emotion="natural",
)
assert result["audio_url"] == "https://cosy.example.com/tts.mp3"
assert result["duration"] == 5.0
assert isinstance(result["sentence_timings"], list)
assert len(result["sentence_timings"]) == 2
assert result["sentence_timings"][0]["text"] == "你好"
cosy = svc.__dict__["_test_cosy"]
cosy.submit_synthesize_task.assert_called_once()
kwargs = cosy.submit_synthesize_task.call_args.kwargs
assert kwargs["text"] == "你好,世界"
assert kwargs["voice_id"] == "longxiaochun"
finally:
patch.stopall()
def test_preview_tts_cosyvoice_error():
"""CosyVoice 抛错:应该包装成 MediaKitError 抛出。"""
from app.services.mediakit_client import MediaKitError
from packages.application.cosyvoice_service import CosyVoiceError
svc = _make_service(cosyvoice=CosyVoiceError("cosyvoice down"))
try:
with pytest.raises(MediaKitError):
svc.preview_tts(
user_id="user-1",
voice_id="longxiaochun",
script_text="你好",
)
finally:
patch.stopall()
def test_preview_tts_download_fail_still_returns_url():
"""音频下载失败:不抛错,返回 audio_url + 空 timings,前端仍能继续(降级)。"""
svc = _make_service(download_error=RuntimeError("network down"))
try:
result = svc.preview_tts(
user_id="user-1",
voice_id="longxiaochun",
script_text="你好,世界",
)
assert result["audio_url"] == "https://cosy.example.com/tts.mp3"
assert result["duration"] == 0.0
assert result["sentence_timings"] == []
finally:
patch.stopall()
def test_preview_tts_ffprobe_zero_duration():
"""ffprobe 返回 0timings 为空,不抛错。"""
svc = _make_service(ffprobe_duration=0.0)
try:
result = svc.preview_tts(
user_id="user-1",
voice_id="longxiaochun",
script_text="你好",
)
assert result["audio_url"]
assert result["duration"] == 0.0
assert result["sentence_timings"] == []
finally:
patch.stopall()
def test_preview_tts_no_audio_url_in_response():
"""CosyVoice 返回无 audio_url:抛 MediaKitError TTSNoAudio。"""
from app.services.mediakit_client import MediaKitError
svc = _make_service(cosyvoice={"audio_url": ""})
try:
with pytest.raises(MediaKitError) as exc_info:
svc.preview_tts(
user_id="user-1",
voice_id="longxiaochun",
script_text="你好",
)
assert exc_info.value.code == "TTSNoAudio"
finally:
patch.stopall()
+4 -6
View File
@@ -1,4 +1,4 @@
"""Tests for sentence timing functions in lipsync_tts."""
"""Tests for sentence timing functions (now in packages/domain/sentence_timings.py)."""
import os
import subprocess
@@ -6,11 +6,9 @@ import tempfile
import unittest
from unittest.mock import MagicMock, patch
from apps.api.app.tasks.lipsync_tts import (
_compute_sentence_timings,
_estimate_sentence_timings_by_chars,
_split_script_into_sentences,
)
from packages.domain.sentence_timings import compute_sentence_timings as _compute_sentence_timings
from packages.domain.sentence_timings import estimate_sentence_timings_by_chars as _estimate_sentence_timings_by_chars
from packages.domain.sentence_timings import split_script_into_sentences as _split_script_into_sentences
class TestSplitScriptIntoSentences(unittest.TestCase):