Files
xiaoxia-saas/packages/domain/voice_duration_planner.py
T
xiaoxia ee4636e087
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 1s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 30s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 55s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 1m27s
CI/CD Pipeline / Unit Tests (pull_request) Successful in 1m30s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 1m32s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Successful in 1m48s
CI/CD Pipeline / Validate - Style (pull_request) Successful in 1m56s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 2m51s
AI Code Review / AI Code Review (pull_request) Successful in 6m18s
CI/CD Pipeline / Validate - Security (pull_request) Successful in 6m17s
CI/CD Pipeline / CI Gate (pull_request) Successful in 4s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
ACR Cleanup / ACR Image Cleanup (pull_request_target) Successful in 8s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Successful in 4m13s
Preview Cleanup / Cleanup Preview Environment (pull_request) Successful in 34s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 196h20m23s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 196h20m26s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 196h20m26s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 196h20m23s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 196h26m36s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 196h26m40s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 196h26m35s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 196h26m42s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 196h26m42s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 196h26m43s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 196h26m43s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 196h26m46s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 196h26m47s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 196h26m46s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 196h55m10s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 197h1m19s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 197h1m25s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 197h1m26s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 197h1m30s
feat: #1768 节奏曲线模板多样化 — 8种预设 + 时长钳制 + 误差校验
变更:
- RHYTHM_TEMPLATES 从 6 种扩展到 8 种(新增 [2,2,1,1,2] 和 [3,2,1,2,1])
- 修复 MIN_CLIP_DURATION 被重复定义为 1.0 的 bug(恢复为 2.0)
- plan_clip_durations 新增 asset_durations 参数,钳制最大片段时长 <= 素材可用时长 × 90%
- 新增时长总和误差校验(成片净时长与配音时长误差 <= 0.5s),超限时末段补偿修正
- edit_plan_service.apply_voice_duration_to_plan 传入素材时长参与钳制
- 48 个新增单元测试全部通过

向后兼容:asset_durations 默认 None,不影响现有调用方
2026-09-08 10:57:16 +08:00

262 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""配音时长 → 片段时长分配纯函数(#1749 + #1764 节奏模板)。
定稿规则(工单 #1749):
1. 片段数 = 模板片段数,定死,不因素材增减;
2. 成片总时长 = 配音时长:逐段分配段长,含转场重叠扣减
(cut 零重叠;xfade 等转场按转场时长重叠),误差 < 0.5s;
3. 素材短于段长 → 末帧冻结(tpad)/ 音频补静音(apad)铺满,
禁止慢放、禁止截断配音;
4. 任何情况下不得因素材时长/数量报错打断用户。
#1764 节奏模板 + #1768 多样化增强:
- 预设 8 种权重序列,不同变体用不同节奏模板
- 片段时长 = 配音总时长 × 该片段权重 / 权重总和
- 平均分配作为权重全 1 的特例保留
- 每个片段 >= MIN_CLIP_DURATION(2秒)
- #1768:最大片段时长 <= 素材可用时长 × 90%
- #1768:成片总时长与配音时长误差 <= TOTAL_DURATION_TOLERANCE(0.5s)
本模块为纯函数:输入片段骨架(每段转场效果/时长)与配音总时长,
输出每段目标时长(target duration)与成片总时长。不碰 DB、不碰素材。
"""
from __future__ import annotations
import logging
import random
from typing import Optional
logger = logging.getLogger(__name__)
#: 单段最小时长(秒):低于此值播放器/渲染链路易出问题
MIN_CLIP_DURATION = 2.0
#: 成片总时长与配音时长的可接受误差(秒)
TOTAL_DURATION_TOLERANCE = 0.5
# ── #1764 节奏模板池 ──────────────────────────────────────────────────────
# 每种模板是权重序列,权重值代表相对时长比例
# 变体基于 variant_seed 随机选一个模板,实现不同变体时长结构不同
RHYTHM_TEMPLATES: list[list[int]] = [
[1, 1, 1, 1, 1], # 平均(基准)
[2, 1, 3, 1, 2], # 中间长,两端短
[1, 2, 1, 2, 1], # 偶数段长
[3, 1, 1, 1, 3], # 两端长,中间短
[1, 1, 3, 2, 1], # 后段渐长
[2, 1, 1, 3, 1], # 前段较长 + 第4段最长
[2, 2, 1, 1, 2], # #1768 前重后轻
[3, 2, 1, 2, 1], # #1768 渐弱节奏
]
def get_rhythm_template(variant_seed: int | None = None) -> list[int]:
"""根据 variant_seed 选择一个节奏模板。
Args:
variant_seed: 变体随机种子;None 时返回平均模板
Returns:
权重序列(list[int])
"""
if variant_seed is None:
return RHYTHM_TEMPLATES[0] # 默认平均
rng = random.Random(variant_seed)
return rng.choice(RHYTHM_TEMPLATES)
def adapt_template_length(template: list[int], clip_count: int) -> list[int]:
"""将节奏模板适配到实际片段数。
片段数 != 模板长度时:
- clip_count < len(template): 截断
- clip_count > len(template): 循环填充
Args:
template: 原始权重序列
clip_count: 实际片段数
Returns:
适配后的权重序列(长度 == clip_count)
"""
if clip_count <= 0:
return []
if clip_count == len(template):
return template[:]
if clip_count < len(template):
return template[:clip_count]
# clip_count > len(template): 循环填充
result = []
for i in range(clip_count):
result.append(template[i % len(template)])
return result
def transition_overlap_seconds(transition_effect: Optional[str], transition_duration: float) -> float:
"""转场导致的相邻片段重叠时长。
cut(或空/None)无重叠;其余转场(xfade/fade/slide 等)按转场时长重叠。
"""
effect = (transition_effect or "cut").strip().lower()
if effect in ("cut", "", "none"):
return 0.0
dur = float(transition_duration or 0.0)
return max(0.0, dur)
def plan_clip_durations(
clip_count: int,
voice_duration: float,
transition_effects: Optional[list[Optional[str]]] = None,
transition_durations: Optional[list[float]] = None,
rhythm_template: Optional[list[int]] = None,
asset_durations: Optional[list[float]] = None,
) -> list[float]:
"""把配音总时长分配到 clip_count 段,返回每段目标时长(秒)。
#1764:支持节奏模板,按权重比例分配时长;无模板时平均分配(向后兼容)。
分配口径:Σ段长 − Σ转场重叠 = 配音时长(成片净时长 = 配音)。
转场重叠发生在相邻片段之间,共 clip_count-1 处;第 i 处重叠取
**后一段(i+1)** 的转场设置(与 xfade 构建口径一致:转场挂在后段)。
舍入误差全部由最后一段吸收,保证 total_output_duration(结果) ≈ 配音。
Args:
clip_count: 片段数(模板定死,必须 > 0)。
voice_duration: 配音总时长(秒);<=0 返回空列表表示"无配音"。
transition_effects: 每段转场效果(长度 clip_count,index 0 的转场无效)。
transition_durations: 每段转场时长(长度 clip_count)。
asset_durations: #1768 每段可用素材时长(秒),用于钳制最大片段时长
<= 素材可用时长 × 90%。长度 clip_count;None 或空则不钳制上限。
Returns:
每段目标时长列表(长度 clip_count);无配音/非法输入返回 []。
"""
if not clip_count or clip_count <= 0:
return []
try:
voice = float(voice_duration)
except (TypeError, ValueError):
return []
if voice <= 0:
return []
effects = transition_effects or [None] * clip_count
durations = transition_durations or [0.0] * clip_count
# 相邻片段间的转场重叠总和(转场挂在后段,取 i=1..clip_count-1)
total_overlap = 0.0
for i in range(1, clip_count):
effect = effects[i] if i < len(effects) else None
tdur = durations[i] if i < len(durations) else 0.0
total_overlap += transition_overlap_seconds(effect, tdur)
# 需要的段长总和 = 配音 + 重叠(重叠部分被算了两次,扣回一次)
gross = voice + total_overlap
if gross < clip_count * MIN_CLIP_DURATION:
# 配音极短:保底每段 MIN_CLIP_DURATION(成片略长于配音,末段可冻结)
gross = clip_count * MIN_CLIP_DURATION
logger.info(
"配音时长 %.2fs 过短,%d 段按最小段长 %.1fs 保底(成片将略长于配音)",
voice,
clip_count,
MIN_CLIP_DURATION,
)
# #1764:按节奏模板权重分配(无模板时全 1 = 平均分配)
weights = rhythm_template if rhythm_template and len(rhythm_template) == clip_count else [1] * clip_count
# 确保每个片段 >= MIN_CLIP_DURATION
# 先按权重分配,再检查最小值
total_weight = sum(weights)
raw_durations = [(w / total_weight) * gross for w in weights]
# 保底检查:如果有片段 < MIN_CLIP_DURATION,提升它并从最长片段扣
result = [round(d, 3) for d in raw_durations]
for _ in range(3): # 最多迭代 3 次
min_idx = min(range(len(result)), key=lambda i: result[i])
if result[min_idx] >= MIN_CLIP_DURATION:
break
# 从最长片段借时长
max_idx = max(range(len(result)), key=lambda i: result[i])
if max_idx == min_idx or result[max_idx] <= MIN_CLIP_DURATION:
# 无法再调整,强制保底
result[min_idx] = MIN_CLIP_DURATION
break
deficit = MIN_CLIP_DURATION - result[min_idx]
result[min_idx] = MIN_CLIP_DURATION
result[max_idx] = round(result[max_idx] - deficit, 3)
# #1768:最大片段时长钳制(<= 素材可用时长 × 90%)
if asset_durations and len(asset_durations) == clip_count:
for _ in range(3): # 迭代收敛
clamped = False
for i in range(len(result)):
try:
max_dur = float(asset_durations[i]) * 0.9
except (TypeError, ValueError, IndexError):
continue
if result[i] > max_dur and max_dur >= MIN_CLIP_DURATION:
excess = result[i] - max_dur
result[i] = round(max_dur, 3)
# 将多余时长分配给最短的未超限片段
candidates = [
j
for j in range(len(result))
if j != i
and (
not asset_durations
or j >= len(asset_durations)
or result[j] < float(asset_durations[j]) * 0.9
)
]
if candidates:
shortest = min(candidates, key=lambda j: result[j])
result[shortest] = round(result[shortest] + excess, 3)
clamped = True
if not clamped:
break
# 末段吸收舍入误差
total_assigned = sum(result[:-1])
result[-1] = round(gross - total_assigned, 3)
if result[-1] < MIN_CLIP_DURATION:
result[-1] = MIN_CLIP_DURATION
# #1768:时长总和误差校验(成片净时长 ≈ 配音时长)
net_total = total_output_duration(result, transition_effects, transition_durations)
deviation = abs(net_total - voice)
if deviation > TOTAL_DURATION_TOLERANCE:
logger.warning(
"#1768 时长总和误差 %.3fs 超过阈值 %.1fs(voice=%.2fs, net=%.2fs),末段补偿修正",
deviation,
TOTAL_DURATION_TOLERANCE,
voice,
net_total,
)
# 修正末段使净时长回归配音时长
result[-1] = round(result[-1] + (voice - net_total), 3)
if result[-1] < MIN_CLIP_DURATION:
result[-1] = MIN_CLIP_DURATION
return result
def total_output_duration(
clip_durations: list[float],
transition_effects: Optional[list[Optional[str]]] = None,
transition_durations: Optional[list[float]] = None,
) -> float:
"""按分配后的段长与转场计算成片净时长 = Σ段长 − Σ转场重叠。"""
if not clip_durations:
return 0.0
effects = transition_effects or [None] * len(clip_durations)
durations = transition_durations or [0.0] * len(clip_durations)
total = sum(float(d) for d in clip_durations)
for i in range(1, len(clip_durations)):
effect = effects[i] if i < len(effects) else None
tdur = durations[i] if i < len(durations) else 0.0
total -= transition_overlap_seconds(effect, tdur)
return round(max(0.0, total), 3)