"""配音时长 → 片段时长分配纯函数(#1749 + #1764 节奏模板)。 定稿规则(工单 #1749): 1. 片段数 = 模板片段数,定死,不因素材增减; 2. 成片总时长 = 配音时长:逐段分配段长,含转场重叠扣减 (cut 零重叠;xfade 等转场按转场时长重叠),误差 < 0.5s; 3. 素材短于段长 → 末帧冻结(tpad)/ 音频补静音(apad)铺满, 禁止慢放、禁止截断配音; 4. 任何情况下不得因素材时长/数量报错打断用户。 #1764 节奏模板 + #1768 多样化增强: - 预设 8 种权重序列,不同变体用不同节奏模板 - 片段时长 = 配音总时长 × 该片段权重 / 权重总和 - 平均分配作为权重全 1 的特例保留 - 每个片段 >= MIN_CLIP_DURATION(2秒) - #1768:最大片段时长 <= 素材可用时长 × 90% - #1768:成片总时长与配音时长误差 <= TOTAL_DURATION_TOLERANCE(0.5s) 本模块为纯函数:输入片段骨架(每段转场效果/时长)与配音总时长, 输出每段目标时长(target duration)与成片总时长。不碰 DB、不碰素材。 """ from __future__ import annotations import logging import random from typing import Optional logger = logging.getLogger(__name__) #: 单段最小时长(秒):低于此值播放器/渲染链路易出问题 MIN_CLIP_DURATION = 2.0 #: 成片总时长与配音时长的可接受误差(秒) TOTAL_DURATION_TOLERANCE = 0.5 # ── #1764 节奏模板池 ────────────────────────────────────────────────────── # 每种模板是权重序列,权重值代表相对时长比例 # 变体基于 variant_seed 随机选一个模板,实现不同变体时长结构不同 RHYTHM_TEMPLATES: list[list[int]] = [ [1, 1, 1, 1, 1], # 平均(基准) [2, 1, 3, 1, 2], # 中间长,两端短 [1, 2, 1, 2, 1], # 偶数段长 [3, 1, 1, 1, 3], # 两端长,中间短 [1, 1, 3, 2, 1], # 后段渐长 [2, 1, 1, 3, 1], # 前段较长 + 第4段最长 [2, 2, 1, 1, 2], # #1768 前重后轻 [3, 2, 1, 2, 1], # #1768 渐弱节奏 ] def get_rhythm_template(variant_seed: int | None = None) -> list[int]: """根据 variant_seed 选择一个节奏模板。 Args: variant_seed: 变体随机种子;None 时返回平均模板 Returns: 权重序列(list[int]) """ if variant_seed is None: return RHYTHM_TEMPLATES[0] # 默认平均 rng = random.Random(variant_seed) return rng.choice(RHYTHM_TEMPLATES) def adapt_template_length(template: list[int], clip_count: int) -> list[int]: """将节奏模板适配到实际片段数。 片段数 != 模板长度时: - clip_count < len(template): 截断 - clip_count > len(template): 循环填充 Args: template: 原始权重序列 clip_count: 实际片段数 Returns: 适配后的权重序列(长度 == clip_count) """ if clip_count <= 0: return [] if clip_count == len(template): return template[:] if clip_count < len(template): return template[:clip_count] # clip_count > len(template): 循环填充 result = [] for i in range(clip_count): result.append(template[i % len(template)]) return result def transition_overlap_seconds(transition_effect: Optional[str], transition_duration: float) -> float: """转场导致的相邻片段重叠时长。 cut(或空/None)无重叠;其余转场(xfade/fade/slide 等)按转场时长重叠。 """ effect = (transition_effect or "cut").strip().lower() if effect in ("cut", "", "none"): return 0.0 dur = float(transition_duration or 0.0) return max(0.0, dur) def plan_clip_durations( clip_count: int, voice_duration: float, transition_effects: Optional[list[Optional[str]]] = None, transition_durations: Optional[list[float]] = None, rhythm_template: Optional[list[int]] = None, asset_durations: Optional[list[float]] = None, ) -> list[float]: """把配音总时长分配到 clip_count 段,返回每段目标时长(秒)。 #1764:支持节奏模板,按权重比例分配时长;无模板时平均分配(向后兼容)。 分配口径:Σ段长 − Σ转场重叠 = 配音时长(成片净时长 = 配音)。 转场重叠发生在相邻片段之间,共 clip_count-1 处;第 i 处重叠取 **后一段(i+1)** 的转场设置(与 xfade 构建口径一致:转场挂在后段)。 舍入误差全部由最后一段吸收,保证 total_output_duration(结果) ≈ 配音。 Args: clip_count: 片段数(模板定死,必须 > 0)。 voice_duration: 配音总时长(秒);<=0 返回空列表表示"无配音"。 transition_effects: 每段转场效果(长度 clip_count,index 0 的转场无效)。 transition_durations: 每段转场时长(长度 clip_count)。 asset_durations: #1768 每段可用素材时长(秒),用于钳制最大片段时长 <= 素材可用时长 × 90%。长度 clip_count;None 或空则不钳制上限。 Returns: 每段目标时长列表(长度 clip_count);无配音/非法输入返回 []。 """ if not clip_count or clip_count <= 0: return [] try: voice = float(voice_duration) except (TypeError, ValueError): return [] if voice <= 0: return [] effects = transition_effects or [None] * clip_count durations = transition_durations or [0.0] * clip_count # 相邻片段间的转场重叠总和(转场挂在后段,取 i=1..clip_count-1) total_overlap = 0.0 for i in range(1, clip_count): effect = effects[i] if i < len(effects) else None tdur = durations[i] if i < len(durations) else 0.0 total_overlap += transition_overlap_seconds(effect, tdur) # 需要的段长总和 = 配音 + 重叠(重叠部分被算了两次,扣回一次) gross = voice + total_overlap if gross < clip_count * MIN_CLIP_DURATION: # 配音极短:保底每段 MIN_CLIP_DURATION(成片略长于配音,末段可冻结) gross = clip_count * MIN_CLIP_DURATION logger.info( "配音时长 %.2fs 过短,%d 段按最小段长 %.1fs 保底(成片将略长于配音)", voice, clip_count, MIN_CLIP_DURATION, ) # #1764:按节奏模板权重分配(无模板时全 1 = 平均分配) weights = rhythm_template if rhythm_template and len(rhythm_template) == clip_count else [1] * clip_count # 确保每个片段 >= MIN_CLIP_DURATION # 先按权重分配,再检查最小值 total_weight = sum(weights) raw_durations = [(w / total_weight) * gross for w in weights] # 保底检查:如果有片段 < MIN_CLIP_DURATION,提升它并从最长片段扣 result = [round(d, 3) for d in raw_durations] for _ in range(3): # 最多迭代 3 次 min_idx = min(range(len(result)), key=lambda i: result[i]) if result[min_idx] >= MIN_CLIP_DURATION: break # 从最长片段借时长 max_idx = max(range(len(result)), key=lambda i: result[i]) if max_idx == min_idx or result[max_idx] <= MIN_CLIP_DURATION: # 无法再调整,强制保底 result[min_idx] = MIN_CLIP_DURATION break deficit = MIN_CLIP_DURATION - result[min_idx] result[min_idx] = MIN_CLIP_DURATION result[max_idx] = round(result[max_idx] - deficit, 3) # #1768:最大片段时长钳制(<= 素材可用时长 × 90%) if asset_durations and len(asset_durations) == clip_count: for _ in range(3): # 迭代收敛 clamped = False for i in range(len(result)): try: max_dur = float(asset_durations[i]) * 0.9 except (TypeError, ValueError, IndexError): continue if result[i] > max_dur and max_dur >= MIN_CLIP_DURATION: excess = result[i] - max_dur result[i] = round(max_dur, 3) # 将多余时长分配给最短的未超限片段 candidates = [ j for j in range(len(result)) if j != i and ( not asset_durations or j >= len(asset_durations) or result[j] < float(asset_durations[j]) * 0.9 ) ] if candidates: shortest = min(candidates, key=lambda j: result[j]) result[shortest] = round(result[shortest] + excess, 3) clamped = True if not clamped: break # 末段吸收舍入误差 total_assigned = sum(result[:-1]) result[-1] = round(gross - total_assigned, 3) if result[-1] < MIN_CLIP_DURATION: result[-1] = MIN_CLIP_DURATION # #1768:时长总和误差校验(成片净时长 ≈ 配音时长) net_total = total_output_duration(result, transition_effects, transition_durations) deviation = abs(net_total - voice) if deviation > TOTAL_DURATION_TOLERANCE: logger.warning( "#1768 时长总和误差 %.3fs 超过阈值 %.1fs(voice=%.2fs, net=%.2fs),末段补偿修正", deviation, TOTAL_DURATION_TOLERANCE, voice, net_total, ) # 修正末段使净时长回归配音时长 result[-1] = round(result[-1] + (voice - net_total), 3) if result[-1] < MIN_CLIP_DURATION: result[-1] = MIN_CLIP_DURATION return result def total_output_duration( clip_durations: list[float], transition_effects: Optional[list[Optional[str]]] = None, transition_durations: Optional[list[float]] = None, ) -> float: """按分配后的段长与转场计算成片净时长 = Σ段长 − Σ转场重叠。""" if not clip_durations: return 0.0 effects = transition_effects or [None] * len(clip_durations) durations = transition_durations or [0.0] * len(clip_durations) total = sum(float(d) for d in clip_durations) for i in range(1, len(clip_durations)): effect = effects[i] if i < len(effects) else None tdur = durations[i] if i < len(durations) else 0.0 total -= transition_overlap_seconds(effect, tdur) return round(max(0.0, total), 3)