Files
xiaoxia-saas/packages/domain/sentence_timings.py
T
xiaoxia 053b00634a
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 3s
CI/CD Pipeline / Check push changed paths (push) Successful in 4s
CI/CD Pipeline / Build Staging API Image (push) Successful in 23s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 23s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 26s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 49s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m32s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m38s
CI/CD Pipeline / Integration Tests (push) Successful in 3m50s
CI/CD Pipeline / Validate - Style (push) Successful in 4m2s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m9s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m43s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m26s
CI/CD Pipeline / Unit Tests (push) Successful in 8m55s
CI/CD Pipeline / Validate - Security (push) Successful in 9m7s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Failing after 82h52m10s
CI/CD Pipeline / Deploy Production (push) Failing after 82h52m6s
CI/CD Pipeline / Frontend Lint (push) Failing after 83h1m19s
CI/CD Pipeline / PR Build Web Image (push) Failing after 83h0m59s
CI/CD Pipeline / PR Build API Image (push) Failing after 83h0m59s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 83h0m28s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 83h0m28s
CI/CD Pipeline / CI Gate (push) Failing after 82h51m46s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 83h1m1s
CI/CD Pipeline / Canary Release to Production (push) Failing after 82h51m42s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 82h51m46s
CI/CD Pipeline / Build Production API Image (push) Failing after 82h51m46s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 83h0m59s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 83h0m28s
feat(ai-avatar): 配音前置 (#1875)
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-09-13 04:22:21 +08:00

206 lines
6.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""共享的句子时间戳计算工具 — 供 Celery TTS 任务和 /lipsync/tts-preview 同步接口复用.
- `_split_script_into_sentences`: 按标点分句(中英文逗号/句号/问号/感叹号/分号/换行)
- `_estimate_sentence_timings_by_chars`: 按字数比例估算(静音检测失败时降级)
- `_probe_audio_duration`: ffprobe 读取音频时长
- `compute_sentence_timings`: 基于 ffmpeg silencedetect 精确计算每句起止时间
"""
from __future__ import annotations
import logging
import os
import re
import subprocess
import tempfile
from typing import Optional
logger = logging.getLogger(__name__)
def split_script_into_sentences(script_text: str) -> list[str]:
"""按句号/问号/感叹号/分号/逗号/换行分句(与前端 SENTENCE_SPLIT_RE 一致).
中文短视频文案习惯用「,」断小句(如"卖花的叫花无缺,卖姜的叫姜子牙"),
必须把逗号也纳入分隔符,否则多句文案会被识别成一整句,导致 B-roll 时间戳错位。
"""
text = (script_text or "").strip()
if not text:
return []
parts = re.split(r"[。!?!??!;;,,\n\r]+", text)
return [p.strip() for p in parts if p.strip()]
def estimate_sentence_timings_by_chars(sentences: list[str], total_duration: float) -> list[dict]:
"""降级方案:按字数比例估算句子时间(与原前端逻辑一致)."""
if not sentences or total_duration <= 0:
return []
total_chars = sum(len(s.replace(r"\s", "")) for s in sentences)
if total_chars == 0:
return []
timings = []
acc = 0
for i, sent in enumerate(sentences):
chars = len(sent.replace(r"\s", ""))
start = (acc / total_chars) * total_duration
end = ((acc + chars) / total_chars) * total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
acc += chars
return timings
def probe_audio_duration(audio_data: bytes, timeout: int = 10) -> float:
"""用 ffprobe 读取音频字节流的时长(秒).
Returns:
时长(秒),失败返回 0.0
"""
if not audio_data:
return 0.0
tmp_path: Optional[str] = None
try:
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp:
tmp.write(audio_data)
tmp_path = tmp.name
result = subprocess.run(
[
"ffprobe",
"-v",
"error",
"-show_entries",
"format=duration",
"-of",
"default=noprint_wrappers=1:nokey=1",
tmp_path,
],
capture_output=True,
text=True,
timeout=timeout,
)
stdout = (result.stdout or "").strip()
if not stdout:
logger.warning("[sentence_timings] ffprobe 无输出: stderr=%s", (result.stderr or "")[:200])
return 0.0
return float(stdout)
except Exception as exc:
logger.warning("[sentence_timings] ffprobe 时长探测失败: %s", exc)
return 0.0
finally:
if tmp_path:
try:
os.unlink(tmp_path)
except Exception:
pass
def compute_sentence_timings(audio_data: bytes, script_text: str, total_duration: float) -> list[dict]:
"""基于 TTS 音频的静音检测,精确计算每句文案的起止时间.
使用 ffmpeg silencedetect 检测静音段,将静音点与句子边界对齐。
比字数比例估算准确得多。
Args:
audio_data: TTS 音频二进制数据(MP3)
script_text: 文案全文
total_duration: 音频总时长(秒)
Returns:
list[{"index": int, "text": str, "start_time": float, "end_time": float}]
"""
sentences = split_script_into_sentences(script_text)
if not sentences:
return []
tmp_path: Optional[str] = None
try:
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp:
tmp.write(audio_data)
tmp_path = tmp.name
result = subprocess.run(
[
"ffmpeg",
"-i",
tmp_path,
"-af",
"silencedetect=noise=-25dB:d=0.3",
"-f",
"null",
"-",
],
capture_output=True,
text=True,
timeout=30,
)
stderr = result.stderr or ""
silence_ends = []
for match in re.finditer(r"silence_end:\s*([\d.]+)", stderr):
t = float(match.group(1))
if 0 < t < total_duration:
silence_ends.append(t)
if len(silence_ends) < len(sentences) - 1:
logger.warning(
"[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算",
len(silence_ends),
len(sentences) - 1,
)
return estimate_sentence_timings_by_chars(sentences, total_duration)
n_boundaries = len(sentences) - 1
boundaries = []
used_indices = set()
for i in range(n_boundaries):
expected_pos = (i + 1) / len(sentences) * total_duration
best_idx = None
best_dist = float("inf")
for j, t in enumerate(silence_ends):
if j in used_indices:
continue
dist = abs(t - expected_pos)
if dist < best_dist:
best_dist = dist
best_idx = j
if best_idx is not None:
used_indices.add(best_idx)
boundaries.append(silence_ends[best_idx])
boundaries.sort()
timings = []
prev_end = 0.0
for i, sent in enumerate(sentences):
start = prev_end
end = boundaries[i] if i < len(boundaries) else total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
prev_end = end
return timings
except Exception as exc:
logger.warning("[sentence_timings] 静音检测异常,降级为字数比例估算: %s", exc)
return estimate_sentence_timings_by_chars(sentences, total_duration)
finally:
if tmp_path:
try:
os.unlink(tmp_path)
except Exception:
pass