Files
xiaoxia-saas/packages/domain/sentence_timings.py
T
xiaoxia 26f3abab72
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 18s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 12s
CI/CD Pipeline / Build Staging API Image (push) Successful in 1m17s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 1m23s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m22s
CI/CD Pipeline / Validate - Style (push) Successful in 3m41s
CI/CD Pipeline / Integration Tests (push) Successful in 4m1s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 4m16s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m27s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m36s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 1m42s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m24s
CI/CD Pipeline / Validate - Security (push) Successful in 8m48s
CI/CD Pipeline / Unit Tests (push) Successful in 9m33s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Failing after 44h58m30s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 45h6m21s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 45h8m6s
CI/CD Pipeline / PR Build Web Image (push) Failing after 45h7m38s
CI/CD Pipeline / PR Build API Image (push) Failing after 45h7m39s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 45h5m53s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 45h5m53s
CI/CD Pipeline / CI Gate (push) Failing after 44h58m2s
CI/CD Pipeline / Deploy Production (push) Failing after 44h57m58s
CI/CD Pipeline / Build Production API Image (push) Failing after 44h58m2s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 45h7m40s
CI/CD Pipeline / Canary Release to Production (push) Failing after 44h57m58s
CI/CD Pipeline / Build Production Web Image (push) Failing after 44h58m2s
CI/CD Pipeline / Frontend Lint (push) Failing after 45h7m34s
fix(backend): #1892 画面插入按句号分句 (#1905)
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-09-14 18:15:38 +08:00

207 lines
6.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""共享的句子时间戳计算工具 — 供 Celery TTS 任务和 /lipsync/tts-preview 同步接口复用.
- `split_script_into_sentences`: 按标点分句(中英文句号/问号/感叹号/分号/换行,不含逗号,与前端 sentences.ts 保持一致)
- `_estimate_sentence_timings_by_chars`: 按字数比例估算(静音检测失败时降级)
- `_probe_audio_duration`: ffprobe 读取音频时长
- `compute_sentence_timings`: 基于 ffmpeg silencedetect 精确计算每句起止时间
"""
from __future__ import annotations
import logging
import os
import re
import subprocess
import tempfile
from typing import Optional
logger = logging.getLogger(__name__)
def split_script_into_sentences(script_text: str) -> list[str]:
"""按句号/问号/感叹号/分号/换行分句(与前端 sentences.ts 的 SENTENCE_SPLIT_RE 一致).
仅在句末标点(。!?!?;;)和换行处分句,**不再用逗号(,,)切分**。
按逗号切分会把连贯句子拆得过碎,导致 B-roll/口播画面按句插入时句数过多、
时长过短,效果不符合预期(Issue #1892)。
"""
text = (script_text or "").strip()
if not text:
return []
parts = re.split(r"[。!?!??!;;\n\r]+", text)
return [p.strip() for p in parts if p.strip()]
def estimate_sentence_timings_by_chars(sentences: list[str], total_duration: float) -> list[dict]:
"""降级方案:按字数比例估算句子时间(与原前端逻辑一致)."""
if not sentences or total_duration <= 0:
return []
total_chars = sum(len(s.replace(r"\s", "")) for s in sentences)
if total_chars == 0:
return []
timings = []
acc = 0
for i, sent in enumerate(sentences):
chars = len(sent.replace(r"\s", ""))
start = (acc / total_chars) * total_duration
end = ((acc + chars) / total_chars) * total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
acc += chars
return timings
def probe_audio_duration(audio_data: bytes, timeout: int = 10) -> float:
"""用 ffprobe 读取音频字节流的时长(秒).
Returns:
时长(秒),失败返回 0.0
"""
if not audio_data:
return 0.0
tmp_path: Optional[str] = None
try:
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp:
tmp.write(audio_data)
tmp_path = tmp.name
result = subprocess.run(
[
"ffprobe",
"-v",
"error",
"-show_entries",
"format=duration",
"-of",
"default=noprint_wrappers=1:nokey=1",
tmp_path,
],
capture_output=True,
text=True,
timeout=timeout,
)
stdout = (result.stdout or "").strip()
if not stdout:
logger.warning("[sentence_timings] ffprobe 无输出: stderr=%s", (result.stderr or "")[:200])
return 0.0
return float(stdout)
except Exception as exc:
logger.warning("[sentence_timings] ffprobe 时长探测失败: %s", exc)
return 0.0
finally:
if tmp_path:
try:
os.unlink(tmp_path)
except Exception:
pass
def compute_sentence_timings(audio_data: bytes, script_text: str, total_duration: float) -> list[dict]:
"""基于 TTS 音频的静音检测,精确计算每句文案的起止时间.
使用 ffmpeg silencedetect 检测静音段,将静音点与句子边界对齐。
比字数比例估算准确得多。
Args:
audio_data: TTS 音频二进制数据(MP3)
script_text: 文案全文
total_duration: 音频总时长(秒)
Returns:
list[{"index": int, "text": str, "start_time": float, "end_time": float}]
"""
sentences = split_script_into_sentences(script_text)
if not sentences:
return []
tmp_path: Optional[str] = None
try:
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp:
tmp.write(audio_data)
tmp_path = tmp.name
result = subprocess.run(
[
"ffmpeg",
"-i",
tmp_path,
"-af",
"silencedetect=noise=-25dB:d=0.3",
"-f",
"null",
"-",
],
capture_output=True,
text=True,
timeout=30,
)
stderr = result.stderr or ""
silence_ends = []
for match in re.finditer(r"silence_end:\s*([\d.]+)", stderr):
t = float(match.group(1))
if 0 < t < total_duration:
silence_ends.append(t)
if len(silence_ends) < len(sentences) - 1:
logger.warning(
"[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算",
len(silence_ends),
len(sentences) - 1,
)
return estimate_sentence_timings_by_chars(sentences, total_duration)
n_boundaries = len(sentences) - 1
boundaries = []
used_indices = set()
for i in range(n_boundaries):
expected_pos = (i + 1) / len(sentences) * total_duration
best_idx = None
best_dist = float("inf")
for j, t in enumerate(silence_ends):
if j in used_indices:
continue
dist = abs(t - expected_pos)
if dist < best_dist:
best_dist = dist
best_idx = j
if best_idx is not None:
used_indices.add(best_idx)
boundaries.append(silence_ends[best_idx])
boundaries.sort()
timings = []
prev_end = 0.0
for i, sent in enumerate(sentences):
start = prev_end
end = boundaries[i] if i < len(boundaries) else total_duration
timings.append(
{
"index": i,
"text": sent,
"start_time": round(start, 2),
"end_time": round(end, 2),
}
)
prev_end = end
return timings
except Exception as exc:
logger.warning("[sentence_timings] 静音检测异常,降级为字数比例估算: %s", exc)
return estimate_sentence_timings_by_chars(sentences, total_duration)
finally:
if tmp_path:
try:
os.unlink(tmp_path)
except Exception:
pass