aps-agent/video/gen_audio.py

37 lines
1.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""从 narration.md 解析 SEG 分段,逐段调用 TTS 生成 MP3。"""
import os
import re
import subprocess
import sys
from pathlib import Path
WS = Path(__file__).resolve().parent
plugin_dir = (os.environ.get("APS_AUDIO_GENERATION_PLUGIN") or "").strip()
voice_id = (os.environ.get("APS_TTS_VOICE_ID") or "").strip()
if not plugin_dir or not voice_id:
raise SystemExit("请配置 APS_AUDIO_GENERATION_PLUGIN 和 APS_TTS_VOICE_ID 后再生成配音")
PLUGIN = Path(plugin_dir).expanduser().resolve()
VOICE = voice_id
text = (WS / "narration.md").read_text(encoding="utf-8")
segments = re.findall(r"SEG (\d+)([^)]*)\s*\n(.*?)(?=\n\nSEG |\Z)", text, re.S)
print(f"共解析 {len(segments)} 段")
for num, body in segments:
out = WS / "audio" / f"seg{num}.mp3"
if out.exists() and out.stat().st_size > 10_000:
print(f"seg{num} 已存在,跳过")
continue
body = " ".join(body.split())
result = subprocess.run(
[sys.executable, str(PLUGIN / "scripts" / "audio_generation_tool.py"), "speech",
"--text", body, "--voice-id", VOICE, "--output", str(out)],
cwd=PLUGIN, capture_output=True, text=True, timeout=600,
)
ok = out.exists() and out.stat().st_size > 10_000
print(f"seg{num}: {'OK' if ok else 'FAIL'} ({len(body)} 字)")
if not ok:
print(result.stdout[-500:], result.stderr[-500:])
sys.exit(1)
print("全部配音生成完成")