Files
zerto-ai-rewind/demo/vo/build_narration.py
T

36 lines
1.9 KiB
Python

"""Parse the approved markdown script into clips + durations."""
import json, os, re, subprocess, sys
from pathlib import Path
SP = Path(__file__).parent
SRC = Path(os.environ.get("NARRATION_MD", SP.parent / "narration.md"))
KEY = Path(os.environ.get("XAI_KEY_FILE", "~/xai-api.key")).expanduser().read_text().strip()
VOICE = os.environ.get("XAI_VOICE_ID", "") # a custom voice id from /v1/custom-voices
SPEED = float(sys.argv[1]) if len(sys.argv) > 1 else 1.0
beats = re.findall(r"\[([id]\d+)\]\s*(.*?)(?=\n\n|\Z)", SRC.read_text(), re.S)
lines = {k: " ".join(v.split()) for k, v in beats}
(SP / "narration.json").write_text(json.dumps(lines, indent=2) + "\n")
# Do NOT map an acronym to run-together phonetics: {"VM": "vee em"} is spoken
# as one word, "vem". Expand it instead, or leave it alone.
REPLACE = {"VM": "virtual machine"}
out = {}
for key, text in lines.items():
body = {"text": text, "voice_id": VOICE, "language": "en", "speed": SPEED,
"replace": REPLACE, "output_format": {"codec": "wav", "sample_rate": 24000}}
req = SP / "req.json"; req.write_text(json.dumps(body))
dest = SP / f"{key}.wav"
code = subprocess.run(["curl", "-sS", "-o", str(dest), "-w", "%{http_code}", "-X", "POST",
"https://api.x.ai/v1/tts", "-H", f"Authorization: Bearer {KEY}",
"-H", "Content-Type: application/json", "--data-binary", f"@{req}"],
capture_output=True, text=True).stdout.strip()
req.unlink()
if code != "200":
print(f" {key}: HTTP {code} FAILED"); sys.exit(1)
d = float(subprocess.run(["ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "default=nw=1:nk=1", str(dest)], capture_output=True, text=True).stdout.strip())
out[key] = round(d, 2)
print(f" {key:5} {d:5.2f}s {len(text.split()):3} words")
(SP / "durations.json").write_text(json.dumps(out, indent=1))
print(f"\n {len(out)} clips, {sum(out.values()):.1f}s of narration")