36 lines
1.9 KiB
Python
36 lines
1.9 KiB
Python
"""Parse the approved markdown script into clips + durations."""
|
|
import json, os, re, subprocess, sys
|
|
from pathlib import Path
|
|
SP = Path(__file__).parent
|
|
SRC = Path(os.environ.get("NARRATION_MD", SP.parent / "narration.md"))
|
|
KEY = Path(os.environ.get("XAI_KEY_FILE", "~/xai-api.key")).expanduser().read_text().strip()
|
|
VOICE = os.environ.get("XAI_VOICE_ID", "") # a custom voice id from /v1/custom-voices
|
|
SPEED = float(sys.argv[1]) if len(sys.argv) > 1 else 1.0
|
|
|
|
beats = re.findall(r"\[([id]\d+)\]\s*(.*?)(?=\n\n|\Z)", SRC.read_text(), re.S)
|
|
lines = {k: " ".join(v.split()) for k, v in beats}
|
|
(SP / "narration.json").write_text(json.dumps(lines, indent=2) + "\n")
|
|
|
|
# Do NOT map an acronym to run-together phonetics: {"VM": "vee em"} is spoken
|
|
# as one word, "vem". Expand it instead, or leave it alone.
|
|
REPLACE = {"VM": "virtual machine"}
|
|
out = {}
|
|
for key, text in lines.items():
|
|
body = {"text": text, "voice_id": VOICE, "language": "en", "speed": SPEED,
|
|
"replace": REPLACE, "output_format": {"codec": "wav", "sample_rate": 24000}}
|
|
req = SP / "req.json"; req.write_text(json.dumps(body))
|
|
dest = SP / f"{key}.wav"
|
|
code = subprocess.run(["curl", "-sS", "-o", str(dest), "-w", "%{http_code}", "-X", "POST",
|
|
"https://api.x.ai/v1/tts", "-H", f"Authorization: Bearer {KEY}",
|
|
"-H", "Content-Type: application/json", "--data-binary", f"@{req}"],
|
|
capture_output=True, text=True).stdout.strip()
|
|
req.unlink()
|
|
if code != "200":
|
|
print(f" {key}: HTTP {code} FAILED"); sys.exit(1)
|
|
d = float(subprocess.run(["ffprobe", "-v", "error", "-show_entries", "format=duration",
|
|
"-of", "default=nw=1:nk=1", str(dest)], capture_output=True, text=True).stdout.strip())
|
|
out[key] = round(d, 2)
|
|
print(f" {key:5} {d:5.2f}s {len(text.split()):3} words")
|
|
(SP / "durations.json").write_text(json.dumps(out, indent=1))
|
|
print(f"\n {len(out)} clips, {sum(out.values()):.1f}s of narration")
|