#!/usr/bin/env -S uv run --script
# /// script
# requires-python = ">=3.10,<3.13"
# dependencies = [
#   "kokoro>=0.9.4",
#   "transformers>=4.45",
#   "torch>=2.4",
#   "numpy",
#   "soundfile",
#   "mutagen>=1.47",
#   "pip",
# ]
# ///
"""
article_episode.py: read an article aloud as a glanceable podcast episode.

Takes the article.json written by fetch_article.py and speaks it with Kokoro
(open-source TTS, Apache-2.0): the body in one voice, image captions in a
second voice. Each time the article shows an image, a soft chime plays and that
image becomes the episode's chapter artwork, so a glance at the podcast app
shows what the text is talking about.

  narrator   af_heart          captions   am_puck, a little faster
  images     chime, then the caption; bunched images get a quieter tick each
             and are held on screen for at least --min-dwell seconds;
             galleries of more than four are shown as one grid

article.json fields this script reads beyond fetch_article.py's output:
  pronounce      {"McKnight": "Mick-Nite"} respellings, or {"Kauffer": "/kˈɔfəɹ/"}
                 Kokoro phonemes, applied wherever the word appears
  block.speak    replaces a block's text when spoken (keep the meaning!)
  image.speak    what to say for an image (default: the tidied caption or alt text)
  block.skip     true to leave a block out of the audio
  math_say       {"ΔM": "Delta M", "(t′,q′,k′)": null} how each inline formula is
                 spoken; null (or missing) says "shown" while its card is on screen

Writes episode.mp3 and episode.json next to article.json, ready for
publish.py add.

Usage:
  article_episode.py path/to/article.json [--voice af_heart] [--caption-voice am_puck]
"""

import argparse
import json
import re
import subprocess
import sys
import time
from pathlib import Path

import numpy as np

from glance import duration_ms, fail, fmt_ts, run, write_tags

SR = 24000

# Pauses, in seconds, chosen by ear during the TTS bake-off.
PAUSE = {
  "after_paragraph": 0.75,
  "before_heading": 0.7,     # on top of the previous block's pause
  "after_heading": 0.6,
  "after_list_item": 0.45,
  "break": 1.2,
  "before_image": 0.35,
  "after_cue": 0.3,
  "after_image": 0.6,
  "after_intro": 1.2,
}


# --- Sound -----------------------------------------------------------------

def silence(seconds):
  return np.zeros(int(seconds * SR), dtype=np.float32)


def note(freq, duration, volume, delay=0.0):
  """A soft bell-like note: 10ms attack, exponential decay."""
  t = np.arange(int(duration * SR)) / SR
  envelope = np.minimum(t / 0.01, 1.0) * np.exp(-t * 7.5)
  tone = volume * envelope * np.sin(2 * np.pi * freq * t)
  return np.concatenate([silence(delay), tone]).astype(np.float32)


def mix(*sounds):
  out = np.zeros(max(len(s) for s in sounds), dtype=np.float32)
  for s in sounds:
    out[:len(s)] += s
  return out


# Two-note chime (E6 then A5) for "look: a new image"; a single quieter
# note for each further image in a bunch.
CHIME = mix(note(1318.5, 0.45, 0.35), note(880.0, 0.6, 0.35, delay=0.14))
TICK = note(880.0, 0.3, 0.22)
# Barely-there cue for a paragraph's formula card: maths papers have dozens.
FAINT_TICK = note(1046.5, 0.22, 0.09)

MATH_TOKEN = re.compile(r"⟦(m\d+)⟧")


# --- Text ------------------------------------------------------------------

def tidy_caption(text):
  """Captions are fragments: fold bracketed asides into a clause and end
  with a full stop, so they're read as a sentence rather than a label."""
  text = re.sub(r"\s*\(([^)]+)\)", r", \1", text.strip())
  text = re.sub(r"\s+,", ",", re.sub(r"\s+", " ", text)).strip(" ,")
  return text if text.endswith((".", "!", "?", "…")) else text + "."


def meaningful_alt(alt):
  return bool(alt) and len(alt) >= 8 and not re.search(r"\.(jpe?g|png|webp|gif)$|^(image|img|photo)\b", alt, re.I)


def image_words(image):
  """What to say for one image: Claude's `speak`, else caption, else alt text."""
  if image.get("speak") is not None:
    return image["speak"].strip()
  if image.get("caption"):
    return tidy_caption(image["caption"])
  if meaningful_alt(image.get("alt")):
    return tidy_caption(image["alt"])
  return ""


def apply_pronunciations(text, pronounce):
  """Respell words Kokoro gets wrong. Values wrapped in slashes are Kokoro
  phonemes ("[word](/phonemes/)"); anything else is a plain respelling."""
  for word, say in pronounce.items():
    pattern = re.compile(rf"(?<![\w\[]){re.escape(word)}(?![\w\]])")
    replacement = f"[{word}]({say})" if say.startswith("/") and say.endswith("/") else say
    text = pattern.sub(lambda m: replacement, text)
  return text


# --- Rendering -------------------------------------------------------------

class Episode:
  """Accumulates audio and records where chapters and sections start."""

  def __init__(self, pipeline, args, pronounce, maths=None, math_say=None):
    self.pipeline, self.args, self.pronounce = pipeline, args, pronounce
    self.maths, self.math_say = maths or {}, math_say or {}
    self.parts, self.samples = [], 0
    self.chapters, self.sections = [], []

  @property
  def now(self):
    return self.samples / SR

  def add(self, audio):
    self.parts.append(audio)
    self.samples += len(audio)

  def pause(self, seconds):
    self.add(silence(seconds))

  def speak_maths(self, text):
    """Inline formulas are spoken by name, or as "shown" while on screen."""
    def name(match):
      formula = self.maths.get(match[1], {}).get("text", "")
      return self.math_say.get(formula) or "shown"
    return MATH_TOKEN.sub(name, text)

  def say(self, text, caption=False):
    text = apply_pronunciations(self.speak_maths(text).strip(), self.pronounce)
    if not text:
      return
    voice = self.args.caption_voice if caption else self.args.voice
    speed = self.args.caption_speed if caption else self.args.speed
    for _, _, audio in self.pipeline(text, voice=voice, speed=speed):
      if audio is not None:
        self.add(audio.numpy().astype(np.float32) if hasattr(audio, "numpy") else np.asarray(audio, np.float32))

  def chapter(self, image_path):
    self.chapters.append((self.now, image_path))

  def audio(self):
    return np.concatenate(self.parts) if self.parts else silence(1)


def render_images(ep, block, workdir):
  images = block["images"]
  ep.pause(PAUSE["before_image"])
  if block.get("grid"):
    ep.chapter(workdir / block["grid"])
    ep.add(CHIME)
    ep.pause(PAUSE["after_cue"])
    ep.say(tidy_caption(block["caption"]) if block.get("caption") else f"A gallery of {len(images)} images.",
           caption=True)
  else:
    for i, image in enumerate(images):
      shown_at = ep.now
      ep.chapter(workdir / image["file"])
      ep.add(CHIME if i == 0 else TICK)
      ep.pause(PAUSE["after_cue"])
      ep.say(image_words(image), caption=True)
      # Hold each image in a bunch long enough to glance at before the next.
      if i < len(images) - 1:
        ep.pause(max(0.35, ep.args.min_dwell - (ep.now - shown_at)))
    if block.get("caption"):
      ep.pause(0.3)
      ep.say(tidy_caption(block["caption"]), caption=True)
  ep.pause(PAUSE["after_image"])


def render(article, workdir, pipeline, args):
  ep = Episode(pipeline, args, article.get("pronounce") or {}, article.get("maths"), article.get("math_say"))
  plain = lambda t: MATH_TOKEN.sub(lambda m: (article.get("maths") or {}).get(m[1], {}).get("text", ""), t)
  ep.chapter(workdir / article["cover"])

  # Intro: title, subtitle, byline.
  ep.say(article.get("speak_title") or tidy_caption(article["title"]))
  if article.get("subtitle"):
    ep.pause(0.35)
    ep.say(tidy_caption(article["subtitle"]))
  # Many newsletters are named after their author; don't say the name twice.
  author, site = (article.get("author") or "").strip(), (article.get("site") or "").strip()
  if site.casefold() == author.casefold():
    site = ""
  byline = ", ".join(v for v in (f"By {author}" if author else None, site) if v)
  if byline:
    ep.pause(0.35)
    ep.say(byline + ".")
  ep.pause(PAUSE["after_intro"])

  blocks = [b for b in article["blocks"] if not b.get("skip")]
  for n, block in enumerate(blocks, 1):
    kind = block["type"]
    text = block.get("speak") or block.get("text") or ""
    if block.get("maths_card"):
      # This block's formulas go on screen as it starts.
      ep.chapter(workdir / block["maths_card"])
      ep.add(FAINT_TICK)
      ep.pause(0.15)
    if kind == "heading":
      ep.pause(PAUSE["before_heading"])
      ep.sections.append({"start": int(ep.now), "title": plain(block["text"])})
      ep.say(tidy_caption(text))
      ep.pause(PAUSE["after_heading"])
    elif kind in ("paragraph", "quote"):
      ep.say(text)
      ep.pause(PAUSE["after_paragraph"])
    elif kind == "list":
      for item in block.get("speak_items") or block["items"]:
        ep.say(tidy_caption(item))
        ep.pause(PAUSE["after_list_item"])
      ep.pause(PAUSE["after_paragraph"] - PAUSE["after_list_item"])
    elif kind == "note":
      ep.say(text, caption=True)
      ep.pause(PAUSE["after_paragraph"])
    elif kind == "break":
      ep.pause(PAUSE["break"])
    elif kind == "images" and block["images"]:
      render_images(ep, block, workdir)
    elif kind == "equation" and block.get("file"):
      # Display equations are shown, never spoken; a short silence to look.
      ep.pause(0.3)
      ep.chapter(workdir / block["file"])
      ep.add(TICK)
      ep.pause(1.6)
    if n % 10 == 0:
      print(f"  {n}/{len(blocks)} blocks, {fmt_ts(ep.now)} of audio", file=sys.stderr, flush=True)
  return ep


def encode(audio, mp3, bitrate):
  """Normalise to podcast loudness (-16 LUFS) and encode as mono MP3."""
  import soundfile as sf
  wav = mp3.with_suffix(".wav")
  sf.write(wav, audio, SR)
  run(["ffmpeg", "-y", "-v", "error", "-i", str(wav), "-af", "loudnorm=I=-16:TP=-1.5:LRA=11",
       "-ar", "44100", "-ac", "1", "-map_metadata", "-1", "-c:a", "libmp3lame", "-b:a", bitrate, str(mp3)])
  wav.unlink()


# --- Main ------------------------------------------------------------------

def main():
  p = argparse.ArgumentParser(description="Read an article aloud as a glanceable podcast episode.")
  p.add_argument("article", type=Path, help="article.json from fetch_article.py")
  p.add_argument("--voice", default="af_heart", help="Kokoro voice for the article (default af_heart)")
  p.add_argument("--caption-voice", default="am_puck", help="Kokoro voice for captions (default am_puck)")
  p.add_argument("--speed", type=float, default=1.0, help="speaking speed for the article (default 1.0)")
  p.add_argument("--caption-speed", type=float, default=1.1, help="speaking speed for captions (default 1.1)")
  p.add_argument("--min-dwell", type=float, default=4.0,
                 help="minimum seconds each image in a bunch stays on screen (default 4)")
  p.add_argument("--bitrate", default="64k", help="MP3 bitrate (default 64k)")
  args = p.parse_args()

  article_path = args.article.expanduser().resolve()
  if not article_path.is_file():
    fail(f"not found: {article_path}")
  article = json.loads(article_path.read_text())
  workdir = article_path.parent

  print("Loading Kokoro…", file=sys.stderr)
  from kokoro import KPipeline
  pipeline = KPipeline(lang_code="a", repo_id="hexgrad/Kokoro-82M")

  print("Speaking…", file=sys.stderr)
  started = time.time()
  ep = render(article, workdir, pipeline, args)
  audio = ep.audio()
  took = time.time() - started

  mp3 = workdir / "episode.mp3"
  print("Encoding…", file=sys.stderr)
  encode(audio, mp3, args.bitrate)
  total_ms = duration_ms(mp3)

  meta = {
    "kind": "article",
    "source_id": article.get("link"),
    "title": article["title"],
    "author": article.get("author") or article.get("site"),
    "link": article.get("link"),
    "original_date": article.get("date"),
  }
  write_tags(mp3, ep.chapters, total_ms, workdir / article["cover"], meta)

  image_count = len(ep.chapters) - 1  # the first chapter is the cover
  episode = {
    **meta,
    "site": article.get("site"),
    "description": "",
    "notes": article.get("notes") or article.get("subtitle") or "",
    "sections": ep.sections,
    "duration_seconds": round(total_ms / 1000),
    "mp3": str(mp3),
    "cover": str(workdir / article["cover"]),
    "frames": {"count": image_count, "mode": "article images"},
    "transcript": None,
  }
  episode_json = workdir / "episode.json"
  episode_json.write_text(json.dumps(episode, indent=2, ensure_ascii=False) + "\n")

  speech = len(audio) / SR
  print(f"""
Episode ready: {article['title']}
  audio:     {fmt_ts(total_ms / 1000)}, {mp3.stat().st_size / 1_000_000:.1f} MB (spoken in {took:.0f}s, {speech / took:.1f}x real time)
  images:    {image_count} chapters, after the cover
  sections:  {len(ep.sections)} (from headings)
  voices:    {args.voice} (article), {args.caption_voice} (captions)
  metadata:  {episode_json}

Next: publish.py add {episode_json} --cleanup""")


if __name__ == "__main__":
  main()
