#!/usr/bin/env -S uv run --script
# /// script
# requires-python = ">=3.10"
# dependencies = [
#   "yt-dlp[default,curl-cffi]",
#   "mutagen>=1.47",
# ]
# ///
"""
video_episode.py: turn a video into a "glanceable" podcast episode.

Downloads a YouTube video (or takes a local video file), extracts the audio as
an MP3, grabs a frame every few seconds (or at each scene change), and embeds
those frames as ID3 chapter images. Podcast apps that show chapter artwork
(e.g. Overcast) then play the video as a slow slideshow while you listen.

An independent reimplementation of Alex Chan's glancecast technique:
  https://alexwlchan.net/2026/glancecast/ (write-up)
  https://alexwlchan.net/projects/glancecast/ (their original tool, MIT)

Output goes to a work folder (default ~/Library/Caches/glanceable-podcast/<id>/):
  episode.mp3     the finished episode
  cover.jpg       episode artwork (the video thumbnail, or the first frame)
  episode.json    metadata for publish.py; edit title/notes/sections first
  transcript.txt  (with --transcript) timestamped captions, for writing notes

Usage:
  video_episode.py URL_OR_FILE [--interval 5 | --scene 0.3] [--transcript] ...
"""

import argparse
import html
import json
import re
import shutil
import subprocess
import sys
import time
import urllib.request
from pathlib import Path

from glance import CACHE, SCALE, duration_ms, fail, fmt_ts, run, slugify, write_tags

# --- Download --------------------------------------------------------------

def ytdl_options(args):
  """Options shared by the video and caption downloads."""
  opts = {"noplaylist": True, "quiet": True, "noprogress": True}
  if args.cookies_from_browser:
    opts["cookiesfrombrowser"] = (args.cookies_from_browser,)
  return opts


def substack_video(url):
  """The HLS stream of a Substack post's uploaded video, or None.

  yt-dlp's Substack extractor only sees a video post's podcast audio. The page
  names the upload (`"videoUpload":{"id":…}`), and Substack's video endpoint
  redirects any visitor to a short-lived signed stream URL for it."""
  if "youtube.com" in url or "youtu.be" in url:
    return None
  headers = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7)"}
  try:
    with urllib.request.urlopen(urllib.request.Request(url, headers=headers), timeout=30) as resp:
      page = resp.read().decode("utf-8", errors="replace")
    match = re.search(r'\\?"videoUpload\\?":\{\\?"id\\?":\\?"([0-9a-f-]{36})', page)
    if not match:
      return None
    src = f"https://substack.com/api/v1/video/upload/{match[1]}/src?type=hls"
    with urllib.request.urlopen(urllib.request.Request(src, headers=headers), timeout=30) as resp:
      return resp.geturl()
  except Exception:
    return None


def download(url, args):
  """Download the video (≤720p is plenty for 945px frames) and its
  thumbnail. Returns (workdir, video_path, info)."""
  import yt_dlp

  base = args.workdir or CACHE
  opts = {
    **ytdl_options(args),
    "format": "bv*[height<=720]+ba/b[height<=720]/bv*+ba/b",
    "merge_output_format": "mkv",
    "paths": {"home": str(base)},
    "outtmpl": {
      "default": "%(id)s/video.%(ext)s",
      "thumbnail": "%(id)s/thumb.%(ext)s",
    },
    "writethumbnail": True,
  }
  if args.remove_sponsors:
    categories = ["sponsor", "selfpromo", "interaction"]
    opts["postprocessors"] = [
      {"key": "SponsorBlock", "categories": categories},
      {"key": "ModifyChapters", "remove_sponsor_segments": categories},
    ]
  stream = substack_video(url)
  try:
    with yt_dlp.YoutubeDL(opts) as ydl:
      if stream:
        # Metadata (title, author, date) from the post; the pictures from its video.
        print("Found the post's uploaded video.", file=sys.stderr)
        post = ydl.sanitize_info(ydl.extract_info(url, download=False))
        opts["outtmpl"] = {k: v.replace("%(id)s", str(post["id"])) for k, v in opts["outtmpl"].items()}
        with yt_dlp.YoutubeDL(opts) as video_ydl:
          info = video_ydl.sanitize_info(video_ydl.extract_info(stream, download=True))
        for key in ("id", "title", "channel", "uploader", "upload_date", "description", "webpage_url", "chapters"):
          info[key] = post.get(key)
        info["webpage_url"] = info["webpage_url"] or url
        thumb_url = post.get("thumbnail")
        workdir = Path(info["requested_downloads"][0]["filepath"]).parent
        if thumb_url and not any(workdir.glob("thumb.*")):
          try:
            urllib.request.urlretrieve(thumb_url, workdir / "thumb.jpg")
          except Exception:
            pass  # the cover falls back to the first frame
      else:
        info = ydl.sanitize_info(ydl.extract_info(url, download=True))
  except yt_dlp.utils.DownloadError as err:
    fail(
      f"download failed: {err}\n"
      "- 'Sign in to confirm you're not a bot': retry with --cookies-from-browser safari (or chrome/firefox).\n"
      "- Extraction/format/signature errors: update yt-dlp with "
      "`uv run --upgrade-package yt-dlp video_episode.py ...` and check `deno` is installed."
    )
  if info.get("_type") == "playlist":
    fail("that's a playlist; pass a single video URL")
  if info.get("is_live"):
    fail("live streams aren't supported")

  video = Path(info["requested_downloads"][0]["filepath"])
  return video.parent, video, info


def download_captions(url, workdir, args, attempts=3):
  """Fetch English captions as VTT. YouTube often rate-limits caption
  requests (HTTP 429), so retry with a pause, and give up quietly: an
  episode without a transcript is still an episode. Returns a path or None."""
  import yt_dlp

  opts = {
    **ytdl_options(args),
    "skip_download": True,
    "writesubtitles": True,
    "writeautomaticsub": True,
    "subtitleslangs": ["en", "en-GB", "en-US"],
    "subtitlesformat": "vtt",
    "outtmpl": {"subtitle": str(workdir / "captions.%(ext)s")},
    "no_warnings": True,
  }
  for attempt in range(1, attempts + 1):
    try:
      with yt_dlp.YoutubeDL(opts) as ydl:
        ydl.extract_info(url, download=True)
      break
    except yt_dlp.utils.DownloadError as err:
      if attempt == attempts:
        print(f"warning: no transcript; captions download failed: {err}", file=sys.stderr)
        return None
      time.sleep(10 * attempt)
  return next(iter(sorted(workdir.glob("captions*.vtt"))), None)


# --- Media processing ------------------------------------------------------

def extract_audio(video, mp3, args):
  cmd = ["ffmpeg", "-y", "-v", "error", "-i", str(video), "-map", "0:a:0", "-vn",
         "-map_metadata", "-1", "-c:a", "libmp3lame", "-b:a", args.bitrate]
  if not args.stereo:
    cmd += ["-ac", "1"]  # speech is fine in mono, and it halves the size
  run(cmd + [str(mp3)])


def extract_frames(video, frames_dir, args):
  """Return [(start_seconds, jpg_path), ...] in time order."""
  if frames_dir.exists():
    shutil.rmtree(frames_dir)
  frames_dir.mkdir(parents=True)
  scale = SCALE.format(s=args.max_size)
  pattern = str(frames_dir / "f_%05d.jpg")

  if args.scene is None:
    run(["ffmpeg", "-y", "-v", "error", "-i", str(video), "-an",
         "-vf", f"fps=1/{args.interval},{scale}", "-q:v", "3", pattern])
    files = sorted(frames_dir.glob("f_*.jpg"))
    return [(i * args.interval, f) for i, f in enumerate(files)]

  # Scene mode: always keep the first frame, then one per scene change.
  # showinfo logs each selected frame's timestamp to stderr.
  vf = rf"select=eq(n\,0)+gt(scene\,{args.scene}),showinfo,{scale}"
  proc = run(["ffmpeg", "-y", "-hide_banner", "-i", str(video), "-an",
              "-vf", vf, "-fps_mode", "vfr", "-q:v", "3", pattern])
  times = [float(t) for t in re.findall(r"pts_time:\s*(-?[\d.]+)", proc.stderr)]
  files = sorted(frames_dir.glob("f_*.jpg"))
  if len(times) != len(files):
    fail(f"scene detection mismatch ({len(times)} timestamps, {len(files)} frames)")

  # Transitions (fades, builds) fire several changes in quick succession.
  # Collapse each burst into one chapter that starts at the burst's first
  # change but shows its last, settled frame.
  frames = []
  for t, f in zip(times, files):
    if frames and t - frames[-1][0] < args.min_gap:
      frames[-1][1].unlink()
      frames[-1] = (frames[-1][0], f)
    else:
      frames.append((max(t, 0.0), f))
  return frames


def make_cover(thumb, fallback, cover):
  if thumb:
    run(["ffmpeg", "-y", "-v", "error", "-i", str(thumb),
         "-vf", "scale=min(iw\\,1400):-2", "-q:v", "2", str(cover)])
  else:
    shutil.copyfile(fallback, cover)


# --- Transcript ------------------------------------------------------------

def vtt_to_text(vtt, out, every=30):
  """Flatten captions into timestamped ~30s paragraphs. Auto-captions repeat
  each line in the next cue as they roll, so consecutive duplicates go."""
  paras, current, para_start, cue_start, last = [], [], None, 0, None
  for line in vtt.read_text(encoding="utf-8", errors="replace").splitlines():
    m = re.match(r"^(?:(\d+):)?(\d\d):(\d\d)\.\d{3}\s+-->", line)
    if m:
      cue_start = int(m[1] or 0) * 3600 + int(m[2]) * 60 + int(m[3])
      continue
    text = html.unescape(re.sub(r"<[^>]+>", "", line)).strip()
    if (not text or text == "WEBVTT" or text.isdigit()
        or text.startswith(("Kind:", "Language:", "NOTE"))):
      continue
    if text == last:
      continue
    last = text
    if para_start is None:
      para_start = cue_start
    elif cue_start - para_start >= every and current:
      paras.append((para_start, " ".join(current)))
      current, para_start = [], cue_start
    current.append(text)
  if current:
    paras.append((para_start, " ".join(current)))
  out.write_text("\n\n".join(f"[{fmt_ts(t)}] {txt}" for t, txt in paras) + "\n")


# --- Main ------------------------------------------------------------------

def main():
  p = argparse.ArgumentParser(description="Turn a video into a glanceable podcast episode.")
  p.add_argument("source", help="YouTube (or other yt-dlp supported) URL, or a local video file")
  mode = p.add_mutually_exclusive_group()
  mode.add_argument("--interval", type=float, default=5,
                    help="seconds between frames (default 5)")
  mode.add_argument("--scene", type=float, metavar="THRESHOLD",
                    help="one frame per scene change instead, e.g. 0.3 (good for slides)")
  p.add_argument("--min-gap", type=float, default=2,
                 help="scene mode: merge changes closer together than this (default 2s)")
  p.add_argument("--max-size", type=int, default=945, help="max frame width/height (default 945)")
  p.add_argument("--bitrate", default="64k", help="MP3 bitrate (default 64k)")
  p.add_argument("--stereo", action="store_true", help="keep stereo (default is mono)")
  p.add_argument("--transcript", action="store_true", help="also save captions as transcript.txt")
  p.add_argument("--remove-sponsors", action="store_true",
                 help="cut sponsor/self-promo segments using SponsorBlock")
  p.add_argument("--cookies-from-browser", metavar="BROWSER",
                 help="use YouTube cookies from safari/chrome/firefox (for bot checks)")
  p.add_argument("--title", help="override the episode title")
  p.add_argument("--author", help="override the author")
  p.add_argument("--workdir", type=Path, help=f"base folder for output (default {CACHE})")
  p.add_argument("--keep-video", action="store_true", help="don't delete the downloaded video")
  args = p.parse_args()

  for tool in ("ffmpeg", "ffprobe"):
    if not shutil.which(tool):
      fail(f"{tool} not found; install it with: brew install ffmpeg")

  local = Path(args.source).expanduser()
  if local.is_file():
    workdir = (args.workdir or CACHE) / slugify(local.stem, "video")
    workdir.mkdir(parents=True, exist_ok=True)
    video, info, downloaded = local.resolve(), {}, False
    meta = {"kind": "video", "source_id": None, "title": local.stem, "author": None, "link": None,
            "original_date": None, "description": "", "sections": []}
  elif re.match(r"https?://", args.source):
    print("Downloading…", file=sys.stderr)
    workdir, video, info = download(args.source, args)
    downloaded = True
    date = info.get("upload_date") or ""
    meta = {
      "kind": "video",
      "source_id": info.get("id"),
      "title": info.get("title") or "Untitled",
      "author": info.get("channel") or info.get("uploader"),
      "link": info.get("webpage_url") or args.source,
      "original_date": f"{date[:4]}-{date[4:6]}-{date[6:]}" if len(date) == 8 else None,
      "description": info.get("description") or "",
      "sections": [{"start": int(c["start_time"]), "title": c["title"]}
                   for c in info.get("chapters") or []],
    }
  else:
    fail(f"not a URL or an existing file: {args.source}")

  if args.title:
    meta["title"] = args.title
  if args.author:
    meta["author"] = args.author

  mp3 = workdir / "episode.mp3"
  cover = workdir / "cover.jpg"
  frames_dir = workdir / "frames"

  print("Extracting audio…", file=sys.stderr)
  extract_audio(video, mp3, args)
  total_ms = duration_ms(mp3)

  print("Extracting frames…", file=sys.stderr)
  frames = extract_frames(video, frames_dir, args)
  if not frames:
    fail("no frames were extracted; is this an audio-only file?")

  thumb = next((t for t in sorted(workdir.glob("thumb.*"))), None)
  make_cover(thumb, frames[0][1], cover)

  print("Writing chapters…", file=sys.stderr)
  write_tags(mp3, frames, total_ms, cover, meta)

  transcript = None
  if args.transcript and downloaded:
    print("Fetching captions…", file=sys.stderr)
    vtt = download_captions(args.source, workdir, args)
    if vtt:
      transcript = workdir / "transcript.txt"
      vtt_to_text(vtt, transcript)

  # Tidy up intermediates; the frames now live inside the MP3.
  shutil.rmtree(frames_dir, ignore_errors=True)
  for leftover in [*workdir.glob("thumb.*"), *workdir.glob("captions*.vtt")]:
    leftover.unlink(missing_ok=True)
  if downloaded and not args.keep_video:
    video.unlink(missing_ok=True)

  frame_mode = (f"scene changes (threshold {args.scene})" if args.scene is not None
                else f"every {args.interval:g}s")
  episode = {
    **meta,
    "notes": "",
    "duration_seconds": round(total_ms / 1000),
    "mp3": str(mp3),
    "cover": str(cover),
    "frames": {"count": len(frames), "mode": frame_mode},
    "transcript": str(transcript) if transcript else None,
  }
  episode_json = workdir / "episode.json"
  episode_json.write_text(json.dumps(episode, indent=2, ensure_ascii=False) + "\n")

  size_mb = mp3.stat().st_size / 1_000_000
  print(f"""
Episode ready: {meta['title']}
  audio:      {fmt_ts(total_ms / 1000)}, {size_mb:.1f} MB
  frames:     {len(frames)} ({frame_mode})
  sections:   {len(meta['sections'])}{' (from YouTube chapters)' if meta['sections'] else ''}
  transcript: {transcript or ('none available' if args.transcript else 'not requested')}
  metadata:   {episode_json}

Next: review episode.json (title, notes, sections), then run
  publish.py add {episode_json} --cleanup""")


if __name__ == "__main__":
  main()
