#!/usr/bin/env -S uv run --script
# /// script
# requires-python = ">=3.10"
# dependencies = [
#   "beautifulsoup4>=4.12",
#   "lxml>=5",
#   "readability-lxml>=0.8.1",
#   "lxml_html_clean",
#   "pillow>=10.1",
#   "playwright>=1.45",
# ]
# ///
"""
fetch_article.py: extract an article (text, headings and images) for reading aloud.

Fetches a web article (or reads a saved .html file), keeps only the article
itself, downloads its images, and writes article.json: an ordered list of
blocks (headings, paragraphs, quotes, lists and image groups) that
article_episode.py turns into a glanceable podcast episode.

Substack posts get a dedicated extractor (galleries, captions, full-size
images), as do journal pages on the Atypon platform (PNAS and others: abstract,
theorem boxes); other sites go through Readability.

Maths (MathML, as written by MathJax and KaTeX) is rendered to images with
MathJax in your installed Chrome, via Playwright. Display equations become
their own blocks; inline formulas become ⟦m001⟧ tokens in the text, with a card
image of each paragraph's formulas. `math_say` maps each distinct formula to
how it's spoken: short ones get a name ("ΔM" → "Delta M"), longer ones are
null and spoken as "shown".

Output goes to a work folder (default ~/Library/Caches/glanceable-podcast/article-<slug>/):
  article.json    the article; review it (pronunciations, captions, notes) before rendering
  images/         each image as a JPEG fitted to 945px, plus grids for big galleries
  cover.jpg       episode artwork (the article's share image, or a title card)

Usage:
  fetch_article.py URL_OR_HTML_FILE [--workdir DIR]
"""

import argparse
import io
import json
import math
import re
import sys
import urllib.parse
import urllib.request
from pathlib import Path

from glance import CACHE, MAX_SIZE, fail, slugify

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/128.0 Safari/537.36")

# Galleries bigger than this become one grid image rather than a run of chapters.
GRID_OVER = 4

# Page furniture to drop before walking the article body.
JUNK = [
  "script", "style", "noscript", "svg", "button", "form", "iframe",
  "s", "del", "strike",  # struck-through text: authors' jokey corrections read as nonsense aloud
  ".subscription-widget-wrap", ".subscription-widget", ".button-wrapper",
  ".captioned-button-wrap", ".share-dialog", ".header-anchor-parent",
  ".image-link-expand", "a.footnote-anchor", ".footnote", ".embedded-post-wrap",
  ".tweet", ".youtube-wrap", ".paywall", ".poll-embed", ".install-substack-app-embed",
]

BLOCK_TAGS = {"p", "h1", "h2", "h3", "h4", "h5", "h6", "blockquote", "ul", "ol",
              "figure", "hr", "pre", "table", "img", "div", "section", "article"}


# --- Fetching --------------------------------------------------------------

def http_get(url, binary=False):
  req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept": "*/*"})
  with urllib.request.urlopen(req, timeout=60) as resp:
    data = resp.read()
    if binary:
      return data
    return data.decode(resp.headers.get_content_charset() or "utf-8", errors="replace"), resp.geturl()


def load_page(source):
  path = Path(source).expanduser()
  if path.is_file():
    return path.read_text(encoding="utf-8", errors="replace"), None
  if not re.match(r"https?://", source):
    fail(f"not a URL or an existing file: {source}")
  try:
    return http_get(source)
  except Exception as err:
    fail(f"couldn't fetch the page: {err}")


# --- Metadata --------------------------------------------------------------

def meta_content(soup, *keys):
  for key in keys:
    tag = soup.find("meta", attrs={"property": key}) or soup.find("meta", attrs={"name": key})
    if tag and tag.get("content"):
      return tag["content"].strip()
  return None


def json_ld_article(soup):
  """The first schema.org Article-like object on the page, if any, with
  author/publisher references ({"@id": ...}) resolved within the same graph."""
  for script in soup.find_all("script", type="application/ld+json"):
    try:
      data = json.loads(script.string or "")
    except ValueError:
      continue
    items = data if isinstance(data, list) else data.get("@graph", [data])
    by_id = {i["@id"]: i for i in items if isinstance(i, dict) and "@id" in i}
    resolve = lambda v: by_id.get(v["@id"], v) if isinstance(v, dict) and "@id" in v and "name" not in v else v
    for item in items:
      kind = item.get("@type") if isinstance(item, dict) else None
      kinds = kind if isinstance(kind, list) else [kind]
      if any(k and ("Article" in k or k == "BlogPosting") for k in kinds):
        item = dict(item)
        for key in ("author", "publisher"):
          value = item.get(key)
          item[key] = [resolve(v) for v in value] if isinstance(value, list) else resolve(value)
        return item
  return {}


def read_metadata(soup, final_url):
  ld = json_ld_article(soup)
  author = ld.get("author")
  if isinstance(author, list):
    author = ", ".join(a.get("name", "") for a in author if isinstance(a, dict)) or None
  elif isinstance(author, dict):
    author = author.get("name")
  # Journals publish Google Scholar tags: one citation_author per author.
  scholar_authors = [m["content"].strip() for m in soup.find_all("meta", attrs={"name": "citation_author"})
                     if m.get("content")]
  if scholar_authors:
    names = [" ".join(reversed(a.split(", "))) if ", " in a else a for a in scholar_authors]
    author = names[0] if len(names) == 1 else ", ".join(names[:-1]) + " and " + names[-1]
  canonical = soup.find("link", rel="canonical")
  # The share title (og:title) rarely carries the " | Site name" suffix that <title> does.
  title_tag = soup.select_one("h1.post-title")
  subtitle_tag = soup.select_one("h3.subtitle")
  date = (ld.get("datePublished") or meta_content(soup, "article:published_time")
          or (meta_content(soup, "citation_publication_date", "citation_date") or "").replace("/", "-"))
  return {
    "title": ((title_tag.get_text(" ", strip=True) if title_tag else None) or meta_content(soup, "citation_title")
              or meta_content(soup, "og:title")
              or (soup.find("h1").get_text(" ", strip=True) if soup.find("h1") else None)
              or (soup.find("title").get_text(" ", strip=True) if soup.find("title") else None) or "Untitled"),
    "subtitle": subtitle_tag.get_text(" ", strip=True) if subtitle_tag else None,
    "author": author or meta_content(soup, "author", "article:author"),
    "site": (meta_content(soup, "citation_journal_title", "og:site_name")
             or (ld.get("publisher") or {}).get("name")),
    "date": date[:10] or None,
    "link": (canonical.get("href") if canonical else None) or meta_content(soup, "og:url") or final_url,
    "cover_url": meta_content(soup, "og:image", "twitter:image"),
  }


# --- Maths -----------------------------------------------------------------

MATH_TOKEN = re.compile(r"⟦(m\d+)⟧")
SHORT_MATH = 4  # formulas this short (in characters) are spoken by name

SYMBOL_NAMES = {
  "α": "alpha", "β": "beta", "γ": "gamma", "Γ": "Gamma", "δ": "delta", "Δ": "Delta", "ε": "epsilon",
  "ϵ": "epsilon", "ζ": "zeta", "η": "eta", "θ": "theta", "Θ": "Theta", "ι": "iota", "κ": "kappa",
  "λ": "lambda", "Λ": "Lambda", "μ": "mu", "ν": "nu", "ξ": "xi", "Ξ": "Xi", "π": "pi", "Π": "Pi",
  "ρ": "rho", "σ": "sigma", "Σ": "Sigma", "τ": "tau", "υ": "upsilon", "φ": "phi", "ϕ": "phi",
  "Φ": "Phi", "χ": "chi", "ψ": "psi", "Ψ": "Psi", "ω": "omega", "Ω": "Omega",
  "′": " prime", "″": " double prime", "∈": " in ", "∉": " not in ", "⊂": " subset of ",
  "⊆": " subset of ", "∂": "boundary ", "×": " times ", "=": " equals ", "≠": " not equal to ",
  "≤": " at most ", "≥": " at least ", "<": " less than ", ">": " greater than ", "+": " plus ",
  "−": " minus ", "→": " to ", "↦": " maps to ", "∞": "infinity", "∅": "the empty set",
  "ℤ": "Z", "ℕ": "N", "ℝ": "R", "ℚ": "Q", "ℂ": "C", "∘": " composed with ", "⋅": " times ", "·": " times ",
}


def auto_name(text):
  """A first guess at speaking a short formula; Claude reviews these."""
  out = ""
  for ch in text:
    if ch in SYMBOL_NAMES:
      # Greek letters are words, so keep them apart from what follows: ΔM -> "Delta M".
      out += SYMBOL_NAMES[ch] + (" " if ch.isalpha() else "")
    elif ch.isdigit() and out and out[-1].isalpha():
      out += " " + ch  # q0 -> "q 0"; superscripts and subscripts both flatten to digits
    elif ch in "(){}[]⟨⟩|,":
      out += " "
    else:
      out += ch
  return re.sub(r"\s+", " ", out).strip()


def extract_maths(body):
  """Swap each formula for a placeholder: inline ones become ⟦m001⟧ text tokens,
  display ones become <figure data-equation="m001">. Returns {id: details}."""
  from bs4 import BeautifulSoup, NavigableString

  maker = BeautifulSoup("", "lxml")
  maths = {}
  for math in list(body.find_all("math")):
    if math.find_parent("math") or not math.parent:
      continue
    wrapper = math.find_parent("mjx-container") or math.find_parent("span", class_="katex") or math
    display = (math.get("display") == "block" or wrapper.get("display") == "true"
               or bool(math.find_parent(class_=re.compile(r"katex-display|display-formula|disp-formula"))))
    mid = f"m{len(maths) + 1:03}"
    maths[mid] = {"mathml": str(math), "text": math.get_text("", strip=True), "display": display}
    if display:
      container = math.find_parent(class_=re.compile(r"display-formula|disp-formula|katex-display")) or wrapper
      label = container.find(class_=re.compile(r"\blabel\b"))
      maths[mid]["label"] = label.get_text("", strip=True).strip("()") if label else None
      placeholder = maker.new_tag("figure")
      placeholder["data-equation"] = mid
      container.replace_with(placeholder)
    else:
      wrapper.replace_with(NavigableString(f"⟦{mid}⟧"))
  return maths


def maths_html(mathml, size, lead=False):
  return f'<div class="item{" lead" if lead else ""}" style="font-size:{size}px">{mathml}</div>'


def render_maths(maths, blocks, workdir):
  """Render display equations and per-paragraph formula cards to JPEGs with
  MathJax in Chrome. Cards also repeat a display equation just before the
  paragraph, so the picture doesn't jump away from it mid-thought."""
  from playwright.sync_api import sync_playwright

  images_dir = workdir / "images"
  images_dir.mkdir(parents=True, exist_ok=True)
  jobs = []  # (element id, output path, inner html)
  previous = None
  for n, block in enumerate(blocks):
    if block["type"] == "equation":
      path = images_dir / f"eq-{block['id']}.jpg"
      block["file"] = str(path.relative_to(workdir))
      jobs.append((f"j{len(jobs)}", path, maths_html(maths[block["id"]]["mathml"], 60)))
    ids = list(dict.fromkeys(MATH_TOKEN.findall(block.get("text", "") + " ".join(block.get("items", [])))))
    if ids:
      seen, items = set(), []
      if previous and previous["type"] == "equation":
        items.append(maths_html(maths[previous["id"]]["mathml"], 46, lead=True))
      for mid in ids:
        text = maths[mid]["text"]
        if text not in seen:
          seen.add(text)
          items.append(maths_html(maths[mid]["mathml"], 46 if len(ids) <= 8 else 38))
      path = images_dir / f"card-{n:03}.jpg"
      block["maths_card"] = str(path.relative_to(workdir))
      jobs.append((f"j{len(jobs)}", path, f'<div class="card">{"".join(items)}</div>'))
    previous = block
  if not jobs:
    return 0

  page_html = """<!doctype html><html><head><meta charset="utf-8"><style>
    body { margin: 0; background: #fff; color: #111; }
    .job { display: inline-block; padding: 36px 48px; background: #fff; }
    .card { display: flex; flex-wrap: wrap; gap: 30px 52px; align-items: center; max-width: 860px; }
    .card .lead { flex-basis: 100%; margin-bottom: 14px; }
    </style>
    <script>window.MathJax = { svg: { fontCache: 'global' } };</script>
    <script src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/mml-svg.js"></script>
    </head><body>""" + "".join(f'<div><div class="job" id="{jid}">{inner}</div></div>' for jid, _, inner in jobs) + "</body></html>"
  html_path = workdir / "maths.html"
  html_path.write_text(page_html, encoding="utf-8")

  with sync_playwright() as p:
    try:
      browser = p.chromium.launch(channel="chrome")
    except Exception:
      try:
        browser = p.chromium.launch()
      except Exception:
        fail("rendering maths needs Google Chrome installed (or run `uvx playwright install chromium`)")
    page = browser.new_page(viewport={"width": 1700, "height": 1000}, device_scale_factor=2)
    page.goto(html_path.as_uri())
    try:
      page.wait_for_function("window.MathJax && MathJax.startup && MathJax.startup.promise", timeout=20000)
      page.evaluate("MathJax.startup.promise")
    except Exception:
      print("warning: MathJax didn't load (offline?); using Chrome's own MathML rendering", file=sys.stderr)
    for jid, path, _ in jobs:
      png = page.locator(f"#{jid}").screenshot()
      save_on_square(open_image(png), path)
    browser.close()
  html_path.unlink()
  return len(jobs)


def save_on_square(img, path, size=MAX_SIZE, margin=40):
  """Centre on a white square, the shape podcast apps show chapter art in.
  Renders are at 2x, so small equations can be enlarged up to 2x and stay sharp."""
  from PIL import Image
  room = size - 2 * margin
  scale = min(room / img.width, room / img.height, 2.0)
  img = img.resize((max(1, round(img.width * scale)), max(1, round(img.height * scale))), Image.LANCZOS)
  canvas = Image.new("RGB", (size, size), "white")
  canvas.paste(img, ((size - img.width) // 2, (size - img.height) // 2))
  canvas.save(path, "JPEG", quality=90)


# --- Body extraction -------------------------------------------------------

def find_body(soup, page):
  """The element holding the article itself, and which extractor found it."""
  from bs4 import BeautifulSoup

  body = soup.select_one("div.available-content div.body.markup") or soup.select_one("div.body.markup")
  if body:
    return body, "Substack"
  journal = soup.select_one("section#bodymatter")  # Atypon (PNAS, Science, ACS…)
  if journal:
    abstract = soup.select_one("section#abstract")
    if abstract:
      heading = abstract.find(["h2", "h3"])
      if heading:
        heading.string = "Abstract"
      journal.insert(0, abstract.extract())
    return journal, "Journal (Atypon)"
  from readability import Document
  summary = Document(page).summary(html_partial=True)
  return BeautifulSoup(summary, "lxml"), "Readability"


def clean_text(el):
  """Visible text of an element with inline markup flattened and spacing tidied."""
  for br in el.find_all("br"):
    br.replace_with("\n")
  text = el.get_text("")
  return re.sub(r"[ \t\r\f\v ]+", " ", re.sub(r"\s*\n\s*", " ", text)).strip()


def image_from(el, base_url):
  """{'url', 'caption', 'alt'} for a figure or image container, or None."""
  img = el if el.name == "img" else el.find("img")
  if not img:
    return None
  url = None
  link = el.find("a", class_="image-link")
  if link and link.get("href", "").startswith("http"):
    url = link["href"]  # Substack links each image to its full-size original
  if not url and img.get("data-attrs"):
    try:
      url = json.loads(img["data-attrs"]).get("src")
    except ValueError:
      pass
  if not url:
    srcset = img.get("srcset") or img.get("data-srcset")
    if srcset:
      url = srcset.split(",")[-1].strip().split(" ")[0]  # usually the largest
  url = url or img.get("src") or img.get("data-src")
  if not url or url.startswith("data:"):
    return None
  caption = el.find("figcaption")
  return {
    "url": urllib.parse.urljoin(base_url or "", url),
    "caption": clean_text(caption) if caption else None,
    "alt": (img.get("alt") or "").strip() or None,
  }


def gallery_from(el, base_url):
  """Substack image galleries carry their images as JSON in data-attrs."""
  try:
    data = json.loads(el.get("data-attrs") or "{}")
  except ValueError:
    return None
  gallery = data.get("gallery", data)
  images = [{"url": urllib.parse.urljoin(base_url or "", i["src"]), "caption": None,
             "alt": (i.get("alt") or "").strip() or None}
            for i in gallery.get("images", []) if i.get("src")]
  if not images:
    return None
  return {"type": "images", "caption": (gallery.get("caption") or "").strip() or None, "images": images}


def has_blocks(el):
  return any(getattr(d, "name", None) in BLOCK_TAGS for d in el.descendants)


def walk(el, blocks, base_url):
  """Append blocks for el's children, in document order. Runs of inline text
  between block elements (e.g. a paragraph split by a display equation)
  become paragraphs of their own."""
  run = []

  def flush():
    text = re.sub(r"[ \t\r\f\v\u00a0]+", " ", re.sub(r"\s*\n\s*", " ", "".join(run))).strip()
    run.clear()
    if re.search(r"\w", text):
      blocks.append({"type": "paragraph", "text": text})

  for child in el.children:
    name = getattr(child, "name", None)
    if name is None:
      run.append(str(child))
      continue
    if name not in BLOCK_TAGS and not has_blocks(child):
      for br in child.find_all("br"):
        br.replace_with("\n")
      run.append(child.get_text(""))
      continue
    flush()
    classes = child.get("class") or []
    if name in ("h1", "h2", "h3", "h4", "h5", "h6"):
      text = clean_text(child)
      if text:
        blocks.append({"type": "heading", "text": text, "level": int(name[1])})
    elif child.get("data-equation"):
      blocks.append({"type": "equation", "id": child["data-equation"]})
    elif "image-gallery-embed" in classes:
      gallery = gallery_from(child, base_url)
      if gallery:
        blocks.append(gallery)
    elif name == "figure" or "captioned-image-container" in classes or name == "img":
      image = image_from(child, base_url)
      if image:
        blocks.append({"type": "images", "caption": None, "images": [image]})
      elif name == "figure":
        walk(child, blocks, base_url)  # a figure of text: theorem, definition, pull quote
    elif (name in ("ul", "ol") or child.get("role") == "list") and child.find(attrs={"data-equation": True}):
      # Items with display equations in them can't be flattened to one line of text each.
      for li in child.find_all(lambda t: t.name == "li" or t.get("role") == "listitem", recursive=False):
        walk(li.find(class_="content") or li, blocks, base_url)
    elif child.get("role") == "list":  # ARIA lists, as journals write them
      items = [clean_text(li.find(class_="content") or li)
               for li in child.find_all(attrs={"role": "listitem"}, recursive=False)]
      items = [i for i in items if i]
      if items:
        blocks.append({"type": "list", "ordered": True, "items": items})
    elif name == "p" or child.get("role") == "paragraph":
      if child.find("img") and len(clean_text(child)) < 20:
        image = image_from(child, base_url)
        if image:
          blocks.append({"type": "images", "caption": None, "images": [image]})
        continue
      if has_blocks(child):
        walk(child, blocks, base_url)
        continue
      text = clean_text(child)
      if text:
        blocks.append({"type": "paragraph", "text": text})
    elif name == "blockquote":
      text = clean_text(child)
      if text:
        blocks.append({"type": "quote", "text": text})
    elif name in ("ul", "ol"):
      items = [clean_text(li) for li in child.find_all("li", recursive=False)]
      items = [i for i in items if i]
      if items:
        blocks.append({"type": "list", "ordered": name == "ol", "items": items})
    elif name == "hr":
      blocks.append({"type": "break"})
    elif name == "pre":
      blocks.append({"type": "note", "text": "There's a code example here; see the original article."})
    elif name == "table":
      blocks.append({"type": "note", "text": "There's a table here; see the original article."})
    elif has_blocks(child):
      walk(child, blocks, base_url)
    else:
      text = clean_text(child)
      if len(text) > 40:  # a bare <div> of prose
        blocks.append({"type": "paragraph", "text": text})
  flush()


def tidy_blocks(blocks, meta):
  """Drop headings that repeat the title, merge back-to-back image blocks
  into groups, and squash repeated breaks."""
  def norm(s):
    return re.sub(r"[^a-z0-9]+", " ", (s or "").lower()).strip()

  # Posts often repeat their title and subtitle as headings, sometimes after
  # a short note; drop those among the opening blocks, before real prose.
  title, subtitle = norm(meta["title"]), norm(meta["subtitle"])
  for i, block in enumerate(blocks[:6]):
    if block["type"] == "paragraph" and len(block["text"].split()) > 25:
      break
    text = norm(block.get("text"))
    if block["type"] == "heading" and text and (title.startswith(text) or text == subtitle):
      block["type"] = "drop"
  blocks = [b for b in blocks if b["type"] != "drop"]

  tidied = []
  for block in blocks:
    prev = tidied[-1] if tidied else None
    if block["type"] == "images" and prev and prev["type"] == "images":
      prev["images"] += block["images"]
      prev["caption"] = prev["caption"] or block["caption"]
    elif block["type"] == "break" and (not prev or prev["type"] == "break"):
      continue
    else:
      tidied.append(block)
  while tidied and tidied[-1]["type"] == "break":
    tidied.pop()
  return tidied


# --- Images ----------------------------------------------------------------

def open_image(data):
  from PIL import Image
  img = Image.open(io.BytesIO(data))
  img.seek(0)  # first frame of an animated GIF/WebP
  if img.mode in ("RGBA", "LA", "P"):
    img = img.convert("RGBA")
    background = Image.new("RGB", img.size, "white")
    background.paste(img, mask=img.split()[-1])
    return background
  return img.convert("RGB")


def save_fitted(img, path, size=MAX_SIZE):
  img = img.copy()
  img.thumbnail((size, size))
  img.save(path, "JPEG", quality=85)


def make_grid(paths, out, size=MAX_SIZE, gap=12):
  """Tile images into one square-ish grid, for galleries too big to step through."""
  from PIL import Image
  cols = math.ceil(math.sqrt(len(paths)))
  rows = math.ceil(len(paths) / cols)
  cell = (size - gap * (cols + 1)) // cols
  grid = Image.new("RGB", (size, gap + rows * (cell + gap)), "white")
  for i, p in enumerate(paths):
    img = Image.open(p)
    img.thumbnail((cell, cell))
    r, c = divmod(i, cols)
    x = gap + c * (cell + gap) + (cell - img.width) // 2
    y = gap + r * (cell + gap) + (cell - img.height) // 2
    grid.paste(img, (x, y))
  grid.save(out, "JPEG", quality=85)


def title_card(meta, out, size=1400):
  """Plain cover art for articles without a share image."""
  from PIL import Image, ImageDraw, ImageFont
  card = Image.new("RGB", (size, size), (24, 24, 27))
  draw = ImageDraw.Draw(card)
  big, small = ImageFont.load_default(size=96), ImageFont.load_default(size=48)
  words, lines, line = meta["title"].split(), [], ""
  for word in words:
    trial = f"{line} {word}".strip()
    if draw.textlength(trial, font=big) > size - 200 and line:
      lines.append(line)
      line = word
    else:
      line = trial
  lines.append(line)
  y = size // 2 - len(lines) * 60
  for line in lines:
    draw.text((100, y), line, font=big, fill="white")
    y += 120
  byline = " · ".join(v for v in (meta.get("author"), meta.get("site")) if v)
  if byline:
    draw.text((100, y + 40), byline, font=small, fill=(161, 161, 170))
  card.save(out, "JPEG", quality=90)


def download_images(blocks, workdir):
  """Fetch every image, save it fitted to 945px, and number it img01, img02…"""
  images_dir = workdir / "images"
  images_dir.mkdir(parents=True, exist_ok=True)
  n = 0
  for block in blocks:
    if block["type"] != "images":
      continue
    kept = []
    for image in block["images"]:
      n += 1
      image["id"] = f"img{n:02}"
      try:
        img = open_image(http_get(image["url"], binary=True))
      except Exception as err:
        print(f"warning: skipped {image['id']} ({err})", file=sys.stderr)
        continue
      path = images_dir / f"{image['id']}.jpg"
      save_fitted(img, path)
      image.update({"file": str(path.relative_to(workdir)), "width": img.width, "height": img.height,
                    "speak": None})
      kept.append(image)
    block["images"] = kept
    block["grid"] = None
    if len(kept) > GRID_OVER:
      grid = images_dir / f"{kept[0]['id']}-grid.jpg"
      make_grid([workdir / i["file"] for i in kept], grid)
      block["grid"] = str(grid.relative_to(workdir))
  return [b for b in blocks if b["type"] != "images" or b["images"]]


def meaningful_alt(alt):
  return bool(alt) and len(alt) >= 8 and not re.search(r"\.(jpe?g|png|webp|gif)$|^(image|img|photo)\b", alt, re.I)


# --- Main ------------------------------------------------------------------

def main():
  p = argparse.ArgumentParser(description="Extract an article for a glanceable podcast episode.")
  p.add_argument("source", help="article URL, or a saved .html file (e.g. for paywalled posts)")
  p.add_argument("--workdir", type=Path, help=f"base folder for output (default {CACHE})")
  args = p.parse_args()

  from bs4 import BeautifulSoup

  page, final_url = load_page(args.source)
  soup = BeautifulSoup(page, "lxml")
  meta = read_metadata(soup, final_url)
  local = Path(args.source).expanduser()
  # A saved page's images sit in the folder saved next to it, not on the web.
  base_url = local.resolve().parent.as_uri() + "/" if local.is_file() else (meta["link"] or final_url)

  body, extractor = find_body(soup, page)
  paywalled = bool(soup.select_one(".paywall, .paywall-content")) or "This post is for paid subscribers" in page
  maths = extract_maths(body)
  for selector in JUNK:
    for junk in body.select(selector):
      junk.decompose()

  blocks = []
  walk(body, blocks, base_url)
  blocks = tidy_blocks(blocks, meta)
  if not any(b["type"] == "paragraph" for b in blocks):
    fail("couldn't find the article text on that page. If it needs a login or runs on "
         "JavaScript, save the page as HTML from your browser and pass the file instead.")

  path_slug = urllib.parse.urlparse(meta["link"] or "").path.rstrip("/").split("/")[-1]
  workdir = (args.workdir or CACHE) / f"article-{slugify(path_slug or meta['title'], 'article')}"
  workdir.mkdir(parents=True, exist_ok=True)

  print("Downloading images…", file=sys.stderr)
  blocks = download_images(blocks, workdir)
  if maths:
    print(f"Rendering {len(maths)} formulas…", file=sys.stderr)
    render_maths(maths, blocks, workdir)

  cover = workdir / "cover.jpg"
  try:
    if not meta["cover_url"]:
      raise ValueError("no share image")
    save_fitted(open_image(http_get(meta["cover_url"], binary=True)), cover, size=1400)
  except Exception:
    title_card(meta, cover)

  article = {
    "kind": "article",
    **{k: v for k, v in meta.items() if k != "cover_url"},
    "cover": "cover.jpg",
    "notes": "",
    "pronounce": {},
    # How each distinct inline formula is spoken; null means "shown".
    "math_say": {m["text"]: (auto_name(m["text"]) if len(m["text"]) <= SHORT_MATH else None)
                 for m in maths.values() if not m["display"]},
    "maths": {mid: {k: v for k, v in m.items() if k != "mathml"} for mid, m in maths.items()},
    "blocks": blocks,
  }
  article_json = workdir / "article.json"
  article_json.write_text(json.dumps(article, indent=2, ensure_ascii=False) + "\n")

  words = sum(len(b.get("text", "").split()) + sum(len(i.split()) for i in b.get("items", []))
              for b in blocks)
  groups = [b for b in blocks if b["type"] == "images"]
  images = [i for g in groups for i in g["images"]]
  bunched = [g for g in groups if len(g["images"]) > 1]
  # Images shown only inside a grid are covered by the gallery caption.
  silent = [i for g in groups if not g["grid"] for i in g["images"]
            if not i.get("caption") and not meaningful_alt(i.get("alt"))]
  print(f"""
Article ready: {meta['title']}
  by:        {' · '.join(v for v in (meta['author'], meta['site'], meta['date']) if v) or 'unknown'}
  extractor: {extractor}
  text:      {words} words, about {round(words / 155)} min of speech, {sum(b['type'] == 'heading' for b in blocks)} headings
  images:    {len(images)} in {len(groups)} groups ({len(bunched)} with several images; {sum(1 for g in groups if g['grid'])} shown as a grid)
  uncaptioned (no caption or useful alt text): {', '.join(f"{i['id']} ({i['file']})" for i in silent) or 'none'}
  maths:     {sum(b['type'] == 'equation' for b in blocks)} display equations, {sum(1 for m in maths.values() if not m['display'])} inline formulas ({len(article['math_say'])} distinct; {sum(1 for v in article['math_say'].values() if v is None)} spoken as "shown"), {sum(1 for b in blocks if b.get('maths_card'))} formula cards
  metadata:  {article_json}""")
  if paywalled:
    print("\nwarning: this looks paywalled, so the text may be cut short. Save the full page while "
          "signed in and pass the .html file instead.")
  print(f"""
Next: review article.json (pronounce, math_say, image `speak` text, notes), then run
  article_episode.py {article_json}""")


if __name__ == "__main__":
  main()
