#!/bin/zsh
# baton-yt — YouTube ingestion for the video-comprehension pipeline.
#
# Grabs content with yt-dlp and feeds the same decomposition every other
# footage source gets (see the video-comprehension skill): audio -> Apple
# on-device STT transcript, video -> ffmpeg contact sheet, on-screen text ->
# Apple OCR timeline. So any YouTube URL becomes readable text + one
# glanceable image — no video "watched", no cloud model needed.
#
# PATH-dispatched baton module: `baton yt ...` finds this via the git-style
# module contract; --describe feeds `baton help`.
#
#   baton yt audio <url>              audio only (m4a) -> cache dir
#   baton yt video <url>              720p mp4 -> cache dir
#   baton yt brief <url> [--sheets]   transcript: creator subtitle track if one
#                                     exists (ground truth), else Apple STT
#                                     (+ video -> contact sheet with --sheets)
#   baton yt ocr <url>                on-screen text timeline: frames every 2s
#                                     -> Apple OCR -> timecoded unique blocks
#                                     (the citation layer narration never reads)
#   baton yt stills <url> [--threshold 0.3]
#                                     scene-change stills -> <cache>/stills/ + one
#                                     contact sheet + a count of CUTS vs visually
#                                     DISTINCT illustrations (how many pictures the
#                                     edit actually used, not how many times it cut)
#   baton yt subs <url> [--lang xx]   YouTube's own subtitle track as timecoded
#                                     text (creator track preferred, auto-caption
#                                     fallback) — no media download, no STT
#   Cache: ~/.baton/var/yt/<title-slug>--<video-id>/ (re-runs reuse downloads + OCR)
set -e -u -o pipefail

describe="yt — YouTube ingestion: audio/video grab + transcript/contact-sheet/OCR-timeline/subs brief (video-comprehension pipeline)"
[[ "${1:-}" == "--describe" ]] && { print -r -- "$describe"; exit 0; }

verb="${1:-}"; shift 2>/dev/null || true
url="${1:-}"; shift 2>/dev/null || true
[[ -z "$verb" || "$verb" == "--help" || "$verb" == "-h" ]] && {
  sed -n '2,29p' "$0" | sed 's/^# \{0,1\}//'; exit 0; }
[[ -z "$url" ]] && { echo "baton-yt: $verb needs a URL" >&2; exit 2; }

# Title -> filesystem-safe slug. The video id still carries identity, so this
# only has to be readable: transliterate to ASCII, fold to lowercase, collapse
# everything else to single dashes, and cap the length so the slug plus the id
# stays well inside the 255-byte filename limit.
slugify() {
  local s
  s=$(print -r -- "$1" | iconv -c -t 'ascii//TRANSLIT' 2>/dev/null || true)
  [[ -n "$s" ]] || s="$1"
  s="${s:l}"
  s=$(print -r -- "$s" | sed -E 's/[^a-z0-9]+/-/g; s/^-+//; s/-+$//')
  s="${s[1,60]}"
  print -r -- "${s%-}"
}

# Every yt-dlp call goes through yt(): anonymous first and, only when YouTube
# refuses (429 rate limit, 403, the "not a bot" sign-in wall), again with the
# operator's real browser cookies; a 429 that survives the cookies gets one
# more try after a pause. Anonymous goes first because the cookie path
# decrypts the whole browser jar (a Keychain read) and ties the fetch to the
# operator's Google account, and most fetches never need it. Once one call
# needed cookies, the rest of the run uses them without re-earning the refusal.
#   BATON_YT_BROWSER  browser for --cookies-from-browser (default: brave)
#   BATON_YT_COOKIES  0 = never use browser cookies; always = from the first call
# Cookies stay in yt-dlp's memory: no --cookies file (yt-dlp writes the jar
# back to it), no --write-info-json / --dump-json (their http_headers carry a
# Cookie header), no -v, and the stderr kept for diagnosis is held in a shell
# variable, never written to the cache. --ignore-config keeps a user config
# file from adding any of those.
browser="${BATON_YT_BROWSER:-brave}"
use_cookies=0
[[ "${BATON_YT_COOKIES:-}" == always ]] && use_cookies=1
yt_err=""
yt() {
  local rc attempt
  local -a jar
  for attempt in 1 2 3; do
    jar=()
    (( use_cookies )) && jar=(--cookies-from-browser "$browser")
    { yt_err=$(yt-dlp --ignore-config --no-playlist "${jar[@]}" "$@" 2>&1 1>&3 3>&-) && rc=0 || rc=$?; } 3>&1
    (( rc == 0 )) && return 0
    refused || return $rc
    if (( ! use_cookies )) && [[ "${BATON_YT_COOKIES:-1}" != 0 ]]; then
      echo "baton-yt: $(yt_why) anonymously; retrying with $browser cookies" >&2
      use_cookies=1
    elif (( attempt < 3 )) && [[ "$yt_err" == *"HTTP Error 429"* ]]; then
      echo "baton-yt: $(yt_why); retrying in 20s" >&2
      sleep 20
    else
      return $rc
    fi
  done
  return $rc
}

# True when the last yt() failure was YouTube refusing service rather than the
# video being unavailable or a local fault.
refused() {
  [[ "$yt_err" == *"HTTP Error 429"* || "$yt_err" == *"HTTP Error 403"* ||
     "$yt_err" == *"not a bot"* || "$yt_err" == *"Sign in to confirm"* ]]
}

# The last yt() failure in words: refusals by kind, else yt-dlp's last ERROR line.
yt_why() {
  local -a e
  case "$yt_err" in
    *"HTTP Error 429"*) print -r -- "rate-limited by YouTube (HTTP 429)" ;;
    *"not a bot"*|*"Sign in to confirm"*) print -r -- "blocked by YouTube's bot wall (sign-in required)" ;;
    *"HTTP Error 403"*) print -r -- "blocked by YouTube (HTTP 403)" ;;
    *) e=("${(@f)yt_err}"); e=("${(@M)e:#ERROR:*}")
       print -r -- "${${e[-1]:-yt-dlp failed}#ERROR: }" ;;
  esac
}

# 'subs --lang xx' is read here because the metadata call reports whether that
# track exists; brief always uses en.
lang="en"
[[ "$verb" == subs && "${1:-}" == "--lang" && -n "${2:-}" ]] && lang="$2"

# One metadata call serves the folder name, the brief header, the canonical
# URL and which subtitle tracks exist. Separate --print lines rather than one
# delimited line: titles routinely contain the separator ("WWDC23: … | Apple")
# but never a newline. Track presence is read by template traversal, never by
# printing the caption maps (automatic_captions alone runs to ~10 MB). A
# playlist URL resolves to its first video, and every later call uses that
# video's own URL, so no download walks a playlist into one output file.
metaf=$(mktemp -t baton-yt)
trap 'rm -f "$metaf"' EXIT
yt --no-warnings --playlist-items 1 --print '%(id)s' --print '%(title)s' \
  --print '%(uploader)s' --print '%(duration_string)s' --print '%(webpage_url)s' \
  --print "%(subtitles.$lang.0.ext|)s" --print "%(automatic_captions.$lang.0.ext|)s" \
  "$url" > "$metaf" || { echo "baton-yt: cannot resolve $url: $(yt_why)" >&2; exit 1; }
meta_lines=("${(@f)$(<"$metaf")}")
vid="${meta_lines[1]:-}"
title="${meta_lines[2]:-}"
uploader="${meta_lines[3]:-}"
duration="${meta_lines[4]:-}"
[[ -n "${meta_lines[5]:-}" && "${meta_lines[5]}" != NA ]] && url="${meta_lines[5]}"
has_creator_subs="${meta_lines[6]:-}"
has_auto_subs="${meta_lines[7]:-}"
[[ -n "$vid" && "$vid" != NA ]] || { echo "baton-yt: cannot resolve $url (no video id)" >&2; exit 1; }

# A folder is named <title-slug>--<id>, but the ID alone is the identity: the
# slug is cosmetic and a title can be edited after upload. So resolve an
# existing cache by id first (renaming it up to the current title), and only
# fall back to a bare id when there is no title to slug.
root="$HOME/.baton/var/yt"
slug=$(slugify "$title")
existing=("$root"/*"--$vid"(N))
if [[ -n "$slug" ]]; then
  dir="$root/$slug--$vid"
  if [[ ! -d "$dir" ]]; then
    if (( $#existing )); then
      mv "${existing[1]}" "$dir"          # title changed since the last run
    elif [[ -d "$root/$vid" ]]; then
      mv "$root/$vid" "$dir"              # folder predates title naming
    fi
  fi
elif (( $#existing )); then
  dir="${existing[1]}"                    # keep the readable name when the lookup fails
else
  dir="$root/$vid"
fi
mkdir -p "$dir"

# The fetchers set REPLY rather than print, so they run in this shell and a
# switch to cookies carries over to later calls. Downloads land under their
# final name only once complete (yt-dlp writes .part and renames), so a
# non-empty file is a whole one.
fetch_audio() {
  REPLY="$dir/audio.m4a"
  [[ -s "$REPLY" ]] || yt -q --no-warnings -f "bestaudio[ext=m4a]/bestaudio" -o "$REPLY" "$url" \
    || { echo "baton-yt: audio download failed: $(yt_why)" >&2; exit 1; }
}

fetch_video() {
  REPLY="$dir/video.mp4"
  [[ -s "$REPLY" ]] || yt -q --no-warnings -f "bestvideo[height<=720][ext=mp4]+bestaudio[ext=m4a]/best[height<=720][ext=mp4]/best[ext=mp4]/best" --merge-output-format mp4 -o "$REPLY" "$url" \
    || { echo "baton-yt: video download failed: $(yt_why)" >&2; exit 1; }
}

# Ensure the <lang> subtitle track json3 is cached; sets sub_kind
# (creator|auto) and sub_path. Creator tracks are ground truth; auto-captions are
# STT-grade (name misreads, [ __ ] profanity censoring), so provenance is kept
# in the cached filename. 'creator' as the second argument skips auto-captions.
# On failure sets sub_why to the reason: the track does not exist, or YouTube
# refused the download (the track exists but could not be fetched).
sub_why="" sub_kind="" sub_path=""
fetch_subs() {
  local lang="$1" only="${2:-}"
  local jc="$dir/subs.$lang.creator.json3" ja="$dir/subs.$lang.auto.json3"
  local raw="$dir/subs.$lang.json3"
  [[ -s "$jc" ]] && { sub_kind=creator sub_path="$jc"; return 0; }
  [[ "$only" != creator && -s "$ja" ]] && { sub_kind=auto sub_path="$ja"; return 0; }
  sub_why="no '$lang' subtitle track (creator or auto-caption)"
  [[ "$only" == creator ]] && sub_why="no creator '$lang' subtitle track"
  if [[ -n "$has_creator_subs" && "$has_creator_subs" != NA ]]; then
    if yt -q --no-warnings --skip-download --write-subs \
        --sub-langs "$lang" --sub-format json3 -o "$dir/subs" "$url" && [[ -s "$raw" ]]; then
      mv "$raw" "$jc"; sub_kind=creator sub_path="$jc"; return 0
    fi
    sub_why="creator '$lang' track exists but download failed: $(yt_why)"
  fi
  [[ "$only" == creator ]] && return 1
  if [[ -n "$has_auto_subs" && "$has_auto_subs" != NA ]]; then
    if yt -q --no-warnings --skip-download --write-auto-subs \
        --sub-langs "$lang" --sub-format json3 -o "$dir/subs" "$url" && [[ -s "$raw" ]]; then
      mv "$raw" "$ja"; sub_kind=auto sub_path="$ja"; return 0
    fi
    sub_why="auto-caption '$lang' track exists but download failed: $(yt_why)"
  fi
  return 1
}

# json3 caption events -> "[MM:SS] line" cues on stdout.
subs_to_cues() {
  python3 - "$1" <<'EOF'
import sys, json

def ts(ms):
    s = ms // 1000
    return f"{s//60:02d}:{s%60:02d}"

for e in json.load(open(sys.argv[1]))['events']:
    text = ''.join(seg.get('utf8', '') for seg in e.get('segs', [])).strip()
    if text:   # window-setup and aAppend newline events carry no words
        print(f"[{ts(e['tStartMs'])}] {' '.join(text.split())}")
EOF
}

case "$verb" in
  audio) fetch_audio; print -r -- "$REPLY" ;;
  video) fetch_video; print -r -- "$REPLY" ;;
  brief)
    sheets=0
    [[ "${1:-}" == "--sheets" ]] && sheets=1
    # Title + uploader for the header, from the metadata already resolved above.
    meta=""
    [[ -n "$title" ]] && meta="$title | $uploader | $duration"
    t="$dir/transcript.txt"
    # Every cached artifact is written to a .tmp and renamed into place, so a
    # failed or interrupted run leaves nothing a re-run would mistake for done.
    if [[ ! -s "$t" ]]; then
      # A creator-uploaded track beats STT (it's what was actually said, with
      # timecodes, no audio download). Auto-captions do NOT displace STT —
      # they carry the same misread class plus censoring; 'subs' serves those.
      # Any subtitle failure falls through to STT; the reason is printed.
      if fetch_subs en creator; then
        { [[ -n "$meta" ]] && print -r -- "# $meta"
          print -r -- "# source: creator subtitle track"
          subs_to_cues "$sub_path"; } > "$t.tmp"
      else
        echo "baton-yt: $sub_why; transcribing audio with Apple STT" >&2
        fetch_audio; a="$REPLY"
        wav="$dir/audio-16k.wav"
        if [[ ! -s "$wav" ]]; then
          ffmpeg -v error -y -i "$a" -ac 1 -ar 16000 "$dir/audio-16k.tmp.wav"
          mv "$dir/audio-16k.tmp.wav" "$wav"
        fi
        { [[ -n "$meta" ]] && print -r -- "# $meta"
          print -r -- "# source: apple stt"
          baton apple stt "$wav"; } > "$t.tmp"
      fi
      mv "$t.tmp" "$t"
    fi
    if (( sheets )); then
      fetch_video; v="$REPLY"
      s="$dir/sheet.jpg"
      if [[ ! -s "$s" ]]; then
        d=$(ffprobe -v quiet -show_entries format=duration -of csv=p=0 "$v")
        ffmpeg -v error -y -i "$v" -vf "fps=30/$d,scale=320:-1,tile=6x5" -frames:v 1 "$dir/sheet.tmp.jpg"
        mv "$dir/sheet.tmp.jpg" "$s"
      fi
      print -r -- "sheet: $s"
    fi
    print -r -- "transcript: $t"
    ;;
  ocr)
    # Videos cite sources visually — X-post screenshots, slides, headlines —
    # that the narration never reads aloud, so the transcript alone misses a
    # whole citation layer. This extracts it: sample frames, OCR each on-device,
    # merge near-identical reads into timecoded spans.
    fetch_video; v="$REPLY"
    tl="$dir/ocr-timeline.md"
    if [[ ! -s "$tl" ]]; then
      frames="$dir/frames"
      mkdir -p "$frames"
      # 1 frame every 2 seconds; deterministic coverage beats scene detection
      # (slow overlays fade in without a cut, so scene filters miss them).
      # .extracted marks a complete pass; an interrupted one is redone.
      if [[ ! -e "$frames/.extracted" ]]; then
        ffmpeg -v error -y -i "$v" -vf fps=1/2 "$frames/%05d.png"
        : > "$frames/.extracted"
      fi
      # Per-frame OCR is resumable: empty .txt = prior failure, retried here.
      for f in "$frames"/*.png; do
        txt="${f%.png}.txt"
        [[ -s "$txt" ]] || baton apple ocr "$f" > "$txt" 2>/dev/null || : > "$txt"
      done
      python3 - "$frames" "$dir" <<'EOF'
import sys, os, re, difflib
frames_dir, out_dir = sys.argv[1], sys.argv[2]

def norm(t):
    return re.sub(r'\s+', ' ', t.lower()).strip()

entries = []  # (seconds, raw, normed)
for name in sorted(os.listdir(frames_dir)):
    if not name.endswith('.txt'): continue
    idx = int(name[:-4])
    secs = (idx - 1) * 2   # fps=1/2, frame 1 ~ t=0
    raw = open(os.path.join(frames_dir, name)).read().strip()
    n = norm(raw)
    if len(n) < 20:        # talking head / watermark noise
        continue
    entries.append((secs, raw, n))

# Merge consecutive near-identical texts into spans. Known limit: heavy OCR
# noise on identical screens can defeat the 0.85 gate — treat adjacent
# same-looking blocks as one when reading the output.
spans = []
for secs, raw, n in entries:
    if spans:
        prev = spans[-1]
        sim = difflib.SequenceMatcher(None, prev['norm'], n).ratio()
        if sim > 0.85:
            prev['end'] = secs
            if len(n) > len(prev['norm']):   # keep the fullest OCR read
                prev['norm'], prev['raw'] = n, raw
            continue
    spans.append({'start': secs, 'end': secs, 'raw': raw, 'norm': n})

# A span near-identical to an EARLIER one is a re-shown quote: track its
# timecodes on the original instead of duplicating the text.
uniq = []
for s in spans:
    dup_of = None
    for u in uniq:
        if difflib.SequenceMatcher(None, u['norm'], s['norm']).ratio() > 0.85:
            dup_of = u; break
    if dup_of:
        dup_of.setdefault('also', []).append((s['start'], s['end']))
    else:
        uniq.append(s)

def ts(sec):
    return f"{sec//60:02d}:{sec%60:02d}"

with open(os.path.join(out_dir, 'ocr-timeline.md'), 'w') as f:
    f.write(f"# On-screen text timeline\n\n{len(uniq)} unique text blocks from {len(entries)} text-bearing frames\n\n")
    for s in uniq:
        span = f"[{ts(s['start'])}–{ts(s['end'])}]" if s['end'] != s['start'] else f"[{ts(s['start'])}]"
        f.write(f"## {span}\n\n```\n{s['raw']}\n```\n")
        if 'also' in s:
            f.write("re-shown: " + ", ".join(f"{ts(a)}–{ts(b)}" for a,b in s['also']) + "\n")
        f.write("\n")
print(f"unique blocks: {len(uniq)} (from {len(entries)} text frames)", file=sys.stderr)
EOF
    fi
    print -r -- "ocr-timeline: $tl"
    ;;
  stills)
    # Scene-change keyframes. A cut every few seconds is normal for an
    # illustrated explainer, but most cuts are pans/zooms over the SAME
    # picture; the number a producer needs is distinct illustrations, so both
    # are printed. The distinct count clusters 32x18 grey thumbnails by mean
    # absolute difference (numpy) — crude, deterministic, and good to ±10%.
    thr=0.3
    for ((i=1; i<=$#; i++)); do [[ "${@[$i]}" == "--threshold" ]] && thr="${@[$((i+1))]}"; done
    fetch_video; v="$REPLY"
    out="$dir/stills"; mkdir -p "$out"
    if [[ ! -e "$out/.extracted" ]]; then
      ffmpeg -v error -y -i "$v" -vf "select='gt(scene,$thr)',scale=640:-1" -vsync vfr "$out/s%04d.jpg"
      : > "$out/.extracted"
    fi
    # cuts.tsv: when each cut lands and how long the shot before it ran — the
    # producer's number is the shot-length distribution (a brickfilm cutting
    # every 6.1 s is a channel generating 6 s clips), not the count alone.
    if [[ ! -s "$out/cuts.tsv" ]]; then
      ffmpeg -nostdin -loglevel info -i "$v" -vf "select='gt(scene,$thr)',showinfo" -an -f null - 2>&1 \
        | grep -o 'pts_time:[0-9.]*' | cut -d: -f2 \
        | python3 -c 'import sys;t=[0.0]+[float(x) for x in sys.stdin.read().split()];print("still\tcut_at\tshot_len");[print(f"s{i:04d}\t{b:.1f}\t{b-a:.1f}") for i,(a,b) in enumerate(zip(t,t[1:]),1)]' > "$out/cuts.tsv.tmp"
      mv "$out/cuts.tsv.tmp" "$out/cuts.tsv"
    fi
    shotstat=$(python3 -c 'import sys,statistics as s;d=[float(l.split("\t")[2]) for l in open(sys.argv[1]).read().splitlines()[1:]];print(f"median shot {s.median(d):.1f}s, {sum(1 for x in d if abs(x-s.median(d))<0.5)}/{len(d)} within ±0.5s of it") if d else print("")' "$out/cuts.tsv")
    n=$(ls "$out"/s*.jpg 2>/dev/null | wc -l | tr -d ' ')
    cols=8; rows=$(( (n+cols-1)/cols ))
    [[ -s "$dir/stills-sheet.jpg" ]] || ffmpeg -v error -y -pattern_type glob -i "$out/s*.jpg" -vf "scale=200:-1,tile=${cols}x${rows}" -frames:v 1 "$dir/stills-sheet.jpg"
    distinct=$(python3 - "$out" <<'PY' 2>/dev/null
import subprocess,glob,sys
try:
    import numpy as np
except ImportError:
    print("n/a (pip install numpy)"); sys.exit(0)
files=sorted(glob.glob(sys.argv[1]+"/s*.jpg")); reps=[]
def thumb(f):
    raw=subprocess.run(["ffmpeg","-nostdin","-loglevel","error","-i",f,"-vf","scale=32:18,format=gray","-f","rawvideo","-"],capture_output=True).stdout
    return np.frombuffer(raw,dtype=np.uint8).astype(float)
for f in files:
    t=thumb(f)
    if not reps or min(np.mean(np.abs(t-r))/255 for r in reps[-6:])>0.18: reps.append(t)
print(len(reps))
PY
)
    dur=$(ffprobe -v error -show_entries format=duration -of csv=p=0 "$v" | cut -d. -f1)
    echo "$dir/stills: $n scene cuts over ${dur}s (one every $(( dur / (n>0?n:1) ))s); ~$distinct visually distinct illustrations; $shotstat"
    echo "$out/cuts.tsv"
    echo "$dir/stills-sheet.jpg"
    ;;
  subs)
    out="$dir/subs.$lang.txt"
    if [[ ! -s "$out" ]]; then
      fetch_subs "$lang" || { echo "baton-yt: $sub_why ($url)" >&2; exit 1; }
      cues=$(subs_to_cues "$sub_path")
      n=$(print -r -- "$cues" | wc -l | tr -d ' ')
      { print -r -- "# subtitles ($sub_kind, $lang) — $n cues"
        print -r -- "$cues"; } > "$out.tmp"
      mv "$out.tmp" "$out"
      echo "$sub_kind track, $n cues" >&2
    fi
    print -r -- "subs: $out"
    ;;
  *) echo "baton-yt: unknown verb '$verb' (audio|video|brief|ocr|stills|subs)" >&2; exit 2 ;;
esac
