"""
Transcribe Wayne's audio and video into text for the Trail Guide, using
open-source Whisper locally — free, and nothing leaves your Mac.

One-time setup:
    python3 -m pip install faster-whisper
    # only if you'll pull from the web (podcast feed or Vimeo/site links):
    python3 -m pip install yt-dlp

Modes:
    python3 transcribe.py --files media/                         # a folder of audio/video files
    python3 transcribe.py --feed https://www.thegodjourney.com/feed/podcast/
    python3 transcribe.py --feeds godjourney-feeds.txt           # many feeds (e.g. one per year)
    python3 transcribe.py --urls vimeo-links.txt                 # one media/Vimeo URL per line

Always start small: add --limit 2 to try a couple before committing to hundreds.
Transcripts land in content/transcripts/, one .txt per item, with the source URL
as the first line so the Trail Guide can cite and link back. It resumes safely —
anything already transcribed is skipped.

Model size (accuracy vs. speed) via WHISPER_MODEL: base | small (default) |
medium | large-v3. Bigger = more accurate but slower.
"""

import os
import re
import argparse
import tempfile
import xml.etree.ElementTree as ET

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
OUT_DIR = os.path.join(BASE_DIR, "content", "transcripts")
MODEL_NAME = os.environ.get("WHISPER_MODEL", "small")
AUDIO_EXTS = (".mp3", ".m4a", ".wav", ".aac", ".flac", ".ogg",
              ".mp4", ".m4v", ".mov", ".mkv", ".webm")

_model = None


def _get_model():
    global _model
    if _model is None:
        from faster_whisper import WhisperModel
        print(f"Loading Whisper model '{MODEL_NAME}' (first run downloads it)...")
        _model = WhisperModel(MODEL_NAME, device="cpu", compute_type="int8")
    return _model


def _slug(s):
    s = re.sub(r"^https?://", "", s.lower())
    s = re.sub(r"[^a-z0-9]+", "-", s).strip("-")
    return s[-120:] or "item"


def _done(source):
    return os.path.exists(os.path.join(OUT_DIR, _slug(source) + ".txt"))


def transcribe(path):
    model = _get_model()
    segments, _info = model.transcribe(path, vad_filter=True)
    return " ".join(seg.text.strip() for seg in segments).strip()


def save(source, text):
    if len(text) < 100:
        return "thin"
    with open(os.path.join(OUT_DIR, _slug(source) + ".txt"), "w", encoding="utf-8") as f:
        f.write(source + "\n\n" + text)
    return "saved"


def parse_feed(xml_text):
    """Return [{title, page, media}] from a podcast RSS feed."""
    root = ET.fromstring(xml_text)
    items = []
    for it in root.iter("item"):
        title = (it.findtext("title") or "").strip()
        link = (it.findtext("link") or "").strip()
        enc = it.find("enclosure")
        url = enc.get("url") if enc is not None else ""
        if url:
            items.append({"title": title, "page": link or url, "media": url})
    return items


def _get_text(url):
    import requests
    r = requests.get(url, timeout=60, headers={"User-Agent": "WayneTrailGuide/1.0"})
    r.raise_for_status()
    return r.text


def download(url):
    """Download media to a temp file and return its path. Direct audio links are
    fetched plainly; pages (Vimeo, etc.) go through yt-dlp for best audio."""
    import requests
    tmp = tempfile.mkdtemp()
    if url.lower().endswith(AUDIO_EXTS):
        local = os.path.join(tmp, _slug(url)[:60] + (os.path.splitext(url)[1] or ".mp3"))
        with requests.get(url, stream=True, timeout=180,
                          headers={"User-Agent": "WayneTrailGuide/1.0"}) as r:
            r.raise_for_status()
            with open(local, "wb") as f:
                for chunk in r.iter_content(1 << 16):
                    f.write(chunk)
        return local
    import yt_dlp
    opts = {"format": "bestaudio/best",
            "outtmpl": os.path.join(tmp, "%(id)s.%(ext)s"),
            "quiet": True, "noprogress": True}
    with yt_dlp.YoutubeDL(opts) as ydl:
        info = ydl.extract_info(url, download=True)
        return ydl.prepare_filename(info)


def run(items, limit):
    os.makedirs(OUT_DIR, exist_ok=True)
    if limit:
        items = items[:limit]
    counts = {"saved": 0, "skip": 0, "thin": 0, "fail": 0}
    for i, it in enumerate(items, 1):
        source = it["source"]
        if _done(source):
            counts["skip"] += 1
            continue
        try:
            path = it.get("path") or download(it["media"])
            counts[save(source, transcribe(path))] += 1
            print(f"  [{i}/{len(items)}] {source}")
        except Exception as e:
            counts["fail"] += 1
            print(f"  ! {source}: {e}")
    print(f"\nDone. {counts}")
    print("Next: python3 ingest.py")


def _read_list(path):
    with open(path) as f:
        return [ln.strip() for ln in f if ln.strip() and not ln.strip().startswith("#")]


def _from_feeds(feed_urls):
    items, seen = [], set()
    for fu in feed_urls:
        for it in parse_feed(_get_text(fu)):
            if it["page"] in seen:
                continue
            seen.add(it["page"])
            items.append({"source": it["page"], "media": it["media"]})
    return items


def gather(args):
    if args.files:
        root = args.files
        if not os.path.isdir(root):
            print(f"Folder not found: {root}   (your audio is in content/media/ — "
                  f"try: python3 transcribe.py --files content/media/)")
            return []
        items = []
        for dirpath, _dirs, names in os.walk(root):
            for n in sorted(names):
                if n.lower().endswith(AUDIO_EXTS):
                    full = os.path.join(dirpath, n)
                    source = os.path.splitext(os.path.relpath(full, root))[0]
                    items.append({"source": source, "path": full})
        items.sort(key=lambda d: d["source"])
        return items
    if args.feed:
        return _from_feeds([args.feed])
    if args.feeds:
        return _from_feeds(_read_list(args.feeds))
    if args.urls:
        return [{"source": u, "media": u} for u in _read_list(args.urls)]
    return None


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--files", help="folder of local audio/video files")
    ap.add_argument("--feed", help="a single podcast RSS feed URL")
    ap.add_argument("--feeds", help="text file of feed URLs, one per line (e.g. by year)")
    ap.add_argument("--urls", help="text file of media/Vimeo URLs (one per line)")
    ap.add_argument("--limit", type=int, default=0, help="0 = all; use a small number to test")
    args = ap.parse_args()

    items = gather(args)
    if items is None:
        print("Choose a mode: --files DIR | --feed URL | --urls FILE")
        return
    print(f"{len(items)} item(s) to consider. Transcribing with Whisper '{MODEL_NAME}'.")
    run(items, args.limit)


if __name__ == "__main__":
    main()
