"""Download all Berkshire Hathaway shareholder letters (1977-present).

Older letters are plain HTML pages; 1998+ are PDFs. Both are saved to
corpus/buffett/ and converted to plain text alongside.
"""
import re
import sys
from pathlib import Path
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup

INDEX = "https://www.berkshirehathaway.com/letters/letters.html"
OUT = Path(__file__).resolve().parents[2] / "corpus" / "buffett"
HEADERS = {"User-Agent": "Mozilla/5.0 (research corpus builder)"}


def html_to_text(html: str) -> str:
    soup = BeautifulSoup(html, "html.parser")
    for tag in soup(["script", "style"]):
        tag.decompose()
    text = soup.get_text("\n")
    return re.sub(r"\n{3,}", "\n\n", text).strip()


def pdf_to_text(pdf_path: Path) -> str:
    import fitz  # pymupdf

    with fitz.open(pdf_path) as doc:
        return "\n".join(page.get_text() for page in doc)


def main() -> None:
    OUT.mkdir(parents=True, exist_ok=True)
    resp = requests.get(INDEX, headers=HEADERS, timeout=30)
    resp.raise_for_status()
    soup = BeautifulSoup(resp.text, "html.parser")

    links = {}
    for a in soup.find_all("a", href=True):
        m = re.search(r"(19[7-9]\d|20\d\d)", a["href"])
        if m and ("letters/" in a["href"] or a["href"].endswith((".html", ".pdf"))):
            links.setdefault(m.group(1), urljoin(INDEX, a["href"]))

    if not links:
        sys.exit("No letter links found — the index page layout may have changed.")

    for year, url in sorted(links.items()):
        txt_path = OUT / f"buffett_letter_{year}.txt"
        if txt_path.exists() and len(txt_path.read_text(encoding="utf-8")) > 2000:
            continue
        try:
            r = requests.get(url, headers=HEADERS, timeout=60)
            r.raise_for_status()
            if not url.lower().endswith(".pdf"):
                # Some years are chooser pages that link to the real PDF/HTML letter.
                page = BeautifulSoup(r.text, "html.parser")
                sub = next((a["href"] for a in page.find_all("a", href=True)
                            if a["href"].lower().endswith(".pdf")), None)
                if sub and len(html_to_text(r.text)) < 2000:
                    url = urljoin(url, sub)
                    r = requests.get(url, headers=HEADERS, timeout=60)
                    r.raise_for_status()
            if url.lower().endswith(".pdf"):
                raw = OUT / f"buffett_letter_{year}.pdf"
                raw.write_bytes(r.content)
                text = pdf_to_text(raw)
            else:
                text = html_to_text(r.text)
            txt_path.write_text(text, encoding="utf-8")
            print(f"{year}: {len(text):,} chars")
        except Exception as e:  # noqa: BLE001 — keep going, report at end
            print(f"{year}: FAILED ({e})")

    print(f"\nDone. Letters in {OUT}")


if __name__ == "__main__":
    main()
