"""Extract the Training Data PDFs (Valuation, Dalio, Shiller) to plain text
chunks under corpus/books/ for use as grounding excerpts in teacher prompts.
"""
import re
from pathlib import Path

import fitz  # pymupdf

SOURCE_DIR = Path(r"C:\Users\cheed\OneDrive\Desktop\Training Data")
OUT = Path(__file__).resolve().parents[2] / "corpus" / "books"
CHUNK_CHARS = 6000  # ~1.5k tokens per chunk, paragraph-aligned


def extract(pdf: Path) -> str:
    with fitz.open(pdf) as doc:
        pages = [page.get_text() for page in doc]
    text = "\n".join(pages)
    text = re.sub(r"[ \t]+", " ", text)
    return re.sub(r"\n{3,}", "\n\n", text).strip()


def chunk(text: str) -> list[str]:
    chunks, buf = [], ""
    for para in text.split("\n\n"):
        if len(buf) + len(para) > CHUNK_CHARS and buf:
            chunks.append(buf.strip())
            buf = ""
        buf += para + "\n\n"
    if buf.strip():
        chunks.append(buf.strip())
    return chunks


def main() -> None:
    OUT.mkdir(parents=True, exist_ok=True)
    for pdf in sorted(SOURCE_DIR.glob("*.pdf")):
        stem = re.sub(r"\W+", "_", pdf.stem).strip("_").lower()
        text = extract(pdf)
        parts = chunk(text)
        for i, part in enumerate(parts):
            (OUT / f"{stem}_{i:04d}.txt").write_text(part, encoding="utf-8")
        print(f"{pdf.name}: {len(text):,} chars -> {len(parts)} chunks")
    print(f"\nDone. Chunks in {OUT}")


if __name__ == "__main__":
    main()
