"""Assemble teacher prompts: persona + point-in-time fundamentals + price context
+ a few grounding excerpts from the corpus.

Output: datasets/teacher_prompts.jsonl, one record per (ticker, as_of):
    {"ticker", "as_of", "system", "user"}

Price context comes from yfinance and is truncated at the as-of date (again:
no lookahead). Corpus excerpts are sampled from corpus/books + corpus/buffett
so the teacher writes analyses grounded in the source thinking.
"""
import json
import random
from pathlib import Path

import pandas as pd
import yfinance as yf

ROOT = Path(__file__).resolve().parents[2]
FUND = ROOT / "datasets" / "fundamentals"
OUT = ROOT / "datasets" / "teacher_prompts.jsonl"
PERSONA = (ROOT / "persona" / "system_prompt.txt").read_text(encoding="utf-8")

EXCERPTS_PER_PROMPT = 3
random.seed(7)


_history_cache: dict[str, pd.Series] = {}


def price_context(ticker: str, as_of: str) -> str:
    if ticker not in _history_cache:
        hist = yf.Ticker(ticker).history(start="2000-01-01", auto_adjust=True)
        _history_cache[ticker] = hist["Close"] if not hist.empty else pd.Series(dtype=float)
    close = _history_cache[ticker].loc[:as_of]  # truncate at as-of date: no lookahead
    if close.empty:
        return "No price history available."
    lines = [f"Price on {as_of} (last close): ${close.iloc[-1]:.2f}"]
    for label, days in [("1y", 252), ("3y", 756), ("5y", 1260)]:
        if len(close) > days:
            ret = close.iloc[-1] / close.iloc[-days] - 1
            lines.append(f"{label} return to date: {ret:+.1%}")
    lines.append(f"52w high/low: ${close.iloc[-252:].max():.2f} / ${close.iloc[-252:].min():.2f}"
                 if len(close) >= 252 else "")
    return "\n".join(l for l in lines if l)


def corpus_excerpts() -> list[str]:
    pool = list((ROOT / "corpus" / "books").glob("*.txt"))
    pool += list((ROOT / "corpus" / "buffett").glob("*.txt"))
    picks = random.sample(pool, min(EXCERPTS_PER_PROMPT, len(pool)))
    return [p.read_text(encoding="utf-8", errors="ignore")[:3000] for p in picks]


def main() -> None:
    records = []
    for f in sorted(FUND.glob("*.json")):
        snap = json.loads(f.read_text())
        ticker, as_of = snap["ticker"], snap["as_of"]
        fundamentals = json.dumps(snap["fundamentals"], indent=2)
        excerpts = "\n\n---\n\n".join(corpus_excerpts())
        user = f"""Today's date is {as_of}. You know NOTHING about events after this date.

Analyze {ticker} using only the data below and your investment process.

## Point-in-time fundamentals (SEC filings available as of {as_of})
{fundamentals}

## Price context (as of {as_of})
{price_context(ticker, as_of)}

## Reference reading (excerpts from your library)
{excerpts}

Deliver the full analysis in your standard structure. End with a machine-readable
verdict block exactly like:
VERDICT: {{"direction": "long|short|pass", "conviction": 1-10, "horizon_months": N}}

Calibrate conviction honestly ACROSS all the companies you analyze: a typical
decent idea is a 4-6, a genuinely strong setup is 7-8, and 9-10 is reserved for
the once-a-decade fat pitch. Weak or marginal ideas are 1-3 or a pass. If every
analysis you write scores 9-10, your scores carry no information."""
        records.append({"ticker": ticker, "as_of": as_of,
                        "system": PERSONA, "user": user})
        print(f"{ticker} {as_of}")

    with OUT.open("w", encoding="utf-8") as fh:
        for r in records:
            fh.write(json.dumps(r) + "\n")
    print(f"\n{len(records)} prompts -> {OUT}")


if __name__ == "__main__":
    main()
