"""Phase 7 — honest evaluation of the distilled student on a held-out TIME window.

Point the student (running locally) at (ticker, as_of) pairs dated AFTER the
training cutoff, parse its verdicts, and compare a simple equal-weight long
portfolio of its 'long' calls against SPY over each horizon.

Works with any OpenAI-compatible server:
    llama-server ... (speculative decoding setup — see README)
        python src/eval/backtest_verdicts.py --cutoff 2023-01-01 --base-url http://localhost:8080/v1
    ollama serve (with the deepfeline model created)
        python src/eval/backtest_verdicts.py --cutoff 2023-01-01

This reuses the same prompt builder as training, so results are apples-to-apples.
"""
import argparse
import json
import re
from pathlib import Path

import requests

from src.eval.score_and_filter import forward_return  # same return math

ROOT = Path(__file__).resolve().parents[2]
PROMPTS = ROOT / "datasets" / "teacher_prompts_holdout.jsonl"  # build with Phase 2+3 on post-cutoff dates
VERDICT_RE = re.compile(r'VERDICT:\s*(\{.*?\})', re.DOTALL)


def ask_student(base_url: str, system: str, user: str) -> str:
    r = requests.post(f"{base_url.rstrip('/')}/chat/completions", json={
        "model": "deepfeline",
        "messages": [{"role": "system", "content": system},
                     {"role": "user", "content": user}],
        "temperature": 0.3,
    }, timeout=600)
    r.raise_for_status()
    return r.json()["choices"][0]["message"]["content"]


def main() -> None:
    ap = argparse.ArgumentParser()
    ap.add_argument("--cutoff", required=True, help="training cutoff date; only prompts after this are scored")
    ap.add_argument("--base-url", default="http://localhost:11434/v1",
                    help="OpenAI-compatible endpoint (Ollama default; llama-server: http://localhost:8080/v1)")
    args = ap.parse_args()

    longs, results = [], []
    for line in PROMPTS.read_text(encoding="utf-8").splitlines():
        rec = json.loads(line)
        if rec["as_of"] <= args.cutoff:
            continue
        reply = ask_student(args.base_url, rec["system"], rec["user"])
        m = VERDICT_RE.search(reply)
        if not m:
            continue
        try:
            v = json.loads(m.group(1))
        except json.JSONDecodeError:
            continue
        results.append({**rec, "verdict": v})
        if v.get("direction") == "long":
            horizon = int(v.get("horizon_months", 12))
            stock = forward_return(rec["ticker"], rec["as_of"], horizon)
            spy = forward_return("SPY", rec["as_of"], horizon)
            if stock is not None and spy is not None:
                longs.append((rec["ticker"], rec["as_of"], stock, spy))
                print(f"{rec['ticker']} {rec['as_of']}: stock {stock:+.1%} vs SPY {spy:+.1%}")

    if longs:
        avg_stock = sum(l[2] for l in longs) / len(longs)
        avg_spy = sum(l[3] for l in longs) / len(longs)
        wins = sum(1 for l in longs if l[2] > l[3])
        print(f"\n{len(longs)} long calls | avg {avg_stock:+.1%} vs SPY {avg_spy:+.1%} "
              f"| beat SPY {wins}/{len(longs)}")
    else:
        print("No scoreable long verdicts.")

    (ROOT / "datasets" / "eval_results.json").write_text(json.dumps(results, indent=2))


if __name__ == "__main__":
    main()
