"""Fixed data prep + evaluation for the autoresearch loop.

FROZEN — the research agent never edits this file (see program.md).
Run once to build the price cache:  .venv/bin/python autoresearch/prepare.py
"""
from __future__ import annotations

import os
import sys

import numpy as np
import pandas as pd

HERE = os.path.dirname(os.path.abspath(__file__))
ROOT = os.path.dirname(HERE)
sys.path.insert(0, ROOT)

CLOSE_CSV = os.path.join(HERE, "close.csv")
VOLUME_CSV = os.path.join(HERE, "volume.csv")


def fetch() -> None:
    import yfinance as yf

    from screener.universe import get_sp500

    tickers = get_sp500()
    df = yf.download(tickers, period="5y", auto_adjust=True)
    close = df["Close"].dropna(axis=1, thresh=int(len(df) * 0.9))
    volume = df["Volume"][close.columns]
    close.ffill().to_csv(CLOSE_CSV)
    volume.to_csv(VOLUME_CSV)
    print(f"close panel: {close.shape[0]} rows x {close.shape[1]} tickers, "
          f"{close.index[0].date()}..{close.index[-1].date()}")
    if close.shape[1] < 100:
        print("WARNING: small universe — offline fallback or throttled fetch?")


def load() -> tuple[pd.DataFrame, pd.DataFrame]:
    if not os.path.exists(CLOSE_CSV):
        sys.exit("price cache missing — run: .venv/bin/python autoresearch/prepare.py")
    kw = dict(index_col=0, parse_dates=True)
    return pd.read_csv(CLOSE_CSV, **kw), pd.read_csv(VOLUME_CSV, **kw)


def sharpe(monthly_rets: pd.Series) -> float:
    return float(np.sqrt(12) * monthly_rets.mean() / monthly_rets.std())


def evaluate(scores_fn, top_n: int = 20, holdout_months: int = 12) -> float:
    """Monthly-rebalanced equal-weight top-N backtest. Returns -Sharpe(in-sample).

    scores_fn(close, volume) receives panels truncated to the rebalance date
    (the last row IS the as-of date), so lookahead is structurally impossible.
    """
    close, volume = load()
    idx = close.index
    month_ends = [d.iloc[-1] for _, d in idx.to_series().groupby(idx.to_period("M"))]

    rets = []
    for d, d_next in zip(month_ends, month_ends[1:]):
        if idx.get_loc(d) < 260:  # warm-up for 252d lookbacks
            continue
        picks = scores_fn(close.loc[:d], volume.loc[:d]).dropna().nlargest(top_n).index
        rets.append(float((close.loc[d_next, picks] / close.loc[d, picks] - 1).mean()))
    rets = pd.Series(rets)

    ins, hold = rets.iloc[:-holdout_months], rets.iloc[-holdout_months:]
    print(f"{len(rets)} monthly returns: {len(ins)} in-sample, {len(hold)} holdout")
    print(f"in-sample Sharpe: {sharpe(ins):.4f}")
    print(f"holdout Sharpe:   {sharpe(hold):.4f}  (reported for honesty — never optimize against it)")
    return -sharpe(ins)


if __name__ == "__main__":
    fetch()
