"""AI sentiment read on a single ticker — advisory only, feeds no score.

Pulls recent headlines from Yahoo (via yfinance), hands them to Claude together
with every strategy signal the screener already computed, and gets back a
sentiment score plus a reconciliation of news against numbers.

Deliberately standalone: nothing here is called from engine.py or data.py, so a
screen stays offline-fast. See SCORING_SEAM at the bottom before changing that.
"""
from __future__ import annotations

import json
import os
import re
from dataclasses import asdict

from .strategies import DESCRIPTIONS

MODEL = "claude-opus-5"
_TAGS = re.compile(r"<[^>]+>")

_LENSES = "\n".join(f"- {name}: {desc}" for name, desc in DESCRIPTIONS.items())

SYSTEM = f"""You are an equity-research assistant for an educational stock screener.

The screener scores every stock through four lenses:
{_LENSES}

Each lens is a weighted mean of individual signals. Read the payload carefully:
- `signals[].score` is 0..1. `strategies[].score` and `verdict` are 0..100.
- A signal `score` of null means the metric was MISSING from Yahoo's data and was
  EXCLUDED from the mean. It was NOT scored neutrally. Never treat null as 0.5,
  and never read a null as a bad result — it is an absent one.
- `signals[].note` explains what the metric is meant to capture.

`headlines` is Yahoo Finance's publisher feed for this ticker. It is a narrow
source, not a survey of the whole internet, and be aware of two things:
- Some items will be about OTHER companies (peers, suppliers, sector stories).
  Ignore those. Do not attribute them to this company.
- Ten headlines is not a sentiment survey. If coverage is thin, sparse, or stale,
  say so plainly and keep your score near neutral. Never invent sentiment or
  imply a consensus you cannot see. Absence of news is not bullish or bearish.

Your job is to reconcile what the press says against what the numbers say.

Return:
- `score`: 0-100 sentiment. 0 = maximally bearish, 50 = neutral, 100 = maximally
  bullish. Judge the mood of the coverage against the fundamentals, and stay near
  50 when evidence is weak or conflicting.
- `verdict`: two to four words, e.g. "Cautiously bearish".
- `bullets`: 3-6 items. Each must name a SPECIFIC metric from the payload by its
  exact signal name (e.g. ROE, ShortFloat%, MeanRev_Z, EV/EBITDA) and say whether
  the news agrees with it, contradicts it, or is silent on it. Cite the publisher
  when you lean on a headline. Where the news and the metric disagree, say so —
  that tension is the most useful thing you can surface.
- `suggestion`: one short paragraph on what a reader should watch, verify, or
  investigate next, and which single number would most change the picture.

Educational framing only. Never tell the reader to buy, sell, or hold, and never
predict a price. Be concise and concrete: no preamble, no restating the payload,
no hedging boilerplate."""

SCHEMA = {
    "type": "object",
    "properties": {
        "score": {"type": "integer", "description": "0-100 sentiment, 50 = neutral"},
        "verdict": {"type": "string", "description": "Two to four word summary"},
        "bullets": {"type": "array", "items": {"type": "string"}},
        "suggestion": {"type": "string"},
    },
    "required": ["score", "verdict", "bullets", "suggestion"],
    "additionalProperties": False,
}


def available() -> bool:
    """True when an API key is reachable.

    Streamlit promotes every top-level key in .streamlit/secrets.toml into
    os.environ at server start, so an env var and a secrets file both land here.
    """
    return bool(os.environ.get("ANTHROPIC_API_KEY"))


def headlines(ticker: str, limit: int = 10) -> list[dict]:
    """Recent Yahoo headlines for a ticker, flattened. Never raises.

    yfinance >= 1.x nests each item as {"id": ..., "content": {...}}; older 0.2.x
    releases returned flat dicts. Read defensively so a shape change degrades to
    an empty list rather than silently yielding blank strings.
    """
    try:
        import yfinance as yf

        items = yf.Ticker(ticker).news or []
    except Exception:
        return []

    out = []
    for item in items[:limit]:
        c = item.get("content") or item or {}
        title = (c.get("title") or "").strip()
        if not title:
            continue
        summary = (c.get("summary") or c.get("description") or "").strip()
        url = (c.get("canonicalUrl") or c.get("clickThroughUrl") or {}).get("url")
        out.append({
            "title": title,
            "summary": _TAGS.sub("", summary)[:600],
            "publisher": (c.get("provider") or {}).get("displayName"),
            "published": c.get("pubDate") or c.get("displayTime"),
            "kind": c.get("contentType"),
            "url": url,
        })
    return out


def payload(sd, results: dict, news: list[dict]) -> str:
    """Serialise ticker + every StrategyResult/Signal + headlines to JSON.

    Returns a str, so it is hashable and usable directly as a cache key.
    StrategyResult and Signal are plain dataclasses, so asdict() is the whole
    encoder. StockData itself is never passed through it — it holds a DataFrame.
    """
    return json.dumps(
        {
            "ticker": sd.ticker,
            "name": sd.name,
            "sector": sd.sector,
            "price": sd.price,
            "market_cap": sd.market_cap,
            "strategies": {name: asdict(res) for name, res in results.items()},
            "headlines": news,
        },
        default=str,
    )


def analyze(payload_json: str) -> dict:
    """Claude's sentiment read. Returns keys: score, verdict, bullets, suggestion.

    Raises on network/API failure — callers should surface that, not swallow it.
    """
    import anthropic  # local: a missing dep breaks this panel, not the whole app

    r = anthropic.Anthropic().messages.create(
        model=MODEL,
        max_tokens=16000,  # caps thinking + text together; thinking is on by default
        system=SYSTEM,
        thinking={"type": "adaptive"},
        output_config={
            "effort": "medium",
            "format": {"type": "json_schema", "schema": SCHEMA},
        },
        messages=[{"role": "user", "content": payload_json}],
    )
    if r.stop_reason == "refusal":  # check before touching content — it may be empty
        return {
            "score": 50,
            "verdict": "No read available",
            "bullets": ["Claude declined to analyse this request."],
            "suggestion": "Try a different ticker.",
        }
    text = next(b.text for b in r.content if b.type == "text")
    return json.loads(text)


# --- SCORING_SEAM ------------------------------------------------------------
# `analyze()["score"]` is already a 0-100 number, so folding sentiment into the
# numeric scoring later is one call: scoring.hump_better(score, 20, 65, 100)
# wrapped in a Signal (hump_better is currently unused and exists for exactly
# this inverted-U "narrative virality" shape).
#
# Do NOT register that as a fifth entry in STRATEGIES without deciding
# deliberately: strategies run inside screen_one() under ThreadPoolExecutor(8),
# so it would fire one network + LLM call per ticker — minutes of latency and
# real money per screen — and it would break smoke_test.py, which is offline.


if __name__ == "__main__":  # offline sanity check: python -m screener.sentiment
    from .strategies import STRATEGIES

    class _Fake:  # minimal StockData stand-in, no network
        ticker, name, sector, price, market_cap = "TEST", "Test Co", "Tech", 10.0, 1e9
        info: dict = {"returnOnEquity": 0.2, "trailingPE": 12.0}
        get = staticmethod(lambda k: _Fake.info.get(k))
        returns = staticmethod(lambda d: 0.1)

        def __init__(self):
            import pandas as pd
            self.history = pd.DataFrame()

    sd = _Fake()
    results = {n: f(sd) for n, f in STRATEGIES.items()}
    p = json.loads(payload(sd, results, []))
    assert set(p["strategies"]) == set(STRATEGIES), p["strategies"]
    assert any(s["name"] == "ROE" for s in p["strategies"]["Buffett"]["signals"])
    assert available.__doc__ and SCHEMA["additionalProperties"] is False
    print(f"payload OK — {len(json.dumps(p))} chars, key set {sorted(p)}")
    print(f"api key present: {available()}")
