"""Point-in-time fundamentals from SEC EDGAR.

For each (ticker, as_of_date) pair, pulls XBRL company facts and keeps ONLY
values from filings with filed <= as_of_date. This is what prevents lookahead
bias: the snapshot contains exactly what an investor could have read that day.

Usage:
    set SEC_EMAIL=you@example.com
    python src/data/edgar.py --tickers datasets/tickers.csv --dates datasets/asof_dates.txt

tickers.csv: one ticker per line.  asof_dates.txt: one YYYY-MM-DD per line.
Output: datasets/fundamentals/{TICKER}_{DATE}.json
"""
import argparse
import json
import os
import time
from datetime import date
from pathlib import Path

import requests

ROOT = Path(__file__).resolve().parents[2]
OUT = ROOT / "datasets" / "fundamentals"

SEC_EMAIL = os.environ.get("SEC_EMAIL", "")
HEADERS = {"User-Agent": f"DeepFelineValue research {SEC_EMAIL}"}

# Core concepts for a value-investing snapshot (us-gaap taxonomy).
CONCEPTS = {
    "Revenues": "revenue",
    "RevenueFromContractWithCustomerExcludingAssessedTax": "revenue",
    "SalesRevenueNet": "revenue",  # pre-2018 tag (e.g. AAPL)
    "NetIncomeLoss": "net_income",
    "OperatingIncomeLoss": "operating_income",
    "NetCashProvidedByUsedInOperatingActivities": "operating_cash_flow",
    "NetCashProvidedByUsedInOperatingActivitiesContinuingOperations": "operating_cash_flow",
    "PaymentsToAcquirePropertyPlantAndEquipment": "capex",
    "Assets": "total_assets",
    "Liabilities": "total_liabilities",
    "StockholdersEquity": "equity",
    "LongTermDebtNoncurrent": "long_term_debt",
    "LongTermDebt": "long_term_debt",
    "CashAndCashEquivalentsAtCarryingValue": "cash",
    "CommonStockSharesOutstanding": "shares_outstanding",
    "EntityCommonStockSharesOutstanding": "shares_outstanding",
}


def load_ticker_map() -> dict[str, str]:
    r = requests.get("https://www.sec.gov/files/company_tickers.json",
                     headers=HEADERS, timeout=30)
    r.raise_for_status()
    return {v["ticker"].upper(): f"{v['cik_str']:010d}" for v in r.json().values()}


def company_facts(cik: str) -> dict:
    r = requests.get(f"https://data.sec.gov/api/xbrl/companyfacts/CIK{cik}.json",
                     headers=HEADERS, timeout=60)
    r.raise_for_status()
    return r.json()


def point_in_time_snapshot(facts: dict, as_of: str) -> dict:
    """Latest value per concept among facts FILED on or before as_of."""
    snapshot: dict[str, dict] = {}
    gaap = facts.get("facts", {}).get("us-gaap", {})
    dei = facts.get("facts", {}).get("dei", {})
    for concept, label in CONCEPTS.items():
        node = gaap.get(concept) or dei.get(concept)
        if not node:
            continue
        best = None
        for unit_vals in node.get("units", {}).values():
            for item in unit_vals:
                if item.get("filed", "9999") > as_of:
                    continue  # not yet public on the as-of date
                # Annual figures preferred; fall back to quarterly.
                key = (item.get("end", ""), item.get("form") == "10-K")
                if best is None or key > (best.get("end", ""), best.get("form") == "10-K"):
                    best = item
        if best and (label not in snapshot or best["end"] > snapshot[label]["period_end"]):
            snapshot[label] = {
                "value": best["val"],
                # start+end reveal the duration (quarter vs full year) for flow items
                "period_start": best.get("start"),
                "period_end": best["end"],
                "filed": best["filed"],
                "form": best.get("form"),
            }
    return snapshot


def main() -> None:
    ap = argparse.ArgumentParser()
    ap.add_argument("--tickers", default=str(ROOT / "datasets" / "tickers.csv"))
    ap.add_argument("--dates", default=str(ROOT / "datasets" / "asof_dates.txt"))
    args = ap.parse_args()

    if not SEC_EMAIL:
        raise SystemExit("Set SEC_EMAIL to your email first (SEC requires it in the User-Agent).")

    tickers = [t.strip().upper() for t in
               Path(args.tickers).read_text(encoding="utf-8-sig").splitlines() if t.strip()]
    dates = [d.strip() for d in
             Path(args.dates).read_text(encoding="utf-8-sig").splitlines() if d.strip()]
    for d in dates:
        date.fromisoformat(d)  # validate early

    OUT.mkdir(parents=True, exist_ok=True)
    cik_map = load_ticker_map()

    for ticker in tickers:
        cik = cik_map.get(ticker)
        if not cik:
            print(f"{ticker}: no CIK found, skipping")
            continue
        try:
            facts = company_facts(cik)
        except Exception as e:  # noqa: BLE001
            print(f"{ticker}: EDGAR fetch failed ({e})")
            continue
        for as_of in dates:
            out_path = OUT / f"{ticker}_{as_of}.json"
            if out_path.exists():
                continue
            snap = point_in_time_snapshot(facts, as_of)
            if not snap:
                continue
            out_path.write_text(json.dumps(
                {"ticker": ticker, "as_of": as_of, "fundamentals": snap}, indent=2))
        print(f"{ticker}: done")
        time.sleep(0.15)  # SEC fair-access limit is 10 req/s; stay well under


if __name__ == "__main__":
    main()
