"""Convert policy documents (PDF / Word) into citable markdown for the RAG pipeline.

    python scripts/ingest_docs.py --input data/source_docs --config config/config.yaml

Reads every .pdf/.docx in --input and writes one markdown file per document into data.raw_dir, so
scripts/prepare_data.py can chunk + index it. We deliberately preserve:
  - headings  -> citation anchors (e.g. mypolicy#emergency-medical)
  - tables    -> markdown tables, because benefit/coverage LIMITS in travel insurance live in
                 tables and would otherwise be flattened into unreadable text.

PDFs: text + tables via pdfplumber, sectioned by page. Word: heading styles -> md headings,
paragraphs + tables preserved via python-docx.
"""
from __future__ import annotations

import argparse
import re
import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))

from agent.settings import load_config  # noqa: E402


def rows_to_markdown(rows: list[list]) -> str:
    """Render a table (list of rows of cells) as a GitHub markdown table."""
    clean = [[("" if c is None else str(c)).replace("\n", " ").strip() for c in row] for row in rows]
    clean = [r for r in clean if any(cell for cell in r)]
    if not clean:
        return ""
    width = max(len(r) for r in clean)
    clean = [r + [""] * (width - len(r)) for r in clean]
    header, *body = clean
    out = ["| " + " | ".join(header) + " |", "| " + " | ".join(["---"] * width) + " |"]
    out += ["| " + " | ".join(r) + " |" for r in body]
    return "\n".join(out)


def convert_pdf(path: Path) -> str:
    import pdfplumber

    parts = [f"# {path.stem}"]
    with pdfplumber.open(path) as pdf:
        for i, page in enumerate(pdf.pages, 1):
            parts.append(f"\n## Page {i}")
            text = (page.extract_text() or "").strip()
            if text:
                parts.append(text)
            for table in page.extract_tables() or []:
                md = rows_to_markdown(table)
                if md:
                    parts.append("\n" + md)
    return "\n\n".join(parts) + "\n"


def convert_docx(path: Path) -> str:
    from docx import Document
    from docx.document import Document as _Doc
    from docx.oxml.ns import qn
    from docx.table import Table
    from docx.text.paragraph import Paragraph

    doc = Document(str(path))
    parts = [f"# {path.stem}"]

    # Iterate body blocks in document order so tables land where they belong.
    parent = doc.element.body
    for child in parent.iterchildren():
        if child.tag == qn("w:p"):
            para = Paragraph(child, doc)
            text = para.text.strip()
            if not text:
                continue
            style = (para.style.name or "").lower()
            m = re.match(r"heading (\d)", style)
            if m:
                level = min(int(m.group(1)) + 1, 6)  # Heading 1 -> ##, under the file's # title
                parts.append(f"\n{'#' * level} {text}")
            else:
                parts.append(text)
        elif child.tag == qn("w:tbl"):
            table = Table(child, doc)
            rows = [[cell.text for cell in row.cells] for row in table.rows]
            md = rows_to_markdown(rows)
            if md:
                parts.append("\n" + md)
    return "\n\n".join(parts) + "\n"


def main() -> None:
    ap = argparse.ArgumentParser()
    ap.add_argument("--input", default="data/source_docs", help="folder with .pdf/.docx files")
    ap.add_argument("--config", default="config/config.yaml")
    args = ap.parse_args()
    cfg = load_config(args.config)

    from agent.settings import PROJECT_ROOT

    in_dir = Path(args.input)
    if not in_dir.is_absolute():
        in_dir = PROJECT_ROOT / in_dir
    out_dir = cfg.resolve("data.raw_dir")
    out_dir.mkdir(parents=True, exist_ok=True)

    if not in_dir.exists():
        raise SystemExit(f"Input folder not found: {in_dir}")

    converted = 0
    for path in sorted(in_dir.rglob("*")):
        ext = path.suffix.lower()
        try:
            if ext == ".pdf":
                md = convert_pdf(path)
            elif ext in (".docx",):
                md = convert_docx(path)
            elif ext == ".doc":
                print(f"  ! {path.name}: legacy .doc not supported — save as .docx first")
                continue
            else:
                continue
        except Exception as e:  # noqa: BLE001
            print(f"  ! failed on {path.name}: {e}")
            continue
        out_path = out_dir / f"{path.stem}.md"
        out_path.write_text(md, encoding="utf-8")
        converted += 1
        print(f"  {path.name} -> {out_path.relative_to(PROJECT_ROOT)}")

    print(f"\nConverted {converted} document(s) -> {out_dir}")
    if converted:
        print("Next: python scripts/prepare_data.py --config", args.config)
    else:
        print(f"No .pdf/.docx found in {in_dir}. Drop your policy documents there and re-run.")


if __name__ == "__main__":
    main()
