#!/usr/bin/env python3
"""One-time build of knowledge.json for the Talk tab's grounded answers.

    python3 build-knowledge.py

Needs pdftotext (poppler-utils) and a local Ollama with nomic-embed-text pulled.
Sources are U.S. government works (public domain). Chunks are ~700 chars, tagged
with source and page; embeddings are the first 256 dims of nomic-embed-text v1.5
(Matryoshka: truncation keeps the ranking, cosine renormalises), stored as ints
scaled by 1000 to keep the JSON small. The page compares the query embedding
against these with plain cosine, so the scale never matters.
"""
import json, os, re, subprocess, tempfile, requests

SOURCES = {
  'VA Brief CBT Guide (Cully & Teten 2008)': 'https://www.mirecc.va.gov/visn16/docs/therapists_guide_to_brief_cbtmanual.pdf',
  'VA CBT for Depression Therapist Manual (Wenzel, Brown & Karlin 2011)': 'https://www.mirecc.va.gov/docs/cbt-d_manual_depression.pdf',
}
OLLAMA = os.environ.get('OLLAMA_HOST', 'http://127.0.0.1:11434')
DIMS, CHUNK = 256, 700
OUT = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'knowledge.json')

chunks = []
for src, url in SOURCES.items():
    pdf = os.path.join(tempfile.gettempdir(), re.sub(r'\W+', '_', src) + '.pdf')
    if not os.path.exists(pdf):
        print('downloading', url, flush=True)
        open(pdf, 'wb').write(requests.get(url, timeout=300).content)
    text = subprocess.run(['pdftotext', '-layout', pdf, '-'], capture_output=True, text=True, check=True).stdout
    for page_no, page in enumerate(text.split('\f'), 1):
        buf = ''
        for p in re.split(r'\n\s*\n', page):
            p = re.sub(r'\s+', ' ', p).strip()
            if not p:
                continue
            if buf and len(buf) + len(p) > CHUNK:
                chunks.append({'s': '%s, p.%d' % (src, page_no), 't': buf}); buf = ''
            buf = (buf + ' ' + p).strip()
        if len(buf) > 200:
            chunks.append({'s': '%s, p.%d' % (src, page_no), 't': buf})

# Drop tables of contents, tables, reference lists, and anything carrying a
# copyright notice (the CBT-D manual embeds one copyrighted instrument).
chunks = [c for c in chunks
          if len(re.findall(r'\b[a-z]{3,}\b', c['t'])) > 25
          and 'copyright' not in c['t'].lower() and '©' not in c['t']]
print(len(chunks), 'chunks to embed', flush=True)

for i in range(0, len(chunks), 32):
    batch = chunks[i:i+32]
    r = requests.post(OLLAMA + '/api/embed', timeout=900,
                      json={'model': 'nomic-embed-text', 'input': ['search_document: ' + c['t'] for c in batch]})
    r.raise_for_status()
    for c, e in zip(batch, r.json()['embeddings']):
        c['e'] = [int(round(x * 1000)) for x in e[:DIMS]]
    print('embedded', min(i + 32, len(chunks)), '/', len(chunks), flush=True)

with open(OUT, 'w', encoding='utf-8') as f:
    json.dump({'model': 'nomic-embed-text', 'dims': DIMS, 'chunks': chunks}, f, separators=(',', ':'), ensure_ascii=False)
print(OUT, os.path.getsize(OUT) // 1024, 'KB')
