"""COLAB/CLOUD SCRIPT — distill onto Gemma 4 E4B (sequence-level distillation:
SFT the student on the filtered teacher outputs).

Run on Google Colab (free T4 works — Unsloth's Gemma 4 E4B notebook is the
reference: https://unsloth.ai/docs/models/gemma-4/train):
    pip install unsloth
    # upload: datasets/student_sft.jsonl (from Phase 5)

After training, exports GGUF Q4_K_M for local Ollama on the GTX 1070.
"""
from datasets import load_dataset
from trl import SFTConfig, SFTTrainer
from unsloth import FastModel

MAX_SEQ = 8192

model, tokenizer = FastModel.from_pretrained(
    "unsloth/gemma-4-E4B-it",  # fall back to "google/gemma-4-E4B-it" if unavailable
    max_seq_length=MAX_SEQ,
    load_in_4bit=True,
)
model = FastModel.get_peft_model(
    model,
    r=16,
    lora_alpha=16,
    lora_dropout=0,
    target_modules=["q_proj", "k_proj", "v_proj", "o_proj",
                    "gate_proj", "up_proj", "down_proj"],
)

ds = load_dataset("json", data_files="datasets/student_sft.jsonl", split="train")


def to_text(ex):
    msgs = [
        {"role": "system", "content": ex["system"]},
        {"role": "user", "content": ex["user"]},
        {"role": "assistant", "content": ex["assistant"]},
    ]
    return {"text": tokenizer.apply_chat_template(msgs, tokenize=False)}


ds = ds.map(to_text)

trainer = SFTTrainer(
    model=model,
    tokenizer=tokenizer,
    train_dataset=ds,
    args=SFTConfig(
        dataset_text_field="text",
        max_seq_length=MAX_SEQ,
        per_device_train_batch_size=2,
        gradient_accumulation_steps=4,
        num_train_epochs=3,
        learning_rate=2e-4,
        lr_scheduler_type="cosine",
        warmup_ratio=0.03,
        logging_steps=5,
        output_dir="student_ckpt",
        save_strategy="epoch",
        bf16=True,
    ),
)
trainer.train()

# GGUF export for local inference on the GTX 1070 via Ollama.
model.save_pretrained_gguf("deepfeline_e4b_gguf", tokenizer,
                           quantization_method="q4_k_m")
print("GGUF saved — download deepfeline_e4b_gguf/ and run: ollama create deepfeline -f Modelfile")
