{
 "nbformat": 4,
 "nbformat_minor": 0,
 "metadata": {
  "colab": {"provenance": [], "gpuType": "T4"},
  "kernelspec": {"name": "python3", "display_name": "Python 3"},
  "accelerator": "GPU"
 },
 "cells": [
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
    "# DeepFeline Value — Gemma 4 E4B fine-tune\n",
    "Runtime → Change runtime type → **T4 GPU**, then Runtime → **Run all**.\n",
    "You'll be prompted to upload `student_sft.jsonl` in step 2.\n",
    "Total time on a free T4: roughly 2–4 hours. Keep the tab open."
   ]
  },
  {
   "cell_type": "code",
   "metadata": {},
   "source": [
    "%pip install -q unsloth"
   ],
   "execution_count": null,
   "outputs": []
  },
  {
   "cell_type": "code",
   "metadata": {},
   "source": [
    "from google.colab import files\n",
    "print('Upload student_sft.jsonl (from datasets/ on your PC):')\n",
    "uploaded = files.upload()\n",
    "assert 'student_sft.jsonl' in uploaded, 'please upload student_sft.jsonl'"
   ],
   "execution_count": null,
   "outputs": []
  },
  {
   "cell_type": "code",
   "metadata": {},
   "source": [
    "# Preprocess: strip the teacher-only 'Reference reading' excerpts so training\n",
    "# prompts match what the model will see at inference (ticker + fundamentals only).\n",
    "import json, re\n",
    "\n",
    "def strip_reference(user: str) -> str:\n",
    "    i = user.find('## Reference reading')\n",
    "    j = user.find('Deliver the full analysis')\n",
    "    if i != -1 and j != -1:\n",
    "        return user[:i] + user[j:]\n",
    "    return user\n",
    "\n",
    "rows = [json.loads(l) for l in open('student_sft.jsonl', encoding='utf-8')]\n",
    "for r in rows:\n",
    "    r['user'] = strip_reference(r['user'])\n",
    "with open('student_sft_clean.jsonl', 'w', encoding='utf-8') as fh:\n",
    "    for r in rows:\n",
    "        fh.write(json.dumps(r) + '\\n')\n",
    "print(len(rows), 'examples ready')"
   ],
   "execution_count": null,
   "outputs": []
  },
  {
   "cell_type": "code",
   "metadata": {},
   "source": [
    "from unsloth import FastModel\n",
    "from datasets import load_dataset\n",
    "from trl import SFTConfig, SFTTrainer\n",
    "\n",
    "MAX_SEQ = 5120\n",
    "\n",
    "model, tokenizer = FastModel.from_pretrained(\n",
    "    'unsloth/gemma-4-E4B-it',  # fallback: 'google/gemma-4-E4B-it'\n",
    "    max_seq_length=MAX_SEQ,\n",
    "    load_in_4bit=True,\n",
    ")\n",
    "model = FastModel.get_peft_model(\n",
    "    model, r=16, lora_alpha=16, lora_dropout=0,\n",
    "    target_modules=['q_proj', 'k_proj', 'v_proj', 'o_proj',\n",
    "                    'gate_proj', 'up_proj', 'down_proj'],\n",
    ")\n",
    "\n",
    "ds = load_dataset('json', data_files='student_sft_clean.jsonl', split='train')\n",
    "\n",
    "def to_text(ex):\n",
    "    msgs = [\n",
    "        {'role': 'system', 'content': ex['system']},\n",
    "        {'role': 'user', 'content': ex['user']},\n",
    "        {'role': 'assistant', 'content': ex['assistant']},\n",
    "    ]\n",
    "    return {'text': tokenizer.apply_chat_template(msgs, tokenize=False)}\n",
    "\n",
    "ds = ds.map(to_text)\n",
    "\n",
    "trainer = SFTTrainer(\n",
    "    model=model, tokenizer=tokenizer, train_dataset=ds,\n",
    "    args=SFTConfig(\n",
    "        dataset_text_field='text', max_seq_length=MAX_SEQ,\n",
    "        per_device_train_batch_size=1, gradient_accumulation_steps=8,\n",
    "        num_train_epochs=2, learning_rate=2e-4,\n",
    "        lr_scheduler_type='cosine', warmup_ratio=0.03,\n",
    "        logging_steps=5, output_dir='ckpt', save_strategy='no',\n",
    "    ),\n",
    ")\n",
    "trainer.train()"
   ],
   "execution_count": null,
   "outputs": []
  },
  {
   "cell_type": "code",
   "metadata": {},
   "source": [
    "# Quick smoke test before export\n",
    "from unsloth import FastModel\n",
    "FastModel.for_inference(model)\n",
    "msgs = [\n",
    "    {'role': 'system', 'content': rows[0]['system']},\n",
    "    {'role': 'user', 'content': rows[0]['user']},\n",
    "]\n",
    "inputs = tokenizer.apply_chat_template(msgs, add_generation_prompt=True, return_tensors='pt').to('cuda')\n",
    "out = model.generate(input_ids=inputs, max_new_tokens=400, temperature=0.7)\n",
    "print(tokenizer.decode(out[0][inputs.shape[1]:], skip_special_tokens=True))"
   ],
   "execution_count": null,
   "outputs": []
  },
  {
   "cell_type": "code",
   "metadata": {},
   "source": [
    "# Export GGUF (Q4_K_M, ~4.5GB) and download\n",
    "model.save_pretrained_gguf('deepfeline_e4b_gguf', tokenizer, quantization_method='q4_k_m')\n",
    "import glob\n",
    "gguf = glob.glob('deepfeline_e4b_gguf/*.gguf')[0]\n",
    "print('Download this file:', gguf)\n",
    "from google.colab import files\n",
    "files.download(gguf)"
   ],
   "execution_count": null,
   "outputs": []
  }
 ]
}
