"""Local speech-to-text via faster-whisper.

Audio is transcribed on THIS machine (CPU or GPU) — it is never sent to a cloud speech service.
The model is loaded lazily on first use and reused. Small models (tiny.en/base.en) run fine on CPU
with int8 compute, keeping with the energy-efficiency goal.
"""
from __future__ import annotations

import io
import os

# ctranslate2 and onnxruntime each bundle their own OpenMP runtime (libiomp5md); on Windows this
# trips "OMP: Error #15: ... already initialized". Allowing the duplicate is the standard, safe-for-
# inference workaround. Must be set before faster_whisper (-> ctranslate2) is imported.
os.environ.setdefault("KMP_DUPLICATE_LIB_OK", "TRUE")

from .settings import Config


class Transcriber:
    def __init__(self, cfg: Config):
        self.model_id = cfg.get("stt.model", "base.en")
        self.device = cfg.get("stt.device", "cpu")
        self.compute_type = cfg.get("stt.compute_type", "int8")
        self.language = cfg.get("stt.language", "en")
        self._model = None

    def _load(self):
        if self._model is None:
            from faster_whisper import WhisperModel  # lazy: only needed when transcribing

            self._model = WhisperModel(self.model_id, device=self.device, compute_type=self.compute_type)
        return self._model

    def transcribe(self, audio_bytes: bytes) -> str:
        """Transcribe raw audio bytes (any ffmpeg-decodable format, e.g. webm/opus) to text."""
        model = self._load()
        # vad_filter drops silence so short/empty clips don't hallucinate text.
        segments, _info = model.transcribe(
            io.BytesIO(audio_bytes), language=self.language, vad_filter=True
        )
        return " ".join(seg.text.strip() for seg in segments).strip()
