o
    °§Tj:  ã                   @  sV   d Z ddlmZ ddlmZ ddlmZ eG dd„ dƒƒZddd„ZG dd„ dƒZ	dS )uš  LLM client â€” talks to an OpenAI-compatible endpoint.

Works with either serving engine (both expose an OpenAI-compatible /v1 API):

  vLLM (int4 w4a16, best throughput/watt on the GPU):
    python -m vllm.entrypoints.openai.api_server         --model outputs/model-int4 --max-model-len 4096 --port 8001

  Ollama (GGUF Q4_0; simplest, great for the E-series):
    ollama run gemma4:e4b-it-qat        # or your fine-tuned GGUF via a Modelfile
    # Ollama serves OpenAI-compat at http://localhost:11434/v1

Point serving.base_url / serving.served_model in config.yaml at whichever you run.
Generation is capped (max_tokens, stop sequences) to avoid wasted compute.
é    )Úannotations)Ú	dataclassé   )ÚConfigc                   @  s<   e Zd ZU ded< dZded< dZded< eddd	„ƒZd
S )Ú	LLMResultÚstrÚtextr   ÚintÚprompt_tokensÚcompletion_tokensÚreturnc                 C  s   | j | j S )N)r
   r   )Úself© r   úsrc/agent/llm.pyÚtotal_tokens   s   zLLMResult.total_tokensN)r   r	   )Ú__name__Ú
__module__Ú__qualname__Ú__annotations__r
   r   Úpropertyr   r   r   r   r   r      s   
 r   Úvaluer   r   c                 C  s\   ddl }t| tƒr,|  d¡r,| dd… }|j |¡}|s*td| › d|› d|› d�ƒ‚|S | S )	a  Resolve an 'env:VAR_NAME' reference to the environment variable's value.

    Keeps real API keys out of config files: config holds 'env:OPENROUTER_API_KEY', the actual
    key lives only in the environment. Plain (non 'env:') values pass through unchanged.
    r   Nzenv:é   zserving.api_key is 'z' but $z& is not set. Set it first, e.g.  setx z' "<your-key>"  (then open a new shell).)ÚosÚ
isinstancer   Ú
startswithÚenvironÚgetÚ
SystemExit)r   r   ÚnameZsecretr   r   r   Ú_resolve_secret"   s   ÿÿr   c                   @  s    e Zd Zddd„Zdd	d
„ZdS )Ú	LLMClientÚcfgr   c                 C  sœ   ddl m} | d¡p| d¡| _t| dd¡ƒ| _t| dd¡ƒ| _t| d	d
¡ƒ| _| dd ¡| _	| d¡p9d }|| dd¡t
| dd¡ƒ|d�| _d S )Nr   )ÚOpenAIzserving.served_modelzmodel.base_idzgeneration.temperaturegš™™™™™É?zgeneration.top_pgÍÌÌÌÌÌì?zgeneration.max_tokensi   zgeneration.stopzserving.headerszserving.base_urlzhttp://localhost:8001/v1zserving.api_keyz
not-needed)Zbase_urlZapi_keyZdefault_headers)Zopenair"   r   ÚmodelÚfloatÚtemperatureÚtop_pr	   Ú
max_tokensÚstopr   Ú_client)r   r!   r"   Zheadersr   r   r   Ú__init__7   s   
ýzLLMClient.__init__Úmessagesú
list[dict]r   r   c                 C  sj   | j jjj| j|| j| j| j| jd�}t	|dd ƒ}t
|jd jjp"d ¡ t	|ddƒp+dt	|ddƒp2dd�S )N)r#   r+   r%   r&   r'   r(   Úusager   Ú r
   r   )r   r
   r   )r)   ZchatZcompletionsZcreater#   r%   r&   r'   r(   Úgetattrr   ÚchoicesÚmessageZcontentÚstrip)r   r+   Zrespr-   r   r   r   ÚgenerateG   s   
úýzLLMClient.generateN)r!   r   )r+   r,   r   r   )r   r   r   r*   r3   r   r   r   r   r    6   s    
r    N)r   r   r   r   )
Ú__doc__Z
__future__r   Zdataclassesr   Zsettingsr   r   r   r    r   r   r   r   Ú<module>   s    

