o
    Ë TjN  ã                   @  s–   d Z ddlmZ ddlZddlZddlZddlZddlmZ ej	 
deeeƒ ¡ jd d ƒ¡ ddlmZ ddd„Zddd„ZedkrIeƒ  dS dS )u#  Launch the vLLM OpenAI server with the project's energy/token efficiency settings applied.

    python scripts/serve_vllm.py --config config/config.yaml            # start the server
    python scripts/serve_vllm.py --config config/config.yaml --dry-run  # just print the command

Reads config and assembles the vLLM command with:
  - the int4 (w4a16) model + capped context length
  - fp8 KV cache            (efficiency.kv_cache_dtype)   -> less memory/energy per token
  - prefix caching          (efficiency.enable_prefix_caching) -> static system prompt not re-prefilled
  - speculative decoding     (efficiency.speculative)       -> lower latency/energy per request
      * method=ngram : prompt-lookup, no extra model, ideal for RAG's verbatim copying
      * method=eagle : uses a DeepSpec-trained Eagle3 draft model (efficiency.speculative.draft_model)

Speculative decoding does NOT change how many tokens are generated â€” it makes generating them
cheaper/faster. Token COUNT is minimized elsewhere (router, cache, compact prompt, context budget).
é    )ÚannotationsN)ÚPathé   Úsrc)Úload_configÚportÚintÚreturnú	list[str]c           	      C  s0  t |  d¡ƒ}t|  dd¡ƒ}tjddd|d|  d|¡d	t |ƒd
t |ƒg}|  d¡}|r2|d|g7 }t|  dd¡ƒr?|dg7 }t |  dd¡ƒ ¡ }|dkrqdt|  dd¡ƒt|  dd¡ƒt|  dd¡ƒdœ}|dt 	|¡g7 }|S |dkr–|  dd¡}|s�t
dƒ‚d|t|  dd¡ƒdœ}|dt 	|¡g7 }|S ) Nzmodel.quantized_dirzmodel.max_model_leni   z-mz"vllm.entrypoints.openai.api_serverz--modelz--served-model-namezserving.served_modelz--max-model-lenú--portzefficiency.kv_cache_dtypez--kv-cache-dtypez efficiency.enable_prefix_cachingFz--enable-prefix-cachingzefficiency.speculative.methodZnoneZngramz-efficiency.speculative.num_speculative_tokensé   z(efficiency.speculative.prompt_lookup_maxé   z(efficiency.speculative.prompt_lookup_miné   )ÚmethodÚnum_speculative_tokensZprompt_lookup_maxZprompt_lookup_minz--speculative-configZeaglez"efficiency.speculative.draft_modelÚ z@efficiency.speculative.draft_model is required when method=eagle)r   Zmodelr   )ÚstrÚresolver   ÚgetÚsysÚ
executableÚboolÚlowerÚjsonÚdumpsÚ
SystemExit)	Úcfgr   Z	model_dirZmax_lenÚcmdZkvr   ÚspecZdraft© r   úscripts/serve_vllm.pyÚbuild_command   s@   û

üôýr!   ÚNonec                  C  s’   t  ¡ } | jddd� | jdtdd� | jddd	d
� |  ¡ }t|jƒ}t||jƒ}d 	dd„ |D ƒ¡}t
|d ƒ |jr?d S t |d |¡ d S )Nz--configzconfig/config.yaml)Údefaultr   iA  )Útyper#   z	--dry-runÚ
store_truez'print the command instead of running it)ÚactionÚhelpú c                 s  s(   � | ]}d |v rd|› d�n|V  qdS )r(   ú'Nr   )Ú.0Úcr   r   r    Ú	<genexpr>R   s   €& zmain.<locals>.<genexpr>Ú
r   )ÚargparseÚArgumentParserÚadd_argumentr   Ú
parse_argsr   Zconfigr!   r   ÚjoinÚprintÚdry_runÚosÚexecvp)ZapÚargsr   r   Z	printabler   r   r    ÚmainI   s   
r8   Ú__main__)r   r   r	   r
   )r	   r"   )Ú__doc__Z
__future__r   r.   r   r5   r   Zpathlibr   ÚpathÚinsertr   Ú__file__r   ÚparentsZagent.settingsr   r!   r8   Ú__name__r   r   r   r    Ú<module>   s    $

+
ÿ