o
    ·£Tj¼  ã                   @  s€   d Z ddlmZ ddlmZ dZeG dd„ dƒƒZd"dd„Zd#dd„Zd$dd„Z	d%dd„Z
	d&d'dd„Zd(dd„Zd)dd „Zd!S )*uŸ  System prompt + RAG context assembly, tuned for minimal token usage.

Because QLoRA fine-tuning *internalizes* the behavioral rules (cite sources, use only context,
no personalized advice), the served system prompt can be short â€” every token here is paid on
EVERY request, so a compact prompt is a direct, permanent cost/energy saving. Training and serving
use the same compact prompt to avoid a train/serve mismatch.

Context assembly also minimizes input tokens: near-duplicate chunks are dropped and the context is
capped to a token budget (lowest-scoring chunks fall off first). This is the application-layer
analogue of the paper's sparse top-k context selection.
é    )Úannotations)Ú	dataclasszÒYou are a travel-insurance assistant. Use ONLY the CONTEXT; if the answer isn't there, say so and suggest a licensed advisor. Cite the passage number(s) you use, e.g. [1]. Never say which plan to buy. Be brief.c                   @  s&   e Zd ZU ded< ded< ded< dS )ÚRetrievedChunkÚstrÚtextÚsourceÚfloatÚscoreN)Ú__name__Ú
__module__Ú__qualname__Ú__annotations__© r   r   úsrc/agent/prompt.pyr      s   
 r   r   r   ÚreturnÚintc                 C  s   t dt| ƒd ƒS )zRCheap, dependency-free token estimate (~4 chars/token). Good enough for budgeting.é   é   N)ÚmaxÚlen©r   r   r   r   Úestimate_tokens    s   r   c                 C  s   d  |  ¡  ¡ ¡S )Nú )ÚjoinÚlowerÚsplitr   r   r   r   Ú
_normalize%   s   r   Úchunksúlist[RetrievedChunk]c                   sL   g }g }| D ]}t |jƒ‰ t‡ fdd„|D ƒƒrq| |¡ | ˆ ¡ q|S )zRDrop exact/subsumed duplicates (e.g. a table that also appears as flattened text).c                 3  s(   � | ]}ˆ |kpˆ |v p|ˆ v V  qd S ©Nr   )Ú.0Ús©Znormr   r   Ú	<genexpr>/   s   €& zdedup_chunks.<locals>.<genexpr>N)r   r   ÚanyÚappend)r   ZkeptZ	seen_normÚcr   r"   r   Údedup_chunks)   s   

r'   Úbudget_tokensc                 C  sX   t | dd„ dd�}g d}}|D ]}t|jƒ}|r || |kr q| |¡ ||7 }q|S )zUKeep highest-scoring chunks until the token budget is hit (always keep at least one).c                 S  s   | j S r   )r	   )r&   r   r   r   Ú<lambda>8   s    z"fit_token_budget.<locals>.<lambda>T)ÚkeyÚreverser   N)Úsortedr   r   r%   )r   r(   ZorderedÚoutÚusedr&   Zcostr   r   r   Úfit_token_budget6   s   



r/   TÚdedupÚboolc                 C  s   |rt | ƒ} t| |ƒS )z@Dedup + budget-trim retrieved chunks to minimize context tokens.N)r'   r/   )r   r(   r0   r   r   r   Úcompress_contextC   s   
r2   c              
   C  sP   | sdS dg}t | dƒD ]\}}| d|› d|j› d|j ¡ › �¡ qd |¡S )	z=Render retrieved chunks as a numbered, citable CONTEXT block.zCONTEXT: (none found)zCONTEXT:r   ú[z] (z)
z

N)Ú	enumerater%   r   r   Ústripr   )r   ÚlinesÚir&   r   r   r   Úbuild_context_blockL   s   &
r8   Úqueryú
list[dict]c                 C  s0   t |ƒ}|› d|  ¡ › d�}dtdœd|dœgS )zIAssemble the chat messages for the LLM (kept terse to save input tokens).z

Q: zH
Answer from CONTEXT only; cite the passage number(s) you use, e.g. [1].Úsystem)ZroleZcontentÚuserN)r8   r5   ÚSYSTEM_PROMPT)r9   r   ÚcontextZuser_contentr   r   r   Úbuild_messagesV   s
   þr?   N)r   r   r   r   )r   r   r   r   )r   r   r   r   )r   r   r(   r   r   r   )T)r   r   r(   r   r0   r1   r   r   )r   r   r   r   )r9   r   r   r   r   r:   )Ú__doc__Z
__future__r   Zdataclassesr   r=   r   r   r   r'   r/   r2   r8   r?   r   r   r   r   Ú<module>   s    ÿ



ÿ
	
