+
    TV-jx  ã                   óš   € ^ RI t ^ RIt^ RIt^ RIt^ RIHt ^RIHt ^RI	H
t
Ht ^RIHt RtR tR t]R8X  d   ]! R	4       ]! 4        R# R# )
é    N)Úgenerate_step)Úmake_prompt_cacheÚsave_prompt_cache)Úloadiˆ  c                 ó  € \         P                  ! RR7      p V P                  R\        RRR7       V P                  R\        RR	7       V P                  R
RRR7       V P                  R\        RRR7       V P                  R\        RRR7       V P                  RRRR7       V P                  RRRR7       V P                  R\        RRR7       V P                  R\        R^@R7       V P                  RR \        \
        R!7       V # )"z&Set up and return the argument parser.z=Cache the state of a prompt to be reused with mlx_lm.generate)Údescriptionz--modelÚ	mlx_modelz;The path to the local model directory or Hugging Face repo.)ÚtypeÚdefaultÚhelpz--adapter-pathz9Optional path for the trained adapter weights and config.)r
   r   z--trust-remote-codeÚ
store_truez)Enable trusting remote code for tokenizer)Úactionr   z--eos-tokenNz#End of sequence token for tokenizerz--max-kv-sizez$Set the maximum key-value cache sizez--prompt-cache-filez$The file to save the prompt cache inT)r   Úrequiredz--promptz;Message to be processed by the model ('-' reads from stdin))r   r   z	--kv-bitszFNumber of bits for KV cache quantization. Defaults to no quantization.)r
   r   r   z--kv-group-sizez%Group size for KV cache quantization.z--quantized-kv-startzLWhen --kv-bits is set, start quantizing the KV cache from this step onwards.)r   r
   r   )ÚargparseÚArgumentParserÚadd_argumentÚstrÚintÚDEFAULT_QUANTIZED_KV_START)Úparsers    Úd/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/mlx_lm/cache_prompt.pyÚsetup_arg_parserr      s^  € ä×$Ò$ØSô€Fð ×ÑØÜØØJð	 ô ð ×ÑØÜØHð ô ð
 ×ÑØØØ8ð ô ð
 ×ÑØÜØØ2ð	 ô ð ×ÑØÜØØ3ð	 ô ð ×ÑØØ3Øð ô ð
 ×ÑØØØJð ô ð
 ×ÑØÜð'àð ô ð ×ÑØÜØ4Øð	 ô ð ×ÑØð"äÜ*ð ô ð €Mó    c                  ó  aa€ \        4       p V P                  4       pR VP                  '       d   RMR/pVP                  e   VP                  VR&   \	        VP
                  VP                  VR7      w  r4VP                  R8X  d   \        P                  P                  4       MVP                  Vn        VP                  '       d'   RRRVP                  /.pVP                  VR	RR
7      pMVP                  VP                  4      p\        W1P                  4      p\         P"                  ! V4      p\$        P$                  ! 4       o^ oVV3R lp	\'        VV^ VVP(                  VP*                  VP,                  V	R7       F  p
K  	  \/        4        \/        R\         P0                  ! 4       R,          R R24       \/        R4       / pVP
                  VR&   \2        P4                  ! V4      VR&   \7        VP8                  W{4       R# )Útrust_remote_codeTNÚ	eos_token)Úadapter_pathÚtokenizer_configÚ-ÚroleÚuserÚcontentF)Úadd_generation_promptÚcontinue_final_messagec                 óè   <€ \         P                   ! 4       pWS,
          ,          pR V R RVR R2p\        S\        V4      4      o\        VRS\        V4      ,
          ,          ,           RRR7       R	# )
zProcessed Ú6dz	 tokens (z6.2fz tok/s)Ú Ú T)ÚendÚflushN)ÚtimeÚmaxÚlenÚprint)Ú	processedÚtotal_tokensÚcurrentÚspeedÚmsgÚmax_msg_lenÚstarts   &&   €€r   ÚcallbackÚmain.<locals>.callbackv   sa   ø€ Ü—)’)“+ˆØ u�_Õ-ˆØ˜Y r˜N¨)°E¸$°<¸wÐGˆä˜+¤s¨3£xÓ0ˆÜˆc�C˜;¬¨S«Õ1Õ2Õ2¸À$×Gr   )Ú
max_tokensÚprompt_cacheÚkv_bitsÚkv_group_sizeÚquantized_kv_startÚprompt_progress_callbackzPeak memory: g    eÍÍAz.3fz GBz	Saving...Úmodelr   )r   Ú
parse_argsr   r   r   r>   r   ÚpromptÚsysÚstdinÚreadÚhas_chat_templateÚapply_chat_templateÚencoder   Úmax_kv_sizeÚmxÚarrayr+   r   r:   r;   r<   r.   Úget_peak_memoryÚjsonÚdumpsr   Úprompt_cache_file)r   Úargsr   r>   Ú	tokenizerÚmessagesr@   ÚcacheÚyr6   Ú_Úmetadatar4   r5   s               @@r   ÚmainrU   S   sÄ  ù€ ÜÓ€FØ×ÑÓ€Dð ,°T×5K×5KÐ5K©TÐQUÐVÐØ‡~�~Ò!Ø(,¯©Ð˜Ñ%äØ�
‰
Ø×&Ñ&Ø)ôÑ€Eð '+§k¡k°SÔ&8”#—)‘)—.‘.Ô"¸d¿k¹k€D„Kà×"×"Ð"Ø˜V Y°·±Ð<Ð=ˆØ×.Ñ.ØØ"'Ø#'ð /ó 
‰ð ×!Ñ! $§+¡+Ó.ˆä˜e×%5Ñ%5Ó6€EÜ
�Š�Ó€Aô �IŠI‹K€EØ€KöHô Ø	ØØØØ—‘Ø×(Ñ(Ø×2Ñ2Ø!)÷	ˆñ 	ñ	ô 
„GÜ	ˆMœ"×,Ò,Ó.°Õ4°SÐ9¸Ð
=Ô>ä	ˆ+ÔØ€HØŸ
™
€HˆWÑÜ#'§:¢:Ð.>Ó#?€HÐÑ Ü�d×,Ñ,¨eÖ>r   Ú__main__z�Calling `python -m mlx_lm.cache_prompt...` directly is deprecated. Use `mlx_lm.cache_prompt...` or `python -m mlx_lm cache_prompt ...` instead.)r   rK   rA   r+   Úmlx.coreÚcorerH   Úgenerater   Úmodels.cacher   r   Úutilsr   r   r   rU   Ú__name__r.   © r   r   Ú<module>r^      sU   ðó Û Û 
Û å å #ß >Ý à!Ð ò?òD>?ðB ˆzÔÙ	ð	Xôñ 	†Fñ r   