+
    QV-j»O ã                   óª  a € R6 t-0 t ^ RIHtHt ^ RIHt ^ RIt^RIHt ^RI	H
t
HtHtHtHtHt ]
! 4       '       d   ^ RIHt ]! RRR	7      t]P(                  ! ]4      t/ t] ^ k / t] ^k  ! R
 R]4      t ! R R]4      t ! R R]4      t ! R R]4      t ! R R]4      t ! R R]4      t ! R R]4      t ! R R]4      t  ! R R] 4      t! ! R R] 4      t" ! R R]4      t# ! R  R!]#4      t$ ! R" R#]$]4      t%]PM                  R$]R%]R&]R']$R(]$R)]$R*]$R+]%/4        ! R, R-4      t' ! R. R/]'4      t( ! R0 R1]'4      t) ! R2 R3]'4      t* ! R4 R5]'4      t+])t,R# )7é    )ÚABCÚabstractmethod)ÚIterableN©ÚPreTrainedConfig)Úis_hqq_availableÚis_optimum_quanto_availableÚis_quanto_greaterÚis_torch_greater_or_equalÚis_torchdynamo_compilingÚlogging)Ú	Quantizerz2.7T©Ú
accept_devc                   ó  a a€ ] tR t^+t oRtRtRtV 3R ltR tR t	]
V3R lR l4       t]
V3R	 lR
 l4       t]
V3R lR l4       t]
V3R lR l4       t]
V3R lR l4       tR tR tV3R lR ltV3R lR ltV3R ltRtVtV ;t# )ÚCacheLayerMixinz0Base, abstract class for a single layer's cache.FNc                óð   <€ \         SV `  ! R/ VB  V P                  P                  R R4      pVeE   \	        4       P                  R4      pVe   \        W4      '       d   V \        V&   R# V \        V&   R# R# )Ú
layer_typeNÚStaticLayer© )ÚsuperÚ__init_subclass__Ú__dict__ÚgetÚglobalsÚ
issubclassÚLAYER_TYPE_STATIC_CACHE_MAPPINGÚLAYER_TYPE_CACHE_MAPPING)ÚclsÚkwargsr   Ústatic_baseÚ	__class__s   &,  €Úi/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/cache_utils.pyr   Ú!CacheLayerMixin.__init_subclass__4   sj   ø€ Ü‰Ò!Ñ+ FÒ+Ø—\‘\×%Ñ% l°DÓ9ˆ
ØÒ!Ü!›)Ÿ-™-¨Ó6ˆKØÒ&¬:°c×+GÒ+GØ>AÔ/°
Ó;à7:Ô(¨Ó4ñ "ó    c                ó0   € R V n         R V n        RV n        R # ©NF)ÚkeysÚvaluesÚis_initialized©Úselfs   &r#   Ú__init__ÚCacheLayerMixin.__init__>   s   € Ø)-ˆŒ	Ø+/ˆŒØ#ˆÖr%   c                ó0   € V P                   P                   # ©N©r"   Ú__name__r+   s   &r#   Ú__repr__ÚCacheLayerMixin.__repr__C   ó   € Ø—.‘.×)Ñ)Ð*Ð+r%   c                óR   <€ V ^8„  d   QhRS[ P                  RS[ P                  RR/# ©é   Ú
key_statesÚvalue_statesÚreturnN©ÚtorchÚTensor)ÚformatÚ__classdict__s   "€r#   Ú__annotate__ÚCacheLayerMixin.__annotate__G   s!   ø€ ×dÑd©e¯l©lÐdÉ%Ï,É,ÐdÐ[_Ñdr%   c                ó   € R # r0   r   ©r,   r9   r:   s   &&&r#   Úlazy_initializationÚ#CacheLayerMixin.lazy_initializationF   s   € Ùadr%   c                ó’   <€ V ^8„  d   QhRS[ P                  RS[ P                  RS[S[ P                  S[ P                  3,          /# ©r8   r9   r:   r;   ©r=   r>   Útuple)r?   r@   s   "€r#   rA   rB   J   s?   ø€ ÷ 0ñ 0ÙŸ,™,ð0Ù6;·l±lð0á	‰u�|‰|™UŸ\™\Ð)Õ	*ñ0r%   c                ó   € R # r0   r   ©r,   r9   r:   Úargsr    s   &&&*,r#   ÚupdateÚCacheLayerMixin.updateI   s   € ñ -0r%   c                ó<   <€ V ^8„  d   QhRS[ RS[S[ S[ 3,          /# ©r8   Úquery_lengthr;   ©ÚintrJ   )r?   r@   s   "€r#   rA   rB   O   s   ø€ ×GÑG©3ÐG±5¹¹c¸µ?ÑGr%   c                ó   € R # r0   r   )r,   rR   s   &&r#   Úget_mask_sizesÚCacheLayerMixin.get_mask_sizesN   s   € ÙDGr%   c                ó    <€ V ^8„  d   QhRS[ /# ©r8   r;   ©rT   )r?   r@   s   "€r#   rA   rB   R   s   ø€ ×(Ñ(¡Ñ(r%   c                ó   € R # r0   r   r+   s   &r#   Úget_seq_lengthÚCacheLayerMixin.get_seq_lengthQ   ó   € Ù%(r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   rB   U   s   ø€ ×-Ñ-¡SÑ-r%   c                ó   € R # r0   r   r+   s   &r#   Úget_max_cache_shapeÚ#CacheLayerMixin.get_max_cache_shapeT   s   € Ù*-r%   c                ó¶   € V P                   '       dG   V P                  P                  RRR7      V n        V P                  P                  RRR7      V n        R# R# ©z(Offload this layer's data to CPU device.ÚcpuT©Únon_blockingN)r*   r(   Útor)   r+   s   &r#   ÚoffloadÚCacheLayerMixin.offloadW   sC   € à××ÐØŸ	™	Ÿ™ U¸˜Ó>ˆDŒIØŸ+™+Ÿ.™.¨¸T˜.ÓBˆDŽKñ r%   c                ó,  € V P                   '       d‚   V P                  P                  V P                  8w  d[   V P                  P                  V P                  RR7      V n        V P                  P                  V P                  RR7      V n        R# R# R# ©zcIn case of layer offloading, this allows to move the data back to the layer's device ahead of time.Trf   N)r*   r(   Údevicerh   r)   r+   s   &r#   ÚprefetchÚCacheLayerMixin.prefetch]   sd   € à××Ð 4§9¡9×#3Ñ#3°t·{±{Ô#BØŸ	™	Ÿ™ T§[¡[¸t˜ÓDˆDŒIØŸ+™+Ÿ.™.¨¯©À4˜.ÓHˆDŽKñ $CÑr%   c                ó   <€ V ^8„  d   QhRR/# ©r8   r;   Nr   )r?   r@   s   "€r#   rA   rB   c   s   ø€ ÷ /ñ /�tñ /r%   c                ó@  € V P                   '       d5   V P                  P                  4        V P                  P                  4        \	        V R4      '       dF   \        V P                  \        4      '       d
   ^ V n        R# V P                  P                  4        R# R# )ú4Resets the cache values while preserving the objectsÚcumulative_lengthN)r*   r(   Úzero_r)   ÚhasattrÚ
isinstancert   rT   r+   s   &r#   ÚresetÚCacheLayerMixin.resetc   sl   € à××ÐØ�I‰I�O‰OÔØ�K‰K×ÑÔä�4Ð,×-Ò-ä˜$×0Ñ0´#×6Ò6Ø)*�Ö&à×&Ñ&×,Ñ,Ö.ñ .r%   c                ó8   <€ V ^8„  d   QhRS[ P                  RR/# ©r8   Úbeam_idxr;   N©r=   Ú
LongTensor)r?   r@   s   "€r#   rA   rB   p   s%   ø€ ÷ Wñ W¡e×&6Ñ&6ð W¸4ñ Wr%   c                óD  € V P                  4       ^ 8”  d‹   V P                  P                  ^ VP                  V P                  P                  4      4      V n        V P
                  P                  ^ VP                  V P
                  P                  4      4      V n        R# R# )z,Reorders this layer's cache for beam search.N)r\   r(   Úindex_selectrh   rm   r)   ©r,   r|   s   &&r#   Úreorder_cacheÚCacheLayerMixin.reorder_cachep   sn   € à×ÑÓ  1Ô$ØŸ	™	×.Ñ.¨q°(·+±+¸d¿i¹i×>NÑ>NÓ2OÓPˆDŒIØŸ+™+×2Ñ2°1°h·k±kÀ$Ç+Á+×BTÑBTÓ6UÓVˆDŽKñ %r%   c                ó4   <€ V ^8„  d   Qh/ S[ R,          ;R&   # )r8   Nr   ©Ústr)r?   r@   s   "€r#   rA   rB   +   s   ø‡ ‚ ñ �d•
Ñ!ò r%   )rt   r*   r(   r)   )r2   Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úis_compileabler   r   r-   r3   r   rE   rN   rV   r\   ra   ri   rn   rx   r‚   Ú__annotate_func__Ú__static_attributes__Ú__classdictcell__Ú__classcell__©r"   r@   s   @@r#   r   r   +   s›   ù‡ € Ù:à€Nð "€Jõ;ò$ò
,ð ßdó Ødà÷0ó ð0ð ßGó ØGàß(ó Ø(àß-ó Ø-òCòI÷/ð /÷Wð W÷K … r%   r   c                   óÚ   a a€ ] tR t^wt oRtRtRV3R lV 3R llltV3R lR ltV3R lR ltV3R	 lR
 lt	V3R lR lt
V3R lR ltV3R lR ltV3R lR ltV3R lR ltRtVtV ;t# )ÚDynamicLayerzÔ
A cache layer that grows dynamically as more tokens are generated. This is the default for generative models.
It stores the key and value states as tensors of shape `[batch_size, num_heads, seq_len, head_dim]`.
Fc                ó.   <€ V ^8„  d   QhRS[ R,          /# ©r8   ÚconfigNr   )r?   r@   s   "€r#   rA   ÚDynamicLayer.__annotate__   ó   ø€ ÷ ñ Ñ/°$Õ6ñ r%   c                ó$   <€ \         SV `  4        R # r0   ©r   r-   ©r,   r•   r"   s   &&€r#   r-   ÚDynamicLayer.__init__   ó   ø€ Ü‰ÑÖr%   c                óR   <€ V ^8„  d   QhRS[ P                  RS[ P                  RR/# r7   r<   )r?   r@   s   "€r#   rA   r–   ‚   s+   ø€ ÷ #ñ #©e¯l©lð #É%Ï,É,ð #Ð[_ñ #r%   c                ó"  € VP                   VP                  uV n         V n        \        P                  ! . V P                   V P                  R 7      V n        \        P                  ! . V P                   V P                  R 7      V n        RV n        R# ©©Údtyperm   TN)r¡   rm   r=   Útensorr(   r)   r*   rD   s   &&&r#   rE   Ú DynamicLayer.lazy_initialization‚   s^   € Ø",×"2Ñ"2°J×4EÑ4EÐˆŒ
�D”KÜ—L’L ¨4¯:©:¸d¿k¹kÔJˆŒ	Ü—l’l 2¨T¯Z©ZÀÇÁÔLˆŒØ"ˆÖr%   c                ó’   <€ V ^8„  d   QhRS[ P                  RS[ P                  RS[S[ P                  S[ P                  3,          /# rH   rI   )r?   r@   s   "€r#   rA   r–   ˆ   s?   ø€ ÷ &ñ &ÙŸ,™,ð&Ù6;·l±lð&á	‰u�|‰|™UŸ\™\Ð)Õ	*ñ&r%   c                ó  € V P                   '       g   V P                  W4       \        P                  ! V P                  V.RR7      V n        \        P                  ! V P
                  V.RR7      V n        V P                  V P
                  3# )á1  
Update the key and value caches in-place, and return the necessary keys and value states.

Args:
    key_states (`torch.Tensor`): The new key states to cache.
    value_states (`torch.Tensor`): The new value states to cache.

Returns:
    tuple[`torch.Tensor`, `torch.Tensor`]: The key and value states.
©Údiméþÿÿÿ)r*   rE   r=   Úcatr(   r)   rL   s   &&&*,r#   rN   ÚDynamicLayer.updateˆ   sg   € ð ×"×"Ð"Ø×$Ñ$ ZÔ>ä—I’I˜tŸy™y¨*Ð5¸2Ô>ˆŒ	Ü—i’i §¡¨lÐ ;ÀÔDˆŒØ�y‰y˜$Ÿ+™+Ð%Ð%r%   c                ó<   <€ V ^8„  d   QhRS[ RS[S[ S[ 3,          /# rQ   rS   )r?   r@   s   "€r#   rA   r–   �   ó#   ø€ ÷ $ñ $©3ð $±5¹¹c¸µ?ñ $r%   c                ó:   € ^ pV P                  4       V,           pW23# )zDReturn the length and offset of the cache, used to generate the mask)r\   ©r,   rR   Ú	kv_offsetÚ	kv_lengths   &&  r#   rV   ÚDynamicLayer.get_mask_sizes�   s#   € àˆ	Ø×'Ñ'Ó)¨LÕ8ˆ	ØÐ#Ð#r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   r–   £   s   ø€ ÷ #ñ #¡ñ #r%   c                ó¢   € V P                   '       d    V P                  P                  4       ^ 8X  d   ^ # V P                  P                  R,          # )ú1Returns the sequence length of the cached states.r©   )r*   r(   ÚnumelÚshaper+   s   &r#   r\   ÚDynamicLayer.get_seq_length£   s6   € à×"×"Ð" d§i¡i§o¡oÓ&7¸1Ô&<ÙØ�y‰y�‰˜rÕ"Ð"r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   r–   ©   s   ø€ ÷ ñ ¡Sñ r%   c                ó   € R# )zeReturns the maximum sequence length of the cache object. DynamicLayer does not have a maximum length.éÿÿÿÿr   r+   s   &r#   ra   Ú DynamicLayer.get_max_cache_shape©   s   € àˆ	r%   c                ó$   <€ V ^8„  d   QhRS[ RR/# ©r8   Ú
max_lengthr;   NrZ   )r?   r@   s   "€r#   rA   r–   ­   s   ø€ ÷ 7ñ 7™sð 7 tñ 7r%   c                óö   € V^ 8  d!   V P                  4       \        V4      ,
          pV P                  4       V8:  d   R# V P                  RRV1R3,          V n        V P                  RRV1R3,          V n        R# )zˆ
Crop the past key values up to a new `max_length` in terms of tokens. `max_length` can also be negative
to remove `max_length` tokens.
N.ºNNN)r\   Úabsr(   r)   ©r,   r¿   s   &&r#   ÚcropÚDynamicLayer.crop­   sl   € ð
 ˜Œ>Ø×,Ñ,Ó.´°Z³Õ@ˆJà×ÑÓ  JÔ.Ùà—I‘I˜c ; J ;°Ð1Õ2ˆŒ	Ø—k‘k # {¨
 {°AÐ"5Õ6ˆŽr%   c                ó$   <€ V ^8„  d   QhRS[ RR/# ©r8   Úrepeatsr;   NrZ   )r?   r@   s   "€r#   rA   r–   »   s   ø€ ÷ Hñ H©sð H°tñ Hr%   c                ó¼   € V P                  4       ^ 8”  dG   V P                  P                  V^ R7      V n        V P                  P                  V^ R7      V n        R# R# )z8Repeat the cache `repeats` times in the batch dimension.r§   N)r\   r(   Úrepeat_interleaver)   ©r,   rÈ   s   &&r#   Úbatch_repeat_interleaveÚ$DynamicLayer.batch_repeat_interleave»   sN   € à×ÑÓ  1Ô$ØŸ	™	×3Ñ3°GÀÐ3ÓCˆDŒIØŸ+™+×7Ñ7¸ÀQÐ7ÓGˆDŽKñ %r%   c                ó8   <€ V ^8„  d   QhRS[ P                  RR/# ©r8   Úindicesr;   Nr<   )r?   r@   s   "€r#   rA   r–   Á   s   ø€ ÷ 4ñ 4©E¯L©Lð 4¸Tñ 4r%   c                óœ   € V P                  4       ^ 8”  d7   V P                  VR3,          V n        V P                  VR3,          V n        R# R# )z<Only keep the `indices` in the batch dimension of the cache..N)r\   r(   r)   ©r,   rÐ   s   &&r#   Úbatch_select_indicesÚ!DynamicLayer.batch_select_indicesÁ   s@   € à×ÑÓ  1Ô$ØŸ	™	 '¨3 ,Õ/ˆDŒIØŸ+™+ g¨s lÕ3ˆDŽKñ %r%   )rm   r¡   r*   r(   r)   r0   )r2   r‡   rˆ   r‰   rŠ   Ú
is_slidingr-   rE   rN   rV   r\   ra   rÄ   rÌ   rÓ   r�   rŽ   r�   r�   s   @@r#   r’   r’   w   sr   ù‡ € ñð
 €J÷õ ÷#ð #÷&ð &÷*$ð $÷#ð #÷ð ÷7ð 7÷Hð H÷4÷ 4ð 4r%   r’   c                   óÂ   a a€ ] tR t^Èt oRtRtRV3R lV 3R llltV3R lV 3R lltV3R lR ltV3R	 lR
 lt	V3R lR lt
V3R lR ltV3R lV 3R lltRtVtV ;t# )ÚDynamicSlidingWindowLayerzà
A cache layer that grows dynamically as more tokens are generated, up until the sliding window size.
It stores the key and value states as tensors of shape `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.
Tc                óB   <€ V ^8„  d   QhRS[ R,          RS[R,          /# )r8   r•   NÚsliding_window)r   rT   )r?   r@   s   "€r#   rA   Ú&DynamicSlidingWindowLayer.__annotate__Ð   s*   ø€ ÷ 
Zñ 
ZÑ/°$Õ6ð 
ZÉsÐUYÍzñ 
Zr%   c                ó  <€ \         SV `  4        Vf2   Vf   \        R4      h\        VRR 4      ;'       g    \        VRR 4      pW n        ^ V n        \        P                  ! V P                  \        P                  R7      V n	        R # )Nz5Either `config` or `sliding_window` must be provided.rÙ   Úattention_chunk_size©r¡   )
r   r-   Ú
ValueErrorÚgetattrrÙ   rt   r=   r¢   ÚlongÚ_sliding_window_tensor)r,   r•   rÙ   r"   s   &&&€r#   r-   Ú"DynamicSlidingWindowLayer.__init__Ð   su   ø€ Ü‰ÑÔð Ò!ØŠ~Ü Ð!XÓYÐYÜ$ VÐ-=¸tÓD×uÐuÌÐPVÐXnÐptÓHuˆNØ,ÔØ!"ˆÔÜ&+§l¢l°4×3FÑ3FÌeÏjÉjÔ&YˆÖ#r%   c                óR   <€ V ^8„  d   QhRS[ P                  RS[ P                  RR/# r7   r<   )r?   r@   s   "€r#   rA   rÚ   Ü   s0   ø€ ÷ Rñ R©e¯l©lð RÉ%Ï,É,ð RÐ[_ñ Rr%   c                óz   <€ \         SV `  W4       V P                  P                  V P                  4      V n        R # r0   )r   rE   rá   rh   rm   )r,   r9   r:   r"   s   &&&€r#   rE   Ú-DynamicSlidingWindowLayer.lazy_initializationÜ   s-   ø€ Ü‰Ñ# JÔ=Ø&*×&AÑ&A×&DÑ&DÀTÇ[Á[Ó&QˆÖ#r%   c                ó’   <€ V ^8„  d   QhRS[ P                  RS[ P                  RS[S[ P                  S[ P                  3,          /# rH   rI   )r?   r@   s   "€r#   rA   rÚ   à   s?   ø€ ÷ 2ñ 2ÙŸ,™,ð2Ù6;·l±lð2á	‰u�|‰|™UŸ\™\Ð)Õ	*ñ2r%   c                óÊ  € V P                   '       g   V P                  W4       V ;P                  VP                  R,          ,          un        \        P
                  ! V P                  V.RR7      p\        P
                  ! V P                  V.RR7      pVRRV P                  ) ^,           R1R3,          V n        VRRV P                  ) ^,           R1R3,          V n        WV3# )r¦   r§   rÁ   Nr©   )	r*   rE   rt   r·   r=   rª   r(   r)   rÙ   )r,   r9   r:   rM   r    Úfull_key_statesÚfull_value_statess   &&&*,  r#   rN   Ú DynamicSlidingWindowLayer.updateà   sÆ   € ð ×"×"Ð"Ø×$Ñ$ ZÔ>à×Ò *×"2Ñ"2°2Õ"6Õ6Õô  Ÿ)š) T§Y¡Y°
Ð$;ÀÔDˆÜ!ŸIšI t§{¡{°LÐ&AÀrÔJÐà# A q¨4×+>Ñ+>Ð*>ÀÕ*BÑ*DÀaÐ$GÕHˆŒ	Ø'¨¨1¨t×/BÑ/BÐ.BÀQÕ.FÑ.HÈ!Ð(KÕLˆŒð Ð1Ð1r%   c                ó<   <€ V ^8„  d   QhRS[ RS[S[ S[ 3,          /# rQ   rS   )r?   r@   s   "€r#   rA   rÚ   ý   s#   ø€ ÷ 
$ñ 
$©3ð 
$±5¹¹c¸µ?ñ 
$r%   c                ó  € V P                   V P                  8¬  p\        V P                   V P                  ,
          ^,           ^ 4      pV'       d   V P                  ^,
          V,           pWC3# V P                   V,           pWC3# ©zNReturn the length and offset of the cache, used to generate the attention mask)rt   rÙ   Úmax)r,   rR   Úis_fullr°   r±   s   &&   r#   rV   Ú(DynamicSlidingWindowLayer.get_mask_sizesý   sx   € à×(Ñ(¨D×,?Ñ,?Ñ?ˆä˜×.Ñ.°×1DÑ1DÕDÀqÕHÈ!ÓLˆ	ßØ×+Ñ+¨aÕ/°,Õ>ˆIð Ð#Ð#ð ×.Ñ.°Õ=ˆIàÐ#Ð#r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   rÚ   	  ó   ø€ ÷ &ñ &¡ñ &r%   c                ó   € V P                   # ©rµ   ©rt   r+   s   &r#   r\   Ú(DynamicSlidingWindowLayer.get_seq_length	  ó   € à×%Ñ%Ð%r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   rÚ     s   ø€ ÷ #ñ #¡Sñ #r%   c                ó   € V P                   # ©z+Return the maximum cache shape of the cache©rÙ   r+   s   &r#   ra   Ú-DynamicSlidingWindowLayer.get_max_cache_shape  s   € à×"Ñ"Ð"r%   c                ó$   <€ V ^8„  d   QhRS[ RR/# r¾   rZ   )r?   r@   s   "€r#   rA   rÚ     s   ø€ ÷ 5ñ 5™sð 5 tñ 5r%   c                ó¾   <€ V P                  4       V P                  8¼  d   \        R4      h\        SV `  V4       V P
                  P                  R,          V n        R# )zˆ
Crop the past key values up to a new `max_length` in terms of tokens. `max_length` can also be
negative to remove `max_length` tokens.
z�Cannot `crop` a `DynamicSlidingWindowLayer` after it has seen more tokens than itssliding window (otherwise some states are lost)Nr©   )r\   rÙ   rÞ   r   rÄ   r(   r·   rt   )r,   r¿   r"   s   &&€r#   rÄ   ÚDynamicSlidingWindowLayer.crop  sR   ø€ ð
 ×ÑÓ  D×$7Ñ$7Ô7ÜðBóð ô 	‰‰�ZÔ Ø!%§¡§¡°Õ!4ˆÖr%   )rá   rt   r(   rÙ   r)   ©NN)r2   r‡   rˆ   r‰   rŠ   rÕ   r-   rE   rN   rV   r\   ra   rÄ   r�   rŽ   r�   r�   s   @@r#   r×   r×   È   s`   ù‡ € ñð
 €J÷
Zõ 
Z÷Ró R÷2ð 2÷:
$ð 
$÷&ð &÷#ð #÷5÷ 5ó 5r%   r×   c                   óþ   a a€ ] tR tRt oRtRtRV3R lV 3R llltV3R lR ltV3R lR	 ltV 3R
 lt	V 3R lt
V3R lV 3R lltV3R lV 3R lltV3R lV 3R lltV3R lV 3R lltV3R lV 3R lltRtVtV ;t# )ÚDynamicIndexedLayeri  ag  
A cache layer that extends `DynamicLayer` with an extra indexer key cache for Dynamic Sparse Attention (DSA)
models (e.g. GLM MoE DSA, DeepSeek V32).

The main K/V cache stores tensors of shape `[batch_size, num_heads, seq_len, head_dim]` (inherited).
The indexer key cache stores a tensor of shape `[batch_size, seq_len, index_head_dim]` (3D, single-head).
Údeepseek_sparse_attentionc                ó.   <€ V ^8„  d   QhRS[ R,          /# r”   r   )r?   r@   s   "€r#   rA   Ú DynamicIndexedLayer.__annotate__+  s   ø€ ÷ 2ñ 2Ñ/°$Õ6ñ 2r%   c                óB   <€ \         SV `  V4       R V n        RV n        R # r'   )r   r-   Úindexer_keysÚis_indexer_initializedrš   s   &&€r#   r-   ÚDynamicIndexedLayer.__init__+  s    ø€ Ü‰Ñ˜Ô Ø15ˆÔØ,1ˆÖ#r%   c                ó8   <€ V ^8„  d   QhRS[ P                  RR/# ©r8   Úindexer_key_statesr;   Nr<   )r?   r@   s   "€r#   rA   r  0  s   ø€ ÷ +ñ +¹e¿l¹lð +Ètñ +r%   c                ó¾   € VP                   VP                  uV n        V n        \        P
                  ! . V P                  V P                  R 7      V n        RV n        R# rŸ   )r¡   rm   Úindexer_dtypeÚindexer_devicer=   r¢   r  r  ©r,   r  s   &&r#   Úlazy_initialization_indexerÚ/DynamicIndexedLayer.lazy_initialization_indexer0  sJ   € Ø2D×2JÑ2JÐL^×LeÑLeÐ/ˆÔ˜DÔ/Ü!ŸLšL¨°4×3EÑ3EÈd×NaÑNaÔbˆÔØ&*ˆÖ#r%   c                óN   <€ V ^8„  d   QhRS[ P                  RS[ P                  /# ©r8   r  r;   r<   )r?   r@   s   "€r#   rA   r  5  s#   ø€ ÷ !ñ !±·±ð !Á%Ç,Á,ñ !r%   c                ó²   € V P                   '       g   V P                  V4       \        P                  ! V P                  V.^R7      V n        V P                  # )a0  
Update the indexer key cache by concatenation, and return the full indexer keys.

Args:
    indexer_key_states (`torch.Tensor`): New indexer keys, shape `[batch_size, seq_len, index_head_dim]`.

Returns:
    `torch.Tensor`: The full cached indexer keys, shape `[batch_size, total_len, index_head_dim]`.
r§   )r  r  r=   rª   r  r  s   &&r#   Úupdate_indexerÚ"DynamicIndexedLayer.update_indexer5  sK   € ð ×*×*Ð*Ø×,Ñ,Ð-?Ô@Ü!ŸIšI t×'8Ñ'8Ð:LÐ&MÐSTÔUˆÔØ× Ñ Ð r%   c                ó�   <€ \         SV `  4        V P                  '       d%   V P                  P	                  R RR7      V n        R# R# )re   Trf   N)r   ri   r  r  rh   ©r,   r"   s   &€r#   ri   ÚDynamicIndexedLayer.offloadD  s<   ø€ Ü‰‰ÔØ×&×&Ð&Ø $× 1Ñ 1× 4Ñ 4°UÈÐ 4Ó NˆDÖñ 'r%   c                óò   <€ \         SV `  4        V P                  '       dV   V P                  P                  V P                  8w  d/   V P                  P                  V P                  R R7      V n        R# R# R# )Trf   N)r   rn   r  r  rm   rh   r  s   &€r#   rn   ÚDynamicIndexedLayer.prefetchI  s\   ø€ Ü‰ÑÔØ×&×&Ð&¨4×+<Ñ+<×+CÑ+CÀtÇ{Á{Ô+RØ $× 1Ñ 1× 4Ñ 4°T·[±[ÈtÐ 4Ó TˆDÖñ ,SÑ&r%   c                ó   <€ V ^8„  d   QhRR/# rq   r   )r?   r@   s   "€r#   rA   r  N  s   ø€ ÷ &ñ &�tñ &r%   c                ó€   <€ \         SV `  4        V P                  '       d   V P                  P	                  4        R # R # r0   )r   rx   r  r  ru   r  s   &€r#   rx   ÚDynamicIndexedLayer.resetN  s/   ø€ Ü‰‰ŒØ×&×&Ð&Ø×Ñ×#Ñ#Ö%ñ 'r%   c                ó8   <€ V ^8„  d   QhRS[ P                  RR/# r{   r}   )r?   r@   s   "€r#   rA   r  S  s%   ø€ ÷ iñ i¡e×&6Ñ&6ð i¸4ñ ir%   c                ó  <€ \         SV `  V4       V P                  '       dh   V P                  P	                  4       ^ 8”  dG   V P                  P                  ^ VP                  V P                  P                  4      4      V n        R# R# R# ©r   N)r   r‚   r  r  r¶   r€   rh   rm   )r,   r|   r"   s   &&€r#   r‚   Ú!DynamicIndexedLayer.reorder_cacheS  sk   ø€ Ü‰Ñ˜hÔ'Ø×&×&Ð&¨4×+<Ñ+<×+BÑ+BÓ+DÀqÔ+HØ $× 1Ñ 1× >Ñ >¸qÀ(Ç+Á+Èd×N_ÑN_×NfÑNfÓBgÓ hˆDÖñ ,IÑ&r%   c                ó$   <€ V ^8„  d   QhRS[ RR/# r¾   rZ   )r?   r@   s   "€r#   rA   r  X  s   ø€ ÷ Dñ D™sð D tñ Dr%   c                óz  <€ \         SV `  V4       V P                  '       d    V P                  P	                  4       ^ 8X  d   R# V^ 8¼  d   TM,V P                  P
                  ^,          \        V4      ,
          pV P                  P
                  ^,          V8”  d    V P                  RRV1R3,          V n        R# R# )r   NrÁ   )r   rÄ   r  r  r¶   r·   rÂ   )r,   r¿   Ú	effectiver"   s   && €r#   rÄ   ÚDynamicIndexedLayer.cropX  sš   ø€ Ü‰‰�ZÔ Ø×*×*Ð*¨d×.?Ñ.?×.EÑ.EÓ.GÈ1Ô.LÙØ",°¤/‘J°t×7HÑ7H×7NÑ7NÈqÕ7QÔTWÐXbÓTcÕ7cˆ	Ø×Ñ×"Ñ" 1Õ%¨	Ô1Ø $× 1Ñ 1°!°Z°i°ZÀÐ2BÕ CˆDÖñ 2r%   c                ó$   <€ V ^8„  d   QhRS[ RR/# rÇ   rZ   )r?   r@   s   "€r#   rA   r  `  s   ø€ ÷ Tñ T©sð T°tñ Tr%   c                óÔ   <€ \         SV `  V4       V P                  '       dF   V P                  P	                  4       ^ 8”  d%   V P                  P                  V^ R7      V n        R# R# R# )r   r§   N)r   rÌ   r  r  r¶   rÊ   )r,   rÈ   r"   s   &&€r#   rÌ   Ú+DynamicIndexedLayer.batch_repeat_interleave`  sZ   ø€ Ü‰Ñ'¨Ô0Ø×&×&Ð&¨4×+<Ñ+<×+BÑ+BÓ+DÀqÔ+HØ $× 1Ñ 1× CÑ CÀGÐQRÐ CÓ SˆDÖñ ,IÑ&r%   c                ó8   <€ V ^8„  d   QhRS[ P                  RR/# rÏ   r<   )r?   r@   s   "€r#   rA   r  e  s#   ø€ ÷ @ñ @©E¯L©Lð @¸Tñ @r%   c                óÄ   <€ \         SV `  V4       V P                  '       d>   V P                  P	                  4       ^ 8”  d   V P                  VR3,          V n        R# R# R# )r   .N)r   rÓ   r  r  r¶   )r,   rÐ   r"   s   &&€r#   rÓ   Ú(DynamicIndexedLayer.batch_select_indicese  sR   ø€ Ü‰Ñ$ WÔ-Ø×&×&Ð&¨4×+<Ñ+<×+BÑ+BÓ+DÀqÔ+HØ $× 1Ñ 1°'¸3°,Õ ?ˆDÖñ ,IÑ&r%   )r  r  r  r  r0   )r2   r‡   rˆ   r‰   rŠ   r   r-   r  r  ri   rn   rx   r‚   rÄ   rÌ   rÓ   r�   rŽ   r�   r�   s   @@r#   r  r    s{   ù‡ € ñð -€J÷2õ 2÷
+ð +÷
!ð !õOõ
U÷
&ó &÷
ió i÷
Dó D÷Tó T÷
@÷ @ó @r%   r  c                   ó¤   a a€ ] tR tRt oRtRtRtV3R lV 3R lltV3R lR ltV3R	 lR
 lt	V3R lR lt
V3R lR ltV3R lR ltRtVtV ;t# )r   ik  ar  
A static cache layer that stores the key and value states as static tensors of shape `[batch_size, num_heads, max_cache_len), head_dim]`.
It lazily allocates its full backing tensors, and then mutates them in-place. Built for `torch.compile` support.

Args:
    max_cache_len (`int`):
        Maximum number of tokens that can be stored, used for tensor preallocation.
TFc                ó    <€ V ^8„  d   QhRS[ /# ©r8   Úmax_cache_lenrZ   )r?   r@   s   "€r#   rA   ÚStaticLayer.__annotate__x  s   ø€ ÷ >ñ >¡cñ >r%   c                ót   <€ \         SV `  4        Wn        \        P                  ! ^ .\
        R7      V n        R# )r   rÝ   N)r   r-   r1  r=   r¢   rT   rt   ©r,   r1  r"   s   &&€r#   r-   ÚStaticLayer.__init__x  s)   ø€ Ü‰ÑÔØ*Ôä!&§¢¨q¨c¼Ô!=ˆÖr%   c                óR   <€ V ^8„  d   QhRS[ P                  RS[ P                  RR/# r7   r<   )r?   r@   s   "€r#   rA   r2  ~  s+   ø€ ÷ (#ñ (#©e¯l©lð (#É%Ï,É,ð (#Ð[_ñ (#r%   c                óÜ  € VP                   VP                  uV n         V n        VP                  R,          w  V n        V n        VP                  R,          V n        VP                  R,          V n        \        P                  ! V P                  V P                  V P                  V P                  3V P                   V P                  R7      V n
        \        P                  ! V P                  V P                  V P                  V P
                  3V P                   V P                  R7      V n        V P                  P                  V P                  4      V n        \        4       '       g|   \        P                  P!                  V P                  4       \        P                  P!                  V P                  4       \        P                  P!                  V P                  4       RV n        R# )aÞ  
Lazy initialization of the keys and values tensors. This allows to get all properties (dtype, device,
num_heads in case of TP etc...) at runtime directly, which is extremely practical as it avoids moving
devices, dtypes etc later on for each `update` (which could break the static dynamo addresses as well).

If this is unwanted, one can call `early_initialization(...)` on the Cache directly, which will call this
function ahead-of-time (this is required for `torch.export` for example). Note that for `compile`, as we
internally don't compile the prefill, this is guaranteed to have been called already when compiling.
If compiling the prefill as well, e.g. calling `model.compile(...)` before `generate` with a static cache,
it is still supported in general, but without guarantees depending on the compilation options (e.g. cuda graphs,
i.e. `mode="reduce-overhead"` is known to fail). But it will in general work correctly, and prefill should
not be compiled anyway for performances!
ºNr8   Nr    TNr»   )r¡   rm   r·   Úmax_batch_sizeÚ	num_headsÚ
v_head_dimÚ
k_head_dimr=   Úzerosr1  r(   r)   rt   rh   r   Ú_dynamoÚmark_static_addressr*   rD   s   &&&r#   rE   ÚStaticLayer.lazy_initialization~  sR  € ð #-×"2Ñ"2°J×4EÑ4EÐˆŒ
�D”KØ.8×.>Ñ.>¸rÕ.BÑ+ˆÔ˜Tœ^Ø&×,Ñ,¨RÕ0ˆŒØ$×*Ñ*¨2Õ.ˆŒä—K’KØ× Ñ  $§.¡.°$×2DÑ2DÀdÇoÁoÐVØ—*‘*Ø—;‘;ô
ˆŒ	ô
 —k’kØ× Ñ  $§.¡.°$×2DÑ2DÀdÇoÁoÐVØ—*‘*Ø—;‘;ô
ˆŒð
 "&×!7Ñ!7×!:Ñ!:¸4¿;¹;Ó!GˆÔô (×)Ò)Ü�M‰M×-Ñ-¨d¯i©iÔ8Ü�M‰M×-Ñ-¨d¯k©kÔ:Ü�M‰M×-Ñ-¨d×.DÑ.DÔEà"ˆÖr%   c                ó’   <€ V ^8„  d   QhRS[ P                  RS[ P                  RS[S[ P                  S[ P                  3,          /# rH   rI   )r?   r@   s   "€r#   rA   r2  ¨  s?   ø€ ÷  &ñ  &ÙŸ,™,ð &Ù6;·l±lð &á	‰u�|‰|™UŸ\™\Ð)Õ	*ñ &r%   c                ó  € V P                   '       g   V P                  W4       VP                  R,          p\        P                  ! WPP
                  R7      V P                  ,           pV P                  P                  V4        V P                  P                  ^Wa4       V P                  P                  ^Wb4       V P                  V P                  3#   \         d&    YP                  RRT3&   Y P                  RRT3&    LGi ; i)r¦   ©rm   rÁ   r©   )r*   rE   r·   r=   Úarangerm   rt   Úadd_r(   Úindex_copy_r)   ÚNotImplementedError)r,   r9   r:   rM   r    r±   Úcache_positions   &&&*,  r#   rN   ÚStaticLayer.update¨  sÚ   € ð ×"×"Ð"Ø×$Ñ$ ZÔ>ð ×$Ñ$ RÕ(ˆ	ÜŸš i¿¹ÔDÀt×G]ÑG]Õ]ˆà×Ñ×#Ñ# IÔ.ð	=Ø�I‰I×!Ñ! ! ^Ô@Ø�K‰K×#Ñ# A ~ÔDð �y‰y˜$Ÿ+™+Ð%Ð%øô #ô 	=à.8�I‰I�a˜˜NÐ*Ñ+Ø0<�K‰K˜˜1˜nÐ,Ó-ð	=ús   Â8C Ã-DÄDc                ó<   <€ V ^8„  d   QhRS[ RS[S[ S[ 3,          /# rQ   rS   )r?   r@   s   "€r#   rA   r2  Ê  r­   r%   c                ó$   € ^ pV P                   pW23# rí   ©r1  r¯   s   &&  r#   rV   ÚStaticLayer.get_mask_sizesÊ  s   € àˆ	Ø×&Ñ&ˆ	ØÐ#Ð#r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   r2  Ð  s   ø€ ÷ Dñ D¡ñ Dr%   c                óB   € V P                   '       d   V P                  # ^ # rô   )r*   rt   r+   s   &r#   r\   ÚStaticLayer.get_seq_lengthÐ  s   € à)-×)<×)<Ð)<ˆt×%Ñ%ÐCÀ!ÐCr%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   r2  Ô  s   ø€ ÷ "ñ "¡Sñ "r%   c                ó   € V P                   # rú   rL  r+   s   &r#   ra   ÚStaticLayer.get_max_cache_shapeÔ  s   € à×!Ñ!Ð!r%   )rt   rm   r¡   r*   r<  r(   r9  r1  r:  r;  r)   )r2   r‡   rˆ   r‰   rŠ   r‹   rÕ   r-   rE   rN   rV   r\   ra   r�   rŽ   r�   r�   s   @@r#   r   r   k  s[   ù‡ € ñð €NØ€J÷>ó >÷(#ð (#÷T &ð  &÷D$ð $÷Dð D÷"÷ "ð "r%   r   c                   óˆ   a a€ ] tR tRt oRtRtV3R lV 3R lltV3R lR ltV3R lR	 ltV3R
 lR lt	V 3R lt
RtVtV ;t# )ÚStaticSlidingWindowLayeriÙ  aÊ  
A static cache layer that stores the key and value states as static tensors of shape
`[batch_size, num_heads, min(max_cache_len, sliding_window), head_dim]`. It lazily allocates its full backing
tensors, and then mutates them in-place. Built for `torch.compile` support.

Args:
    max_cache_len (`int`):
        Maximum number of tokens that can be stored, used for tensor preallocation.
    sliding_window (`int`):
        The size of the sliding window.
Tc                ó&   <€ V ^8„  d   QhRS[ RS[ /# )r8   r1  rÙ   rZ   )r?   r@   s   "€r#   rA   Ú%StaticSlidingWindowLayer.__annotate__è  s   ø€ ÷ 'ñ '¡cð '¹3ñ 'r%   c                óL   <€ \        W!4      p\        SV `	  VR 7       ^ V n        R# )rL  N)Úminr   r-   Úcumulative_length_int)r,   r1  rÙ   Úeffective_max_cache_lenr"   s   &&& €r#   r-   Ú!StaticSlidingWindowLayer.__init__è  s'   ø€ Ü"% nÓ"DÐÜ‰ÑÐ'>ÐÔ?à%&ˆÖ"r%   c                ó’   <€ V ^8„  d   QhRS[ P                  RS[ P                  RS[S[ P                  S[ P                  3,          /# rH   rI   )r?   r@   s   "€r#   rA   rW  î  sD   ø€ ÷ M2ñ M2ÙŸ,™,ðM2Ù6;·l±lðM2á	‰u�|‰|™UŸ\™\Ð)Õ	*ñM2r%   c                óÄ  € V P                   '       g   V P                  W4       VP                  R,          pV P                  pW`P                  8¬  pV ;P                  V,          un        V'       Ed/   VP                  R,          ^8X  d¿   V P
                  P                  RRR7      pV P                  P                  RRR7      p	\        P                  ! R.\        V P                  R7      p
WRRV
3&   W)RRV
3&   V P
                  P                  V4       V P                  P                  V	4       V P
                  V P                  3# \        P                  ! V P
                  R	,          V3RR7      p\        P                  ! V P                  R	,          V3RR7      pEM%We,           V P                  8”  dq   V^ 8X  d   TpTpEM\        P                  ! V P
                  RRRV1R3,          V3RR7      p\        P                  ! V P                  RRRV1R3,          V3RR7      pMž\        P                  ! WPP                  R7      V P                  ,           p V P
                  P!                  ^WÑ4       V P                  P!                  ^WÒ4       V P                  P%                  V4       V P
                  V P                  3# V P
                  P                  VRRV P                  ) R1R3,          4       V P                  P                  VRRV P                  ) R1R3,          4       W¼3#   \"         d&    YP
                  RRT3&   Y P                  RRT3&    LËi ; i)
r¦   )Údimsr    rÁ   r§   NrC  r©   r»   )rÁ   rÁ   :é   NNrÁ   )r*   rE   r·   rZ  r1  r(   Úrollr)   r=   r¢   rT   rm   Úcopy_rª   rD  rt   rF  rG  rE  )r,   r9   r:   rM   r    r±   Úcurrent_lengthrï   Únew_keysÚ
new_valuesÚindexrè   ré   rH  s   &&&*,         r#   rN   ÚStaticSlidingWindowLayer.updateî  sä  € ð ×"×"Ð"Ø×$Ñ$ ZÔ>à×$Ñ$ RÕ(ˆ	Ø×3Ñ3ˆØ ×$6Ñ$6Ñ6ˆà×"Ò" iÕ/Õ"çˆ7ð ×Ñ Õ# qÔ(àŸ9™9Ÿ>™>¨"°2˜>Ó6�Ø!Ÿ[™[×-Ñ-¨b°rÐ-Ó:�
ô Ÿš b T´¸T¿[¹[ÔI�Ø(2˜˜A˜u˜Ñ%Ø*6˜1˜a ˜;Ñ'ð —	‘	—‘ Ô)Ø—‘×!Ñ! *Ô-ð —y‘y $§+¡+Ð-Ð-ô #(§)¢)¨T¯Y©Y°{Õ-CÀZÐ,PÐVXÔ"Y�Ü$)§I¢I¨t¯{©{¸;Õ/GÈÐ.VÐ\^Ô$_Ò!àÕ'¨$×*<Ñ*<Ô<à Ô"Ø",�Ø$0Ò!ä"'§)¢)¨T¯Y©Y°q¸!¸_¸n¸_ÈaÐ7OÕ-PÐR\Ð,]ÐceÔ"f�Ü$)§I¢I¨t¯{©{¸1¸aÀÀ.ÀÐRSÐ;SÕ/TÐVbÐ.cÐikÔ$lÑ!ô #Ÿ\š\¨)¿K¹KÔHÈ4×KaÑKaÕaˆNðAØ—	‘	×%Ñ% a¨ÔDØ—‘×'Ñ'¨¨>ÔHð ×"Ñ"×'Ñ'¨	Ô2ð —9‘9˜dŸk™kÐ)Ð)ð 	�	‰	�‰˜¨¨1¨t×/AÑ/AÐ.AÑ.CÀQÐ(FÕGÔHØ�‰×ÑÐ+¨A¨q°4×3EÑ3EÐ2EÑ2GÈÐ,JÕKÔLàÐ1Ð1øô 'ô AØ2<—	‘	˜!˜Q Ð.Ñ/Ø4@—‘˜A˜q .Ð0Ó1ðAús   É8L/ Ì/-MÍMc                ó<   <€ V ^8„  d   QhRS[ RS[S[ S[ 3,          /# rQ   rS   )r?   r@   s   "€r#   rA   rW  =  s#   ø€ ÷ $ñ $©3ð $±5¹¹c¸µ?ñ $r%   c                ó.  € V P                   pV P                  V P                   8¬  p\        V P                  V,
          ^,           ^ 4      pV'       d   W!,           ^,
          pWT3# V P                  V,           V8”  d   V P                  V,           pWT3# TpWT3# rí   )r1  rZ  rî   )r,   rR   rÙ   rï   r°   r±   s   &&    r#   rV   Ú'StaticSlidingWindowLayer.get_mask_sizes=  sš   € à×+Ñ+ˆØ×,Ñ,°×0BÑ0BÑBˆä˜×2Ñ2°^ÕCÀaÕGÈÓKˆ	çØ&Õ5¸Õ9ˆIð Ð#Ð#ð ×'Ñ'¨,Õ6¸ÔGØ×2Ñ2°\ÕAˆIð
 Ð#Ð#ð 'ˆIàÐ#Ð#r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   rW  O  s   ø€ ÷ *ñ *¡ñ *r%   c                ó   € V P                   # rô   ©rZ  r+   s   &r#   r\   Ú'StaticSlidingWindowLayer.get_seq_lengthO  s   € à×)Ñ)Ð)r%   c                ó2   <€ \         SV `  4        ^ V n        R# r"  )r   rx   rZ  r  s   &€r#   rx   ÚStaticSlidingWindowLayer.resetS  s   ø€ Ü‰‰ŒØ%&ˆÖ"r%   rm  )r2   r‡   rˆ   r‰   rŠ   rÕ   r-   rN   rV   r\   rx   r�   rŽ   r�   r�   s   @@r#   rU  rU  Ù  sF   ù‡ € ñ
ð €J÷'ó '÷M2ð M2÷^$ð $÷$*ð *÷'õ 'r%   rU  c                   ó~   a a€ ] tR tRt oRtV3R lV 3R lltV3R lR ltV3R lR ltV3R	 lV 3R
 lltRt	Vt
V ;t# )ÚStaticIndexedLayeriX  a  
A `StaticLayer` with an additional statically-allocated indexer key cache for Dynamic Sparse
Attention (DSA) models (e.g. GLM MoE DSA, DeepSeek V32). This is the static, `torch.compile`-friendly
counterpart of `DynamicIndexedLayer`: the indexer key buffer is preallocated once and mutated in-place.

The main K/V cache is inherited from `StaticLayer` (`[batch_size, num_heads, max_cache_len, head_dim]`).
The indexer key cache stores a tensor of shape `[batch_size, max_cache_len, index_head_dim]` (3D, single-head).
c                ó    <€ V ^8„  d   QhRS[ /# r0  rZ   )r?   r@   s   "€r#   rA   ÚStaticIndexedLayer.__annotate__b  s   ø€ ÷ Fñ F¡cñ Fr%   c                óˆ   <€ \         SV `  VR 7       RV n        RV n        \        P
                  ! ^ .\        R7      V n        R# )rL  NFrÝ   )r   r-   r  r  r=   r¢   rT   Úindexer_cumulative_lengthr4  s   &&€r#   r-   ÚStaticIndexedLayer.__init__b  s:   ø€ Ü‰Ñ }ÐÔ5Ø15ˆÔØ,1ˆÔ#ô */¯ª°q°cÄÔ)EˆÖ&r%   c                ó8   <€ V ^8„  d   QhRS[ P                  RR/# r  r<   )r?   r@   s   "€r#   rA   rt  j  s   ø€ ÷ +ñ +¹e¿l¹lð +Ètñ +r%   c                ó  € VP                   VP                  uV n        V n        VP                  w  r#p\
        P                  ! W P                  V3V P                  V P                  R 7      V n        V P                  P                  V P                  4      V n	        \        4       '       gS   \
        P                  P                  V P                  4       \
        P                  P                  V P                  4       RV n        R# rŸ   )r¡   rm   r  r  r·   r=   r=  r1  r  rv  rh   r   r>  r?  r  )r,   r  r9  Ú_Úindex_head_dims   &&   r#   r  Ú.StaticIndexedLayer.lazy_initialization_indexerj  sÄ   € Ø2D×2JÑ2JÐL^×LeÑLeÐ/ˆÔ˜DÔ/Ø,>×,DÑ,DÑ)ˆ˜>Ü!ŸKšKØ×/Ñ/°Ð@Ø×$Ñ$Ø×&Ñ&ô
ˆÔð
 *.×)GÑ)G×)JÑ)JÈ4×K^ÑK^Ó)_ˆÔ&ä'×)Ò)Ü�M‰M×-Ñ-¨d×.?Ñ.?Ô@Ü�M‰M×-Ñ-¨d×.LÑ.LÔMØ&*ˆÖ#r%   c                óN   <€ V ^8„  d   QhRS[ P                  RS[ P                  /# r  r<   )r?   r@   s   "€r#   rA   rt  y  s#   ø€ ÷ !ñ !±·±ð !Á%Ç,Á,ñ !r%   c                ó²  € V P                   '       g   V P                  V4       VP                  ^,          p\        P                  ! W P
                  R7      V P                  ,           pV P                  P                  V4        V P                  P                  ^W14       V P                  #   \         d    YP                  RT3&    T P                  # i ; i)aî  
Update the indexer key cache in-place at the current positions, and return the full static buffer.

Args:
    indexer_key_states (`torch.Tensor`): New indexer keys, shape `[batch_size, seq_len, index_head_dim]`.

Returns:
    `torch.Tensor`: The full static indexer key cache, shape `[batch_size, max_cache_len, index_head_dim]`.
        Unfilled positions are masked out downstream by the indexer's attention mask, exactly as the
        main `StaticLayer` returns its full preallocated K/V.
rC  rÁ   )r  r  r·   r=   rD  r  rv  rE  r  rF  rG  )r,   r  Úseq_lenrH  s   &&  r#   r  Ú!StaticIndexedLayer.update_indexery  s¾   € ð ×*×*Ð*Ø×,Ñ,Ð-?Ô@à$×*Ñ*¨1Õ-ˆÜŸš g×6IÑ6IÔJÈT×MkÑMkÕkˆà×&Ñ&×+Ñ+¨GÔ4ð	FØ×Ñ×)Ñ)¨!¨^ÔPð
 × Ñ Ð øô	 #ô 	Fà3E×Ñ˜a Ð/Ò0à× Ñ Ð ð		Fús   ÂB- Â-CÃCc                ó   <€ V ^8„  d   QhRR/# rq   r   )r?   r@   s   "€r#   rA   rt  ”  s   ø€ ÷ 3ñ 3�tñ 3r%   c                ó´   <€ \         SV `  4        V P                  '       d7   V P                  P	                  4        V P
                  P	                  4        R # R # r0   )r   rx   r  r  ru   rv  r  s   &€r#   rx   ÚStaticIndexedLayer.reset”  sA   ø€ Ü‰‰ŒØ×&×&Ð&Ø×Ñ×#Ñ#Ô%Ø×*Ñ*×0Ñ0Ö2ñ 'r%   )rv  r  r  r  r  )r2   r‡   rˆ   r‰   rŠ   r-   r  r  rx   r�   rŽ   r�   r�   s   @@r#   rr  rr  X  s9   ù‡ € ñ÷Fó F÷+ð +÷!ð !÷63÷ 3ó 3r%   rr  c                   óŠ   a a€ ] tR tRt oRtRV3R lV 3R llltV3R lR lt]R 4       t]R 4       t	V3R	 lR
 lt
RtVtV ;t# )ÚQuantizedLayeri›  aò  
A quantized layer similar to what is described in the [KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache paper](https://huggingface.co/papers/2402.02750).
It allows the model to generate longer sequence length without allocating too much memory for the key and value caches by
applying quantization.

The cache has two types of storage, one for original precision and one for the quantized cache. A `residual length`
is set as a maximum capacity for the original precision cache. When the length goes beyond maximum capacity, the original
precision cache is discarded and moved into the quantized cache. The quantization is done per-channel with a set `q_group_size`
for both Keys and Values, in contrast to what was described in the paper.
c          
      ó8   <€ V ^8„  d   QhRS[ RS[ RS[ RS[ RS[ /# ©r8   ÚnbitsÚaxis_keyÚ
axis_valueÚq_group_sizeÚresidual_lengthrZ   )r?   r@   s   "€r#   rA   ÚQuantizedLayer.__annotate__§  s=   ø€ ÷ #ñ #áð#ñ ð#ñ ð	#ñ
 ð#ñ ñ#r%   c                ón   <€ \         SV `  4        Wn        W n        W0n        W@n        WPn        ^ V n        R# r"  )r   r-   rˆ  r‰  rŠ  r‹  rŒ  rt   ©r,   rˆ  r‰  rŠ  r‹  rŒ  r"   s   &&&&&&€r#   r-   ÚQuantizedLayer.__init__§  s3   ø€ ô 	‰ÑÔØŒ
Ø ŒØ$ŒØ(ÔØ.ÔØ!"ˆÖr%   c                ó’   <€ V ^8„  d   QhRS[ P                  RS[ P                  RS[S[ P                  S[ P                  3,          /# rH   rI   )r?   r@   s   "€r#   rA   r�  ·  s?   ø€ ÷ #0ñ #0ÙŸ,™,ð#0Ù6;·l±lð#0á	‰u�|‰|™UŸ\™\Ð)Õ	*ñ#0r%   c                ó:  € V ;P                   VP                  R,          ,          un         V P                  '       gu   V P                  W4       V P	                  VP                  4       V P                  R7      V n        V P	                  VP                  4       V P                  R7      V n	        W3# V P                  V P                  4      pV P                  V P                  4      p\        P                  ! WPP                  V.RR7      p\        P                  ! W`P                  V.RR7      pV P                  P                  4       ^8X  dû   V P                  P                  R,          ^,           V P                   8¼  dÈ   V P	                  VP                  4       V P                  R7      V n        V P	                  VP                  4       V P                  R7      V n	        \        P"                  ! . VP$                  VP&                  R7      V n        \        P"                  ! . VP$                  VP&                  R7      V n        Wx3# \        P                  ! V P                  V.RR7      V n        \        P                  ! V P                  V.RR7      V n        Wx3# )r¦   )Úaxisr§   r    r©   )rt   r·   r*   rE   Ú	_quantizeÚ
contiguousr‰  Ú_quantized_keysrŠ  Ú_quantized_valuesÚ_dequantizer=   rª   r(   r)   r¨   rŒ  r¢   r¡   rm   )	r,   r9   r:   rM   r    Údequant_keysÚdequant_valuesÚkeys_to_returnÚvalues_to_returns	   &&&*,    r#   rN   ÚQuantizedLayer.update·  s÷  € ð 	×Ò *×"2Ñ"2°2Õ"6Õ6Õð ×"×"Ð"Ø×$Ñ$ ZÔ>Ø#'§>¡>°*×2GÑ2GÓ2IÐPT×P]ÑP] >Ó#^ˆDÔ Ø%)§^¡^°L×4KÑ4KÓ4MÐTX×TcÑTc ^Ó%dˆDÔ"ØÐ+Ð+à×'Ñ'¨×(<Ñ(<Ó=ˆØ×)Ñ)¨$×*@Ñ*@ÓAˆÜŸš L·)±)¸ZÐ#HÈbÔQˆÜ Ÿ9š9 n·k±kÀ<Ð%PÐVXÔYÐØ�9‰9�=‰=‹?˜aÔ D§I¡I§O¡O°BÕ$7¸!Õ$;¸t×?SÑ?SÔ$SØ#'§>¡>°.×2KÑ2KÓ2MÐTX×TaÑTa >Ó#bˆDÔ Ø%)§^¡^Ð4D×4OÑ4OÓ4QÐX\×XgÑXg ^Ó%hˆDÔ"ÜŸš R¨z×/?Ñ/?È
×HYÑHYÔZˆDŒIÜŸ,š, r°×1AÑ1AÈ*×J[ÑJ[Ô\ˆDŒKð
 Ð/Ð/ô Ÿ	š	 4§9¡9¨jÐ"9¸rÔBˆDŒIÜŸ)š) T§[¡[°,Ð$?ÀRÔHˆDŒKàÐ/Ð/r%   c                ó   € R # r0   r   )r,   r¢   r“  s   &&&r#   r”  ÚQuantizedLayer._quantizeÜ  s   € Ù'*r%   c                ó   € R # r0   r   )r,   Úq_tensors   &&r#   r˜  ÚQuantizedLayer._dequantizeß  r^   r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   r�  â  rò   r%   c                ó   € V P                   # rô   rõ   r+   s   &r#   r\   ÚQuantizedLayer.get_seq_lengthâ  r÷   r%   )
r–  r—  r‰  rŠ  rt   r(   rˆ  r‹  rŒ  r)   ©é   r   r   é@   é€   )r2   r‡   rˆ   r‰   rŠ   r-   rN   r   r”  r˜  r\   r�   rŽ   r�   r�   s   @@r#   r…  r…  ›  sL   ù‡ € ñ	÷#õ #÷ #0ð #0ðJ Ù*ó Ø*àÙ(ó Ø(÷&÷ &ð &r%   r…  c                   óN   a a€ ] tR tRt oRV3R lV 3R llltR tR tRtVtV ;t	# )ÚQuantoQuantizedLayeriç  c          
      ó8   <€ V ^8„  d   QhRS[ RS[ RS[ RS[ RS[ /# r‡  rZ   )r?   r@   s   "€r#   rA   Ú!QuantoQuantizedLayer.__annotate__è  s=   ø€ ÷ )(ñ )(áð)(ñ ð)(ñ ð	)(ñ
 ð)(ñ ñ)(r%   c                óú  <€ \         S	V `  VVVVVR 7       \        4       '       g   \        R4      h\	        RRR7      '       d   ^ RIHpHpHp M\        R4      hV P                  R9  d   \        RV P                   24      hV P                  R9  d   \        RV P                   24      hV P                  R9  d   \        R	V P                   24      hV P                  ^8X  d   TMTV n        V! 4       V n        R
# )©rˆ  r‰  rŠ  r‹  rŒ  zžYou need to install optimum-quanto in order to use KV cache quantization with optimum-quanto backend. Please install it via  with `pip install optimum-quanto`z0.2.5Tr   )ÚMaxOptimizerÚqint2Úqint4ziYou need optimum-quanto package version to be greater or equal than 0.2.5 to use `QuantoQuantizedLayer`. zA`nbits` for `quanto` backend has to be one of [`2`, `4`] but got zE`axis_key` for `quanto` backend has to be one of [`0`, `-1`] but got zG`axis_value` for `quanto` backend has to be one of [`0`, `-1`] but got N)r8   r§  )r   r»   )r   r-   r	   ÚImportErrorr
   Úoptimum.quantor°  r±  r²  rˆ  rÞ   r‰  rŠ  ÚqtypeÚ	optimizer)
r,   rˆ  r‰  rŠ  r‹  rŒ  r°  r±  r²  r"   s
   &&&&&&   €r#   r-   ÚQuantoQuantizedLayer.__init__è  s  ø€ ô 	‰ÑØØØ!Ø%Ø+ð 	ô 	
ô +×,Ò,ÜðTóð ô ˜w°4×8Ó8ßAÒAäØ{óð ð �:‰:˜VÔ#ÜÐ`Ðae×akÑakÐ`lÐmÓnÐnà�=‰= Ô'ÜÐdÐei×erÑerÐdsÐtÓuÐuà�?‰? 'Ô)ÜØYÐZ^×ZiÑZiÐYjÐkóð ð #Ÿj™j¨Aœo‘U°5ˆŒ
Ù%›ˆŽr%   c                óž   € ^ RI Hp V P                  WP                  W P                  4      w  rEV! WP                  W$WPP                  4      pV# )r   )Úquantize_weight)r´  r¹  r¶  rµ  r‹  )r,   r¢   r“  r¹  ÚscaleÚ	zeropointÚqtensors   &&&    r#   r”  ÚQuantoQuantizedLayer._quantize  s?   € Ý2àŸ>™>¨&·*±*¸d×DUÑDUÓVÑˆÙ! &¯*©*°dÀ9×N_ÑN_Ó`ˆØˆr%   c                ó"   € VP                  4       # r0   )Ú
dequantize)r,   r¼  s   &&r#   r˜  Ú QuantoQuantizedLayer._dequantize  s   € Ø×!Ñ!Ó#Ð#r%   )r¶  rµ  r¦  ©
r2   r‡   rˆ   r‰   r-   r”  r˜  r�   rŽ   r�   r�   s   @@r#   r«  r«  ç  s   ù‡ € ÷)(õ )(òV÷$ò $r%   r«  c                   óN   a a€ ] tR tRt oRV3R lV 3R llltR tR tRtVtV ;t	# )ÚHQQQuantizedLayeri  c          
      ó8   <€ V ^8„  d   QhRS[ RS[ RS[ RS[ RS[ /# r‡  rZ   )r?   r@   s   "€r#   rA   ÚHQQQuantizedLayer.__annotate__  s=   ø€ ÷ !&ñ !&áð!&ñ ð!&ñ ð	!&ñ
 ð!&ñ ñ!&r%   c                ór  <€ \         SV `  VVVVVR 7       \        4       '       g   \        R4      hV P                  R9  d   \        RV P                   24      hV P                  R9  d   \        RV P                   24      hV P                  R9  d   \        RV P                   24      h\        V n	        R# )r¯  zYou need to install `HQQ` in order to use KV cache quantization with HQQ backend. Please install it via  with `pip install hqq`zM`nbits` for `HQQ` backend has to be one of [`1`, `2`, `3`, `4`, `8`] but got zA`axis_key` for `HQQ` backend has to be one of [`0`, `1`] but got zC`axis_value` for `HQQ` backend has to be one of [`0`, `1`] but got N)r`  r8   é   r§  é   )r   r`  )
r   r-   r   r³  rˆ  rÞ   r‰  rŠ  ÚHQQQuantizerÚ	quantizerr�  s   &&&&&&€r#   r-   ÚHQQQuantizedLayer.__init__  sÈ   ø€ ô 	‰ÑØØØ!Ø%Ø+ð 	ô 	
ô  ×!Ò!Üð@óð ð
 �:‰:˜_Ô,ÜØ_Ð`d×`jÑ`jÐ_kÐlóð ð �=‰= Ô&ÜÐ`Ðae×anÑanÐ`oÐpÓqÐqà�?‰? &Ô(ÜÐbÐcg×crÑcrÐbsÐtÓuÐuä%ˆŽr%   c           	     óî  € V P                   P                  VVV P                  P                  V P                  P                  V P
                  V P                  R 7      w  r4V P                  P                  VR&   V P                   P                  W4V P                  P                  R7       VR,          P                  VP                  4      VR&   VR,          P                  VP                  4      VR&   W43# ))r“  rm   Úcompute_dtyperˆ  Ú
group_sizerÍ  )Úmetarm   rº  Úzero)	rÊ  Úquantizer(   rm   r¡   rˆ  r‹  Úcudarh   )r,   r¢   r“  r¼  rÏ  s   &&&  r#   r”  ÚHQQQuantizedLayer._quantizeB  s¾   € ØŸ™×/Ñ/ØØØ—9‘9×#Ñ#ØŸ)™)Ÿ/™/Ø—*‘*Ø×(Ñ(ð 0ó 
‰ˆð !%§	¡	§¡ˆˆ_ÑØ�‰×Ñ˜G°t·y±y×7GÑ7GÐÔHØ˜W�×(Ñ(¨¯©Ó8ˆˆW‰Ø˜F•|—‘ w§~¡~Ó6ˆˆV‰Øˆ}Ðr%   c                óD   € Vw  r#V P                   P                  W#4      pV# r0   )rÊ  r¿  )r,   r¼  Úquant_tensorrÏ  r¢   s   &&   r#   r˜  ÚHQQQuantizedLayer._dequantizeQ  s#   € Ø$ÑˆØ—‘×*Ñ*¨<Ó>ˆØˆr%   )rÊ  r¦  rÁ  r�   s   @@r#   rÃ  rÃ    s   ù‡ € ÷!&õ !&òF÷ò r%   rÃ  c                   óÎ   a € ] tR tRt o RtRtR tR t]RV 3R lR ll4       t	]V 3R	 lR
 l4       t
]V 3R lR l4       tR tR tV 3R lR ltV 3R lR ltV 3R lR ltRtV tR# )ÚLinearAttentionCacheLayerMixiniW  zABase, abstract class for a linear attention single layer's cache.Tc                óL   € R V n         R V n        RV n        RV n        RV n        R # r'   )Úconv_statesÚrecurrent_statesÚis_conv_states_initializedÚis_recurrent_states_initializedÚhas_previous_stater+   s   &r#   r-   Ú'LinearAttentionCacheLayerMixin.__init__]  s*   € Ø04ˆÔØ59ˆÔØ*/ˆÔ'Ø/4ˆÔ,Ø"'ˆÖr%   c                ó0   € V P                   P                   # r0   r1   r+   s   &r#   r3   Ú'LinearAttentionCacheLayerMixin.__repr__d  r5   r%   Nc                ón   <€ V ^8„  d   QhRS[ P                  R,          RS[ P                  R,          RR/# ©r8   rÚ  NrÛ  r;   r<   )r?   r@   s   "€r#   rA   Ú+LinearAttentionCacheLayerMixin.__annotate__h  s8   ø€ ÷ ñ Ù Ÿ<™<¨$Õ.ðÙINÏÉÐX\ÕI\ðà	ñr%   c                ó   € R # r0   r   ©r,   rÚ  rÛ  s   &&&r#   rE   Ú2LinearAttentionCacheLayerMixin.lazy_initializationg  s   € ñ r%   c                óN   <€ V ^8„  d   QhRS[ P                  RS[ P                  /# ©r8   rÚ  r;   r<   )r?   r@   s   "€r#   rA   rä  m  s   ø€ ×OÑO©U¯\©\ÐO¹e¿l¹lÑOr%   c                ó   € R # r0   r   )r,   rÚ  s   &&r#   Úupdate_conv_stateÚ0LinearAttentionCacheLayerMixin.update_conv_statel  s   € ÙLOr%   c                óN   <€ V ^8„  d   QhRS[ P                  RS[ P                  /# ©r8   rÛ  r;   r<   )r?   r@   s   "€r#   rA   rä  p  s   ø€ ×YÑY±u·|±|ÐYÉÏÉÑYr%   c                ó   € R # r0   r   )r,   rÛ  s   &&r#   Úupdate_recurrent_stateÚ5LinearAttentionCacheLayerMixin.update_recurrent_stateo  s   € ÙVYr%   c                óÚ   € V P                   '       d#   V P                  P                  RRR7      V n        V P                  '       d%   V P                  P                  RRR7      V n        R# R# rd   )rÜ  rÚ  rh   rÝ  rÛ  r+   s   &r#   ri   Ú&LinearAttentionCacheLayerMixin.offloadr  s\   € à×*×*Ð*Ø#×/Ñ/×2Ñ2°5ÀtÐ2ÓLˆDÔØ×/×/Ð/Ø$(×$9Ñ$9×$<Ñ$<¸UÐQUÐ$<Ó$VˆDÖ!ñ 0r%   c                óš  € V P                   '       dR   V P                  P                  V P                  8w  d-   V P                  P                  V P                  RR7      V n        V P                  '       dV   V P
                  P                  V P                  8w  d/   V P
                  P                  V P                  RR7      V n        R# R# R# rl   )rÜ  rÚ  rm   rh   rÝ  rÛ  r+   s   &r#   rn   Ú'LinearAttentionCacheLayerMixin.prefetchy  s™   € à×*×*Ð*¨t×/?Ñ/?×/FÑ/FÈ$Ï+É+Ô/UØ#×/Ñ/×2Ñ2°4·;±;ÈTÐ2ÓRˆDÔØ×/×/Ð/°D×4IÑ4I×4PÑ4PÐTX×T_ÑT_Ô4_Ø$(×$9Ñ$9×$<Ñ$<¸T¿[¹[ÐW[Ð$<Ó$\ˆDÖ!ñ 5`Ñ/r%   c                ó   <€ V ^8„  d   QhRR/# rq   r   )r?   r@   s   "€r#   rA   rä  €  s   ø€ ÷ (ñ (�tñ (r%   c                óÄ   € V P                   '       d   V P                  P                  4        V P                  '       d   V P                  P                  4        RV n        R# )rs   FN)rÜ  rÚ  ru   rÝ  rÛ  rÞ  r+   s   &r#   rx   Ú$LinearAttentionCacheLayerMixin.reset€  sF   € à×*×*Ð*Ø×Ñ×"Ñ"Ô$Ø×/×/Ð/Ø×!Ñ!×'Ñ'Ô)Ø"'ˆÖr%   c                ó4   <€ V ^8„  d   QhRS[ P                  /# ©r8   r|   r}   )r?   r@   s   "€r#   rA   rä  ˆ  s   ø€ ÷ dñ d¡e×&6Ñ&6ñ dr%   c                ó:  € V P                   '       d;   V P                  P                  ^ VP                  V P                  4      4      V n        V P
                  '       d=   V P                  P                  ^ VP                  V P                  4      4      V n        R# R# ©zDReorders the cache for beam search, given the selected beam indices.N)rÜ  rÚ  r€   rh   rm   rÝ  rÛ  r�   s   &&r#   r‚   Ú,LinearAttentionCacheLayerMixin.reorder_cacheˆ  sr   € à×*×*Ð*Ø#×/Ñ/×<Ñ<¸QÀÇÁÈDÏKÉKÓ@XÓYˆDÔà×/×/Ð/Ø$(×$9Ñ$9×$FÑ$FÀqÈ(Ï+É+ÐVZ×VaÑVaÓJbÓ$cˆDÖ!ñ 0r%   c                ó    <€ V ^8„  d   QhRS[ /# ©r8   r¿   rZ   )r?   r@   s   "€r#   rA   rä  �  s   ø€ ÷ ñ ™sñ r%   c                ó   € R # r0   r   rÃ   s   &&r#   rÄ   Ú#LinearAttentionCacheLayerMixin.crop�  s   € ár%   )rÚ  rÞ  rÜ  rÝ  rÛ  r   )r2   r‡   rˆ   r‰   rŠ   r‹   r-   r3   r   rE   rë  rð  ri   rn   rx   r‚   rÄ   r�   rŽ   ©r@   s   @r#   rØ  rØ  W  s|   ø‡ € ÙKð €Nò(ò,ð ÷ñ ó ðð ßOó ØOàßYó ØYòWò]÷(ð (÷dð d÷ö r%   rØ  c                   ó|   a a€ ] tR tRt oRV3R lV 3R llltRV3R lR lltV3R lR ltV3R lR	 ltR
tVt	V ;t
# )ÚLinearAttentionLayeri•  c                ó.   <€ V ^8„  d   QhRS[ R,          /# r”   r   )r?   r@   s   "€r#   rA   Ú!LinearAttentionLayer.__annotate__–  r—   r%   c                ó$   <€ \         SV `  4        R # r0   r™   rš   s   &&€r#   r-   ÚLinearAttentionLayer.__init__–  rœ   r%   c                ón   <€ V ^8„  d   QhRS[ P                  R,          RS[ P                  R,          RR/# rã  r<   )r?   r@   s   "€r#   rA   r  ™  s8   ø€ ÷ 8ñ 8Ù Ÿ<™<¨$Õ.ð8ÙINÏÉÐX\ÕI\ð8à	ñ8r%   c                ó†  € VeÆ   VP                   VP                  uV n         V n        VP                  ^ ,          VP                  R,          uV n        V n        \
        P                  ! WP                   V P                  R7      V n        \        4       '       g*   \
        P                  P                  V P                  4       RV n        Vet   \
        P                  ! W P                   V P                  R7      V n        \        4       '       g*   \
        P                  P                  V P                  4       RV n        R # R # )Nr    Tr»   )r¡   rm   r·   r9  Úconv_kernel_sizer=   Ú
zeros_likerÚ  r   r>  r?  rÜ  rÛ  rÝ  ræ  s   &&&r#   rE   Ú(LinearAttentionLayer.lazy_initialization™  sí   € ð Ò"Ø&1×&7Ñ&7¸×9KÑ9KÐ#ˆDŒJ˜œà9D×9JÑ9JÈ1Õ9MÈ{×O`ÑO`ÐacÕOdÐ6ˆDÔ Ô!6ä$×/Ò/°Ç:Á:ÐVZ×VaÑVaÔbˆDÔä+×-Ò-Ü—‘×1Ñ1°$×2BÑ2BÔCØ.2ˆDÔ+ØÒ'ä$)×$4Ò$4Ð5EÏZÉZÐ`d×`kÑ`kÔ$lˆDÔ!ä+×-Ò-Ü—‘×1Ñ1°$×2GÑ2GÔHØ37ˆDÖ0ñ (r%   c                óN   <€ V ^8„  d   QhRS[ P                  RS[ P                  /# ré  r<   )r?   r@   s   "€r#   rA   r  ¯  s#   ø€ ÷  ñ  ©U¯\©\ð  ÉÏÉñ  r%   c                ó(  € V P                   '       g   V P                  VR7       V P                  '       g/   V P                  P	                  V4       RV n        V P                  # VP
                  R,          pW0P                  8¼  d>   V P                  P	                  VRV P                  ) R13,          4       V P                  # V P                  P                  V) RR7      pWRRV) R13&   V P                  P	                  V4       V P                  # )zÑ
Update the linear attention cache in-place, and return the necessary conv states.

Args:
    conv_states (`torch.Tensor`): The new conv states to cache.

Returns:
    `torch.Tensor`: The updated conv states.
)rÚ  T.N)Úshiftsr_  rÁ   r»   )rÜ  rE   rÞ  rÚ  rb  r·   r  ra  )r,   rÚ  r    Únum_new_tokensÚnew_conv_statess   &&,  r#   rë  Ú&LinearAttentionLayer.update_conv_state¯  s  € ð ×.×.Ð.Ø×$Ñ$°Ð$Ô=à×&×&Ð&à×Ñ×"Ñ" ;Ô/Ø&*ˆDÔ#ð ×ÑÐð )×.Ñ.¨rÕ2ˆNØ×!6Ñ!6Ô6Ø× Ñ ×&Ñ& {°3¸×9NÑ9NÐ8NÑ8PÐ3PÕ'QÔRð ×ÑÐð	 #'×"2Ñ"2×"7Ñ"7À¸ÐUWÐ"7Ó"X�Ø:E  1 ~ oÑ&6Ð 6Ñ7Ø× Ñ ×&Ñ& Ô7à×ÑÐr%   c                óN   <€ V ^8„  d   QhRS[ P                  RS[ P                  /# rî  r<   )r?   r@   s   "€r#   rA   r  Ð  s&   ø€ ÷ %ñ %±u·|±|ð %ÑRW×R^ÑR^ñ %r%   c                ó˜   € V P                   '       g   V P                  VR7       V P                  P                  V4       V P                  # )zÍ
Update the linear attention cache in-place, and return the necessary ssm states.

Args:
    smm_states (`torch.Tensor`): The new ssm states to cache.

Returns:
    `torch.Tensor`: The updated ssm states.
)rÛ  )rÝ  rE   rÛ  rb  )r,   rÛ  r    s   &&,r#   rð  Ú+LinearAttentionLayer.update_recurrent_stateÐ  sC   € ð ×3×3Ð3Ø×$Ñ$Ð6FÐ$ÔGà×Ñ×#Ñ#Ð$4Ô5Ø×$Ñ$Ð$r%   )	r  rÚ  rm   r¡   rÞ  rÜ  rÝ  r9  rÛ  r0   r   )r2   r‡   rˆ   r‰   r-   rE   rë  rð  r�   rŽ   r�   r�   s   @@r#   r  r  •  s3   ù‡ € ÷õ ÷8ò 8÷, ð  ÷B%÷ %ð %r%   r  c                   óp   a € ] tR tRt o RtRV 3R lR lltV 3R lR ltV 3R lR	 ltV 3R
 lR ltRt	V t
R# )Ú$LinearAttentionAndFullAttentionLayeriá  FNc                ó.   <€ V ^8„  d   QhRS[ R,          /# r”   r   )r?   r@   s   "€r#   rA   Ú1LinearAttentionAndFullAttentionLayer.__annotate__å  s   ø€ ÷ ,ñ ,Ñ/°$Õ6ñ ,r%   c                óZ   € \         P                  V 4       \        P                  V 4       R # r0   )r’   r-   r  )r,   r•   s   &&r#   r-   Ú-LinearAttentionAndFullAttentionLayer.__init__å  s   € Ü×Ñ˜dÔ#Ü×%Ñ% dÖ+r%   c                ó   <€ V ^8„  d   QhRR/# rq   r   )r?   r@   s   "€r#   rA   r  é  s   ø€ ÷ Eñ E°dñ Er%   c                óì   € \        V4      ^8X  d)   \        V4      ^ 8X  d   \        P                  ! V .VO5!   \        V4      ^ 8X  d,   \        V4      ^8X  d   \        P                  ! V 3/ VB  R# R# R# )r8   N)Úlenr’   rE   r  )r,   rM   r    s   &*,r#   rE   Ú8LinearAttentionAndFullAttentionLayer.lazy_initializationé  s]   € äˆt‹9˜Œ>œc &›k¨QÔ.Ü×,Ò,¨TÐ9°DÔ9ô ˆt‹9˜Œ>œc &›k¨QÔ.Ü ×4Ò4°TÑD¸VÔDñ /‰>r%   c                ó   <€ V ^8„  d   QhRR/# rq   r   )r?   r@   s   "€r#   rA   r  ò  s   ø€ ÷ !ñ !�tñ !r%   c                óZ   € \         P                  V 4       \        P                  V 4       R # r0   )r  rx   r’   r+   s   &r#   rx   Ú*LinearAttentionAndFullAttentionLayer.resetò  s   € Ü×"Ñ" 4Ô(Ü×Ñ˜4Ö r%   c                ó4   <€ V ^8„  d   QhRS[ P                  /# rú  r}   )r?   r@   s   "€r#   rA   r  ö  s   ø€ ÷ 3ñ 3¡e×&6Ñ&6ñ 3r%   c                óZ   € \         P                  W4       \        P                  W4       R# rü  )r  r‚   r’   r�   s   &&r#   r‚   Ú2LinearAttentionAndFullAttentionLayer.reorder_cacheö  s   € ä×*Ñ*¨4Ô:Ü×"Ñ" 4Ö2r%   r   r0   )r2   r‡   rˆ   r‰   r‹   r-   rE   rx   r‚   r�   rŽ   r  s   @r#   r  r  á  s4   ø‡ € à€N÷,ò ,÷Eð E÷!ð !÷3ö 3r%   r  Úfull_attentionÚsliding_attentionÚchunked_attentionÚmambaÚconvÚlinear_attentionÚmoeÚhybridc                   óú  a € ] tR tRt o RtR2V 3R lR lltR tR3V 3R lR lltR3V 3R	 lR
 lltV 3R lR lt	V 3R lR lt
V 3R lR ltV 3R lR ltV 3R lR ltR4V 3R lR lltR5V 3R lR lltV 3R lR ltR4V 3R lR lltR tV 3R lR ltV 3R  lR! ltV 3R" lR# ltV 3R$ lR% lt]V 3R& lR' l4       t]V 3R( lR) l4       t]V 3R* lR+ l4       t]V 3R, lR- l4       t]V 3R. lR/ l4       tR0 tR1tV tR# )6ÚCachei  as  
A `Cache` is mostly a list of `CacheLayerMixin` objects, one per model layer. It serves as a container for
the Cache of each layer.

Args:
    layers (`Optional`, *optional*):
        A list of pre-created `CacheLayerMixin` or `LinearAttentionCacheLayerMixin`. If omitted (`None`), then `layer_class_to_replicate`
        will be used.
    layer_class_to_replicate (`type[CacheLayerMixin | LinearAttentionCacheLayerMixin]`, *optional*):
        Only used if `layers` is omitted (`None`), in which case it will be used as the base class for each layer,
        and the layers will be added lazily as soon as `update` is called with a `layer_idx` greater than the current
        list of layers.
    offloading (`bool`, *optional*, defaults to `False`):
        Whether to perform offloading of the layers to `cpu`, to save GPU memory.
    offload_only_non_sliding (`bool`, *optional*, defaults to `True`):
        If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
        usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).
Nc                óŽ   <€ V ^8„  d   QhRS[ S[S[,          ,          R,          RS[S[S[,          ,          R,          RS[RS[/# )r8   ÚlayersNÚlayer_class_to_replicateÚ
offloadingÚoffload_only_non_sliding)Úlistr   rØ  ÚtypeÚbool)r?   r@   s   "€r#   rA   ÚCache.__annotate__'  sZ   ø€ ÷ rñ rá‘_Ñ'EÕEÕFÈÕMðrñ #'¡Ñ9WÕ'WÕ"XÐ[_Õ"_ðrñ ð	rñ
 #'ñrr%   c                ó@  € Ve   Ve   \        R4      hVf   Vf   \        R4      hVe   TM. V n        W n        W0n        V P                  '       dM   W@n        \
        '       d   \        P                  ! 4       M\        P                  P                  4       V n	        R # R # )Na  You can construct a Cache either from a list `layers` of all the predefined `CacheLayer`, or from a `layer_class_to_replicate`, in which case the Cache will append a new layer corresponding to `layer_class_to_replicate` for each new call to `update` with an idx not already in the Cache.z_You should provide exactly one of `layers` or `layer_class_to_replicate` to initialize a Cache.)
rÞ   r2  r3  r4  Úonly_non_slidingÚ#_is_torch_greater_or_equal_than_2_7r=   ÚStreamrÒ  Úprefetch_stream)r,   r2  r3  r4  r5  s   &&&&&r#   r-   ÚCache.__init__'  s”   € ð ÒÐ":Ò"FÜðqóð ð
 Š>Ð6Ò>ÜØqóð ð !'Ò 2‘f¸ˆŒØ(@Ô%Ø$ŒØ�?�?ˆ?Ø$<Ô!ß5XÓ5X¤5§<¢<¤>Ô^c×^hÑ^h×^oÑ^oÓ^qˆDÖ ñ r%   c                óN   € V P                   P                   R V P                   R2# )z(layers=Ú))r"   r2   r2  r+   s   &r#   r3   ÚCache.__repr__?  s$   € Ø—.‘.×)Ñ)Ð*¨(°4·;±;°-¸qÐAÐAr%   c                ó&   <€ V ^8„  d   QhRS[ RS[/# ©r8   Ú	layer_idxr;  ©rT   r8  )r?   r@   s   "€r#   rA   r9  B  s   ø€ ÷ .ñ .¡#ð .¹ñ .r%   c                ó  € V'       d'    WP                   VR P                  R4      ,           pMV\        V P                  4      8  d   TM^ p\
        '       d   V P                  M(\        P                  P                  V P                  4      ;_uu_ 4        V P                  V,          P                  4        RRR4       R#   \         d    T P                   P                  R4      p L�i ; i  + '       g   i     R# ; i)a  
Prefetch a given layer on its device. If `only_non_sliding` is True, it will try to prefetch only the layers
which are non-sliding. If the `layer_idx` is outside the range, this will circle back to the first layers.
Note that we use a non-default stream for this, to avoid blocking.
NF)rÕ   rf  rÞ   r  r2  r<  r>  r=   rÒ  Ústreamrn   ©r,   rE  r;  s   &&&r#   rn   ÚCache.prefetchB  sÂ   € ÷ ð9Ø%¯©¸	¸
Ð(C×(IÑ(IÈ%Ó(PÕP‘	ð
 &/´°T·[±[Ó1AÔ%A™	ÀqˆI÷ &IÓ%HˆT×!Ò!ÌeÏjÉj×N_ÑN_Ð`d×`tÑ`tÓNu×uÑuØ�K‰K˜	Õ"×+Ñ+Ô-÷ vÑuøô ô 9Ø ŸO™O×1Ñ1°%Ó8’	ð9ú÷ v×uÐuús   Š$C Â"C.Ã&C+Ã*C+Ã.C?	c                ó&   <€ V ^8„  d   QhRS[ RS[/# rD  rF  )r?   r@   s   "€r#   rA   r9  V  s   ø€ ÷ -ñ -¡ð -¹ñ -r%   c                óŽ   € V'       d   V P                   V,          '       g$   V P                  V,          P                  4        R# R# )zþ
Offload a given `layer_idx`. If `only_non_sliding` is True, it will offload `layer_idx` only if it is a
non-sliding layer. Note that we do it on the default stream, so that we ensure all earlier
computation in the layer's `update` methods are finished.
N)rÕ   r2  ri   rI  s   &&&r#   ri   ÚCache.offloadV  s0   € ÷ ! T§_¡_°Y×%?Ô%?Ø�K‰K˜	Õ"×*Ñ*Ö,ñ &@r%   c          
      ó˜   <€ V ^8„  d   QhRS[ P                  RS[ P                  RS[RS[S[ P                  S[ P                  3,          /# )r8   r9   r:   rE  r;   )r=   r>   rT   rJ   )r?   r@   s   "€r#   rA   r9  _  sG   ø€ ÷  ñ  ÙŸ,™,ð Ù6;·l±lð ÙORð á	‰u�|‰|™UŸ\™\Ð)Õ	*ñ r%   c                óH  € V P                   eF   \        V P                  4      V8:  d,   V P                  P                  V P                  4       4       KE  V P                  '       df   \
        P                  P                  VP                  4      P                  V P                  4       V P                  V^,           V P                  4       V P                  V,          P                  ! W.VO5/ VB w  rgV P                  '       d   V P                  W0P                  4       Wg3# )a‰  
Updates the cache with the new `key_states` and `value_states` for the layer `layer_idx`.

Parameters:
    key_states (`torch.Tensor`):
        The new key states to cache.
    value_states (`torch.Tensor`):
        The new value states to cache.
    layer_idx (`int`):
        The index of the layer to cache the states for.

Return:
    A tuple containing the updated key and value states.
)r3  r  r2  Úappendr4  r=   rÒ  Údefault_streamrm   Úwait_streamr>  rn   r;  rN   ri   )r,   r9   r:   rE  rM   r    r(   r)   s   &&&&*,  r#   rN   ÚCache.update_  sÐ   € ð$ ×(Ñ(Ò4Ü�d—k‘kÓ" iÔ/Ø—‘×"Ñ" 4×#@Ñ#@Ó#BÖCà�?�?ˆ?ä�J‰J×%Ñ% j×&7Ñ&7Ó8×DÑDÀT×EYÑEYÔZØ�M‰M˜) a�-¨×)>Ñ)>Ô?à—{‘{ 9Õ-×4Ò4°ZÐ_ÐPTÒ_ÐX^Ñ_‰ˆà�?�?ˆ?Ø�L‰L˜×$9Ñ$9Ô:àˆ|Ðr%   c                óT   <€ V ^8„  d   QhRS[ P                  RS[RS[ P                  /# )r8   rÚ  rE  r;   ©r=   r>   rT   )r?   r@   s   "€r#   rA   r9  �  s-   ø€ ÷ ñ ©U¯\©\ð Ácð ÑX]×XdÑXdñ r%   c                ó²   € \        V P                  V,          \        4      '       g   \        R4      hV P                  V,          P                  ! V3/ VB pV# )a#  
Updates the cache with the new `conv_states` for the layer `layer_idx`.

Parameters:
    conv_states (`torch.Tensor`):
        The new conv states to cache.
    layer_idx (`int`):
        The index of the layer to cache the states for.

Return:
    `torch.Tensor`: The updated conv states.
ú?Cannot call `update_conv_state` on a non-LinearAttention layer!)rw   r2  rØ  rÞ   rë  )r,   rÚ  rE  r    s   &&&,r#   rë  ÚCache.update_conv_state�  sK   € ô ˜$Ÿ+™+ iÕ0Ô2P×QÒQÜÐ^Ó_Ð_Ø—k‘k )Õ,×>Ò>¸{ÑUÈfÑUˆØÐr%   c                óT   <€ V ^8„  d   QhRS[ P                  RS[RS[ P                  /# )r8   rÛ  rE  r;   rU  )r?   r@   s   "€r#   rA   r9  •  s.   ø€ ÷  ñ  ±u·|±|ð  ÑPSð  Ñbg×bnÑbnñ  r%   c                ó²   € \        V P                  V,          \        4      '       g   \        R4      hV P                  V,          P                  ! V3/ VB pV# )a%  
Updates the cache with the new `recurrent_states` for the layer `layer_idx`.

Parameters:
    smm_states (`torch.Tensor`):
        The new ssm states to cache.
    layer_idx (`int`):
        The index of the layer to cache the states for.

Return:
    `torch.Tensor`: The updated ssm states.
rW  )rw   r2  rØ  rÞ   rð  )r,   rÛ  rE  r    s   &&&,r#   rð  ÚCache.update_recurrent_state•  sN   € ô ˜$Ÿ+™+ iÕ0Ô2P×QÒQÜÐ^Ó_Ð_ØŸ;™; yÕ1×HÒHÐIYÑdÐ]cÑdÐØÐr%   c                óT   <€ V ^8„  d   QhRS[ P                  RS[RS[ P                  /# )r8   r  rE  r;   rU  )r?   r@   s   "€r#   rA   r9  ©  s2   ø€ ÷ Iñ I±·±ð IÉ#ð IÑRW×R^ÑR^ñ Ir%   c           	     óø   € \        V P                  V,          R4      '       g7   \        RV R\        V P                  V,          4      P                   R24      hV P                  V,          P                  V4      # )aa  
Updates the indexer key cache for layer `layer_idx`.

Parameters:
    indexer_key_states (`torch.Tensor`):
        The new indexer key states to cache, shape `[batch_size, seq_len, index_head_dim]`.
    layer_idx (`int`):
        The index of the layer to cache the states for.

Return:
    `torch.Tensor`: The updated indexer key states (full cache).
r  z&Cannot call `update_indexer` on layer z which is a zY; it has no indexer key cache (expected a `DynamicIndexedLayer` or `StaticIndexedLayer`).)rv   r2  rÞ   r7  r2   r  )r,   r  rE  s   &&&r#   r  ÚCache.update_indexer©  st   € ô �t—{‘{ 9Õ-Ð/?×@Ò@ÜØ8¸¸À<Ü˜Ÿ™ IÕ.Ó/×8Ñ8Ð9ð :NðOóð ð
 �{‰{˜9Õ%×4Ñ4Ð5GÓHÐHr%   c          
      ó    <€ V ^8„  d   QhRS[ RS[ S[S[ ,          ,          RS[ S[S[ ,          ,          RS[P                  RS[P                  /# )r8   Ú
batch_sizer:  Úhead_dimr¡   rm   )rT   r6  r=   r¡   rm   )r?   r@   s   "€r#   rA   r9  ¾  s\   ø€ ÷ !Fñ !Fáð!Fñ ™™c�•?ð!Fñ ™™S�	•/ð	!Fñ
 �{‰{ð!Fñ —‘ñ!Fr%   c                ó†  € \        V\        4      '       d   V.\        V 4      ,          p\        V\        4      '       d   V.\        V 4      ,          p\        V4      \        V P                  4      8w  d/   \	        R\        V4       R\        V P                  4       R24      h\        V4      \        V P                  4      8w  d/   \	        R\        V4       R\        V P                  4       R24      h\        V P                  W#4       F2  w  rgp\        P                  ! W^ V3WER7      p	VP                  W™4       K4  	  R# )z¸
Initialize all the layers in advance (it's otherwise lazily initialized on the first `update` call).
This is useful for our `export` recipes, as `export` needs everything in advance.
z,`num_head` was provided as a list of length z, but the Cache currently has z layersz,`head_dim` was provided as a list of length r    N)	rw   rT   r  r2  rÞ   Úzipr=   r=  rE   )
r,   r`  r:  ra  r¡   rm   ÚlayerÚlayer_num_headsÚlayer_head_dimÚfake_kv_tensors
   &&&&&&    r#   Úearly_initializationÚCache.early_initialization¾  s%  € ô �i¤×%Ò%Ø"˜¤c¨$£iÕ/ˆIÜ�h¤×$Ò$Ø �z¤C¨£IÕ-ˆHäˆy‹>œS §¡Ó-Ô-ÜØ>¼sÀ9»~Ð>NÐNlÔmpÐqu×q|Ñq|Óm}Ðl~ð  Fð  Góð ô ˆx‹=œC §¡Ó,Ô,ÜØ>¼sÀ9»~Ð>NÐNlÔmpÐqu×q|Ñq|Óm}Ðl~ð  Fð  Góð ô 7:¸$¿+¹+ÀyÖ6[Ñ2ˆE Nô #Ÿ[š[¨*ÀqÈ.Ð)YÐafÔvˆNà×%Ñ% nÖEó 7\r%   c                ó&   <€ V ^8„  d   QhRS[ RS[ /# ©r8   rE  r;   rZ   )r?   r@   s   "€r#   rA   r9  á  s   ø€ ÷ 7ñ 7©ð 7±Cñ 7r%   c                ó|  a € V\        S P                  4      8¼  d   ^ # \        S P                  V,          \        4      '       g?   V^ 8w  d   \	        RV R24      h \        V 3R l\        \        S 4      4       4       4      pS P                  V,          P                  4       #   \         d    \	        R4      hi ; i)z=Returns the sequence length of the cache for the given layer.z+You called `get_seq_length` on layer index úR, but this layer is a LinearAttention layer, which does not track sequence length.c              3   óz   <"  € T F0  p\        SP                  V,          \        4      '       g   K,  Vx € K2  	  R # 5ir0   ©rw   r2  r   ©Ú.0Úidxr,   s   & €r#   Ú	<genexpr>Ú'Cache.get_seq_length.<locals>.<genexpr>ð  ó)   øé € Ð rÑ0@¨ÄJÈtÏ{É{Ð[^ÕO_Ôap×Dq§¢Ó0@ùó   ƒ);±
;z{`get_seq_length` can only be called on Attention layers, and the current Cache seem to only contain LinearAttention layers.)	r  r2  rw   r   rÞ   ÚnextÚrangeÚStopIterationr\   ©r,   rE  s   f&r#   r\   ÚCache.get_seq_lengthá  s°   ø€ àœ˜DŸK™KÓ(Ô(Ùô ˜$Ÿ+™+ iÕ0´/×BÒBà˜AŒ~Ü ØAÀ)Àð M6ð 6óð ðä Ô r´´c¸$³iÔ0@Ó rÓr�	ð �{‰{˜9Õ%×4Ñ4Ó6Ð6øô !ô Ü ð.óð ðús   Á'B$ Â$B;c                ó4   <€ V ^8„  d   QhRS[ R,          RS[/# )r8   rE  Nr;   rF  )r?   r@   s   "€r#   rA   r9  ù  s   ø€ ÷ 9ñ 9©C°$­Jð 9Á$ñ 9r%   c                óŠ  a € Ve   V\        S P                  4      8¼  d   R# Vf3    \        V 3R l\        \        S 4      ^,
          RR4       4       4      pM6\        S P                  V,          \        4      '       g   \        RV R24      hS P                  V,          P                  #   \         d    \        R4      hi ; i)zYReturns whether the LinearAttention layer at index `layer_idx` has previous state or not.Fc              3   óz   <"  € T F0  p\        SP                  V,          \        4      '       g   K,  Vx € K2  	  R # 5ir0   )rw   r2  rØ  rp  s   & €r#   rs  Ú+Cache.has_previous_state.<locals>.<genexpr>  s.   øé € ð !á;˜Ü! $§+¡+¨cÕ"2Ô4R×S÷ ’CÛ;ùrv  z`has_previous_state` can only be called on LinearAttention layers, and the current Cache seem to only contain Attention layers.z/You called `has_previous_state` on layer index zJ, but this layer is an Attention layer, which does not support calling it.r»   )	r  r2  rw  rx  ry  rÞ   rw   rØ  rÞ  rz  s   f&r#   rÞ  ÚCache.has_previous_stateù  sÃ   ø€ àÒ  Y´#°d·k±kÓ2BÔ%BÙð Òð
Ü ô !ä$¤S¨£Y°¥]°B¸Ô;ó!ó ‘	ô ˜DŸK™K¨	Õ2Ô4R×SÒSÜØAÀ)Àð M/ð /óð ð
 �{‰{˜9Õ%×8Ñ8Ð8øô !ô Ü ð5óð ðús   §0B+ Â+Cc                óB   <€ V ^8„  d   QhRS[ RS[ RS[S[ S[ 3,          /# ©r8   rR   rE  r;   rS   )r?   r@   s   "€r#   rA   r9    s/   ø€ ÷ Cñ C©3ð C¹3ð CÁ5ÉÉcÈÅ?ñ Cr%   c                ó‚  a € V\        S P                  4      8¼  d   V^ 3# \        S P                  V,          \        4      '       g?   V^ 8w  d   \	        RV R24      h \        V 3R l\        \        S 4      4       4       4      pS P                  V,          P                  V4      #   \         d    \	        R4      hi ; i)z÷
Return a tuple (kv_length, kv_offset) corresponding to the length and offset that will be returned for
the given layer at `layer_idx`.
The masks are then prepared according to the given lengths (kv_length, kv_offset) and patterns for each layer.
z+You called `get_mask_sizes` on layer index rm  c              3   óz   <"  € T F0  p\        SP                  V,          \        4      '       g   K,  Vx € K2  	  R # 5ir0   ro  rp  s   & €r#   rs  Ú'Cache.get_mask_sizes.<locals>.<genexpr>(  ru  rv  z{`get_mask_sizes` can only be called on Attention layers, and the current Cache seem to only contain LinearAttention layers.)	r  r2  rw   r   rÞ   rw  rx  ry  rV   ©r,   rR   rE  s   f&&r#   rV   ÚCache.get_mask_sizes  s»   ø€ ð œ˜DŸK™KÓ(Ô(Ø �?Ð"ô ˜$Ÿ+™+ iÕ0´/×BÒBà˜AŒ~Ü ØAÀ)Àð M6ð 6óð ðä Ô r´´c¸$³iÔ0@Ó rÓr�	ð �{‰{˜9Õ%×4Ñ4°\ÓBÐBøô !ô Ü ð.óð ðús   Á'B' Â'B>c                ó&   <€ V ^8„  d   QhRS[ RS[ /# rk  rZ   )r?   r@   s   "€r#   rA   r9  1  s   ø€ ÷ <ñ <©Sð <¹ñ <r%   c                ó|   € V\        V P                  4      8¼  d   R# V P                  V,          P                  4       # )zaReturns maximum sequence length of the cache object. Dynamic caches do not have a maximum length.r»   )r  r2  ra   rz  s   &&r#   ra   ÚCache.get_max_cache_shape1  s2   € ð œ˜DŸK™KÓ(Ô(ØˆIØ�{‰{˜9Õ%×9Ñ9Ó;Ð;r%   c                ó’   € \        \        V P                  4      4       F$  pV P                  V,          P                  4        K&  	  R# )z$Recursively reset all layers tensorsN)rx  r  r2  rx   rz  s   & r#   rx   ÚCache.reset9  s/   € äœs 4§;¡;Ó/Ö0ˆIØ�K‰K˜	Õ"×(Ñ(Ö*ó 1r%   c                ó4   <€ V ^8„  d   QhRS[ P                  /# rú  r}   )r?   r@   s   "€r#   rA   r9  >  ó   ø€ ÷ ;ñ ;¡e×&6Ñ&6ñ ;r%   c                ó”   € \        \        V P                  4      4       F%  pV P                  V,          P                  V4       K'  	  R# )z!Reorder the cache for beam searchN)rx  r  r2  r‚   )r,   r|   rE  s   && r#   r‚   ÚCache.reorder_cache>  s1   € äœs 4§;¡;Ó/Ö0ˆIØ�K‰K˜	Õ"×0Ñ0°Ö:ó 1r%   c                ó    <€ V ^8„  d   QhRS[ /# rÿ  rZ   )r?   r@   s   "€r#   rA   r9  C  s   ø€ ÷ 4ñ 4™sñ 4r%   c                ó”   € \        \        V P                  4      4       F%  pV P                  V,          P                  V4       K'  	  R# )z"Crop the cache to the given lengthN)rx  r  r2  rÄ   )r,   r¿   rE  s   && r#   rÄ   Ú
Cache.cropC  s1   € äœs 4§;¡;Ó/Ö0ˆIØ�K‰K˜	Õ"×'Ñ'¨
Ö3ó 1r%   c                ó    <€ V ^8„  d   QhRS[ /# ©r8   rÈ   rZ   )r?   r@   s   "€r#   rA   r9  H  s   ø€ ÷ Dñ D©sñ Dr%   c                ó”   € \        \        V P                  4      4       F%  pV P                  V,          P                  V4       K'  	  R# )zRepeat and interleave the cacheN)rx  r  r2  rÌ   )r,   rÈ   rE  s   && r#   rÌ   ÚCache.batch_repeat_interleaveH  s1   € äœs 4§;¡;Ó/Ö0ˆIØ�K‰K˜	Õ"×:Ñ:¸7ÖCó 1r%   c                ó4   <€ V ^8„  d   QhRS[ P                  /# ©r8   rÐ   r<   )r?   r@   s   "€r#   rA   r9  M  s   ø€ ÷ Añ A©E¯L©Lñ Ar%   c                ó”   € \        \        V P                  4      4       F%  pV P                  V,          P                  V4       K'  	  R# )zSelect indices from the cacheN)rx  r  r2  rÓ   )r,   rÐ   rE  s   && r#   rÓ   ÚCache.batch_select_indicesM  s1   € äœs 4§;¡;Ó/Ö0ˆIØ�K‰K˜	Õ"×7Ñ7¸Ö@ó 1r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   r9  S  s   ø€ ÷ ñ ¡ñ r%   c                ó´   € V P                    Uu. uF  qP                  NK  	  pp\        \        V4      4      ^8”  d   \	        RV 24      hV^ ,          # u upi )z*Return the maximum batch size of the cachez0Max batch size is not consistent across layers: )r2  r9  r  ÚsetrÞ   ©r,   rd  r)   s   &  r#   r9  ÚCache.max_batch_sizeR  sU   € ð 59·K²KÓ@±K¨5×&Ô&±KˆÐ@ÜŒs�6‹{Ó˜aÔÜÐOÐPVÈxÐXÓYÐYØ�a�yÐùò As   �Ac                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   r9  [  s   ø€ ÷ ñ ™sñ r%   c                ój   € V P                    Uu. uF  qP                  NK  	  pp\        V4      # u upi )z,Return the maximum cache length of the cache)r2  r1  rî   rŸ  s   &  r#   r1  ÚCache.max_cache_lenZ  s0   € ð 48·;²;Ó?±;¨%×%Ô%±;ˆÐ?Ü�6‹{Ðùò @s   �0c                ó    <€ V ^8„  d   QhRS[ /# rY   ©r8  )r?   r@   s   "€r#   rA   r9  a  s   ø€ ÷ Bñ B¡ñ Br%   c                óÊ   € \        V P                  4      ^ 8X  d   R# \        ;QJ d&    R V P                   4       F  '       d   K   R# 	  R# ! R V P                   4       4      # )z&Return whether the cache is compilableFc              3   ó8   "  € T F  qP                   x € K  	  R # 5ir0   )r‹   ©rq  rd  s   & r#   rs  Ú'Cache.is_compileable.<locals>.<genexpr>f  s   é € ÐA±[¨E×'Ö'³[ùó   ‚T©r  r2  Úallr+   s   &r#   r‹   ÚCache.is_compileable`  sI   € ô ˆt�{‰{Ó˜qÔ Ùß‹sÑA°T·[²[ÓA�sŒsÐAŠsÐAˆsÑA°T·[²[ÓAÓAÐAr%   c                ó    <€ V ^8„  d   QhRS[ /# rY   r¥  )r?   r@   s   "€r#   rA   r9  i  s   ø€ ÷ [ñ [¡ñ [r%   c                óÒ   € \        V P                  4      ^ 8„  ;'       dI    \        ;QJ d&    R V P                   4       F  '       d   K   R# 	  R# ! R V P                   4       4      # )z,Return whether the cache data is initializedc              3   ó8   "  € T F  qP                   x € K  	  R # 5ir0   )r*   r¨  s   & r#   rs  Ú'Cache.is_initialized.<locals>.<genexpr>k  s   é € Ð+ZÉkÀU×,@Ö,@Ëkùrª  FTr«  r+   s   &r#   r*   ÚCache.is_initializedh  sK   € ô �4—;‘;Ó !Ñ#×ZÐZ¯«Ñ+ZÈdÏkÊkÓ+Z¯¬ÐZªÐZ¨Ñ+ZÈdÏkÊkÓ+ZÓ(ZÐZr%   c                ó0   <€ V ^8„  d   QhRS[ S[,          /# rY   )r6  r8  )r?   r@   s   "€r#   rA   r9  n  s   ø€ ÷ Nñ N™D¡�Jñ Nr%   c                óZ   € V P                    Uu. uF  p\        VRR4      NK  	  up# u upi )z9Return whether the layers of the cache are sliding windowrÕ   F)r2  rß   ©r,   rd  s   & r#   rÕ   ÚCache.is_slidingm  s+   € ð BFÇÂÓMÁ¸”˜˜|¨UÖ3ÁÑMÐMùÒMs   �(c                ó,   € \        V P                  4      # )z>
This value corresponds to the number of layers in the model.
)r  r2  r+   s   &r#   Ú__len__ÚCache.__len__r  s   € ô �4—;‘;ÓÐr%   )r3  r2  r4  r;  r>  )NNFT)T©r   r0   ) r2   r‡   rˆ   r‰   rŠ   r-   r3   rn   ri   rN   rë  rð  r  rh  r\   rÞ  rV   ra   rx   r‚   rÄ   rÌ   rÓ   Úpropertyr9  r1  r‹   r*   rÕ   r¸  r�   rŽ   r  s   @r#   r0  r0    s7  ø‡ € ñ÷&rò rò0B÷.ò .÷(-ò -÷ ð  ÷Dð ÷( ð  ÷(Ið I÷*!Fð !F÷F7ò 7÷09ò 9÷4Cð C÷<<ò <ò+÷
;ð ;÷
4ð 4÷
Dð D÷
Að Að
 ÷ó ðð ÷ó ðð
 ÷Bó ðBð ÷[ó ð[ð ÷Nó ðN÷ ð  r%   r0  c                   óL   a a€ ] tR tRt oRtRV3R lV 3R llltR tRtVtV ;t	# )ÚDynamicCachei{  a¦	  
A cache that grows dynamically as more tokens are generated. This is the default for generative models.
It stores the key and value states as a list of `CacheLayer`, one for each layer. The expected shape for each tensor
in the `CacheLayer`s is `[batch_size, num_heads, seq_len, head_dim]`.
If a config is passed, it will additionally check for sliding or hybrid cache structure, greatly reducing the
memory requirement of the cached tensors to `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.

See `Cache` for details on common methods that are implemented by all cache classes.

Args:
    ddp_cache_data (`Iterable[tuple[torch.Tensor, torch.Tensor]]`, *optional*):
        It was originally added for compatibility with `torch.distributed` (DDP). In a nutshell, it is
        `map(gather_map, zip(*caches))`, i.e. each item in the iterable contains the key and value states
        for a layer gathered across replicas by torch.distributed (shape=[global batch size, num_heads, seq_len, head_dim]).
        Note: it needs to be the 1st arg as well to work correctly
    config (`PreTrainedConfig`, *optional*):
        The config of the model for which this Cache will be used. If passed, it will be used to check for sliding
        or hybrid layer structure, greatly reducing the memory requirement of the cached tensors to
        `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.
    offloading (`bool`, *optional*, defaults to `False`):
        Whether to perform offloading of the layers to `cpu`, to save GPU memory.
    offload_only_non_sliding (`bool`, *optional*, defaults to `False`):
        If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
        usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).

Example:

```python
>>> from transformers import AutoTokenizer, AutoModelForCausalLM, DynamicCache

>>> model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2-0.5B-Instruct")
>>> tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen2-0.5B-Instruct")

>>> inputs = tokenizer(text="My name is Qwen2", return_tensors="pt")

>>> # Prepare a cache class and pass it to model's forward
>>> past_key_values = DynamicCache(config=model.config)
>>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
>>> outputs.past_key_values # access cache filled with key/values from generation
```
c                ó”   <€ V ^8„  d   QhRS[ S[S[P                  R,          R3,          ,          R,          RS[R,          RS[RS[/# )r8   Úddp_cache_dataN.r•   r4  r5  )r   rJ   r=   r>   r   r8  )r?   r@   s   "€r#   rA   ÚDynamicCache.__annotate__¦  s[   ø€ ÷ 9vñ 9vá ¡¡u§|¡|°dÕ':¸CÐ'?Õ!@ÕAÀDÕHð9vñ ! 4Õ'ð9vñ ð	9vñ
 #'ñ9vr%   c                ó´  <€ . pVeè   VP                  RR7      p\        VRR 4      ;'       g    \        VRR 4      p\        VRR 4      pVfG   . p\        VP                  4       F+  p	Ve   VP	                  R4       K  VP	                  R4       K-  	  \        VR4      '       d   VR VP                  )  pV F4  p
\        P                  V
\        4      pVP	                  V! V4      4       K6  	  Ve­   \        V4       F�  w  rÍVfl   \        V4      ^8X  d
   V^,          MR pVe4   V^ ,          P                  4       pVP	                  \        VR	7      4       MVP	                  \        4       4       W\,          P                  V^ ,          V^,          4      w   p	KŸ  	  \        V4      ^ 8X  d   \        SV `A  \        VVR
7       R # \        SV `A  WSVR7       R # )NT©ÚdecoderrÙ   rÜ   Úlayer_typesr(  r'  Únum_kv_shared_layersrû   )r3  r4  r5  ©r2  r4  r5  )Úget_text_configrß   rx  Únum_hidden_layersrP  rv   rÅ  r   r   r’   Ú	enumerater  Úitemr×   rN   r   r-   )r,   r¿  r•   r4  r5  r2  Údecoder_configrÙ   rÄ  rz  r   Ú	cache_clsrE  Úkv_and_optional_slidingÚsliding_window_tensorr"   s   &&&&&          €r#   r-   ÚDynamicCache.__init__¦  sÓ  ø€ ð ˆàÒØ#×3Ñ3¸DÐ3ÓAˆNÜ$ ^Ð5EÀtÓL÷ ð ÔPWØÐ 6¸óQˆNô " .°-ÀÓFˆKØÒ"Ø �Ü˜~×?Ñ?Ö@�AØ%Ò1Ø#×*Ñ*Ð+>Ö?à#×*Ñ*Ð+;Ö<ñ	 Aô �~Ð'=×>Ò>Ø)Ð*P¨^×-PÑ-PÐ,PÐQ�ã)�
Ü4×8Ñ8¸Ä\ÓR�	Ø—‘™i¨Ó7Ö8ñ *ð
 Ò%ä6?ÀÖ6OÑ2�	à’>ô KNÐNeÓJfÐjkÔJkÐ,CÀAÖ,FÐquÐ)à,Ò8à)>¸qÕ)A×)FÑ)FÓ)H˜ØŸ™Ô&?È~Ô&^Õ_àŸ™¤l£nÔ5àÕ(×/Ñ/Ð0GÈÕ0JÐLcÐdeÕLfÓg‘�’1ñ 7Pô" ˆv‹;˜!ÔÜ‰GÑÜ)5Ø%Ø)Að ö ô ‰GÑ FÐ\tÐÖur%   c              #  ó€   "  € V P                    F)  pVP                  VP                  \        VR R4      3x € K+  	  R# 5i)rá   N)r2  r(   r)   rß   rµ  s   & r#   Ú__iter__ÚDynamicCache.__iter__á  s3   é € Ø—[”[ˆEØ—*‘*˜eŸl™l¬G°EÐ;SÐUYÓ,ZÐZÔZó !ùs   ‚<>r   )NNFF)
r2   r‡   rˆ   r‰   rŠ   r-   rÑ  r�   rŽ   r�   r�   s   @@r#   r½  r½  {  s$   ù‡ € ñ(÷T9võ 9v÷v[ò [r%   r½  c                   óF   a a€ ] tR tRt oRtRV3R lV 3R llltRtVtV ;t# )ÚStaticCacheiæ  aD  
Static Cache class to be used with `torch.compile(model)` and `torch.export()`. It will check the `config`
for potential hybrid cache structure, and initialize each layer accordingly.

See `Cache` for details on common methods that are implemented by all cache classes.

Args:
    config (`PreTrainedConfig`):
        The config of the model for which this Cache will be used. It will be used to check for sliding
        or hybrid layer structure, and initialize each layer accordingly.
    max_cache_len (`int`):
        The maximum number of tokens that this Cache should hold.
    offloading (`bool`, *optional*, defaults to `False`):
        Whether to perform offloading of the layers to `cpu`, to save GPU memory.
    offload_only_non_sliding (`bool`, *optional*, defaults to `True`):
        If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
        usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).

Example:

```python
>>> from transformers import AutoTokenizer, AutoModelForCausalLM, StaticCache

>>> model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-2-7b-chat-hf")
>>> tokenizer = AutoTokenizer.from_pretrained("meta-llama/Llama-2-7b-chat-hf")

>>> inputs = tokenizer(text="My name is Llama", return_tensors="pt")

>>> # Prepare a cache class and pass it to model's forward
>>> # Leave empty space for 10 new tokens, which can be used when calling forward iteratively 10 times to generate
>>> max_generated_length = inputs.input_ids.shape[1] + 10
>>> past_key_values = StaticCache(config=model.config, max_cache_len=max_generated_length)
>>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
>>> outputs.past_key_values # access cache filled with key/values from generation
StaticCache()
```
c                ó2   <€ V ^8„  d   QhRS[ RS[RS[RS[/# )r8   r•   r1  r4  r5  )r   rT   r8  )r?   r@   s   "€r#   rA   ÚStaticCache.__annotate__  s9   ø€ ÷ 3rñ 3rá ð3rñ ð3rñ ð	3rñ
 #'ñ3rr%   c                óø  <€ VP                  R R7      p\        VRR4      pVf�   \        VRR4      e&   \        VP                  4       Uu. uF  pRNK  	  ppMX\        VRR4      e&   \        VP                  4       Uu. uF  pRNK  	  ppM$\        VP                  4       Uu. uF  pRNK  	  pp\        VR	^ 4      pV^ 8”  d   VRV)  p\        P                  4        U	U
u0 uF@  w  rš\        V
\        4      '       g   K  \        V
\        4      '       g   K5  V	R8w  g   K>  V	kKB  	  pp	p
. pV F¡  pVR8X  d   \        W!P                  R
7      pMoWÛ9   d   \        W!P                  R
7      pMRVR9   d   \        4       pM@V\        9   d   \        V,          ! VR7      pM VR8X  d   \        VR7      pM\!        VR7      pVP#                  V4       K£  	  \$        SV `M  WÃVR7       R# u upi u upi u upi u up
p	i )TrÂ  rÄ  NrÙ   r(  rÜ   r)  r'  rÅ  )r1  rÙ   rL  r  rÆ  )r*  r+  r,  r-  )rÇ  rß   rx  rÈ  r   Úitemsrw   r7  r   r×   rU  rÜ   rÙ   r  r   rr  r   rP  r   r-   )r,   r•   r1  r4  r5  r    rÄ  rz  rÅ  Únamer   Úsliding_layer_typesr2  r   rd  r"   s   &&&&&,         €r#   r-   ÚStaticCache.__init__  s÷  ø€ ð ×'Ñ'°Ð'Ó5ˆÜ˜f m°TÓ:ˆàÒÜ�vÐ/°Ó6ÒBÜ<AÀ&×BZÑBZÔ<[Ó\Ñ<[°qÓ2Ñ<[�Ð\�Ü˜Ð!7¸Ó>ÒJÜ<AÀ&×BZÑBZÔ<[Ó\Ñ<[°qÓ2Ñ<[�Ð\�ä9>¸v×?WÑ?WÔ9XÓYÑ9X°AÓ/Ñ9X�ÐYä& vÐ/EÀqÓIÐØ !Ô#Ø%Ð&<Ð(<Ð'<Ð=ˆKô 6×;Ñ;Ô=ô
á=‘	�Ü˜#œt×$ô ä)3°CÔ9R×)Sô àX\Ð`sÑXs÷ ˆDÙ=ð 	ñ 
ð
 ˆÛ%ˆJØÐ0Ô0ô 1Ø"/×@[Ñ@[ô‘ð Ô2Ü0¸}×]rÑ]rÔs‘àÐKÔKÜ,Ó.‘àÔ>Ô>Ü7¸
ÖCÐR_Ô`‘ØÐ:Ô:ä*¸ÔG‘ä#°-Ô@�Ø�M‰M˜%Ö ñ) &ô, 	‰Ñ ÐXpÐÖqùòM ]ùâ\ùâYùó
s*   ÁG'Á?G,Â$G1Ã"G6ÄG6ÄG6Ä"G6r   )FT©	r2   r‡   rˆ   r‰   rŠ   r-   r�   rŽ   r�   r�   s   @@r#   rÔ  rÔ  æ  s   ù‡ € ñ$÷N3r÷ 3rõ 3rr%   rÔ  c                   óF   a a€ ] tR tRt oRtRV3R lV 3R llltRtVtV ;t# )ÚQuantizedCacheiD  a2  
A quantizer cache similar to what is described in the
[KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache paper](https://huggingface.co/papers/2402.02750).
It allows the model to generate longer sequence length without allocating too much memory for keys and values
by applying quantization.
The cache has two types of storage, one for original precision and one for the
quantized cache. A `residual length` is set as a maximum capacity for the original precision cache. When the
length goes beyond maximum capacity, the original precision cache is discarded and moved into the quantized cache.
The quantization is done per-channel with a set `q_group_size` for both keys and values, in contrast to what was
described in the paper.

See `Cache` for details on common methods that are implemented by all cache classes.

Args:
    backend (`str`):
        The quantization backend to use. One of `("quanto", "hqq").
    config (`PreTrainedConfig`):
        The config of the model for which this Cache will be used.
    nbits (`int`, *optional*, defaults to 4):
        The number of bits for quantization.
    axis_key (`int`, *optional*, defaults to 0):
        The axis on which to quantize the keys.
    axis_value (`int`, *optional*, defaults to 0):
        The axis on which to quantize the values.
    q_group_size (`int`, *optional*, defaults to 64):
        Quantization is done per-channel according to a set `q_group_size` for both keys and values.
    residual_length (`int`, *optional*, defaults to 128):
        Maximum capacity for the original precision cache
c                óD   <€ V ^8„  d   QhRS[ RS[RS[RS[RS[RS[RS[/# )r8   Úbackendr•   rˆ  r‰  rŠ  r‹  rŒ  )r†   r   rT   )r?   r@   s   "€r#   rA   ÚQuantizedCache.__annotate__c  sQ   ø€ ÷ (ñ (áð(ñ !ð(ñ ð	(ñ
 ð(ñ ð(ñ ð(ñ ñ(r%   c           
     ó  <€ VR 8X  d   \         pMVR8X  d   \        pM\        RV R24      hVP                  RR7      p\	        VP
                  4       U	u. uF  p	V! W4WVV4      NK  	  p
p	\        SV `  V
R7       R# u up	i )ÚquantoÚhqqzUnknown quantization backend `Ú`TrÂ  )r2  N)r«  rÃ  rÞ   rÇ  rx  rÈ  r   r-   )r,   rà  r•   rˆ  r‰  rŠ  r‹  rŒ  Úlayer_classrz  r2  r"   s   &&&&&&&&   €r#   r-   ÚQuantizedCache.__init__c  s”   ø€ ð �hÔÜ.‰KØ˜ÔÜ+‰KäÐ=¸g¸YÀaÐHÓIÐIà×'Ñ'°Ð'Ó5ˆô ˜6×3Ñ3Ô4ó
á4�ñ ˜¨À?ÖSÙ4ð 	ð 
ô 	‰Ñ ÐÖ'ùò	
s   ÁA=r   r¦  rÜ  r�   s   @@r#   rÞ  rÞ  D  s   ù‡ € ñ÷<(÷ (õ (r%   rÞ  c                   ó  a € ] tR tRt o RtV 3R lR ltR tV 3R lR ltR tRV 3R	 lR
 llt	R t
V 3R lR ltV 3R lR ltV 3R lR ltV 3R lR ltV 3R lR ltV 3R lR ltV 3R lR lt]R 4       t]V 3R lR l4       tRtV tR# ) ÚEncoderDecoderCachei|  aQ  
Base, abstract class for all encoder-decoder caches. Can be used to hold combinations of self-attention and
cross-attention caches.

See `Cache` for details on common methods that are implemented by all cache classes.

Args:
    caches (`Iterable`):
        Usually an iterable of length 2, containing 2 `Cache` objects, the first one for self-attention, the
        second one for cross-attention. Can optionally also be an iterable of length 1, containing a
        `tuple[tuple[torch.Tensor]]` (usually used for compatibility with torch dp and ddp).

Example:

```python
>>> from transformers import AutoProcessor, AutoModelForCausalLM, DynamicCache, EncoderDecoderCache

>>> model = AutoModelForCausalLM.from_pretrained("openai/whisper-small")
>>> processor = AutoProcessor.from_pretrained("openai/whisper-small")

>>> inputs = processor(audio=YOUR-AUDIO, return_tensors="pt")

>>> # Prepare cache classes for encoder and decoder and pass it to model's forward
>>> self_attention_cache = DynamicCache(config=self.config)
>>> cross_attention_cache = DynamicCache(config=self.config)
>>> past_key_values = EncoderDecoderCache(self_attention_cache, cross_attention_cache)
>>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
>>> outputs.past_key_values # access cache filled with key/values from generation
EncoderDecoderCache()
```
c                ó   <€ V ^8„  d   QhRR/# rq   r   )r?   r@   s   "€r#   rA   Ú EncoderDecoderCache.__annotate__�  s   ø€ ÷ hñ h 4ñ hr%   c           	     óì  € \        V4      ^8X  dÓ   . . r2V^ ,           F¡  p\        V4      ^8X  d3   VP                  VR,          4       VP                  VR,          4       KE  \        V4      ^8X  d3   VP                  VR,          4       VP                  VR,          4       K‡  \        R\        V4      : RV: 24      h	  \        V4      V n        \        V4      V n        M±\        V4      ^8X  d‹   \        V^ ,          \        4      '       d   \        V^,          \        4      '       g4   \        R\        V^ ,          4      : R\        V^,          4      : 24      hV^ ,          V n        V^,          V n        M\        R	\        V4       24      h/ V n
        \        \        V P
                  4      4       F7  p\        V P
                  P                  V4      ^ 8„  4      V P                  V&   K9  	  R
# )r`  :NrÇ  N:rÇ  NNr8  :r8   NNz$Expected len(combined_cache_data) = z% to be 4 or 6.
combined_cache_data = z;One of the two arguments is not a Cache: type(caches[0]) = z, type(caches[1]) = zExpected 1 or 2 arguments, got N)r  rP  rÞ   r½  Úself_attention_cacheÚcross_attention_cacherw   r0  Ú	TypeErrorr7  Ú
is_updatedrx  r8  r\   )r,   ÚcachesÚself_attention_cache_dataÚcross_attention_cache_dataÚcombined_cache_datarE  s   &*    r#   r-   ÚEncoderDecoderCache.__init__�  s³  € äˆv‹;˜!ÔØDFÈÐ'AØ'-¨a§y yÐ#ÜÐ*Ó+¨qÔ0Ø-×4Ñ4Ð5HÈÕ5LÔMØ.×5Ñ5Ð6IÈ"Õ6MÖNäÐ,Ó-°Ô2Ø-×4Ñ4Ð5HÈÕ5LÔMØ.×5Ñ5Ð6IÈ"Õ6MÖNä$Ð'L´Ð5HÓ1IÑ0MÐMtÐ^qÑ]uÐ%vÓwÐwñ (1ô )5Ð5NÓ(OˆDÔ%Ü)5Ð6PÓ)QˆDÕ&ä�‹[˜AÔÜ˜f Q�i¬×/Ò/´zÀ&ÈÅ)ÌU×7SÒ7SÜÐ"^ÌDÐQWÐXYÕQZËOÑK_Ð_tÔbfÐgmÐnoÕgpÓbqÑauÐ vÓwÐwØ(.¨q­	ˆDÔ%Ø)/°­ˆDÕ&ô Ð>¼sÀ6»{¸mÐLÓMÐMàˆŒÜœs 4×#=Ñ#=Ó>Ö?ˆIÜ)-¨d×.HÑ.H×.WÑ.WÐXaÓ.bÐefÑ.fÓ)gˆD�O‰O˜IÓ&ó @r%   c              #  ót   "  € \        V P                  V P                  4       F  w  rW,           x € K  	  R# 5i)zuReturns tuples of style (self_attn_k, self_attn_v, self_attn_sliding, cross_attn_k, cross_attn_v, cross_attn_sliding)N)rc  rí  rî  )r,   Úself_attention_layerÚcross_attention_layers   &  r#   rÑ  ÚEncoderDecoderCache.__iter__»  s1   é € ä;>¸t×?XÑ?XÐZ^×ZtÑZtÖ;uÑ7Ð Ø&Õ>Ô>ó <vùs   ‚68c                ó    <€ V ^8„  d   QhRS[ /# rY   r…   )r?   r@   s   "€r#   rA   rë  À  s   ø€ ÷ 
ñ 
™#ñ 
r%   c                óh   € V P                   P                   R V P                   RV P                   R2# )z(self_attention_cache=z, cross_attention_cache=rA  )r"   r2   rí  rî  r+   s   &r#   r3   ÚEncoderDecoderCache.__repr__À  s;   € à�~‰~×&Ñ&Ð'Ð'=¸d×>WÑ>WÐ=XÐXpØ×)Ñ)Ð*¨!ð-ð	
r%   c                ó,   € \        V P                  4      # )z–
Support for backwards-compatible `past_key_values` length, e.g. `len(past_key_values)`. This value corresponds
to the number of layers in the model.
)r  rí  r+   s   &r#   r¸  ÚEncoderDecoderCache.__len__Æ  s   € ô
 �4×,Ñ,Ó-Ð-r%   c                ó&   <€ V ^8„  d   QhRS[ RS[ /# rk  rZ   )r?   r@   s   "€r#   rA   rë  Í  s   ø€ ÷ Cñ C©ð C±Cñ Cr%   c                ó8   € V P                   P                  V4      # )zYReturns the sequence length of the cached states. A layer index can be optionally passed.)rí  r\   rz  s   &&r#   r\   Ú"EncoderDecoderCache.get_seq_lengthÍ  s   € à×(Ñ(×7Ñ7¸	ÓBÐBr%   c                ó²   € V P                   P                  4        V P                  P                  4        V P                   F  pR V P                  V&   K  	  R# )FN)rí  rx   rî  rð  rz  s   & r#   rx   ÚEncoderDecoderCache.resetÑ  sB   € Ø×!Ñ!×'Ñ'Ô)Ø×"Ñ"×(Ñ(Ô*ØŸœˆIØ).ˆD�O‰O˜IÓ&ó )r%   c                ó4   <€ V ^8„  d   QhRS[ P                  /# rú  r}   )r?   r@   s   "€r#   rA   rë  ×  rŽ  r%   c                ór   € V P                   P                  V4       V P                  P                  V4       R# rü  )rí  r‚   rî  r�   s   &&r#   r‚   Ú!EncoderDecoderCache.reorder_cache×  s*   € à×!Ñ!×/Ñ/°Ô9Ø×"Ñ"×0Ñ0°Ö:r%   c                ó    <€ V ^8„  d   QhRS[ /# )r8   Úmethodr…   )r?   r@   s   "€r#   rA   rë  Ü  s   ø€ ÷ ñ ©#ñ r%   c           	     ó  € \        V P                  \        4      '       d!   \        V P                  \        4      '       gF   \	        R V RV P                  P                  4        RV P                  P                  4        R24      hR# )rå  z)` is only defined for dynamic cache, got z" for the self attention cache and z for the cross attention cache.N)rw   rí  r½  rî  rï  Ú__str__)r,   r  s   &&r#   Úcheck_dynamic_cacheÚ'EncoderDecoderCache.check_dynamic_cacheÜ  s}   € ä�t×0Ñ0´,×?Ò?Ü˜4×5Ñ5´|×DÒDäØ�F�8ÐDÀT×E^ÑE^×EfÑEfÓEhÐDið j'Ø'+×'AÑ'A×'IÑ'IÓ'KÐ&LÐLkðmóð ñ Er%   c                ó    <€ V ^8„  d   QhRS[ /# )r8   Úmaximum_lengthrZ   )r?   r@   s   "€r#   rA   rë  ç  s   ø€ ÷ 7ñ 7¡3ñ 7r%   c                ó†   € V P                  V P                  P                  4       V P                  P                  V4       R# )zÛ
Crop the past key values up to a new `maximum_length` in terms of tokens. `maximum_length` can also be
negative to remove `maximum_length` tokens. This is used in assisted decoding and contrastive search (on the Hub).
N)r  rÄ   r2   rí  )r,   r  s   &&r#   rÄ   ÚEncoderDecoderCache.cropç  s0   € ð
 	× Ñ  §¡×!3Ñ!3Ô4Ø×!Ñ!×&Ñ& ~Ö6r%   c                ó    <€ V ^8„  d   QhRS[ /# r•  rZ   )r?   r@   s   "€r#   rA   rë  ï  s   ø€ ÷ Dñ D©sñ Dr%   c                ó¼   € V P                  V P                  P                  4       V P                  P                  V4       V P                  P                  V4       R# )zaRepeat the cache `repeats` times in the batch dimension. Used in contrastive search (on the Hub).N)r  rÌ   r2   rí  rî  rË   s   &&r#   rÌ   Ú+EncoderDecoderCache.batch_repeat_interleaveï  sD   € à× Ñ  ×!=Ñ!=×!FÑ!FÔGØ×!Ñ!×9Ñ9¸'ÔBØ×"Ñ"×:Ñ:¸7ÖCr%   c                ó4   <€ V ^8„  d   QhRS[ P                  /# r™  r<   )r?   r@   s   "€r#   rA   rë  õ  s   ø€ ÷ Añ A©E¯L©Lñ Ar%   c                ó¼   € V P                  V P                  P                  4       V P                  P                  V4       V P                  P                  V4       R# )zeOnly keep the `indices` in the batch dimension of the cache. Used in contrastive search (on the Hub).N)r  rÓ   r2   rí  rî  rÒ   s   &&r#   rÓ   Ú(EncoderDecoderCache.batch_select_indicesõ  sD   € à× Ñ  ×!:Ñ!:×!CÑ!CÔDØ×!Ñ!×6Ñ6°wÔ?Ø×"Ñ"×7Ñ7¸Ö@r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   rZ   )r?   r@   s   "€r#   rA   rë  û  s   ø€ ÷ ?ñ ?¡Sñ ?r%   c                ó6   € V P                   P                  4       # )zKReturns the maximum sequence length (i.e. max capacity) of the cache object)rí  ra   r+   s   &r#   ra   Ú'EncoderDecoderCache.get_max_cache_shapeû  s   € à×(Ñ(×<Ñ<Ó>Ð>r%   c                óB   <€ V ^8„  d   QhRS[ RS[ RS[S[ S[ 3,          /# r‚  rS   )r?   r@   s   "€r#   rA   rë  ÿ  s/   ø€ ÷ Qñ Q©3ð Q¹3ð QÁ5ÉÉcÈÅ?ñ Qr%   c                ó8   € V P                   P                  W4      # r0   )rí  rV   r†  s   &&&r#   rV   Ú"EncoderDecoderCache.get_mask_sizesÿ  s   € Ø×(Ñ(×7Ñ7¸ÓPÐPr%   c                ó.   € V P                   P                  # r0   )rí  rÕ   r+   s   &r#   rÕ   ÚEncoderDecoderCache.is_sliding  s   € à×(Ñ(×3Ñ3Ð3r%   c                ó    <€ V ^8„  d   QhRS[ /# rY   r¥  )r?   r@   s   "€r#   rA   rë    s   ø€ ÷ 8ñ 8¡ñ 8r%   c                ó.   € V P                   P                  # r0   )rí  r‹   r+   s   &r#   r‹   Ú"EncoderDecoderCache.is_compileable  s   € à×(Ñ(×7Ñ7Ð7r%   )rî  rð  rí  Nrº  )r2   r‡   rˆ   r‰   rŠ   r-   rÑ  r3   r¸  r\   rx   r‚   r  rÄ   rÌ   rÓ   ra   rV   r»  rÕ   r‹   r�   rŽ   r  s   @r#   ré  ré  |  s°   ø‡ € ñ÷@hð hò<?÷

ð 
ò.÷Cò Cò/÷;ð ;÷
ð ÷7ð 7÷Dð D÷Að A÷?ð ?÷Qð Qð ñ4ó ð4ð ÷8ó ö8r%   ré  c                ó¬   € V ^8„  d   Qh/ ^ \         9   d   \        \        \        3,          ;R&   ^\         9   d   \        \        \        3,          ;R&   # )r8   r   r   )Ú__conditional_annotations__Údictr†   r7  )r?   s   "r#   rA   rA      s@   € × #Ñ #÷B /Ò .œ$œs¤D˜y�/Ñ .ñC $÷N 6Ò 5¤¤c¬4 i¥Ñ 5òO $r%   ).r#  Úabcr   r   Úcollections.abcr   r=   Úconfiguration_utilsr   Úutilsr   r	   r
   r   r   r   Úhqq.core.quantizer   rÉ  r<  Ú
get_loggerr2   Úloggerr   r   r   r’   r×   r  r   rU  rr  r…  r«  rÃ  rØ  r  r  rN   r0  r½  rÔ  rÞ  ré  ÚSlidingWindowCacherA   )r#  s   @r#   Ú<module>r-     sÀ  øðß #Ô #Ý $ã å 1÷÷ ñ ×ÒÝ;á&?ÀÐRVÔ&WÐ #ð 
×	Ò	˜HÓ	%€ð -/Ð Ó .ð 46Ð Ó 5ôIW�cô IWôXN4�?ô N4ôbT5 ô T5ônI@˜,ô I@ôXk"�/ô k"ô\|'˜{ô |'ô~@3˜ô @3ôFI&�\ô I&ôX4$˜>ô 4$ôn6˜ô 6ôr; Sô ;ô|I%Ð9ô I%ôX3Ð+?Àô 3ð> × Ñ à˜,ð 	Ð6ØÐ6ð 	Ð%ØÐ$ØÐ0ØÐ#àÐ6ðô÷&e ñ e ôPh[�5ô h[ôV[r�%ô [rô|5(�Uô 5(ôpL8˜%ô L8ð` !Ò r%   