+
    QV-j]'  ã                  ó   € ^ RI Ht ^ RIt^ RIHt ^ RIHtHt ^ RIt^ RI	H
t
Ht ^ RIHt ^RIHtHtHtHt ]P$                  .t]P(                  ! ]4      t]! RR	R
7      t]! RR	R
7      t]! RR	R
7      t]! RR	R
7      t]! RR	R
7      t]! RR	R
7      t]! RR	R
7      t]! RR	R
7      t]! RR	R
7      t]P@                  PC                  4       t"R t#R R R llt$ ! R R]PJ                  4      t&R R lt'RR/R R llt(R R lt)]! ]4      R 4       t*R# )!é    )ÚannotationsN)ÚCallable©Ú	lru_cacheÚwraps)Ústorage_ptrÚstorage_size)Únn)Úis_torch_greater_or_equalÚis_torch_xla_availableÚis_torchdynamo_compilingÚloggingz2.8T)Ú
accept_devz2.6z2.4z2.3z2.2z2.1z2.0z1.13z1.12c                óJ   € ^ RI Hp V! WV P                  VP                  4      # )z”
A function that calls the internal `_softmax_backward_data` PyTorch method and that adjusts the arguments according
to the torch version detected.
)Ú_softmax_backward_data)Útorchr   ÚdimÚdtype)ÚparentÚgrad_outputÚoutputr   s   &&& Úk/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/pytorch_utils.pyÚsoftmax_backward_datar   4   s   € õ -á! +°v·z±zÀ6Ç<Á<ÓPÐPó    c               ó(   € V ^8„  d   QhRRRRRRRR/# )é   Úlayerz	nn.LinearÚindexztorch.LongTensorr   ÚintÚreturn© )Úformats   "r   Ú__annotate__r#   ?   s*   € ÷ ñ ˜ið Ð0@ð Àsð ÐS\ñ r   c                óì  € VP                  V P                  P                  4      pV P                  P                  W!4      P	                  4       P                  4       pV P                  e`   V^8X  d*   V P                  P	                  4       P                  4       pM/V P                  V,          P	                  4       P                  4       p\        V P                  P                  4       4      p\        V4      WR&   \        P                  ! V^,          V^ ,          V P                  RJR7      P                  V P                  P                  4      pRVP                  n        VP                  P                  VP                  4       4       RVP                  n        V P                  eL   RVP                  n        VP                  P                  XP                  4       4       RVP                  n        V# )a|  
Prune a linear layer to keep only entries in index.

Used to remove heads.

Args:
    layer (`torch.nn.Linear`): The layer to prune.
    index (`torch.LongTensor`): The indices to keep in the layer.
    dim (`int`, *optional*, defaults to 0): The dimension on which to keep the indices.

Returns:
    `torch.nn.Linear`: The pruned layer as a new layer with `requires_grad=True`.
N)ÚbiasFT)ÚtoÚweightÚdeviceÚindex_selectÚdetachÚcloner%   ÚlistÚsizeÚlenr
   ÚLinearÚrequires_gradÚcopy_Ú
contiguous)r   r   r   ÚWÚbÚnew_sizeÚ	new_layers   &&&    r   Úprune_linear_layerr7   ?   sa  € ð �H‰H�U—\‘\×(Ñ(Ó)€EØ�‰×!Ñ! #Ó-×4Ñ4Ó6×<Ñ<Ó>€AØ‡z�zÒØ�!Œ8Ø—
‘
×!Ñ!Ó#×)Ñ)Ó+‰Aà—
‘
˜5Õ!×(Ñ(Ó*×0Ñ0Ó2ˆAÜ�E—L‘L×%Ñ%Ó'Ó(€HÜ˜“J€H�MÜ—	’	˜( 1�+ x°¥{¸¿¹È4Ð9OÔP×SÑSÐTY×T`ÑT`×TgÑTgÓh€IØ%*€I×ÑÔ"Ø×Ñ×Ñ˜1Ÿ<™<›>Ô*Ø%)€I×ÑÔ"Ø‡z�zÒØ',ˆ	�‰Ô$Ø�‰×Ñ˜QŸ\™\›^Ô,Ø'+ˆ	�‰Ô$ØÐr   c                  ó>   a € ] tR t^atRtV 3R ltR R ltR tRtV ;t	# )ÚConv1Da  
1D-convolutional layer as defined by Radford et al. for OpenAI GPT (and also used in GPT-2).

Basically works like a linear layer but the weights are transposed.

Args:
    nf (`int`): The number of output features.
    nx (`int`): The number of input features.
c                	óN  <€ \         SV `  4        Wn        W n        \        P
                  ! \        P                  ! W!4      4      V n        \        P
                  ! \        P                  ! V4      4      V n
        \        P                  P                  V P                  R R7       R# )g{®Gáz”?)ÚstdN)ÚsuperÚ__init__ÚnfÚnxr
   Ú	Parameterr   Úemptyr'   Úzerosr%   ÚinitÚnormal_)Úselfr>   r?   Ú	__class__s   &&&€r   r=   ÚConv1D.__init__l   sa   ø€ Ü‰ÑÔØŒØŒÜ—l’l¤5§;¢;¨rÓ#6Ó7ˆŒÜ—L’L¤§¢¨R£Ó1ˆŒ	Ü
�‰�‰˜Ÿ™¨ˆÖ.r   c               ó   € V ^8„  d   QhRR/# )r   r    Ústrr!   )r"   s   "r   r#   ÚConv1D.__annotate__t   s   € ÷ Bñ B˜#ñ Br   c                	ó:   € R P                   ! R/ V P                  B # )zConv1D(nf={nf}, nx={nx})r!   )r"   Ú__dict__)rE   s   &r   Ú__repr__ÚConv1D.__repr__t   s   € Ø)×0Ò0ÑA°4·=±=ÑAÐAr   c           	     	ó  € VP                  4       R R V P                  3,           p\        P                  ! V P                  VP                  RVP                  R4      4      V P                  4      pVP                  V4      pV# )Néÿÿÿÿ)r-   r>   r   Úaddmmr%   Úviewr'   )rE   ÚxÚsize_outs   && r   ÚforwardÚConv1D.forwardw   s^   € Ø—6‘6“8˜C˜R�= D§G¡G :Õ-ˆÜ�KŠK˜Ÿ	™	 1§6¡6¨"¨a¯f©f°R«jÓ#9¸4¿;¹;ÓGˆØ�F‰F�8ÓˆØˆr   )r%   r>   r?   r'   )
Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r=   rM   rU   Ú__static_attributes__Ú__classcell__)rF   s   @r   r9   r9   a   s   ø† ñõ/õB÷ð r   r9   c               ó(   € V ^8„  d   QhRRRRRRRR/# )r   Ú
forward_fnzCallable[..., torch.Tensor]Ú
chunk_sizer   Ú	chunk_dimr    útorch.Tensorr!   )r"   s   "r   r#   r#   ~   s6   € ÷ K&ñ K&Ø+ðK&àðK&ð ðK&ð
 ñK&r   c                ó²  a aa	€ \        V4      ^ 8”  g   Q V R24       h\        \        P                  ! S 4      P                  4      pV\        V4      8w  d   \	        RV R\        V4       R24      hV^ 8”  EdZ   V^ ,          P
                  S,          pV F=  pVP
                  S,          V8w  g   K  \	        RV RVP
                  S,           24      h	  V^ ,          P
                  S,          V,          ^ 8w  d*   \	        RV^ ,          P
                  S,           RV 24      hV^ ,          P
                  S,          V,          o	\        ;QJ d    . VV	3R	 lV 4       F  NK  	  5M! VV	3R	 lV 4       4      p\        ;QJ d    . V 3R
 l\        V!   4       F  NK  	  5M! V 3R
 l\        V!   4       4      p\        P                  ! VSR7      # S ! V!  # )aö  
This function chunks the `input_tensors` into smaller input tensor parts of size `chunk_size` over the dimension
`chunk_dim`. It then applies a layer `forward_fn` to each chunk independently to save memory.

If the `forward_fn` is independent across the `chunk_dim` this function will yield the same result as directly
applying `forward_fn` to `input_tensors`.

Args:
    forward_fn (`Callable[..., torch.Tensor]`):
        The forward function of the model.
    chunk_size (`int`):
        The chunk size of a chunked tensor: `num_chunks = len(input_tensors[0]) / chunk_size`.
    chunk_dim (`int`):
        The dimension over which the `input_tensors` should be chunked.
    input_tensors (`tuple[torch.Tensor]`):
        The input tensors of `forward_fn` which will be chunked

Returns:
    `torch.Tensor`: A tensor with the same shape as the `forward_fn` would have given if applied`.


Examples:

```python
# rename the usual forward() fn to forward_chunk()
def forward_chunk(self, hidden_states):
    hidden_states = self.decoder(hidden_states)
    return hidden_states


# implement a chunked forward function
def forward(self, hidden_states):
    return apply_chunking_to_forward(self.forward_chunk, self.chunk_size_lm_head, self.seq_len_dim, hidden_states)
```z" has to be a tuple/list of tensorszforward_chunk_fn expects z arguments, but only z input tensors are givenz/All input tenors have to be of the same shape: z, found shape zThe dimension to be chunked z( has to be a multiple of the chunk size c              3  óH   <"  € T F  qP                  SSR 7      x € K  	  R# 5i)©r   N)Úchunk)Ú.0Úinput_tensorra   Ú
num_chunkss   & €€r   Ú	<genexpr>Ú,apply_chunking_to_forward.<locals>.<genexpr>Ã   s%   øé € Ð$uÑgtÐWc×%7Ñ%7¸
È	Ð%7×%RÐ%RÓgtùs   ƒ"c              3  ó0   <"  € T F  pS! V!  x € K  	  R # 5i©Nr!   )rg   Úinput_tensors_chunkr_   s   & €r   rj   rk   Å   s   øé € ÐuÑZtÐCV™jÐ*=Ö>ÓZtùs   ƒre   )
r.   ÚinspectÚ	signatureÚ
parametersÚ
ValueErrorÚshapeÚtupleÚzipr   Úcat)
r_   r`   ra   Úinput_tensorsÚnum_args_in_forward_chunk_fnÚtensor_shaperh   Úinput_tensors_chunksÚoutput_chunksri   s
   f&f*     @r   Úapply_chunking_to_forwardr|   ~   sÈ  ú€ ôR ˆ}Ó Ô!ÐW m _Ð4VÐ#WÓWÐ!ô $'¤w×'8Ò'8¸Ó'D×'OÑ'OÓ#PÐ Ø#¤s¨=Ó'9Ô9ÜØ'Ð(DÐ'EÐEZÔ[^Ð_lÓ[mÐZnð o ð  ó
ð 	
ð
 �A…~Ø$ QÕ'×-Ñ-¨iÕ8ˆÛ)ˆLØ×!Ñ! )Õ,°Ö<Ü ØEÀlÀ^ð T#Ø#/×#5Ñ#5°iÕ#@Ð"AðCóð ñ *ð ˜Õ×!Ñ! )Õ,¨zÕ9¸QÔ>ÜØ.¨}¸QÕ/?×/EÑ/EÀiÕ/PÐ.Qð RØ"�|ð%óð ð
 # 1Õ%×+Ñ+¨IÕ6¸*ÕDˆ
÷  %œuÕ$uÑgtÓ$uŸu™uÕ$uÑgtÓ$uÓuÐçœÔuÔZ]Ð_sÒZtÓuŸ™ÔuÔZ]Ð_sÒZtÓuÓuˆä�yŠy˜¨IÔ6Ð6á�}Ñ%Ð%r   Úindexingc               ó$   € V ^8„  d   QhRRRRRR/# )r   Útensorsz!torch.Tensor | list[torch.Tensor]r}   z
str | Noner    ztuple[torch.Tensor, ...]r!   )r"   s   "r   r#   r#   Ì   s#   € ÷ 7ñ 7Ð8ð 7ÀJð 7ÐZrñ 7r   c                ó.   € \         P                  ! VRV / # )z«
Wrapper around torch.meshgrid to avoid warning messages about the introduced `indexing` argument.

Reference: https://pytorch.org/docs/1.13/generated/torch.meshgrid.html
r}   )r   Úmeshgrid)r}   r   s   $*r   r�   r�   Ì   s   € ô �>Š>˜7Ð6¨XÑ6Ð6r   c               ó    € V ^8„  d   QhRRRR/# )r   Útensorrb   r    ztuple[torch.device, int, int]r!   )r"   s   "r   r#   r#   Õ   s   € ÷ :ñ :˜lð :Ð/Lñ :r   c                óÐ  € \         '       dn   \        R4      '       d]   ^ RIHp \	        W4      '       dF   V P                  4       pV P                  VP                  4       P                  4       V P                  3# V P                  P                  R8X  d1   \        4       '       d!   ^ RIpVP                  P                  V 4      pM\        V 4      pV P                  V\!        V 4      3# )a  
Unique identifier to a tensor storage. Multiple different tensors can share the same underlying storage. For
example, "meta" tensors all share the same storage, and thus their identifier will all be equal. This identifier is
guaranteed to be unique and constant for this tensor's storage during its lifetime. Two tensor storages with
non-overlapping lifetimes may have the same id.
z2.5)ÚDTensorÚxlaN)Ú_torch_distributed_availabler   Útorch.distributed.tensorr…   Ú
isinstanceÚto_localr(   ÚstorageÚdata_ptrÚnbytesÚtyper   Ú	torch_xlaÚ_XLACÚ_xla_get_tensor_idr   r	   )rƒ   r…   Úlocal_tensorr�   Ú	unique_ids   &    r   Úid_tensor_storager”   Õ   sª   € ÷ $Ó#Ô(AÀ%×(HÒ(HÝ4ä�f×&Ò&Ø!Ÿ?™?Ó,ˆLØ—=‘= ,×"6Ñ"6Ó"8×"AÑ"AÓ"CÀVÇ]Á]ÐRÐRà‡}�}×Ñ˜UÔ"Ô'=×'?Ò'?ó
 	à—O‘O×6Ñ6°vÓ>‰	ä Ó'ˆ	à�=‰=˜)¤\°&Ó%9Ð9Ð9r   c                 ó   a a€ V V3R lpV# )z£
LRU cache decorator from standard functools library, but with a workaround to disable
caching when torchdynamo is compiling. Expected to work with class methods.
c                óX   <a a€ \        S/ SB ! S 4      o\        S 4      V V3R  l4       pV# )c                 óD   <€ \        4       '       d	   S! V / VB # S! V / VB # rm   )r   )ÚargsÚkwargsÚfuncÚfunc_with_caches   *,€€r   ÚwrapperÚGcompile_compatible_method_lru_cache.<locals>.decorator.<locals>.wrapperû   s,   ø€ ä'×)Ò)Ù˜TÐ, VÑ,Ð,á&¨Ð7°Ñ7Ð7r   r   )rš   rœ   r›   Úlru_argsÚ
lru_kwargss   f @€€r   Ú	decoratorÚ6compile_compatible_method_lru_cache.<locals>.decoratorø   s4   ú€ Ü# XÐ<°Ò<¸TÓBˆä	ˆt‹õ	8ó 
ð	8ð ˆr   r!   )rž   rŸ   r    s   jl r   Ú#compile_compatible_method_lru_cacher¢   ñ   s   ù€ ö
ð Ðr   )r   )+Ú
__future__r   ro   Úcollections.abcr   Ú	functoolsr   r   r   Úsafetensors.torchr   r	   r
   Úutilsr   r   r   r   Ú	LayerNormÚALL_LAYERNORM_LAYERSÚ
get_loggerrW   ÚloggerÚ"is_torch_greater_or_equal_than_2_8Ú"is_torch_greater_or_equal_than_2_6Ú"is_torch_greater_or_equal_than_2_4Ú"is_torch_greater_or_equal_than_2_3Ú"is_torch_greater_or_equal_than_2_2Ú"is_torch_greater_or_equal_than_2_1Ú"is_torch_greater_or_equal_than_2_0Ú#is_torch_greater_or_equal_than_1_13Ú#is_torch_greater_or_equal_than_1_12ÚdistributedÚis_availabler‡   r   r7   ÚModuler9   r|   r�   r”   r¢   r!   r   r   Ú<module>r¸      s)  ðõ #ã Ý $ß &ã ß 7Ý ÷ó ð Ÿ™�~Ð à	×	Ò	˜HÓ	%€á%>¸uÐQUÔ%VÐ "Ù%>¸uÐQUÔ%VÐ "ñ &?¸uÐQUÔ%VÐ "Ù%>¸uÐQUÔ%VÐ "Ù%>¸uÐQUÔ%VÐ "Ù%>¸uÐQUÔ%VÐ "Ù%>¸uÐQUÔ%VÐ "Ù&?ÀÐSWÔ&XÐ #Ù&?ÀÐSWÔ&XÐ #ð  %×0Ñ0×=Ñ=Ó?Ð òQ÷ôDˆR�Y‰Yô õ:K&ð\7ÐQU÷ 7õ:ñ8 €yÓñó òr   