+
    QV-jQ5  ã                   óö   € R t ^ RIt^ RIHt  ^ RIt^RIHt ^RI	H
t
 ^RIHtHtHt ^RIHtHtHt ]P&                  ! ]4      tRR/tR	t]! ]4       ! R
 R]
4      4       t ! R R4      tR#   ] d    Rt Lei ; i)zT
SentencePiece-based tokenization class for loading from sentencepiece.model files.
N)Úcopyfile)Úimport_protobuf)ÚPreTrainedTokenizer)ÚINIT_TOKENIZER_DOCSTRINGÚ
AddedTokenÚgenerate_merges)Úadd_end_docstringsÚloggingÚrequires_backendsÚ
vocab_fileztokenizer.modelu   â–�c                   óÞ   a a€ ] tR t^,t oRt]tV 3R lt]V3R lR l4       t	R t
RV3R lR lltRV3R lR	 lltR
 tR tR tV3R lR ltRV3R lR lltRV3R lV 3R llltRtVtV ;t# )ÚSentencePieceBackenda.  
Base class for SentencePiece-based tokenizers that load from sentencepiece.model files.

Inherits from [`~tokenization_utils.PreTrainedTokenizer`].

Handle all the shared methods for tokenization and special tokens as well as methods downloading/caching/loading
pretrained tokenizers as well as adding tokens to the vocabulary.

This class also contain the added tokens in a unified way on top of all tokenizers so we don't have to handle the
specific vocabulary augmentation methods of the various underlying dictionary structures (BPE, sentencepiece...).
c                ó   <€ \        V R 4       VP                  R4      V n        VP                  RR4      V n        VP	                  R/ 4      V n        RV9  d   R VR&   \        P                  ! R/ V P
                  B pVP                  V P                  4       V P                  '       g€   \        4       pVP                  P                  VP                  4       4      pVP                  P                  '       d1   RVP                  n        VP                  VP!                  4       4       W n        V P"                  P%                  4       V n        V P
                  VR&   \(        SV `T  ! R/ VB  V P-                  4        R# )	Úsentencepiecer   ÚlegacyTÚsp_model_kwargsÚbackendFN© )r
   Úgetr   r   Úpopr   ÚspmÚSentencePieceProcessorÚLoadr   Ú
ModelProtoÚ
FromStringÚserialized_model_protoÚnormalizer_specÚadd_dummy_prefixÚLoadFromSerializedProtoÚSerializeToStringÚsp_modelÚget_piece_sizeÚtotal_vocab_sizeÚsuperÚ__init__Ú_update_trie)ÚselfÚkwargsÚ	tokenizerÚ	model_pb2ÚprotoÚ	__class__s   &,   €Ú~/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/tokenization_utils_sentencepiece.pyr$   ÚSentencePieceBackend.__init__<   s8  ø€ ä˜$ Ô0ð !Ÿ*™* \Ó2ˆŒØ—j‘j ¨4Ó0ˆŒØ%Ÿz™zÐ*;¸RÓ@ˆÔð ˜FÔ"Ø /ˆF�9Ñô ×.Ò.ÑF°×1EÑ1EÑFˆ	Ø�‰�t—‘Ô'à�{�{ˆ{Ü'Ó)ˆIØ×(Ñ(×3Ñ3°I×4TÑ4TÓ4VÓWˆEØ×$Ñ$×5×5Ð5Ø9>�×%Ñ%Ô6Ø×1Ñ1°%×2IÑ2IÓ2KÔLà!Œð !%§¡× <Ñ <Ó >ˆÔð %)×$8Ñ$8ˆÐ Ñ!ô
 	‰ÒÑ"˜6Ò"Ø×ÑÖó    c                ó    <€ V ^8„  d   QhRS[ /# ©é   Úreturn)Úint)ÚformatÚ__classdict__s   "€r,   Ú__annotate__Ú!SentencePieceBackend.__annotate__d   s   ø€ ÷ .ñ .™Cñ .r.   c                ó6   € V P                   P                  4       # )zReturns vocab size)r    r!   )r&   s   &r,   Ú
vocab_sizeÚSentencePieceBackend.vocab_sizec   s   € ð �}‰}×+Ñ+Ó-Ð-r.   c                ó¬   € \        V P                  4       Uu/ uF  qP                  V4      VbK  	  ppVP                  V P                  4       V# u upi )zReturns vocab as a dict)Úranger9   Úconvert_ids_to_tokensÚupdateÚadded_tokens_encoder)r&   ÚiÚvocabs   &  r,   Ú	get_vocabÚSentencePieceBackend.get_vocabh   sL   € ä;@ÀÇÁÔ;QÓRÑ;Q°a×+Ñ+¨AÓ.°Ò1Ñ;QˆÐRØ�‰�T×.Ñ.Ô/Øˆùò Ss   ˜Ac                ó\   <€ V ^8„  d   QhRS[ S[,          S[ S[,          ,          RS[RS[/# )r1   Ú
new_tokensÚspecial_tokensr2   )ÚlistÚstrr   Úboolr3   )r4   r5   s   "€r,   r6   r7   n   s7   ø€ ÷ Nñ N¡d©3¥i±$±zÕ2BÕ&Bð NÑTXð NÑehñ Nr.   c           	     óX  € V'       g   ^ # \        V 4      p^ pV EFk  p\        V\        \        34      '       g   \	        RV R\        V4       R24      h\        V4      R8X  d   KM  \        V\        4      '       dA   WPP                  9   d   Ku  WPP                  9   ;'       g    Tp\        VRRV'       * VR7      pM'V'       d    VP                  RRR	VP                  /4       WPP                  P                  4       9   d   Kê  VP                  '       gE   VP                  '       d3   \        V R
R4      '       d    VP                  P                  4       Vn        V P                   P#                  VP                  4      pWpP                   P%                  4       8  ;'       d)    V P                   P'                  V4      VP                  8H  pV'       d   Tp	MTp	V^,          pV^,          pVP                  '       d6   \        V4      V P                  9  d   V P(                  P+                  V4       WPP                  V	&   W�P                  VP                  &   V P,                  '       g   EKR  \.        P1                  RV R24       EKn  	  V P3                  4        V P5                  4        V# )aÍ  
Add a list of new tokens to the tokenizer class. If the new tokens are not in the vocabulary, they are added to
it with indices starting from length of the current vocabulary. Special tokens are sometimes already in the
vocab which is why they have to be handled specifically.

Args:
    new_tokens (`list[str]`or `list[tokenizers.AddedToken]`):
        Token(s) to add in vocabulary. A token is counted as added if it's not already in the vocabulary
        (tested by checking if the tokenizer assign the index of the `unk_token` to them). If a token is part
        of the vocabulary then we simply mark this token as an `AddedToken` which allows to control the
        stripping and normalization of this token. This is NOT possible in `tokenizers`.
    special_tokens (`bool`, *optional*, defaults to `False`):
        Whether or not the tokens should be added as special tokens.

Returns:
    `int`: The number of tokens actually added to the vocabulary.

Examples:

```python
# Let's see how to increase the vocabulary of Bert model and tokenizer
tokenizer = BertTokenizer.from_pretrained("google-bert/bert-base-uncased")
model = BertModel.from_pretrained("google-bert/bert-base-uncased")

num_added_toks = tokenizer.add_tokens(["new_tok1", "my_new-tok2"])
print("We have added", num_added_toks, "tokens")
# Note: resize_token_embeddings expects to receive the full size of the new vocabulary, i.e. the length of the tokenizer.
model.resize_token_embeddings(len(tokenizer))
```zToken z is not a string but a Ú.Ú F)ÚrstripÚlstripÚ
normalizedÚspecialrP   TrO   Údo_lower_casezAdding z to the vocabulary)ÚlenÚ
isinstancerH   r   Ú	TypeErrorÚtypeÚ_added_tokens_encoderÚall_special_tokensÚ__setstate__rO   Ú_added_tokens_decoderÚvaluesrP   ÚgetattrÚcontentÚlowerr    Úpiece_to_idr!   Ú	IdToPieceÚ_extra_special_tokensÚappendÚverboseÚloggerÚinfor%   Ú_update_total_vocab_size)
r&   rE   rF   Ú
next_indexÚ	num_addedÚtokenÚ
is_specialÚtok_idÚin_base_vocabÚtoken_indexs
   &&&       r,   Ú_add_tokensÚ SentencePieceBackend._add_tokensn   s#  € ÷< Ùä˜“Yˆ
Øˆ	ÜˆEÜ˜e¤c¬:Ð%6×7Ò7Ü &¨¨Ð/FÄtÈEÃ{ÀmÐSTÐ UÓVÐVÜ�5‹z˜RÔÙÜ˜%¤×%Ò%Ø×6Ñ6Ô6ÙØ"×&=Ñ&=Ñ=×OÐOÀ�
Ü" 5°¸uÐU_ÔQ_ÐisÔt‘ßð ×"Ñ" I¨t°\À5×CSÑCSÐ#TÔUà×2Ñ2×9Ñ9Ó;Ô;ÙØ—=—=�= U×%5×%5Ð%5¼'À$ÈÐY^×:_Ò:_Ø %§¡× 3Ñ 3Ó 5�”ð —]‘]×.Ñ.¨u¯}©}Ó=ˆFàŸ™×5Ñ5Ó7Ñ7×lÐl¸D¿M¹M×<SÑ<SÐTZÓ<[Ð_d×_lÑ_lÑ<lð ÷ Ø$‘à(�Ø˜a•�
Ø˜Q•�	à�}�}ˆ}¤ U£°4×3JÑ3JÔ!JØ×*Ñ*×1Ñ1°%Ô8à6;×&Ñ& {Ñ3Ø8C×&Ñ& u§}¡}Ñ5Ø�|�|‹|Ü—‘˜g e WÐ,>Ð?×@ñO  ðR 	×ÑÔØ×%Ñ%Ô'ØÐr.   c                ó>   <€ V ^8„  d   QhRS[ S[,          R,          /# )r1   Úunique_no_split_tokensN©rG   rH   )r4   r5   s   "€r,   r6   r7   ¾   s   ø€ ÷ ,ñ ,±4¹µ9¸tÕ3Cñ ,r.   c                ó  € V P                   P                  4        FO  pVP                  V P                  P                  9  g   K*  V P                  P                  VP                  4       KQ  	  V P                   F:  pW P                  P                  9  g   K  V P                  P                  V4       K<  	  T;'       g    .  F:  pW P                  P                  9  g   K  V P                  P                  V4       K<  	  R # ©N)rY   rZ   r\   Útokens_trieÚ_tokensÚaddrW   )r&   rp   rh   s   && r,   r%   Ú!SentencePieceBackend._update_trie¾   sÀ   € à×/Ñ/×6Ñ6Ö8ˆEØ�}‰} D×$4Ñ$4×$<Ñ$<Ö<Ø× Ñ ×$Ñ$ U§]¡]Ö3ñ 9ð ×,Ô,ˆEØ×,Ñ,×4Ñ4Ö4Ø× Ñ ×$Ñ$ UÖ+ñ -ð ,×1Ð1¨rÒ1ˆEØ×,Ñ,×4Ñ4Ö4Ø× Ñ ×$Ñ$ UÖ+ó 2r.   c                ó   € V P                   '       g   VP                  \        R34      '       g"   V P                  P	                  V\
        R7      # V P                  P	                  V P                  V,           \
        R7      p\        V P                  P	                  \        V P                  4      4      4      p\        V4      V8¼  d   W4R # T# )uð  
Returns a tokenized string.

We de-activated the `add_dummy_prefix` option, thus the sentencepiece internals will always strip any
SPIECE_UNDERLINE. For example: `self.sp_model.encode(f"{SPIECE_UNDERLINE}Hey", out_type = str)` will give
`['H', 'e', 'y']` instead of `['â–�He', 'y']`. Thus we always encode `f"{unk_token}text"` and strip the
`unk_token`. Here is an example with `unk_token = "<unk>"` and `unk_token_length = 4`.
`self.tokenizer.sp_model.encode("<unk> Hey", out_type = str)[4:]`.
Ú )Úout_typeN)r   Ú
startswithÚSPIECE_UNDERLINEr    ÚencoderH   Ú	unk_tokenrR   )r&   Útextr'   ÚtokensÚunk_token_lengths   &&,  r,   Ú	_tokenizeÚSentencePieceBackend._tokenizeÌ   s    € ð �;�;ˆ;˜dŸo™oÔ/?ÀÐ.E×FÒFØ—=‘=×'Ñ'¨´sÐ'Ó;Ð;ð —‘×%Ñ% d§n¡n°tÕ&;ÄcÐ%ÓJˆä˜tŸ}™}×3Ñ3´C¸¿¹Ó4GÓHÓIÐÜ,/°«KÐ;KÔ,KˆvÐ'Ð(ÐWÐQWÐWr.   c                ó8   € V P                   P                  V4      # )z0Converts a token (str) to an id using the vocab.)r    r^   )r&   rh   s   &&r,   Ú_convert_token_to_idÚ)SentencePieceBackend._convert_token_to_idß   s   € à�}‰}×(Ñ(¨Ó/Ð/r.   c                ó<   € V P                   P                  V4      pV# )z=Converts an index (integer) in a token (str) using the vocab.)r    r_   )r&   Úindexrh   s   && r,   Ú_convert_id_to_tokenÚ)SentencePieceBackend._convert_id_to_tokenã   s   € à—‘×'Ñ'¨Ó.ˆØˆr.   c                ó6   <€ V ^8„  d   QhRS[ S[,          RS[/# )r1   r€   r2   rq   )r4   r5   s   "€r,   r6   r7   è   s   ø€ ÷ ñ ©t±C­yð ¹Sñ r.   c                ól   € RP                  V4      P                  \        R4      P                  4       pV# )z:Converts a sequence of tokens (string) in a single string.rL   ry   )ÚjoinÚreplacer|   Ústrip)r&   r€   Ú
out_strings   && r,   Úconvert_tokens_to_stringÚ-SentencePieceBackend.convert_tokens_to_stringè   s,   € à—W‘W˜V“_×,Ñ,Ô-=¸sÓC×IÑIÓKˆ
ØÐr.   c                óJ   <€ V ^8„  d   QhRS[ RS[ R,          RS[S[ ,          /# )r1   Úsave_directoryÚfilename_prefixNr2   )rH   Útuple)r4   r5   s   "€r,   r6   r7   í   s-   ø€ ÷ !ñ !©cð !ÁCÈ$ÅJð !ÑZ_Ñ`cÕZdñ !r.   c                ó\  € \         P                  P                  V4      '       g   \        P	                  RV R24       R# \         P                  P                  Y'       d
   VR,           MRV P                  R,          ,           4      p\         P                  P                  V P                  4      \         P                  P                  V4      8w  dI   \         P                  P                  V P                  4      '       d   \        V P                  V4       V3# \         P                  P                  V P                  4      '       gL   \        VR4      ;_uu_ 4       pV P                  P                  4       pVP                  V4       RRR4       V3# V3#   + '       g   i     T3# ; i)aD  
Save the sentencepiece vocabulary (copy original file) to a directory.

Args:
    save_directory (`str`):
        The directory in which to save the vocabulary.
    filename_prefix (`str`, *optional*):
        An optional prefix to add to the named of the saved files.

Returns:
    `tuple(str)`: Paths to the files saved.
zVocabulary path (z) should be a directoryNÚ-rL   r   Úwb)ÚosÚpathÚisdirrc   Úerrorr�   Úvocab_files_namesÚabspathr   Úisfiler   Úopenr    r   Úwrite)r&   r”   r•   Úout_vocab_fileÚfiÚcontent_spiece_models   &&&   r,   Úsave_vocabularyÚ$SentencePieceBackend.save_vocabularyí   s7  € ô �w‰w�}‰}˜^×,Ò,Ü�L‰LÐ,¨^Ð,<Ð<SÐTÔUÙÜŸ™Ÿ™Ø¶o˜_¨sÖ2È2ÐQU×QgÑQgÐhtÕQuÕuó
ˆô �7‰7�?‰?˜4Ÿ?™?Ó+¬r¯w©w¯©¸~Ó/NÔNÔSU×SZÑSZ×SaÑSaÐbf×bqÑbq×SrÒSrÜ�T—_‘_ nÔ5ð Ð Ð ô —‘—‘ §¡×0Ò0Ü�n d×+Ô+¨rØ'+§}¡}×'KÑ'KÓ'MÐ$Ø—‘Ð-Ô.÷ ,ð Ð Ð �Ð Ð ÷	 ,Ö+ð Ð Ð ús   Å,FÆF+	c          
      óf   <€ V ^8„  d   QhRS[ S[S[ ,          ,          RS[RS[R,          RS[RS[/# )r1   Ú	token_idsÚskip_special_tokensÚclean_up_tokenization_spacesNÚspaces_between_special_tokensr2   )r3   rG   rI   rH   )r4   r5   s   "€r,   r6   r7   
  sI   ø€ ÷ 
ñ 
á™™c�•?ð
ñ "ð
ñ '+¨T¥kð	
ñ
 (,ð
ñ 
ñ
r.   c           	     ó0   <€ \         SV `  ! RRVRVRV/VB # )z¸
Decode token ids to string.

Uses the generic decode path from PreTrainedTokenizer which works for all vocabularies,
including custom vocabularies that override _convert_id_to_token.
r©   rª   r«   r   )r#   Ú_decode)r&   r©   rª   r«   r¬   r'   r+   s   &&&&&,€r,   r®   ÚSentencePieceBackend._decode
  s;   ø€ ô ‰wŠñ 
Øð
à 3ð
ð *Fð
ð ñ	
ð 	
r.   )r   r    r   r"   r   )Frs   )FNF)Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__ÚVOCAB_FILES_NAMESrž   r$   Úpropertyr9   rB   rm   r%   r‚   r…   r‰   r‘   r¦   r®   Ú__static_attributes__Ú__classdictcell__Ú__classcell__)r+   r5   s   @@r,   r   r   ,   s{   ù‡ € ñ
ð *Ðõ%ðN ÷.ó ð.ò÷Nò N÷`,ò ,òXò&0ò÷
ð ÷
!ò !÷:
÷ 
õ 
r.   r   c                   óL   a € ] tR tRt o RtV 3R lR ltR	V 3R lR lltRtV tR# )
ÚSentencePieceExtractori!  zd
Extractor implementation for SentencePiece trained models. https://github.com/google/sentencepiece
c                ó    <€ V ^8„  d   QhRS[ /# )r1   Úmodel)rH   )r4   r5   s   "€r,   r6   Ú#SentencePieceExtractor.__annotate__&  s   ø€ ÷ ñ ™cñ r.   c                óx   € \        V R 4       ^ RIHp V! 4       V n        V P                  P	                  V4       R# )r   )r   N)r
   r   r   Úspr   )r&   r½   r   s   && r,   r$   ÚSentencePieceExtractor.__init__&  s)   € Ü˜$ Ô0Ý8á(Ó*ˆŒØ�‰�‰�UÖr.   Nc                ó†   <€ V ^8„  d   QhRS[ S[S[S[3,          S[S[ S[S[3,          ,          S[S[ ,          3,          /# r0   )r–   ÚdictrH   r3   rG   Úfloat)r4   r5   s   "€r,   r6   r¾   -  s>   ø€ ÷ 4ñ 4©E±$±s¹C°xµ.Á$ÁuÉSÑRWÈZÕGXÕBYÑ[_Ñ`eÕ[fÐ2fÕ,gñ 4r.   c                óÞ  € V P                   p\        VP                  4       4       Uu/ uF  q2P                  V4      VbK  	  pp\        VP                  4       4       Uu/ uF#  qRP                  V4      VP	                  V4      bK%  	  pp\        WF4      p\        VP                  4       4       Uu. uF$  qRP                  V4      VP	                  V4      3NK&  	  ppWHV3# u upi u upi u upi )zª
By default will return vocab and merges with respect to their order, by sending `vocab_scores` we're going to
order the merges with respect to the piece scores instead.
)rÀ   r<   ÚGetPieceSizeÚid_to_pieceÚ	get_scorer   )	r&   Úvocab_scoresrÀ   rˆ   Ú	vocab_idsr@   Úvocab_scores_dictÚmergesÚvocab_scores_lists	   &&       r,   ÚextractÚSentencePieceExtractor.extract-  sÐ   € ð
 �W‰WˆÜ?DÀRÇ_Á_ÓEVÔ?WÓXÑ?W°e—^‘^ EÓ*¨EÒ1Ñ?Wˆ	ÐXäINÈrÏÉÓO`ÔIaÓbÑIaÀAŸ^™^¨AÓ.°·±¸Q³Ò?ÑIaÐÐbä  Ó>ˆäKPÐQS×Q`ÑQ`ÓQbÔKcÓdÑKcÀaŸn™n¨QÓ/°·±¸a³ÓAÑKcÐÐdà¨VÐ3Ð3ùò Yùâbùò es   ¨C Á)C%Â0*C*)rÀ   rs   )	r°   r±   r²   r³   r´   r$   rÎ   r·   r¸   )r5   s   @r,   r»   r»   !  s#   ø‡ € ñ÷ð ÷4÷ 4ð 4r.   r»   )r´   rš   Úshutilr   r   r   ÚImportErrorÚconvert_slow_tokenizerr   Útokenization_pythonr   Útokenization_utils_baser   r   r   Úutilsr   r	   r
   Ú
get_loggerr°   rc   rµ   r|   r   r»   r   r.   r,   Ú<module>r×      s    ðñó 
Ý ðÛõ 4Ý 4÷ñ ÷
 BÑ Að 
×	Ò	˜HÓ	%€à!Ð#4Ð5Ð àÐ ñ Ð,Ó-ôq
Ð.ó q
ó .ðq
÷h4ó 4øðS ô Ø
‚Cðús   ŽA, Á,	A8Á7A8