+
    QV-jÞo  ã                   ó   € ^ RI t ^ RIt^ RIHtHt ^ RIt^RIHt ^RI	H
t
HtHt ^RIHtHtHtHt ]! 4       '       d   ^ RIt^RIHt  ! R R]4      t ! R	 R
]
4      t]! ]! RR7      R4       ! R R]4      4       t]tR# )é    N)ÚAnyÚoverload)ÚBasicTokenizer)ÚExplicitEnumÚadd_end_docstringsÚis_torch_available)ÚArgumentHandlerÚChunkPipelineÚDatasetÚbuild_pipeline_init_args)Ú,MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING_NAMESc                   ó6   a € ] tR t^t o RtV 3R lR ltRtV tR# )Ú"TokenClassificationArgumentHandlerz-
Handles arguments for token classification.
c                ó@   <€ V ^8„  d   QhRS[ S[S[ ,          ,          /# )é   Úinputs)ÚstrÚlist)ÚformatÚ__classdict__s   "€Ú|/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/pipelines/token_classification.pyÚ__annotate__Ú/TokenClassificationArgumentHandler.__annotate__   s   ø€ ÷ Fñ F™s¡T©#¥Y�ñ Fó    c                ó†  € VP                  R R4      pVP                  R4      pVeD   \        V\        \        34      '       d(   \	        V4      ^ 8”  d   \        V4      p\	        V4      pMj\        V\
        4      '       d   V.p^pMN\        e   \        V\        4      '       g!   \        V\        P                  4      '       d   WRV3# \        R4      hVP                  R4      pV'       dR   \        V\        4      '       d!   \        V^ ,          \        4      '       d   V.p\	        V4      V8w  d   \        R4      hWWd3# )Úis_split_into_wordsFÚ	delimiterNzAt least one input is required.Úoffset_mappingz;offset_mapping should have the same batch size as the input)
ÚgetÚ
isinstancer   ÚtupleÚlenr   r   ÚtypesÚGeneratorTypeÚ
ValueError)Úselfr   Úkwargsr   r   Ú
batch_sizer   s   &&,    r   Ú__call__Ú+TokenClassificationArgumentHandler.__call__   s	  € Ø$Ÿj™jÐ)>ÀÓFÐØ—J‘J˜{Ó+ˆ	àÒ¤*¨V´d¼E°]×"CÒ"CÌÈFËÐVWÌÜ˜&“\ˆFÜ˜V›‰JÜ˜¤×$Ò$Ø�XˆFØ‰JÜÒ ¤Z°¼×%@Ò%@ÄJÈvÔW\×WjÑWj×DkÒDkØ°°iÐ?Ð?äÐ>Ó?Ð?àŸ™Ð$4Ó5ˆßÜ˜.¬$×/Ò/´J¸~ÈaÕ?PÔRW×4XÒ4XØ"0Ð!1�Ü�>Ó" jÔ0Ü Ð!^Ó_Ð_Ø¨NÐEÐEr   © N)Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r)   Ú__static_attributes__Ú__classdictcell__)r   s   @r   r   r      s   ø‡ € ñ÷Fö Fr   r   c                   ó.   € ] tR t^3tRtRtRtRtRtRt	Rt
R# )	ÚAggregationStrategyzDAll the valid aggregation strategies for TokenClassificationPipelineÚnoneÚsimpleÚfirstÚaverageÚmaxr+   N)r,   r-   r.   r/   r0   ÚNONEÚSIMPLEÚFIRSTÚAVERAGEÚMAXr1   r+   r   r   r4   r4   3   s   † ÙNà€DØ€FØ€EØ€GØ
„Cr   r4   T)Úhas_tokenizeraÙ	  
        ignore_labels (`list[str]`, defaults to `["O"]`):
            A list of labels to ignore.
        stride (`int`, *optional*):
            If stride is provided, the pipeline is applied on all the text. The text is split into chunks of size
            model_max_length. Works only with fast tokenizers and `aggregation_strategy` different from `NONE`. The
            value of this argument defines the number of overlapping tokens between chunks. In other words, the model
            will shift forward by `tokenizer.model_max_length - stride` tokens each step.
        aggregation_strategy (`str`, *optional*, defaults to `"none"`):
            The strategy to fuse (or not) tokens based on the model prediction.

                - "none" : Will simply not do any aggregation and simply return raw results from the model
                - "simple" : Will attempt to group entities following the default schema. (A, B-TAG), (B, I-TAG), (C,
                  I-TAG), (D, B-TAG2) (E, B-TAG2) will end up being [{"word": ABC, "entity": "TAG"}, {"word": "D",
                  "entity": "TAG2"}, {"word": "E", "entity": "TAG2"}] Notice that two consecutive B tags will end up as
                  different entities. On word based languages, we might end up splitting words undesirably : Imagine
                  Microsoft being tagged as [{"word": "Micro", "entity": "ENTERPRISE"}, {"word": "soft", "entity":
                  "NAME"}]. Look for FIRST, MAX, AVERAGE for ways to mitigate that and disambiguate words (on languages
                  that support that meaning, which is basically tokens separated by a space). These mitigations will
                  only work on real words, "New york" might still be tagged with two different entities.
                - "first" : (works only on word based models) Will use the `SIMPLE` strategy except that words, cannot
                  end up with different tags. Words will simply use the tag of the first token of the word when there
                  is ambiguity.
                - "average" : (works only on word based models) Will use the `SIMPLE` strategy except that words,
                  cannot end up with different tags. scores will be averaged first across tokens, and then the maximum
                  label is applied.
                - "max" : (works only on word based models) Will use the `SIMPLE` strategy except that words, cannot
                  end up with different tags. Word entity will simply be the token with the maximum score.c                   óz  a a€ ] tR t^=t oRtRtRtRtRtRt	]
! 4       3V 3R lltR"V3R lR llt]V3R	 lR
 l4       t]V3R lR l4       tV3R lV 3R lltR#R ltR t]P$                  R3R ltR tR$V3R lR lltV3R lR ltV3R lR ltV3R lR ltV3R lR ltV3R lR ltV3R lR  ltR!tVtV ;t# )%ÚTokenClassificationPipelineu	  
Named Entity Recognition pipeline using any `ModelForTokenClassification`. See the [named entity recognition
examples](../task_summary#named-entity-recognition) for more information.

Example:

```python
>>> from transformers import pipeline

>>> token_classifier = pipeline(model="Jean-Baptiste/camembert-ner", aggregation_strategy="simple")
>>> sentence = "Je m'appelle jean-baptiste et je vis Ã  montrÃ©al"
>>> tokens = token_classifier(sentence)
>>> tokens
[{'entity_group': 'PER', 'score': 0.9931, 'word': 'jean-baptiste', 'start': 12, 'end': 26}, {'entity_group': 'LOC', 'score': 0.998, 'word': 'montrÃ©al', 'start': 38, 'end': 47}]

>>> token = tokens[0]
>>> # Start and end provide an easy way to highlight words in the original text.
>>> sentence[token["start"] : token["end"]]
' jean-baptiste'

>>> # Some models use the same idea to do part of speech.
>>> syntaxer = pipeline(model="vblagoje/bert-english-uncased-finetuned-pos", aggregation_strategy="simple")
>>> syntaxer("My name is Sarah and I live in London")
[{'entity_group': 'PRON', 'score': 0.999, 'word': 'my', 'start': 0, 'end': 2}, {'entity_group': 'NOUN', 'score': 0.997, 'word': 'name', 'start': 3, 'end': 7}, {'entity_group': 'AUX', 'score': 0.994, 'word': 'is', 'start': 8, 'end': 10}, {'entity_group': 'PROPN', 'score': 0.999, 'word': 'sarah', 'start': 11, 'end': 16}, {'entity_group': 'CCONJ', 'score': 0.999, 'word': 'and', 'start': 17, 'end': 20}, {'entity_group': 'PRON', 'score': 0.999, 'word': 'i', 'start': 21, 'end': 22}, {'entity_group': 'VERB', 'score': 0.998, 'word': 'live', 'start': 23, 'end': 27}, {'entity_group': 'ADP', 'score': 0.999, 'word': 'in', 'start': 28, 'end': 30}, {'entity_group': 'PROPN', 'score': 0.999, 'word': 'london', 'start': 31, 'end': 37}]
```

Learn more about the basics of using a pipeline in the [pipeline tutorial](../pipeline_tutorial)

This token recognition pipeline can currently be loaded from [`pipeline`] using the following task identifier:
`"ner"` (for predicting the classes of tokens in a sequence: person, organisation, location or miscellaneous).

The models that this pipeline can use are models that have been fine-tuned on a token classification task. See the
up-to-date list of available models on
[huggingface.co/models](https://huggingface.co/models?filter=token-classification).
Ú	sequencesFTc                ó€   <€ \         SV `  ! R/ VB  V P                  \        4       \	        R R7      V n        Wn        R# )F)Údo_lower_caseNr+   )ÚsuperÚ__init__Úcheck_model_typer   r   Ú_basic_tokenizerÚ_args_parser)r&   Úargs_parserr'   Ú	__class__s   &&,€r   rF   Ú$TokenClassificationPipeline.__init__ˆ   s5   ø€ Ü‰ÒÑ"˜6Ò"à×ÑÔJÔKä .¸UÔ CˆÔØ'Ör   Nc                ó–   <€ V ^8„  d   QhRS[ R,          RS[S[S[S[3,          ,          R,          RS[RS[R,          RS[R,          /# )r   Úaggregation_strategyNr   r   Ústrider   )r4   r   r!   ÚintÚboolr   )r   r   s   "€r   r   Ú(TokenClassificationPipeline.__annotate__�   s^   ø€ ÷ 99ñ 99ñ 2°DÕ8ð99ñ ™U¡3© 8�_Õ-°Õ4ð	99ñ
 "ð99ñ �d•
ð99ñ ˜•:ñ99r   c                óŒ  € / pWGR &   V'       d   Vf   RMTVR&   Ve   W7R&   / pVe‘   \        V\        4      '       d   \        VP                  4       ,          pV\        P                  \        P
                  \        P                  09   d(   V P                  P                  '       g   \        R4      hW(R&   Ve   WR&   Ve~   WPP                  P                  8¼  d   \        R4      hV\        P                  8X  d   \        RV R	24      hV P                  P                  '       d   R
RRRRV/p	W—R&   M\        R4      hV/ V3# )r   Ú r   r   z{Slow tokenizers cannot handle subwords. Please set the `aggregation_strategy` option to `"simple"` or use a fast tokenizer.rN   Úignore_labelszl`stride` must be less than `tokenizer.model_max_length` (or even lower if the tokenizer adds special tokens)zI`stride` was provided to process all the text but `aggregation_strategy="z&"`, please select another one instead.Úreturn_overflowing_tokensTÚpaddingrO   Útokenizer_paramszm`stride` was provided to process all the text but you're using a slow tokenizer. Please use a fast tokenizer.)r    r   r4   Úupperr<   r>   r=   Ú	tokenizerÚis_fastr%   Úmodel_max_lengthr:   )
r&   rU   rN   r   r   rO   r   Úpreprocess_paramsÚpostprocess_paramsrX   s
   &&&&&&&   r   Ú_sanitize_parametersÚ0TokenClassificationPipeline._sanitize_parameters�   s{  € ð ÐØ3FÐ/Ñ0çØ4=Ò4E©SÈ9Ð˜kÑ*àÒ%Ø2@Ð.Ñ/àÐØÒ+ÜÐ.´×4Ò4Ü':Ð;O×;UÑ;UÓ;WÕ'XÐ$à$Ü'×-Ñ-Ô/B×/FÑ/FÔH[×HcÑHcÐdôeàŸ™×.×.Ð.ä ð>óð ð :NÐ5Ñ6ØÒ$Ø2?˜Ñ/ØÒØŸ™×8Ñ8Ô8Ü ð Cóð ð $Ô':×'?Ñ'?Ô?Ü ðØ,Ð-Ð-SðUóð ð
 —>‘>×)×)Ð)à3°TØ! 4Ø  &ð(Ð$ð
 =MÐ&8Ò9ä$ð8óð ð ! "Ð&8Ð8Ð8r   c          	      óR   <€ V ^8„  d   QhRS[ RS[RS[S[S[ S[ 3,          ,          /# ©r   r   r'   Úreturn)r   r   r   Údict)r   r   s   "€r   r   rR   Ì   s%   ø€ ×OÑO™sÐO©cÐO±d¹4ÁÁSÀ½>Õ6JÑOr   c                ó   € R # ©Nr+   ©r&   r   r'   s   &&,r   r)   Ú$TokenClassificationPipeline.__call__Ë   s   € ÙLOr   c          
      ór   <€ V ^8„  d   QhRS[ S[,          RS[RS[ S[ S[S[S[3,          ,          ,          /# rb   )r   r   r   rd   )r   r   s   "€r   r   rR   Ï   s/   ø€ ×[Ñ[™t¡C�yÐ[±CÐ[¹DÁÁdÉ3ÑPSÈ8ÅnÕAUÕ<VÑ[r   c                ó   € R # rf   r+   rg   s   &&,r   r)   rh   Î   s   € ÙX[r   c                ó¸   <€ V ^8„  d   QhRS[ S[S[ ,          ,          RS[RS[S[S[ S[ 3,          ,          S[S[S[S[ S[ 3,          ,          ,          ,          /# rb   )r   r   r   rd   )r   r   s   "€r   r   rR   Ñ   sV   ø€ ÷ #2ñ #2™s¡T©#¥Y�ð #2¹#ð #2Á$ÁtÉCÑQTÈHÅ~ÕBVÑY]Ñ^bÑcgÑhkÑmpÐhpÕcqÕ^rÕYsÕBsñ #2r   c                ó"  <€ V P                   ! V3/ VB w  r4rVWBR&   WbR&   V'       dM   \        ;QJ d    R V 4       F  '       d   K   RM	  RM! R V 4       4      '       g   \        SV `  ! V.3/ VB # V'       d   WRR&   \        SV `  ! V3/ VB # )ak  
Classify each token of the text(s) given as inputs.

Args:
    inputs (`str` or `List[str]`):
        One or several texts (or one list of texts) for token classification. Can be pre-tokenized when
        `is_split_into_words=True`.

Return:
    A list or a list of list of `dict`: Each result comes as a list of dictionaries (one for each token in the
    corresponding input, or each entity if this pipeline was instantiated with an aggregation_strategy) with
    the following keys:

    - **word** (`str`) -- The token/word classified. This is obtained by decoding the selected tokens. If you
      want to have the exact string in the original sentence, use `start` and `end`.
    - **score** (`float`) -- The corresponding probability for `entity`.
    - **entity** (`str`) -- The entity predicted for that token/word (it is named *entity_group* when
      *aggregation_strategy* is not `"none"`.
    - **index** (`int`, only present when `aggregation_strategy="none"`) -- The index of the corresponding
      token in the sentence.
    - **start** (`int`, *optional*) -- The index of the start of the corresponding entity in the sentence. Only
      exists if the offsets are available within the tokenizer
    - **end** (`int`, *optional*) -- The index of the end of the corresponding entity in the sentence. Only
      exists if the offsets are available within the tokenizer
r   r   c              3   óB   "  € T F  p\        V\        4      x € K  	  R # 5irf   )r    r   )Ú.0Úinputs   & r   Ú	<genexpr>Ú7TokenClassificationPipeline.__call__.<locals>.<genexpr>ï   s   é € Ð*WÑPVÀu¬:°e¼T×+BÐ+BÓPVùs   ‚FTr   )rI   ÚallrE   r)   )r&   r   r'   Ú_inputsr   r   r   rK   s   &&,    €r   r)   rh   Ñ   s‘   ø€ ð6 CG×BSÒBSÐTZÑBeÐ^dÑBeÑ?ˆ nØ(;Ð$Ñ%Ø'ˆ{Ñß§s£sÑ*WÑPVÓ*W§s§s¢sÑ*WÑPVÓ*W×'WÒ'WÜ‘7Ò# V HÑ7°Ñ7Ð7ßØ'5Ð#Ñ$ä‰wÒ Ñ1¨&Ñ1Ð1r   c              +  ót  "  € VP                  R / 4      pV P                  P                  ;'       d    V P                  P                  ^ 8„  pRpVR,          pV'       d™   VR,          p\        V\        4      '       g   \        R4      hTp	VP                  V	4      p. p\        V4      p
^ pV	 F>  pVP                  W»\        V4      ,           34       V\        V4      V
,           ,          pK@  	  T	pRVR&   M#\        V\        4      '       g   \        R4      hTpV P                  ! V3RRR	VR
RRV P                  P                  /VB pV'       d(   V P                  P                  '       g   \        R4      hVP                  RR4       \        VR,          4      p\        V4       F…  pVP                  4        UUu/ uF  w  ppVVV,          P                  ^ 4      bK!  	  pppVe   VVR&   V^ 8X  d   TMRVR&   VV^,
          8H  VR&   Ve   VP                  V4      VR&   VVR&   Vx € K‡  	  R# u uppi 5i)rX   Nr   r   zEWhen `is_split_into_words=True`, `sentence` must be a list of tokens.TzKWhen `is_split_into_words=False`, `sentence` must be an untokenized string.Úreturn_tensorsÚptÚ
truncationÚreturn_special_tokens_maskÚreturn_offsets_mappingz@is_split_into_words=True is only supported with fast tokenizers.Úoverflow_to_sample_mappingÚ	input_idsr   ÚsentenceÚis_lastÚword_idsÚword_to_chars_map)ÚpoprZ   r\   r    r   r%   Újoinr"   Úappendr   r[   ÚrangeÚitemsÚ	unsqueezer~   )r&   r|   r   r]   rX   rw   r   r   r   ÚwordsÚdelimiter_lenÚchar_offsetÚwordÚtext_to_tokenizer   Ú
num_chunksÚiÚkÚvÚmodel_inputss   &&&,                r   Ú
preprocessÚ&TokenClassificationPipeline.preprocessö   s4  é € Ø,×0Ñ0Ð1CÀRÓHÐØ—^‘^×4Ñ4×\Ð\¸¿¹×9XÑ9XÐ[\Ñ9\ˆ
à ÐØ/Ð0EÕFÐßØ)¨+Õ6ˆIÜ˜h¬×-Ò-Ü Ð!hÓiÐiØˆEØ —~‘~ eÓ,ˆHà "ÐÜ 	›NˆMØˆKÛ�Ø!×(Ñ(¨+ÄSÈÃYÕ7NÐ)OÔPØœs 4›y¨=Õ8Õ8’ñ ð
  %ÐØ6:ÐÐ2Ò3ä˜h¬×,Ò,Ü Ð!nÓoÐoØ'Ðà—’Øñ
àð
ð "ð
ð (,ð	
ð
 $(§>¡>×#9Ñ#9ð
ð ñ
ˆ÷  t§~¡~×'=×'=Ð'=ÜÐ_Ó`Ð`à�
‰
Ð/°Ô6Ü˜ Õ,Ó-ˆ
ä�zÖ"ˆAØ=C¿\¹\¼^ÔL¹^±T°Q¸˜A˜q �tŸ~™~¨aÓ0Ò0¹^ˆLÑLØÒ)Ø1?�Ð-Ñ.à34¸´6¡x¸tˆL˜Ñ$Ø&'¨:¸­>Ñ&9ˆL˜Ñ#Ø Ò,Ø+1¯?©?¸1Ó+=�˜ZÑ(Ø4E�Ð0Ñ1àÔó #ùÛLùs'   ‚AH8ÁC1H8ÅH8Å-AH8Æ=%H2Ç"AH8c                ól  € VP                  R 4      pVP                  RR4      pVP                  R4      pVP                  R4      pVP                  RR4      pVP                  RR4      pV P                  ! R/ VB p\        V\        4      '       d
   VR,          MV^ ,          p	RV	R VRVRVRVRVRV/VC# )	Úspecial_tokens_maskr   Nr|   r}   r~   r   Úlogitsr+   )r€   Úmodelr    rd   )
r&   r�   r“   r   r|   r}   r~   r   Úoutputr”   s
   &&        r   Ú_forwardÚ$TokenClassificationPipeline._forward.  sÕ   € à*×.Ñ.Ð/DÓEÐØ%×)Ñ)Ð*:¸DÓAˆØ×#Ñ# JÓ/ˆØ×"Ñ" 9Ó-ˆØ×#Ñ# J°Ó5ˆØ(×,Ñ,Ð-@À$ÓGÐà—’Ñ+˜lÑ+ˆÜ%/°¼×%=Ò%=�˜Ö!À6È!Å9ˆð �fØ!Ð#6Ø˜nØ˜Ø�wØ˜ØÐ!2ð	
ð ð	
ð 		
r   c                óN  € Vf   R.p. pV^ ,          P                  R4      pV EFÔ  pVR,          ^ ,          P                  \        P                  \        P                  39   d=   VR,          ^ ,          P                  \        P                  4      P                  4       pMVR,          ^ ,          P                  4       pV^ ,          R,          pVR,          ^ ,          p	VR,          e   VR,          ^ ,          MR p
VR,          ^ ,          P                  4       pVP                  R4      p\        P                  ! VRR	R
7      p\        P                  ! W},
          4      pWîP                  RR	R
7      ,          pV P                  VV	VV
VVVVR7      pV P                  VV4      pV Uu. uF7  pVP                  RR 4      V9  g   K  VP                  RR 4      V9  g   K5  VNK9  	  ppVP                  V4       EK×  	  \        V4      pV^8”  d   V P!                  V4      pV# u upi )NÚOr   r”   r|   r{   r   r“   r~   T)ÚaxisÚkeepdims)r~   r   ÚentityÚentity_groupéÿÿÿÿ)r   ÚdtypeÚtorchÚbfloat16Úfloat16ÚtoÚfloat32ÚnumpyÚnpr9   ÚexpÚsumÚgather_pre_entitiesÚ	aggregateÚextendr"   Úaggregate_overlapping_entities)r&   Úall_outputsrN   rU   Úall_entitiesr   Úmodel_outputsr”   r|   r{   r   r“   r~   ÚmaxesÚshifted_expÚscoresÚpre_entitiesÚgrouped_entitiesr�   Úentitiesr‹   s   &&&&                 r   ÚpostprocessÚ'TokenClassificationPipeline.postprocessE  s  € ØÒ Ø ˜EˆMØˆð (¨�N×.Ñ.Ð/BÓCÐä(ˆMØ˜XÕ& qÕ)×/Ñ/´E·N±NÄEÇMÁMÐ3RÔRØ& xÕ0°Õ3×6Ñ6´u·}±}ÓE×KÑKÓM‘à& xÕ0°Õ3×9Ñ9Ó;�à" 1•~ jÕ1ˆHØ% kÕ2°1Õ5ˆIà6CÐDTÕ6UÒ6a�Ð.Õ/°Ö2Ðgkð ð #0Ð0EÕ"FÀqÕ"I×"OÑ"OÓ"QÐØ$×(Ñ(¨Ó4ˆHä—F’F˜6¨°TÔ:ˆEÜŸ&š& ¥Ó0ˆKØ §?¡?¸ÀT ?Ó#JÕJˆFà×3Ñ3ØØØØØ#Ø$Ø!Ø"3ð 4ó 	ˆLð  $Ÿ~™~¨lÐ<PÓQÐñ /óá.�FØ—:‘:˜h¨Ó-°]ÑBô ð —J‘J˜~¨tÓ4¸MÑI÷ �Ù.ð ð ð ×Ñ ×)ñI )ôJ ˜Ó%ˆ
Ø˜Œ>Ø×>Ñ>¸|ÓLˆLØÐùòs   Æ(H"ÇH"ÇH"c                ó²  € \        V4      ^ 8X  d   V# \        VR R7      p. pV^ ,          pV F”  pVR,          VR,          u;8:  d   VR,          8  d[   M MWVR,          VR,          ,
          pVR,          VR,          ,
          pWV8”  g   WV8X  d   VR,          VR,          8”  d   TpK}  K  K�  VP                  V4       TpK–  	  VP                  V4       V# )r   c                 ó   € V R ,          # )Ústartr+   )Úxs   &r   Ú<lambda>ÚLTokenClassificationPipeline.aggregate_overlapping_entities.<locals>.<lambda>z  s   € °!°G¶*r   ©Úkeyr»   ÚendÚscore)r"   Úsortedr‚   )r&   r¶   Úaggregated_entitiesÚprevious_entityr�   Úcurrent_lengthÚprevious_lengths   &&     r   r­   Ú:TokenClassificationPipeline.aggregate_overlapping_entitiesw  sÒ   € Üˆx‹=˜AÔØˆOÜ˜(Ñ(<Ô=ˆØ ÐØ" 1�+ˆÛˆFØ˜wÕ'¨6°'­?ÖS¸_ÈUÕ=S×SØ!'¨¥°¸µÕ!@�Ø"1°%Õ"8¸?È7Õ;SÕ"S�à"Ô4Ø%Ô8Ø˜w�¨/¸'Õ*BÔBà&,’Oñ Cñ 9ð
 $×*Ñ*¨?Ô;Ø"(’ñ ð 	×"Ñ" ?Ô3Ø"Ð"r   c                ó0  <€ V ^8„  d   QhRS[ RS[P                  RS[P                  RS[S[S[S[3,          ,          R,          RS[P                  RS[RS[S[R,          ,          R,          R	S[S[S[S[3,          ,          R,          R
S[S[,          /	# )r   r|   r{   r³   r   Nr“   rN   r~   r   rc   )r   r§   Úndarrayr   r!   rP   r4   rd   )r   r   s   "€r   r   rR   �  s²   ø€ ÷ Fñ FáðFñ —:‘:ðFñ —
‘
ð	Fñ
 ™U¡3© 8�_Õ-°Õ4ðFñ  ŸZ™ZðFñ 2ðFñ ‘s˜T•zÕ" TÕ)ðFñ  ¡¡c©3 h¥Õ0°4Õ7ðFñ 
‰d�ñFr   c	                óô  € . p	\        V4       EFå  w  r«WZ,          '       d   K  V P                  P                  \        W*,          4      4      pVEe|   WJ,          w  rÞVe.   Ve*   Wz,          pVe   W�,          w  ppVV,          pVV,          p\	        V\        4      '       g!   VP                  4       pVP                  4       pWV p\        V P                  RR4      '       dJ   \        V P                  P                  P                  RR4      '       d   \        V4      \        V4      8g  pMqV\        P                  \        P                  \        P                  09   d   \        P                  ! R\         4       V^ 8„  ;'       d    RW^,
          V^,            9  p\        W*,          4      V P                  P"                  8X  d   TpRpMRpRpRpRVRVR	VR
VRV
RV/pV	P%                  V4       EKè  	  V	# )zTFuse various numpy arrays into dicts with all the information needed for aggregationNÚ
_tokenizerÚcontinuing_subword_prefixz?Tokenizer does not support real words, using fallback heuristicrT   Fr‰   r³   r»   rÁ   ÚindexÚ
is_subword)Ú	enumeraterZ   Úconvert_ids_to_tokensrP   r    ÚitemÚgetattrrÌ   r•   r"   r4   r<   r=   r>   ÚwarningsÚwarnÚUserWarningÚunk_token_idr‚   )r&   r|   r{   r³   r   r“   rN   r~   r   r´   ÚidxÚtoken_scoresr‰   Ú	start_indÚend_indÚ
word_indexÚ
start_charÚ_Úword_refrÏ   Ú
pre_entitys   &&&&&&&&&            r   rª   Ú/TokenClassificationPipeline.gather_pre_entities�  så  € ð ˆÜ!*¨6×!2ÑˆCà"×'Ô'Ùà—>‘>×7Ñ7¼¸I½NÓ8KÓLˆDØÓ)Ø%3Õ%8Ñ"�	ð Ò'Ð,=Ò,IØ!)¥�JØ!Ò-Ø(9Õ(E™˜
 AØ! ZÕ/˜	Ø :Õ-˜ä! )¬S×1Ò1Ø )§¡Ó 0�IØ%Ÿl™l›n�GØ#¨gÐ6�Ü˜4Ÿ>™>¨<¸×>Ò>Ä7Ø—N‘N×-Ñ-×3Ñ3Ð5PÐRV÷Dò Dô
 "% T£¬c°(«mÑ!;‘Jð ,Ü+×1Ñ1Ü+×3Ñ3Ü+×/Ñ/ð0ô ô
 !ŸšØ]Ü'ôð "+¨Q¡×!eÐ!e°3¸hÐSTÅ}ÐW`ÐcdÕWdÐ>eÑ3e�Jä�y•~Ó&¨$¯.©.×*EÑ*EÔEØ#�DØ!&�Jøà �	Ø�Ø"�
ð ˜Ø˜,Ø˜Ø�wØ˜Ø˜jðˆJð ×Ñ 
×+ñq "3ðr Ðr   c                óL   <€ V ^8„  d   QhRS[ S[,          RS[RS[ S[,          /# )r   r´   rN   rc   ©r   rd   r4   )r   r   s   "€r   r   rR   Õ  s.   ø€ ÷ -ñ -¡d©4¥jð -ÑH[ð -Ñ`dÑeiÕ`jñ -r   c                óä  € V\         P                  \         P                  09   d”   . pV FŠ  pVR ,          P                  4       pVR ,          V,          pRV P                  P
                  P                  V,          RVRVR,          RVR,          RVR,          RVR,          /pVP                  V4       KŒ  	  MV P                  W4      pV\         P                  8X  d   V# V P                  V4      # )r³   r�   rÂ   rÎ   r‰   r»   rÁ   )
r4   r:   r;   Úargmaxr•   ÚconfigÚid2labelr‚   Úaggregate_wordsÚgroup_entities)r&   r´   rN   r¶   rà   Ú
entity_idxrÂ   r�   s   &&&     r   r«   Ú%TokenClassificationPipeline.aggregateÕ  sá   € ØÔ$7×$<Ñ$<Ô>Q×>XÑ>XÐ#YÔYØˆHÛ*�
Ø'¨Õ1×8Ñ8Ó:�
Ø" 8Õ,¨ZÕ8�à˜dŸj™j×/Ñ/×8Ñ8¸ÕDØ˜UØ˜Z¨Õ0Ø˜J vÕ.Ø˜Z¨Õ0Ø˜: eÕ,ð�ð —‘ Ö'ò +ð ×+Ñ+¨LÓOˆHàÔ#6×#;Ñ#;Ô;ØˆOØ×"Ñ" 8Ó,Ð,r   c                ó<   <€ V ^8„  d   QhRS[ S[,          RS[RS[/# ©r   r¶   rN   rc   rã   )r   r   s   "€r   r   rR   ë  s(   ø€ ÷ ñ ¡t©D¥zð ÑI\ð Ñaeñ r   c                ó¸  € V P                   P                  V Uu. uF  q3R ,          NK  	  up4      pV\        P                  8X  dR   V^ ,          R,          pVP	                  4       pWV,          pV P
                  P                  P                  V,          pEMV\        P                  8X  dX   \        VR R7      pVR,          pVP	                  4       pWV,          pV P
                  P                  P                  V,          pM¤V\        P                  8X  d…   \        P                  ! V Uu. uF  q3R,          NK  	  up4      p\        P                  ! V^ R7      p	V	P	                  4       p
V P
                  P                  P                  V
,          pWš,          pM\        R4      hRVRVR VRV^ ,          R,          R	VR
,          R	,          /pV# u upi u upi )r‰   r³   c                 ó0   € V R ,          P                  4       # )r³   )r9   )r�   s   &r   r½   Ú<TokenClassificationPipeline.aggregate_word.<locals>.<lambda>ó  s   € ¸&ÀÕ:J×:NÑ:NÔ:Pr   r¿   )r›   zInvalid aggregation_strategyr�   rÂ   r»   rÁ   rŸ   )rZ   Úconvert_tokens_to_stringr4   r<   rå   r•   ræ   rç   r>   r9   r=   r§   ÚstackÚnanmeanr%   )r&   r¶   rN   r�   r‰   r³   rØ   rÂ   Ú
max_entityÚaverage_scoresrê   Ú
new_entitys   &&&         r   Úaggregate_wordÚ*TokenClassificationPipeline.aggregate_wordë  s‹  € Ø�~‰~×6Ñ6ÑU]Ó7^ÑU]È6¸v¿¸ÑU]Ñ7^Ó_ˆØÔ#6×#<Ñ#<Ô<Ø˜a•[ Õ*ˆFØ—-‘-“/ˆCØ•KˆEØ—Z‘Z×&Ñ&×/Ñ/°Õ4ŠFØ!Ô%8×%<Ñ%<Ô<Ü˜XÑ+PÔQˆJØ Õ)ˆFØ—-‘-“/ˆCØ•KˆEØ—Z‘Z×&Ñ&×/Ñ/°Õ4‰FØ!Ô%8×%@Ñ%@Ô@Ü—X’X¹hÓG¹h°F h×/Ð/¹hÑGÓHˆFÜŸZšZ¨°QÔ7ˆNØ'×.Ñ.Ó0ˆJØ—Z‘Z×&Ñ&×/Ñ/°
Õ;ˆFØ"Õ.‰EäÐ;Ó<Ð<à�fØ�UØ�DØ�X˜a•[ Õ)Ø�8˜B•< Õ&ð
ˆ
ð Ðùò7 8_ùò Hs   šGÄ-Gc                óL   <€ V ^8„  d   QhRS[ S[,          RS[RS[ S[,          /# rí   rã   )r   r   s   "€r   r   rR   	  s.   ø€ ÷ ñ ©©T­
ð ÑJ]ð ÑbfÑgkÕblñ r   c                ód  € V\         P                  \         P                  09   d   \        R4      h. pRpV FQ  pVf   V.pK  VR,          '       d   VP	                  V4       K.  VP	                  V P                  WB4      4       V.pKS  	  Ve!   VP	                  V P                  WB4      4       V# )zÚ
Override tokens from a given word that disagree to force agreement on word boundaries.

Example: micro|soft| com|pany| B-ENT I-NAME I-ENT I-ENT will be rewritten with first strategy as microsoft|
company| B-ENT I-ENT
z;NONE and SIMPLE strategies are invalid for word aggregationNrÏ   )r4   r:   r;   r%   r‚   r÷   )r&   r¶   rN   Úword_entitiesÚ
word_groupr�   s   &&&   r   rè   Ú+TokenClassificationPipeline.aggregate_words	  s²   € ð  Ü×$Ñ$Ü×&Ñ&ð$
ô 
ô ÐZÓ[Ð[àˆØˆ
ÛˆFØÒ!Ø$˜X’
Ø˜×%Ô%Ø×!Ñ! &Ö)à×$Ñ$ T×%8Ñ%8¸Ó%ZÔ[Ø$˜X’
ñ ð Ò!Ø× Ñ  ×!4Ñ!4°ZÓ!VÔWØÐr   c                ó6   <€ V ^8„  d   QhRS[ S[,          RS[/# ©r   r¶   rc   ©r   rd   )r   r   s   "€r   r   rR   %  s   ø€ ÷ ñ ©4±­:ð ¹$ñ r   c                ó˜  € V^ ,          R,          P                  R^4      R,          p\        P                  ! V Uu. uF  q"R,          NK  	  up4      pV Uu. uF  q"R,          NK  	  ppRXR\        P                  ! V4      RV P                  P                  V4      RV^ ,          R,          RVR,          R,          /pV# u upi u upi )	zŠ
Group together the adjacent tokens with the same entity predicted.

Args:
    entities (`dict`): The entities predicted by the pipeline.
r�   Ú-rÂ   r‰   rž   r»   rÁ   rŸ   )Úsplitr§   ró   ÚmeanrZ   rñ   )r&   r¶   r�   r³   Útokensrž   s   &&    r   Úgroup_sub_entitiesÚ.TokenClassificationPipeline.group_sub_entities%  s¹   € ð ˜!•˜XÕ&×,Ñ,¨S°!Ó4°RÕ8ˆÜ—’¹8ÓD¹8° GŸ_˜_¹8ÑDÓEˆÙ/7Ó8©x V˜—.�.©xˆÐ8ð ˜FØ”R—W’W˜V“_Ø�D—N‘N×;Ñ;¸FÓCØ�X˜a•[ Õ)Ø�8˜B•< Õ&ð
ˆð Ðùò EùÚ8s   ¼CÁCc                ó<   <€ V ^8„  d   QhRS[ RS[S[ S[ 3,          /# )r   Úentity_namerc   )r   r!   )r   r   s   "€r   r   rR   :  s#   ø€ ÷ ñ ¡3ð ©5±±c°­?ñ r   c                ó¤   € VP                  R 4      '       d   RpVR,          pW#3# VP                  R4      '       d   RpVR,          pW#3# RpTpW#3# )zB-ÚB:r   NNzI-ÚI)Ú
startswith)r&   r	  ÚbiÚtags   &&  r   Úget_tagÚ#TokenClassificationPipeline.get_tag:  sg   € Ø×!Ñ! $×'Ò'ØˆBØ˜b•/ˆCð ˆwˆð ×#Ñ# D×)Ò)ØˆBØ˜b•/ˆCð ˆwˆð ˆBØˆCØˆwˆr   c                óF   <€ V ^8„  d   QhRS[ S[,          RS[ S[,          /# rÿ   r   )r   r   s   "€r   r   rR   H  s#   ø€ ÷ #ñ #¡t©D¥zð #±d¹4µjñ #r   c                ó¢  € . p. pV Fœ  pV'       g   VP                  V4       K  V P                  VR,          4      w  rVV P                  VR,          R,          4      w  rxWh8X  d   VR8w  d   VP                  V4       Ky  VP                  V P                  V4      4       V.pKž  	  V'       d!   VP                  V P                  V4      4       V# )z“
Find and group together the adjacent tokens with the same entity predicted.

Args:
    entities (`dict`): The entities predicted by the pipeline.
r�   r  rŸ   )r‚   r  r  )	r&   r¶   Úentity_groupsÚentity_group_disaggr�   r  r  Úlast_biÚlast_tags	   &&       r   ré   Ú*TokenClassificationPipeline.group_entitiesH  sÂ   € ð ˆØ ÐãˆFß&Ø#×*Ñ*¨6Ô2Ùð —l‘l 6¨(Õ#3Ó4‰GˆBØ $§¡Ð-@ÀÕ-DÀXÕ-NÓ OÑˆGàŒ 2¨¤9à#×*Ñ*¨6Ö2ð ×$Ñ$ T×%<Ñ%<Ð=PÓ%QÔRØ'- hÒ#ñ' ÷( à× Ñ  ×!8Ñ!8Ð9LÓ!MÔNàÐr   )rI   rH   )NNNFNNrf   )NN)r,   r-   r.   r/   r0   Údefault_input_namesÚ_load_processorÚ_load_image_processorÚ_load_feature_extractorÚ_load_tokenizerr   rF   r_   r   r)   r�   r—   r4   r:   r·   r­   rª   r«   r÷   rè   r  r  ré   r1   r2   Ú__classcell__)rK   r   s   @@r   rA   rA   =   sÜ   ù‡ € ñ@"ðH &Ðà€OØ!ÐØ#ÐØ€Oá#EÓ#G÷ (÷99ò 99ðv ßOó ØOàß[ó Ø[÷#2ó #2ôJ6òp
ð. =P×<TÑ<TÐdhô 0òd#÷,Fò F÷P-ð -÷,ð ÷<ð ÷8ð ÷*ð ÷#÷ #ð #r   rA   )r#   rÔ   Útypingr   r   r¦   r§   Ú$models.bert.tokenization_bert_legacyr   Úutilsr   r   r   Úbaser	   r
   r   r   r¡   Úmodels.auto.modeling_autor   r   r4   rA   ÚNerPipeliner+   r   r   Ú<module>r%     s‹   ðÛ Û ß  ã å A÷ñ ÷
 TÓ Sñ ×ÒÛåXôF¨ô Fô:˜,ô ñ Ù¨4Ô0ðnóô>O -ó Oó?ð>Oðd *‚r   