§
    ‚ŠtjÞo  ã                   ó  — d dl Z d dlZd dlmZmZ d dlZddlmZ ddl	m
Z
mZmZ ddlmZmZmZmZ  e¦   «         r
d dlZddlmZ  G d	„ d
e¦  «        Z G d„ de
¦  «        Z e ed¬¦  «        d¦  «         G d„ de¦  «        ¦   «         ZeZdS )é    N)ÚAnyÚoverloadé   )ÚBasicTokenizer)ÚExplicitEnumÚadd_end_docstringsÚis_torch_availableé   )ÚArgumentHandlerÚChunkPipelineÚDatasetÚbuild_pipeline_init_args)Ú,MODEL_FOR_TOKEN_CLASSIFICATION_MAPPING_NAMESc                   ó0   — e Zd ZdZdeee         z  fd„ZdS )Ú"TokenClassificationArgumentHandlerz5
    Handles arguments for token classification.
    Úinputsc                 ó¨  — |                      dd¦  «        }|                      d¦  «        }|�Nt          |t          t          f¦  «        r2t	          |¦  «        dk    rt          |¦  «        }t	          |¦  «        }nft          |t
          ¦  «        r|g}d}nKt          �t          |t          ¦  «        st          |t          j        ¦  «        r||d |fS t          d¦  «        ‚|                      d¦  «        }|rUt          |t          ¦  «        rt          |d         t          ¦  «        r|g}t	          |¦  «        |k    rt          d¦  «        ‚||||fS )	NÚis_split_into_wordsFÚ	delimiterr   r
   zAt least one input is required.Úoffset_mappingz;offset_mapping should have the same batch size as the input)
ÚgetÚ
isinstanceÚlistÚtupleÚlenÚstrr   ÚtypesÚGeneratorTypeÚ
ValueError)Úselfr   Úkwargsr   r   Ú
batch_sizer   s          úi/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/pipelines/token_classification.pyÚ__call__z+TokenClassificationArgumentHandler.__call__   sN  € Ø$ŸjšjÐ)>ÀÑFÔFÐØ—J’J˜{Ñ+Ô+ˆ	àÐ¥*¨Vµd½E°]Ñ"CÔ"CÐÍÈFÉÌÐVWÊÈÝ˜&‘\”\ˆFÝ˜V™œˆJˆJÝ˜¥Ñ$Ô$ð 	@Ø�XˆFØˆJˆJÝÐ ¥Z°½Ñ%@Ô%@Ð ÅJÈvÕW\ÔWjÑDkÔDkÐ ØÐ.°°iÐ?Ð?åÐ>Ñ?Ô?Ð?àŸšÐ$4Ñ5Ô5ˆØð 	`Ý˜.­$Ñ/Ô/ð 2µJ¸~ÈaÔ?PÕRWÑ4XÔ4Xð 2Ø"0Ð!1�Ý�>Ñ"Ô" jÒ0Ð0Ý Ð!^Ñ_Ô_Ð_ØÐ*¨N¸IÐEÐEó    N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r$   © r%   r#   r   r      sH   € € € € € ðð ðF˜s T¨#¤Y™ð Fð Fð Fð Fð Fð Fr%   r   c                   ó&   — e Zd ZdZdZdZdZdZdZdS )ÚAggregationStrategyzDAll the valid aggregation strategies for TokenClassificationPipelineÚnoneÚsimpleÚfirstÚaverageÚmaxN)	r&   r'   r(   r)   ÚNONEÚSIMPLEÚFIRSTÚAVERAGEÚMAXr*   r%   r#   r,   r,   3   s-   € € € € € ØNÐNà€DØ€FØ€EØ€GØ
€C€C€Cr%   r,   T)Úhas_tokenizeraÙ	  
        ignore_labels (`list[str]`, defaults to `["O"]`):
            A list of labels to ignore.
        stride (`int`, *optional*):
            If stride is provided, the pipeline is applied on all the text. The text is split into chunks of size
            model_max_length. Works only with fast tokenizers and `aggregation_strategy` different from `NONE`. The
            value of this argument defines the number of overlapping tokens between chunks. In other words, the model
            will shift forward by `tokenizer.model_max_length - stride` tokens each step.
        aggregation_strategy (`str`, *optional*, defaults to `"none"`):
            The strategy to fuse (or not) tokens based on the model prediction.

                - "none" : Will simply not do any aggregation and simply return raw results from the model
                - "simple" : Will attempt to group entities following the default schema. (A, B-TAG), (B, I-TAG), (C,
                  I-TAG), (D, B-TAG2) (E, B-TAG2) will end up being [{"word": ABC, "entity": "TAG"}, {"word": "D",
                  "entity": "TAG2"}, {"word": "E", "entity": "TAG2"}] Notice that two consecutive B tags will end up as
                  different entities. On word based languages, we might end up splitting words undesirably : Imagine
                  Microsoft being tagged as [{"word": "Micro", "entity": "ENTERPRISE"}, {"word": "soft", "entity":
                  "NAME"}]. Look for FIRST, MAX, AVERAGE for ways to mitigate that and disambiguate words (on languages
                  that support that meaning, which is basically tokens separated by a space). These mitigations will
                  only work on real words, "New york" might still be tagged with two different entities.
                - "first" : (works only on word based models) Will use the `SIMPLE` strategy except that words, cannot
                  end up with different tags. Words will simply use the tag of the first token of the word when there
                  is ambiguity.
                - "average" : (works only on word based models) Will use the `SIMPLE` strategy except that words,
                  cannot end up with different tags. scores will be averaged first across tokens, and then the maximum
                  label is applied.
                - "max" : (works only on word based models) Will use the `SIMPLE` strategy except that words, cannot
                  end up with different tags. Word entity will simply be the token with the maximum score.c                   ór  ‡ — e Zd ZdZdZdZdZdZdZ e	¦   «         fˆ fd„	Z
	 	 	 	 	 	 d'dedz  deeeef                  dz  d	ed
edz  dedz  f
d„Zedededeeeef                  fd„¦   «         Zedee         dedeeeeef                           fd„¦   «         Zdeee         z  dedeeeef                  eeeeef                           z  fˆ fd„Zd(d„Zd„ Zej        dfd„Zd„ Z	 	 d)dedej        dej        deeeef                  dz  dej        dedeedz           dz  deeeef                  dz  dee         fd„Zdee         dedee         fd„Zd ee         dedefd!„Zd ee         dedee         fd"„Z d ee         defd#„Z!d$edeeef         fd%„Z"d ee         dee         fd&„Z#ˆ xZ$S )*ÚTokenClassificationPipelineuv	  
    Named Entity Recognition pipeline using any `ModelForTokenClassification`. See the [named entity recognition
    examples](../task_summary#named-entity-recognition) for more information.

    Example:

    ```python
    >>> from transformers import pipeline

    >>> token_classifier = pipeline(model="Jean-Baptiste/camembert-ner", aggregation_strategy="simple")
    >>> sentence = "Je m'appelle jean-baptiste et je vis Ã  montrÃ©al"
    >>> tokens = token_classifier(sentence)
    >>> tokens
    [{'entity_group': 'PER', 'score': 0.9931, 'word': 'jean-baptiste', 'start': 12, 'end': 26}, {'entity_group': 'LOC', 'score': 0.998, 'word': 'montrÃ©al', 'start': 38, 'end': 47}]

    >>> token = tokens[0]
    >>> # Start and end provide an easy way to highlight words in the original text.
    >>> sentence[token["start"] : token["end"]]
    ' jean-baptiste'

    >>> # Some models use the same idea to do part of speech.
    >>> syntaxer = pipeline(model="vblagoje/bert-english-uncased-finetuned-pos", aggregation_strategy="simple")
    >>> syntaxer("My name is Sarah and I live in London")
    [{'entity_group': 'PRON', 'score': 0.999, 'word': 'my', 'start': 0, 'end': 2}, {'entity_group': 'NOUN', 'score': 0.997, 'word': 'name', 'start': 3, 'end': 7}, {'entity_group': 'AUX', 'score': 0.994, 'word': 'is', 'start': 8, 'end': 10}, {'entity_group': 'PROPN', 'score': 0.999, 'word': 'sarah', 'start': 11, 'end': 16}, {'entity_group': 'CCONJ', 'score': 0.999, 'word': 'and', 'start': 17, 'end': 20}, {'entity_group': 'PRON', 'score': 0.999, 'word': 'i', 'start': 21, 'end': 22}, {'entity_group': 'VERB', 'score': 0.998, 'word': 'live', 'start': 23, 'end': 27}, {'entity_group': 'ADP', 'score': 0.999, 'word': 'in', 'start': 28, 'end': 30}, {'entity_group': 'PROPN', 'score': 0.999, 'word': 'london', 'start': 31, 'end': 37}]
    ```

    Learn more about the basics of using a pipeline in the [pipeline tutorial](../pipeline_tutorial)

    This token recognition pipeline can currently be loaded from [`pipeline`] using the following task identifier:
    `"ner"` (for predicting the classes of tokens in a sequence: person, organisation, location or miscellaneous).

    The models that this pipeline can use are models that have been fine-tuned on a token classification task. See the
    up-to-date list of available models on
    [huggingface.co/models](https://huggingface.co/models?filter=token-classification).
    Ú	sequencesFTc                 ó¦   •—  t          ¦   «         j        di |¤Ž |                      t          ¦  «         t	          d¬¦  «        | _        || _        d S )NF)Údo_lower_caser*   )ÚsuperÚ__init__Úcheck_model_typer   r   Ú_basic_tokenizerÚ_args_parser)r    Úargs_parserr!   Ú	__class__s      €r#   r>   z$TokenClassificationPipeline.__init__ˆ   sV   ø€ Ø�‰ŒÔÐ"Ð"˜6Ð"Ð"Ð"à×ÒÕJÑKÔKÐKå .¸UÐ CÑ CÔ CˆÔØ'ˆÔÐÐr%   NÚaggregation_strategyr   r   Ústrider   c                 ó  — i }||d<   |r	|€dn||d<   |�||d<   i }|�yt          |t          ¦  «        rt          |                     ¦   «                  }|t          j        t          j        t          j        hv r| j        j        st          d¦  «        ‚||d<   |�||d<   |�i|| j        j
        k    rt          d¦  «        ‚|t          j        k    rt          d	|› d
�¦  «        ‚| j        j        rdd|dœ}	|	|d<   nt          d¦  «        ‚|i |fS )Nr   ú r   r   z{Slow tokenizers cannot handle subwords. Please set the `aggregation_strategy` option to `"simple"` or use a fast tokenizer.rD   Úignore_labelszl`stride` must be less than `tokenizer.model_max_length` (or even lower if the tokenizer adds special tokens)zI`stride` was provided to process all the text but `aggregation_strategy="z&"`, please select another one instead.T)Úreturn_overflowing_tokensÚpaddingrE   Útokenizer_paramszm`stride` was provided to process all the text but you're using a slow tokenizer. Please use a fast tokenizer.)r   r   r,   Úupperr4   r6   r5   Ú	tokenizerÚis_fastr   Úmodel_max_lengthr2   )
r    rH   rD   r   r   rE   r   Úpreprocess_paramsÚpostprocess_paramsrK   s
             r#   Ú_sanitize_parametersz0TokenClassificationPipeline._sanitize_parameters�   s²  € ð ÐØ3FÐÐ/Ñ0àð 	UØ4=Ð4E¨S¨SÈ9Ð˜kÑ*àÐ%Ø2@ÐÐ.Ñ/àÐØÐ+ÝÐ.µÑ4Ô4ð YÝ':Ð;O×;UÒ;UÑ;WÔ;WÔ'XÐ$à$Ý'Ô-Õ/BÔ/FÕH[ÔHcÐdðeð eàœÔ.ðeõ !ð>ñô ð ð :NÐÐ5Ñ6ØÐ$Ø2?Ð˜Ñ/ØÐØ˜œÔ8Ò8Ð8Ý ð Cñô ð ð $Õ':Ô'?Ò?Ð?Ý ðUØ,ðUð Uð Uñô ð ð
 ”>Ô)ð à59Ø#'Ø"(ð(ð (Ð$ð
 =MÐ%Ð&8Ñ9Ð9å$ð8ñô ð ð ! "Ð&8Ð8Ð8r%   r   r!   Úreturnc                 ó   — d S ©Nr*   ©r    r   r!   s      r#   r$   z$TokenClassificationPipeline.__call__Ë   s   € ØLOÈCr%   c                 ó   — d S rU   r*   rV   s      r#   r$   z$TokenClassificationPipeline.__call__Î   s   € ØX[ÐX[r%   c                 óì   •—  | j         |fi |¤Ž\  }}}}||d<   ||d<   |r4t          d„ |D ¦   «         ¦  «        s t          ¦   «         j        |gfi |¤ŽS |r||d<    t          ¦   «         j        |fi |¤ŽS )a  
        Classify each token of the text(s) given as inputs.

        Args:
            inputs (`str` or `List[str]`):
                One or several texts (or one list of texts) for token classification. Can be pre-tokenized when
                `is_split_into_words=True`.

        Return:
            A list or a list of list of `dict`: Each result comes as a list of dictionaries (one for each token in the
            corresponding input, or each entity if this pipeline was instantiated with an aggregation_strategy) with
            the following keys:

            - **word** (`str`) -- The token/word classified. This is obtained by decoding the selected tokens. If you
              want to have the exact string in the original sentence, use `start` and `end`.
            - **score** (`float`) -- The corresponding probability for `entity`.
            - **entity** (`str`) -- The entity predicted for that token/word (it is named *entity_group* when
              *aggregation_strategy* is not `"none"`.
            - **index** (`int`, only present when `aggregation_strategy="none"`) -- The index of the corresponding
              token in the sentence.
            - **start** (`int`, *optional*) -- The index of the start of the corresponding entity in the sentence. Only
              exists if the offsets are available within the tokenizer
            - **end** (`int`, *optional*) -- The index of the end of the corresponding entity in the sentence. Only
              exists if the offsets are available within the tokenizer
        r   r   c              3   ó@   K  — | ]}t          |t          ¦  «        V — Œd S rU   )r   r   )Ú.0Úinputs     r#   ú	<genexpr>z7TokenClassificationPipeline.__call__.<locals>.<genexpr>ï   s,   è è € Ð*WÐ*WÀu­:°e½TÑ+BÔ+BÐ*WÐ*WÐ*WÐ*WÐ*WÐ*Wr%   r   )rA   Úallr=   r$   )r    r   r!   Ú_inputsr   r   r   rC   s          €r#   r$   z$TokenClassificationPipeline.__call__Ñ   s¼   ø€ ð6 CTÀ$ÔBSÐTZÐBeÐBeÐ^dÐBeÐBeÑ?ˆÐ$ n°iØ(;ˆÐ$Ñ%Ø'ˆˆ{ÑØð 	8¥sÐ*WÐ*WÐPVÐ*WÑ*WÔ*WÑ'WÔ'Wð 	8Ø#•5‘7”7Ô# V HÐ7Ð7°Ð7Ð7Ð7Øð 	6Ø'5ˆFÐ#Ñ$à�u‰wŒwÔ Ð1Ð1¨&Ð1Ð1Ð1r%   c              +   óÆ  ‡K  — |                      di ¦  «        }| j        j        o| j        j        dk    }d }|d         }|rŸ|d         }t          |t          ¦  «        st          d¦  «        ‚|}	|                     |	¦  «        }g }t          |¦  «        }
d}|	D ]>}|                     ||t          |¦  «        z   f¦  «         |t          |¦  «        |
z   z  }Œ?|	}d|d<   n&t          |t          ¦  «        st          d¦  «        ‚|} | j        |fd|d| j        j
        d	œ|¤Ž}|r| j        j
        st          d
¦  «        ‚|                      dd ¦  «         t          |d         ¦  «        }t          |¦  «        D ]eŠˆfd„|                     ¦   «         D ¦   «         }|�||d<   ‰dk    r|nd |d<   ‰|dz
  k    |d<   |�|                     ‰¦  «        |d<   ||d<   |V — Œfd S )NrK   r   r   r   zEWhen `is_split_into_words=True`, `sentence` must be a list of tokens.TzKWhen `is_split_into_words=False`, `sentence` must be an untokenized string.Úpt)Úreturn_tensorsÚ
truncationÚreturn_special_tokens_maskÚreturn_offsets_mappingz@is_split_into_words=True is only supported with fast tokenizers.Úoverflow_to_sample_mappingÚ	input_idsc                 óN   •— i | ]!\  }}||‰                               d ¦  «        “Œ"S )r   )Ú	unsqueeze)rZ   ÚkÚvÚis      €r#   ú
<dictcomp>z:TokenClassificationPipeline.preprocess.<locals>.<dictcomp>"  s/   ø€ ÐLÐLÐL±T°Q¸˜A˜q œtŸ~š~¨aÑ0Ô0ÐLÐLÐLr%   r   Úsentencer
   Úis_lastÚword_idsÚword_to_chars_map)ÚpoprM   rO   r   r   r   Újoinr   Úappendr   rN   ÚrangeÚitemsro   )r    rm   r   rP   rK   rb   rp   r   r   ÚwordsÚdelimiter_lenÚchar_offsetÚwordÚtext_to_tokenizer   Ú
num_chunksÚmodel_inputsrk   s                    @r#   Ú
preprocessz&TokenClassificationPipeline.preprocessö   sn  øè è € Ø,×0Ò0Ð1CÀRÑHÔHÐØ”^Ô4Ð\¸¼Ô9XÐ[\Ò9\ˆ
à ÐØ/Ð0EÔFÐØð 	(Ø)¨+Ô6ˆIÝ˜h­Ñ-Ô-ð jÝ Ð!hÑiÔiÐiØˆEØ —~’~ eÑ,Ô,ˆHà "ÐÝ 	™NœNˆMØˆKØð 9ð 9�Ø!×(Ò(¨+°{ÅSÈÁYÄYÑ7NÐ)OÑPÔPÐPØ�s 4™yœy¨=Ñ8Ñ8��ð  %ÐØ6:ÐÐ2Ñ3Ð3å˜h­Ñ,Ô,ð pÝ Ð!nÑoÔoÐoØ'Ðà�”Øð
àØ!Ø'+Ø#'¤>Ô#9ð
ð 
ð ð
ð 
ˆð ð 	a t¤~Ô'=ð 	aÝÐ_Ñ`Ô`Ð`à�
Š
Ð/°Ñ6Ô6Ð6Ý˜ Ô,Ñ-Ô-ˆ
å�zÑ"Ô"ð 	ð 	ˆAØLÐLÐLÐL¸V¿\º\¹^¼^ÐLÑLÔLˆLØÐ)Ø1?�Ð-Ñ.à34¸²6°6 x x¸tˆL˜Ñ$Ø&'¨:¸©>Ò&9ˆL˜Ñ#Ø Ð,Ø+1¯?ª?¸1Ñ+=Ô+=�˜ZÑ(Ø4E�Ð0Ñ1àÐÐÐÐð	ð 	r%   c                 ó€  — |                      d¦  «        }|                      dd ¦  «        }|                      d¦  «        }|                      d¦  «        }|                      dd ¦  «        }|                      dd ¦  «        } | j        d
i |¤Ž}t          |t          ¦  «        r|d         n|d         }	|	||||||d	œ|¥S )NÚspecial_tokens_maskr   rm   rn   ro   rp   Úlogitsr   )r€   r   r   rm   rn   ro   rp   r*   )rq   Úmodelr   Údict)
r    r|   r   r   rm   rn   ro   rp   Úoutputr€   s
             r#   Ú_forwardz$TokenClassificationPipeline._forward.  sè   € à*×.Ò.Ð/DÑEÔEÐØ%×)Ò)Ð*:¸DÑAÔAˆØ×#Ò# JÑ/Ô/ˆØ×"Ò" 9Ñ-Ô-ˆØ×#Ò# J°Ñ5Ô5ˆØ(×,Ò,Ð-@À$ÑGÔGÐà�”Ð+Ð+˜lÐ+Ð+ˆÝ%/°½Ñ%=Ô%=ÐL�˜Ô!Ð!À6È!Ä9ˆð Ø#6Ø,Ø ØØ Ø!2ð	
ð 	
ð ð	
ð 		
r%   c                 óÎ  ‡— ‰€dgŠg }|d                               d¦  «        }|D �]“}|d         d         j        t          j        t          j        fv r>|d         d                              t          j        ¦  «                             ¦   «         }n |d         d                              ¦   «         }|d         d         }|d         d         }	|d         �|d         d         nd }
|d         d                              ¦   «         }|                      d	¦  «        }t          j	        |d
d¬¦  «        }t          j
        ||z
  ¦  «        }||                     d
d¬¦  «        z  }|                      ||	||
||||¬¦  «        }|                      ||¦  «        }ˆfd„|D ¦   «         }|                     |¦  «         �Œ•t          |¦  «        }|dk    r|                      |¦  «        }|S )NÚOr   rp   r€   rm   rf   r   r   ro   éÿÿÿÿT)ÚaxisÚkeepdims)ro   rp   c                 ót   •— g | ]4}|                      d d¦  «        ‰vr|                      dd¦  «        ‰v¯2|‘Œ5S )ÚentityNÚentity_group)r   )rZ   r‹   rH   s     €r#   ú
<listcomp>z;TokenClassificationPipeline.postprocess.<locals>.<listcomp>k  sX   ø€ ð ð ð àØ—:’:˜h¨Ñ-Ô-°]ÐBÐBØ—J’J˜~¨tÑ4Ô4¸MÐIÐIð ð JÐIÐIr%   r
   )r   ÚdtypeÚtorchÚbfloat16Úfloat16ÚtoÚfloat32ÚnumpyÚnpr1   ÚexpÚsumÚgather_pre_entitiesÚ	aggregateÚextendr   Úaggregate_overlapping_entities)r    Úall_outputsrD   rH   Úall_entitiesrp   Úmodel_outputsr€   rm   rf   r   r   ro   ÚmaxesÚshifted_expÚscoresÚpre_entitiesÚgrouped_entitiesÚentitiesr{   s      `                r#   Úpostprocessz'TokenClassificationPipeline.postprocessE  s  ø€ ØÐ Ø ˜EˆMØˆð (¨œN×.Ò.Ð/BÑCÔCÐà(ð $	*ñ $	*ˆMØ˜XÔ& qÔ)Ô/µE´NÅEÄMÐ3RÐRÐRØ& xÔ0°Ô3×6Ò6µu´}ÑEÔE×KÒKÑMÔM��à& xÔ0°Ô3×9Ò9Ñ;Ô;�à" 1”~ jÔ1ˆHØ% kÔ2°1Ô5ˆIà6CÐDTÔ6UÐ6a�Ð.Ô/°Ô2Ð2Ðgkð ð #0Ð0EÔ"FÀqÔ"I×"OÒ"OÑ"QÔ"QÐØ$×(Ò(¨Ñ4Ô4ˆHå”F˜6¨°TÐ:Ñ:Ô:ˆEÝœ& ¨%¡Ñ0Ô0ˆKØ  ;§?¢?¸ÀT ?Ñ#JÔ#JÑJˆFà×3Ò3ØØØØØ#Ø$Ø!Ø"3ð 4ñ 	ô 	ˆLð  $Ÿ~š~¨lÐ<PÑQÔQÐðð ð ð à.ðñ ô ˆHð ×Ò Ñ)Ô)Ð)Ñ)Ý˜Ñ%Ô%ˆ
Ø˜Š>ˆ>Ø×>Ò>¸|ÑLÔLˆLØÐr%   c                 ó”  — t          |¦  «        dk    r|S t          |d„ ¬¦  «        }g }|d         }|D ]~}|d         |d         cxk    r|d         k     rFn nC|d         |d         z
  }|d         |d         z
  }||k    s||k    r|d         |d         k    r|}Œg|                     |¦  «         |}Œ|                     |¦  «         |S )Nr   c                 ó   — | d         S )NÚstartr*   )Úxs    r#   ú<lambda>zLTokenClassificationPipeline.aggregate_overlapping_entities.<locals>.<lambda>z  s
   € °!°G´*€ r%   ©Úkeyr¨   ÚendÚscore)r   Úsortedrs   )r    r¤   Úaggregated_entitiesÚprevious_entityr‹   Úcurrent_lengthÚprevious_lengths          r#   r›   z:TokenClassificationPipeline.aggregate_overlapping_entitiesw  s  € Ýˆx‰=Œ=˜AÒÐØˆOÝ˜(Ð(<Ð(<Ð=Ñ=Ô=ˆØ ÐØ" 1œ+ˆØð 	)ð 	)ˆFØ˜wÔ'¨6°'¬?ÐSÐSÒSÐS¸_ÈUÔ=SÒSÐSÐSÐSÐSØ!'¨¤°¸´Ñ!@�Ø"1°%Ô"8¸?È7Ô;SÑ"S�à" _Ò4Ð4Ø%¨Ò8Ð8Ø˜wœ¨/¸'Ô*BÒBÐBà&,�Oøà#×*Ò*¨?Ñ;Ô;Ð;Ø"(��Ø×"Ò" ?Ñ3Ô3Ð3Ø"Ð"r%   rm   rf   r¡   r   ro   rp   c	                 óˆ  — g }	t          |¦  «        D �]®\  }
}||
         rŒ| j                             t          ||
         ¦  «        ¦  «        }|��K||
         \  }}|�!|�||
         }|�||         \  }}||z  }||z  }t	          |t          ¦  «        s(|                     ¦   «         }|                     ¦   «         }|||…         }t          | j        dd¦  «        rAt          | j        j        j        dd¦  «        r!t          |¦  «        t          |¦  «        k    }nW|t          j        t          j        t          j        hv rt          j        dt           ¦  «         |dk    od||dz
  |dz   …         v}t          ||
         ¦  «        | j        j        k    r|}d}nd}d}d}|||||
|d	œ}|	                     |¦  «         �Œ°|	S )
zTFuse various numpy arrays into dicts with all the information needed for aggregationNÚ
_tokenizerÚcontinuing_subword_prefixz?Tokenizer does not support real words, using fallback heuristicr   rG   r
   F)ry   r¡   r¨   r­   ÚindexÚ
is_subword)Ú	enumeraterM   Úconvert_ids_to_tokensÚintr   ÚitemÚgetattrrµ   r�   r   r,   r4   r5   r6   ÚwarningsÚwarnÚUserWarningÚunk_token_idrs   )r    rm   rf   r¡   r   r   rD   ro   rp   r¢   ÚidxÚtoken_scoresry   Ú	start_indÚend_indÚ
word_indexÚ
start_charÚ_Úword_refr¸   Ú
pre_entitys                        r#   r˜   z/TokenClassificationPipeline.gather_pre_entities�  s  € ð ˆÝ!*¨6Ñ!2Ô!2ð 8	,ñ 8	,ÑˆC�à" 3Ô'ð Øà”>×7Ò7½¸IÀc¼NÑ8KÔ8KÑLÔLˆDØÑ)Ø%3°CÔ%8Ñ"�	˜7ð Ð'Ð,=Ð,IØ!)¨#¤�JØ!Ð-Ø(9¸*Ô(E™˜
 AØ! ZÑ/˜	Ø :Ñ-˜å! )­SÑ1Ô1ð -Ø )§¢Ñ 0Ô 0�IØ%Ÿlšl™nœn�GØ# I¨gÐ$5Ô6�Ý˜4œ>¨<¸Ñ>Ô>ð fÅ7Ø”NÔ-Ô3Ð5PÐRVñDô Dð fõ
 "% T¡¤­c°(©m¬mÒ!;�J�Jð ,Ý+Ô1Ý+Ô3Ý+Ô/ð0ð ð õ
 !œØ]Ý'ñô ð ð "+¨Q¢Ð!e°3¸hÀyÐSTÁ}ÐW`ÐcdÑWdÐGdÔ>eÐ3e�Jå�y ”~Ñ&Ô&¨$¬.Ô*EÒEÐEØ#�DØ!&�Jøà �	Ø�Ø"�
ð Ø&Ø"ØØØ(ðð ˆJð ×Ò 
Ñ+Ô+Ð+Ñ+ØÐr%   r¢   c                 ó¦  — |t           j        t           j        hv r{g }|D ]u}|d                              ¦   «         }|d         |         }| j        j        j        |         ||d         |d         |d         |d         dœ}|                     |¦  «         Œvn|                      ||¦  «        }|t           j        k    r|S |  	                    |¦  «        S )Nr¡   r·   ry   r¨   r­   )r‹   r®   r·   ry   r¨   r­   )
r,   r2   r3   Úargmaxr�   ÚconfigÚid2labelrs   Úaggregate_wordsÚgroup_entities)r    r¢   rD   r¤   rÊ   Ú
entity_idxr®   r‹   s           r#   r™   z%TokenClassificationPipeline.aggregateÕ  sî   € ØÕ$7Ô$<Õ>QÔ>XÐ#YÐYÐYØˆHØ*ð (ð (�
Ø'¨Ô1×8Ò8Ñ:Ô:�
Ø" 8Ô,¨ZÔ8�à"œjÔ/Ô8¸ÔDØ"Ø'¨Ô0Ø& vÔ.Ø'¨Ô0Ø% eÔ,ðð �ð —’ Ñ'Ô'Ð'Ð'ð(ð ×+Ò+¨LÐ:NÑOÔOˆHàÕ#6Ô#;Ò;Ð;ØˆOØ×"Ò" 8Ñ,Ô,Ð,r%   r¤   c                 óü  — | j                              d„ |D ¦   «         ¦  «        }|t          j        k    rB|d         d         }|                     ¦   «         }||         }| j        j        j        |         }nå|t          j        k    rNt          |d„ ¬¦  «        }|d         }|                     ¦   «         }||         }| j        j        j        |         }n‡|t          j
        k    rht          j        d„ |D ¦   «         ¦  «        }t          j        |d¬¦  «        }	|	                     ¦   «         }
| j        j        j        |
         }|	|
         }nt          d¦  «        ‚||||d         d	         |d
         d         dœ}|S )Nc                 ó   — g | ]
}|d          ‘ŒS ©ry   r*   ©rZ   r‹   s     r#   r�   z>TokenClassificationPipeline.aggregate_word.<locals>.<listcomp>ì  s   € Ð7^Ð7^Ð7^È6¸¸v¼Ð7^Ð7^Ð7^r%   r   r¡   c                 ó6   — | d                               ¦   «         S )Nr¡   )r1   )r‹   s    r#   rª   z<TokenClassificationPipeline.aggregate_word.<locals>.<lambda>ó  s   € ¸&ÀÔ:J×:NÒ:NÑ:PÔ:P€ r%   r«   c                 ó   — g | ]
}|d          ‘ŒS )r¡   r*   rÕ   s     r#   r�   z>TokenClassificationPipeline.aggregate_word.<locals>.<listcomp>ù  s   € ÐGÐGÐG°F˜v hÔ/ÐGÐGÐGr%   )rˆ   zInvalid aggregation_strategyr¨   r‡   r­   )r‹   r®   ry   r¨   r­   )rM   Úconvert_tokens_to_stringr,   r4   rÌ   r�   rÍ   rÎ   r6   r1   r5   r•   ÚstackÚnanmeanr   )r    r¤   rD   ry   r¡   rÂ   r®   r‹   Ú
max_entityÚaverage_scoresrÑ   Ú
new_entitys               r#   Úaggregate_wordz*TokenClassificationPipeline.aggregate_wordë  s{  € ØŒ~×6Ò6Ð7^Ð7^ÐU]Ð7^Ñ7^Ô7^Ñ_Ô_ˆØÕ#6Ô#<Ò<Ð<Ø˜a”[ Ô*ˆFØ—-’-‘/”/ˆCØ˜3”KˆEØ”ZÔ&Ô/°Ô4ˆFˆFØ!Õ%8Ô%<Ò<Ð<Ý˜XÐ+PÐ+PÐQÑQÔQˆJØ Ô)ˆFØ—-’-‘/”/ˆCØ˜3”KˆEØ”ZÔ&Ô/°Ô4ˆFˆFØ!Õ%8Ô%@Ò@Ð@Ý”XÐGÐG¸hÐGÑGÔGÑHÔHˆFÝœZ¨°QÐ7Ñ7Ô7ˆNØ'×.Ò.Ñ0Ô0ˆJØ”ZÔ&Ô/°
Ô;ˆFØ" :Ô.ˆEˆEåÐ;Ñ<Ô<Ð<àØØØ˜a”[ Ô)Ø˜B”< Ô&ð
ð 
ˆ
ð Ðr%   c                 ó`  — |t           j        t           j        hv rt          d¦  «        ‚g }d}|D ]R}|€|g}Œ|d         r|                     |¦  «         Œ&|                     |                      ||¦  «        ¦  «         |g}ŒS|�)|                     |                      ||¦  «        ¦  «         |S )zú
        Override tokens from a given word that disagree to force agreement on word boundaries.

        Example: micro|soft| com|pany| B-ENT I-NAME I-ENT I-ENT will be rewritten with first strategy as microsoft|
        company| B-ENT I-ENT
        z;NONE and SIMPLE strategies are invalid for word aggregationNr¸   )r,   r2   r3   r   rs   rÞ   )r    r¤   rD   Úword_entitiesÚ
word_groupr‹   s         r#   rÏ   z+TokenClassificationPipeline.aggregate_words	  sÞ   € ð  ÝÔ$ÝÔ&ð$
ð 
ð 
õ ÐZÑ[Ô[Ð[àˆØˆ
Øð 	&ð 	&ˆFØÐ!Ø$˜X�
�
Ø˜Ô%ð &Ø×!Ò! &Ñ)Ô)Ð)Ð)à×$Ò$ T×%8Ò%8¸ÐEYÑ%ZÔ%ZÑ[Ô[Ð[Ø$˜X�
�
àÐ!Ø× Ò  ×!4Ò!4°ZÐAUÑ!VÔ!VÑWÔWÐWØÐr%   c                 ó>  — |d         d                               dd¦  «        d         }t          j        d„ |D ¦   «         ¦  «        }d„ |D ¦   «         }|t          j        |¦  «        | j                             |¦  «        |d         d         |d         d	         d
œ}|S )zª
        Group together the adjacent tokens with the same entity predicted.

        Args:
            entities (`dict`): The entities predicted by the pipeline.
        r   r‹   ú-r
   r‡   c                 ó   — g | ]
}|d          ‘ŒS )r®   r*   rÕ   s     r#   r�   zBTokenClassificationPipeline.group_sub_entities.<locals>.<listcomp>.  s   € ÐDÐDÐD°˜V Gœ_ÐDÐDÐDr%   c                 ó   — g | ]
}|d          ‘ŒS rÔ   r*   rÕ   s     r#   r�   zBTokenClassificationPipeline.group_sub_entities.<locals>.<listcomp>/  s   € Ð8Ð8Ð8 V�&˜”.Ð8Ð8Ð8r%   r¨   r­   )rŒ   r®   ry   r¨   r­   )Úsplitr•   rÚ   ÚmeanrM   rØ   )r    r¤   r‹   r¡   ÚtokensrŒ   s         r#   Úgroup_sub_entitiesz.TokenClassificationPipeline.group_sub_entities%  s§   € ð ˜!”˜XÔ&×,Ò,¨S°!Ñ4Ô4°RÔ8ˆÝ”ÐDÐD¸8ÐDÑDÔDÑEÔEˆØ8Ð8¨xÐ8Ñ8Ô8ˆð #Ý”W˜V‘_”_Ø”N×;Ò;¸FÑCÔCØ˜a”[ Ô)Ø˜B”< Ô&ð
ð 
ˆð Ðr%   Úentity_namec                 óš   — |                      d¦  «        rd}|dd …         }n&|                      d¦  «        rd}|dd …         }nd}|}||fS )NzB-ÚBr   zI-ÚI)Ú
startswith)r    rê   ÚbiÚtags       r#   Úget_tagz#TokenClassificationPipeline.get_tag:  sk   € Ø×!Ò! $Ñ'Ô'ð 
	ØˆBØ˜a˜b˜b”/ˆCˆCØ×#Ò# DÑ)Ô)ð 	ØˆBØ˜a˜b˜b”/ˆCˆCð ˆBØˆCØ�3ˆwˆr%   c                 óº  — g }g }|D ]©}|s|                      |¦  «         Œ|                      |d         ¦  «        \  }}|                      |d         d         ¦  «        \  }}||k    r|dk    r|                      |¦  «         Œ~|                      |                      |¦  «        ¦  «         |g}Œª|r(|                      |                      |¦  «        ¦  «         |S )z³
        Find and group together the adjacent tokens with the same entity predicted.

        Args:
            entities (`dict`): The entities predicted by the pipeline.
        r‹   r‡   rì   )rs   rñ   ré   )	r    r¤   Úentity_groupsÚentity_group_disaggr‹   rï   rð   Úlast_biÚlast_tags	            r#   rÐ   z*TokenClassificationPipeline.group_entitiesH  s  € ð ˆØ Ðàð 	/ð 	/ˆFØ&ð Ø#×*Ò*¨6Ñ2Ô2Ð2Øð —l’l 6¨(Ô#3Ñ4Ô4‰GˆB�Ø $§¢Ð-@ÀÔ-DÀXÔ-NÑ OÔ OÑˆG�Xà�hŠˆ 2¨¢9 9à#×*Ò*¨6Ñ2Ô2Ð2Ð2ð ×$Ò$ T×%<Ò%<Ð=PÑ%QÔ%QÑRÔRÐRØ'- hÐ#Ð#Øð 	Oà× Ò  ×!8Ò!8Ð9LÑ!MÔ!MÑNÔNÐNàÐr%   )NNNFNNrU   )NN)%r&   r'   r(   r)   Údefault_input_namesÚ_load_processorÚ_load_image_processorÚ_load_feature_extractorÚ_load_tokenizerr   r>   r,   r   r   r»   Úboolr   rR   r   r   r‚   r$   r}   r„   r2   r¥   r›   r•   Úndarrayr˜   r™   rÞ   rÏ   ré   rñ   rÐ   Ú__classcell__)rC   s   @r#   r9   r9   =   sõ  ø€ € € € € ð@"ð "ðH &Ðà€OØ!ÐØ#ÐØ€Oà#EÐ#EÑ#GÔ#Gð (ð (ð (ð (ð (ð (ð Ø;?Ø7;Ø$)Ø!Ø $ð99ð 99ð 2°DÑ8ð99ð ˜U 3¨ 8œ_Ô-°Ñ4ð	99ð
 "ð99ð �d‘
ð99ð ˜‘:ð99ð 99ð 99ð 99ðv ØO˜sÐO¨cÐO°d¸4ÀÀSÀ¼>Ô6JÐOÐOÐOñ „XØOàØ[˜t CœyÐ[°CÐ[¸DÀÀdÈ3ÐPSÈ8ÄnÔAUÔ<VÐ[Ð[Ð[ñ „XØ[ð#2˜s T¨#¤Y™ð #2¸#ð #2À$ÀtÈCÐQTÈHÄ~ÔBVÐY]Ð^bÐcgÐhkÐmpÐhpÔcqÔ^rÔYsÑBsð #2ð #2ð #2ð #2ð #2ð #2ðJ6ð 6ð 6ð 6ðp
ð 
ð 
ð. =PÔ<TÐdhð 0ð 0ð 0ð 0ðd#ð #ð #ð< -1Ø:>ðFð FàðFð ”:ðFð ”
ð	Fð
 ˜U 3¨ 8œ_Ô-°Ñ4ðFð  œZðFð 2ðFð �s˜T‘zÔ" TÑ)ðFð    c¨3 h¤Ô0°4Ñ7ðFð 
ˆdŒðFð Fð Fð FðP- d¨4¤jð -ÐH[ð -Ð`dÐeiÔ`jð -ð -ð -ð -ð, t¨D¤zð ÐI\ð Ðaeð ð ð ð ð<¨¨T¬
ð ÐJ]ð ÐbfÐgkÔblð ð ð ð ð8¨4°¬:ð ¸$ð ð ð ð ð* 3ð ¨5°°c°¬?ð ð ð ð ð# t¨D¤zð #°d¸4´jð #ð #ð #ð #ð #ð #ð #ð #r%   r9   )r   r¾   Útypingr   r   r”   r•   Ú$models.bert.tokenization_bert_legacyr   Úutilsr   r   r	   Úbaser   r   r   r   r�   Úmodels.auto.modeling_autor   r   r,   r9   ÚNerPipeliner*   r%   r#   ú<module>r     s¢  ðØ €€€Ø €€€Ø  Ð  Ð  Ð  Ð  Ð  Ð  Ð  à Ð Ð Ð à AÐ AÐ AÐ AÐ AÐ Aðð ð ð ð ð ð ð ð ð ð
 TÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ Sð ÐÑÔð YØ€L€L€LàXÐXÐXÐXÐXÐXðFð Fð Fð Fð F¨ñ Fô Fð Fð:ð ð ð ð ˜,ñ ô ð ð ÐØÐ¨4Ð0Ñ0Ô0ðnñô ð>Oð Oð Oð Oð O -ñ Oô Oñ?ô ð>Oðd *€€€r%   