§
    ‚Štj‘  ã                   óv   — d dl Z d dlZd dlmZmZmZmZmZ d dlm	Z	 ddl
mZ dddœZ G d	„ d
e¦  «        Zd
gZdS )é    N)Ú	TokenizerÚdecodersÚnormalizersÚpre_tokenizersÚ
processors)ÚUnigramé   )ÚTokenizersBackendzspiece.modelztokenizer.json)Ú
vocab_fileÚtokenizer_filec                   ó�   ‡ — e Zd ZdZeZddgZeZ	 	 	 	 	 	 	 	 dˆ fd	„	Z	d
„ Z
d„ Z	 	 	 ddeee         z  dededz  dedef
ˆ fd„Zˆ xZS )ÚLasrTokenizeraÀ  
    Construct a LASR tokenizer (backed by HuggingFace's *tokenizers* library). Based on
    [Unigram](https://huggingface.co/docs/tokenizers/python/latest/components.html?highlight=unigram#models).

    This tokenizer inherits from [`TokenizersBackend`] which contains most of the main methods. Users should
    refer to this superclass for more information regarding those methods.

    Args:
        vocab_file (`str`, *optional*):
            [SentencePiece](https://github.com/google/sentencepiece) file (generally has a *.spm* extension) that
            contains the vocabulary necessary to instantiate a tokenizer.
        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the end of sequence.
            The token used is the `sep_token`.

            </Tip>

        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        extra_ids (`int`, *optional*, defaults to 100):
            Add a number of extra ids added to the vocabulary for use as sentinels. These tokens are accessible as
            "<extra_id_{%d}>" where "{%d}" is a number between 0 and extra_ids-1. These tokens can be retrieved by
            calling get_sentinel_tokens method and token ids can be by calling get_sentinel_token_ids method
        additional_special_tokens (`list[str]`, *optional*):
            Additional special tokens used by the tokenizer.
        vocab (`str`, `dict` or `list`, *optional*):
            Custom vocabulary dict. If not provided, a minimal vocabulary is created using the special tokens.
    Ú	input_idsÚattention_maskú</s>ú<unk>ú<pad>Néd   c	           	      ó  •— || _         |�ld„ |D ¦   «         }
t          |
¦  «        dk     r|d„ t          |¦  «        D ¦   «         z  }nK|dk    r)|t          |
¦  «        k    rt          d|› d|› d�¦  «        ‚nd„ t          |¦  «        D ¦   «         }
|
}|�|| _        not          |¦  «        d	ft          |¦  «        d	ft          |¦  «        d	fd
g| _        t          |dz
  dd¦  «        D ]"}| j                             d|› d�d	f¦  «         Œ#t          t          | j        dd¬¦  «        ¦  «        | _	        |�t          j        |¦  «        | j	        _        t          j        t          j        ¦   «         t          j        ddd¬¦  «        g¦  «        | j	        _        t%          j        ddd¬¦  «        | j	        _         t)          ¦   «         j        d|||||dœ|	¤Ž t-          j        ddgg d¢d| j        fg¬¦  «        | j	        _        d S )Nc                 ó4   — g | ]}d t          |¦  «        v ¯|‘ŒS )ú
<extra_id_)Ústr)Ú.0Úxs     úh/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/lasr/tokenization_lasr.pyú
<listcomp>z*LasrTokenizer.__init__.<locals>.<listcomp>Z   s,   € Ð[Ð[Ð[ !ÀLÕTWÐXYÑTZÔTZÐDZÐDZ˜AÐDZÐDZÐDZó    é   c                 ó   — g | ]}d |› d�‘Œ	S ©r   ú>© ©r   Úis     r   r   z*LasrTokenizer.__init__.<locals>.<listcomp>\   s$   € Ð-ZÐ-ZÐ-ZÀAÐ.?¸1Ð.?Ð.?Ð.?Ð-ZÐ-ZÐ-Zr   r   zBoth extra_ids (z!) and additional_special_tokens (zm) are provided to LasrTokenizer. In this case the additional_special_tokens must include the extra_ids tokensc                 ó   — g | ]}d |› d�‘Œ	S r    r"   r#   s     r   r   z*LasrTokenizer.__init__.<locals>.<listcomp>d   s$   € ÐHÐHÐH°!Ð-¨Ð-Ð-Ð-ÐHÐHÐHr   g        )õ   â–�g       Àéÿÿÿÿr   r!   r	   F)Úunk_idÚbyte_fallbackr&   ÚalwaysT)ÚreplacementÚprepend_schemeÚsplit)Ú	eos_tokenÚ	unk_tokenÚ	pad_tokenÚ	extra_idsÚadditional_special_tokensú$Ar   )r3   r   z$Br   )ÚsingleÚpairÚspecial_tokensr"   )Ú
_extra_idsÚlenÚrangeÚ
ValueErrorÚ_vocab_scoresr   Úappendr   r   Ú
_tokenizerr   ÚPrecompiledÚ
normalizerr   ÚSequenceÚWhitespaceSplitÚ	MetaspaceÚpre_tokenizerr   ÚdecoderÚsuperÚ__init__r   ÚTemplateProcessingÚeos_token_idÚpost_processor)Úselfr.   r/   r0   Ú_spm_precompiled_charsmapr1   r2   Úvocabr   ÚkwargsÚextra_tokensr$   Ú	__class__s               €r   rF   zLasrTokenizer.__init__J   sœ  ø€ ð $ˆŒð %Ð0Ø[Ð[Ð'@Ð[Ñ[Ô[ˆLÝ�<Ñ Ô  1Ò$Ð$Ø)Ð-ZÐ-ZÍÈyÑIYÔIYÐ-ZÑ-ZÔ-ZÑZÐ)Ð)Ø˜Q’� 9µ°LÑ0AÔ0AÒ#AÐ#AÝ ð yð ð ÐSlð ð ð ñô ð øð IÐHµu¸YÑ7GÔ7GÐHÑHÔHˆLØ(4Ð%ð ÐØ!&ˆDÔÐõ �Y‘” Ð%Ý�Y‘” Ð%Ý�Y‘” Ð%Øð	"ˆDÔõ ˜9 q™=¨"¨bÑ1Ô1ð Dð D�ØÔ"×)Ò)Ð+<¸Ð+<Ð+<Ð+<¸cÐ*BÑCÔCÐCÐCÝ#ÝØÔ"ØØ#ðñ ô ñ
ô 
ˆŒð %Ð0Ý)4Ô)@ÐAZÑ)[Ô)[ˆDŒOÔ&å(6Ô(?åÔ.Ñ0Ô0ÝÔ(°UÈ8Ð[_Ð`Ñ`Ô`ðñ)
ô )
ˆŒÔ%õ #+Ô"4ÀÐW_ÐgkÐ"lÑ"lÔ"lˆŒÔà�‰ŒÔð 	
ØØØØØ&?ð	
ð 	
ð ð	
ð 	
ð 	
õ *4Ô)FØ˜&�>Ø-Ð-Ð-à˜Ô*Ð+ðð*
ñ *
ô *
ˆŒÔ&Ð&Ð&r   c                 ób   — t          t          t          d„ | j        ¦  «        ¦  «        ¦  «        S )zQGet the list of sentinel tokens (extra_id tokens) from additional_special_tokens.c                 óJ   — t          t          j        d| ¦  «        ¦  «        d uS )Nz<extra_id_\d+>)ÚboolÚreÚsearch)r   s    r   ú<lambda>z3LasrTokenizer.get_sentinel_tokens.<locals>.<lambda>š   s    € ¥¥b¤iÐ0AÀ1Ñ&EÔ&EÑ!FÔ!FÈdÐ!R€ r   )ÚlistÚsetÚfilterr2   ©rJ   s    r   Úget_sentinel_tokensz!LasrTokenizer.get_sentinel_tokens—   s1   € åÝ•ÐRÐRÐTXÔTrÑsÔsÑtÔtñ
ô 
ð 	
r   c                 óD   ‡ — ˆ fd„‰                       ¦   «         D ¦   «         S )z&Get the token IDs for sentinel tokens.c                 ó:   •— g | ]}‰                      |¦  «        ‘ŒS r"   )Úconvert_tokens_to_ids©r   ÚtokenrJ   s     €r   r   z8LasrTokenizer.get_sentinel_token_ids.<locals>.<listcomp>Ÿ   s'   ø€ ÐZÐZÐZ°e�×*Ò*¨5Ñ1Ô1ÐZÐZÐZr   )rZ   rY   s   `r   Úget_sentinel_token_idsz$LasrTokenizer.get_sentinel_token_ids�   s)   ø€ àZÐZÐZÐZ¸t×?WÒ?WÑ?YÔ?YÐZÑZÔZÐZr   FTÚ	token_idsÚskip_special_tokensÚclean_up_tokenization_spacesÚgroup_tokensÚreturnc                 óÌ   •‡ — t          |t          ¦  «        r|g}|rd„ t          j        |¦  «        D ¦   «         }ˆ fd„|D ¦   «         } t	          ¦   «         j        d|||dœ|¤ŽS )Nc                 ó   — g | ]
}|d          ‘ŒS )r   r"   )r   Útoken_groups     r   r   z)LasrTokenizer._decode.<locals>.<listcomp>¬   s   € ÐXÐXÐX¨K˜ QœÐXÐXÐXr   c                 ó*   •— g | ]}|‰j         k    ¯|‘ŒS r"   )Úpad_token_idr^   s     €r   r   z)LasrTokenizer._decode.<locals>.<listcomp>¯   s&   ø€ ÐPÐPÐP˜u°U¸dÔ>OÒ5OÐ5O�UÐ5OÐ5OÐ5Or   )ra   rb   rc   r"   )Ú
isinstanceÚintÚ	itertoolsÚgroupbyrE   Ú_decode)rJ   ra   rb   rc   rd   rM   rO   s   `     €r   ro   zLasrTokenizer._decode¡   s™   øø€ õ �i¥Ñ%Ô%ð 	$Ø"˜ˆIØð 	YØXÐX½9Ô;LÈYÑ;WÔ;WÐXÑXÔXˆIð QÐPÐPÐP¨	ÐPÑPÔPˆ	à�u‰wŒwŒð 
ØØ 3Ø)Eð
ð 
ð ð	
ð 
ð 	
r   )r   r   r   Nr   NNN)FNT)Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚVOCAB_FILES_NAMESÚvocab_files_namesÚmodel_input_namesr   ÚmodelrF   rZ   r`   rl   rV   rR   r   ro   Ú__classcell__)rO   s   @r   r   r   !   s  ø€ € € € € ð"ð "ðH *ÐØ$Ð&6Ð7ÐØ€Eð ØØØ"&ØØ"&ØØðK
ð K
ð K
ð K
ð K
ð K
ðZ
ð 
ð 
ð[ð [ð [ð %*Ø48Ø!ð
ð 
à˜˜cœ‘?ð
ð "ð
ð '+¨T¡kð	
ð
 ð
ð 
ð
ð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
r   r   )rm   rS   Ú
tokenizersr   r   r   r   r   Útokenizers.modelsr   Útokenization_utils_tokenizersr
   rt   r   Ú__all__r"   r   r   ú<module>r}      s½   ðð* Ð Ð Ð Ø 	€	€	€	à SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SØ %Ð %Ð %Ð %Ð %Ð %à >Ð >Ð >Ð >Ð >Ð >ð $2ÐEUÐVÐVÐ ðU
ð U
ð U
ð U
ð U
Ð%ñ U
ô U
ð U
ðp Ð
€€€r   