§
    ‚ŠtjŒ  ã                   óž   — d Z ddlZddlmZmZmZmZmZ ddlm	Z	 ddl
mZ ddlmZ  ej        e¦  «        Zdd	d
œZ G d„ de¦  «        ZdgZdS )z Tokenization class for model T5.é    N)Ú	TokenizerÚdecodersÚnormalizersÚpre_tokenizersÚ
processors)ÚUnigramé   )ÚTokenizersBackend)Úloggingzspiece.modelztokenizer.json)Ú
vocab_fileÚtokenizer_filec                   ó|   ‡ — e Zd ZdZeZddgZeZ	 	 	 	 	 	 	 dd	e	e
ee	ef                  z  dz  fˆ fd
„Zd„ Zd„ Zˆ xZS )ÚT5Tokenizera¾  
    Construct a T5 tokenizer (backed by HuggingFace's *tokenizers* library). Based on
    [Unigram](https://huggingface.co/docs/tokenizers/python/latest/components.html?highlight=unigram#models).

    This tokenizer inherits from [`TokenizersBackend`] which contains most of the main methods. Users should
    refer to this superclass for more information regarding those methods.

    Args:
        vocab_file (`str`, *optional*):
            [SentencePiece](https://github.com/google/sentencepiece) file (generally has a *.spm* extension) that
            contains the vocabulary necessary to instantiate a tokenizer.
        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the end of sequence.
            The token used is the `sep_token`.

            </Tip>

        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        extra_ids (`int`, *optional*, defaults to 100):
            Add a number of extra ids added to the vocabulary for use as sentinels. These tokens are accessible as
            "<extra_id_{%d}>" where "{%d}" is a number between 0 and extra_ids-1. These tokens can be retrieved by
            calling get_sentinel_tokens method and token ids can be by calling get_sentinel_token_ids method
        additional_special_tokens (`list[str]`, *optional*):
            Additional special tokens used by the tokenizer.
        vocab (`str`, `dict` or `list`, *optional*):
            Custom vocabulary dict. If not provided, a minimal vocabulary is created using the special tokens.
    Ú	input_idsÚattention_maskNú</s>ú<unk>ú<pad>éd   Úvocabc           	      ó  •— || _         |�ld„ |D ¦   «         }	t          |	¦  «        dk     r|d„ t          |¦  «        D ¦   «         z  }nK|dk    r)|t          |	¦  «        k    rt          d|› d|› d�¦  «        ‚nd„ t          |¦  «        D ¦   «         }	|	}|�|| _        not          |¦  «        d	ft          |¦  «        d	ft          |¦  «        d	fd
g| _        t          |dz
  dd¦  «        D ]"}
| j                             d|
› d�d	f¦  «         Œ#t          t          | j        dd¬¦  «        ¦  «        | _	        |�t          j        |¦  «        | j	        _        t          j        t          j        ¦   «         t          j        ddd¬¦  «        g¦  «        | j	        _        t%          j        ddd¬¦  «        | j	        _         t)          ¦   «         j        d|||||dœ|¤Ž t-          j        ddgg d¢d| j        fg¬¦  «        | j	        _        d S )Nc                 ó4   — g | ]}d t          |¦  «        v ¯|‘ŒS )ú
<extra_id_)Ústr)Ú.0Úxs     úd/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/t5/tokenization_t5.pyú
<listcomp>z(T5Tokenizer.__init__.<locals>.<listcomp>V   s,   € Ð[Ð[Ð[ !ÀLÕTWÐXYÑTZÔTZÐDZÐDZ˜AÐDZÐDZÐDZó    é   c                 ó   — g | ]}d |› d�‘Œ	S ©r   ú>© ©r   Úis     r   r   z(T5Tokenizer.__init__.<locals>.<listcomp>X   s$   € Ð-ZÐ-ZÐ-ZÀAÐ.?¸1Ð.?Ð.?Ð.?Ð-ZÐ-ZÐ-Zr   r   zBoth extra_ids (z!) and additional_special_tokens (zk) are provided to T5Tokenizer. In this case the additional_special_tokens must include the extra_ids tokensc                 ó   — g | ]}d |› d�‘Œ	S r"   r$   r%   s     r   r   z(T5Tokenizer.__init__.<locals>.<listcomp>`   s$   € ÐHÐHÐH°!Ð-¨Ð-Ð-Ð-ÐHÐHÐHr   g        )õ   â–�g       Àéÿÿÿÿr   r#   é   F)Úunk_idÚbyte_fallbackr(   ÚalwaysT)ÚreplacementÚprepend_schemeÚsplit)Ú	eos_tokenÚ	unk_tokenÚ	pad_tokenÚ	extra_idsÚadditional_special_tokensú$Ar   )r6   r   z$Br   )ÚsingleÚpairÚspecial_tokensr$   )Ú
_extra_idsÚlenÚrangeÚ
ValueErrorÚ_vocab_scoresr   Úappendr   r   Ú
_tokenizerr   ÚPrecompiledÚ
normalizerr   ÚSequenceÚWhitespaceSplitÚ	MetaspaceÚpre_tokenizerr   ÚdecoderÚsuperÚ__init__r   ÚTemplateProcessingÚeos_token_idÚpost_processor)Úselfr   r1   r2   r3   Ú_spm_precompiled_charsmapr4   r5   ÚkwargsÚextra_tokensr&   Ú	__class__s              €r   rI   zT5Tokenizer.__init__G   sœ  ø€ ð $ˆŒð %Ð0Ø[Ð[Ð'@Ð[Ñ[Ô[ˆLÝ�<Ñ Ô  1Ò$Ð$Ø)Ð-ZÐ-ZÍÈyÑIYÔIYÐ-ZÑ-ZÔ-ZÑZÐ)Ð)Ø˜Q’� 9µ°LÑ0AÔ0AÒ#AÐ#AÝ ð yð ð ÐSlð ð ð ñô ð øð IÐHµu¸YÑ7GÔ7GÐHÑHÔHˆLØ(4Ð%ð ÐØ!&ˆDÔÐõ �Y‘” Ð%Ý�Y‘” Ð%Ý�Y‘” Ð%Øð	"ˆDÔõ ˜9 q™=¨"¨bÑ1Ô1ð Dð D�ØÔ"×)Ò)Ð+<¸Ð+<Ð+<Ð+<¸cÐ*BÑCÔCÐCÐCå#ÝØÔ"ØØ#ðñ ô ñ
ô 
ˆŒð %Ð0Ý)4Ô)@ÐAZÑ)[Ô)[ˆDŒOÔ&å(6Ô(?åÔ.Ñ0Ô0ÝÔ(°UÈ8Ð[_Ð`Ñ`Ô`ðñ)
ô )
ˆŒÔ%õ #+Ô"4ÀÐW_ÐgkÐ"lÑ"lÔ"lˆŒÔà�‰ŒÔð 	
ØØØØØ&?ð	
ð 	
ð ð	
ð 	
ð 	
õ *4Ô)FØ˜&�>Ø-Ð-Ð-à˜Ô*Ð+ðð*
ñ *
ô *
ˆŒÔ&Ð&Ð&r   c                 ób   — t          t          t          d„ | j        ¦  «        ¦  «        ¦  «        S )zQGet the list of sentinel tokens (extra_id tokens) from additional_special_tokens.c                 óJ   — t          t          j        d| ¦  «        ¦  «        d uS )Nz<extra_id_\d+>)ÚboolÚreÚsearch)r   s    r   ú<lambda>z1T5Tokenizer.get_sentinel_tokens.<locals>.<lambda>—   s    € ¥¥b¤iÐ0AÀ1Ñ&EÔ&EÑ!FÔ!FÈdÐ!R€ r   )ÚlistÚsetÚfilterr5   ©rM   s    r   Úget_sentinel_tokenszT5Tokenizer.get_sentinel_tokens”   s1   € åÝ•ÐRÐRÐTXÔTrÑsÔsÑtÔtñ
ô 
ð 	
r   c                 óD   ‡ — ˆ fd„‰                       ¦   «         D ¦   «         S )z&Get the token IDs for sentinel tokens.c                 ó:   •— g | ]}‰                      |¦  «        ‘ŒS r$   )Úconvert_tokens_to_ids)r   ÚtokenrM   s     €r   r   z6T5Tokenizer.get_sentinel_token_ids.<locals>.<listcomp>œ   s'   ø€ ÐZÐZÐZ°e�×*Ò*¨5Ñ1Ô1ÐZÐZÐZr   )r\   r[   s   `r   Úget_sentinel_token_idsz"T5Tokenizer.get_sentinel_token_idsš   s)   ø€ àZÐZÐZÐZ¸t×?WÒ?WÑ?YÔ?YÐZÑZÔZÐZr   )Nr   r   r   Nr   N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚVOCAB_FILES_NAMESÚvocab_files_namesÚmodel_input_namesr   Úmodelr   rX   ÚtupleÚfloatrI   r\   ra   Ú__classcell__)rQ   s   @r   r   r      sË   ø€ € € € € ð"ð "ðH *ÐØ$Ð&6Ð7ÐØ€Eð 7;ØØØØ"&ØØ"&ðK
ð K
à�T˜%  U 
Ô+Ô,Ñ,¨tÑ3ðK
ð K
ð K
ð K
ð K
ð K
ðZ
ð 
ð 
ð[ð [ð [ð [ð [ð [ð [r   r   )re   rU   Ú
tokenizersr   r   r   r   r   Útokenizers.modelsr   Útokenization_utils_tokenizersr
   Úutilsr   Ú
get_loggerrb   Úloggerrf   r   Ú__all__r$   r   r   ú<module>rt      sâ   ðð 'Ð &à 	€	€	€	à SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SÐ SØ %Ð %Ð %Ð %Ð %Ð %à >Ð >Ð >Ð >Ð >Ð >Ø Ð Ð Ð Ð Ð ð 
ˆÔ	˜HÑ	%Ô	%€à#1ÐEUÐVÐVÐ ð~[ð ~[ð ~[ð ~[ð ~[Ð#ñ ~[ô ~[ð ~[ðB ˆ/€€€r   