§
    ‚Štj'  ã                   óp   — d Z ddlZddlmZmZ ddlmZ  ej        e¦  «        Z	 G d„ de¦  «        Z
dgZdS )z"Tokenization class for model ByT5.é    Né   )Ú
AddedTokenÚPreTrainedTokenizer)Úloggingc            
       ó¢  ‡ — e Zd ZdZddgZ	 	 	 	 	 d	 dˆ fd
„Zed„ ¦   «         Zd„ Z	 d de	e
         de	e
         dz  ded	e	e
         fˆ fd„Zde	e
         d	e	e
         fd„Z	 d!de	e
         de	e
         dz  d	e	e
         fd„Z	 d!de	e
         de	e
         dz  d	e	e
         fd„Zded	e	e         fd„Zd„ Zd„ Zd„ Zd!dededz  d	ee         fd„Zˆ xZS )"ÚByT5Tokenizera—  
    Construct a ByT5 tokenizer. ByT5 simply uses raw bytes utf-8 encoding.

    This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods.

    Args:
        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the end of sequence.
            The token used is the `sep_token`.

            </Tip>

        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        extra_ids (`int`, *optional*, defaults to 125):
            Add a number of extra ids added to the end of the vocabulary for use as sentinels. These tokens are
            accessible as "<extra_id_{%d}>" where "{%d}" is a number between 0 and extra_ids-1. Extra tokens are
            indexed from the end of the vocabulary up to beginning ("<extra_id_0>" is the last token in the vocabulary
            like in ByT5 preprocessing see
            [here](https://github.com/google-research/text-to-text-transfer-transformer/blob/9fd7b14a769417be33bc6c850f9598764913c833/t5/data/preprocessors.py#L2117)).
        additional_special_tokens (`list[str]`, *optional*):
            Additional special tokens used by the tokenizer.
    Ú	input_idsÚattention_maskú</s>ú<unk>ú<pad>é}   NÚreturnc           	      óš  •— |dk    r|€d„ t          |¦  «        D ¦   «         }nb|dk    r\|�Zt          |¦  «        dk    rGt          t          t          d„ |¦  «        ¦  «        ¦  «        }||k    rt	          d|› d|› d�¦  «        ‚t          |t          ¦  «        rt          |dd¬¦  «        n|}t          |t          ¦  «        rt          |dd¬¦  «        n|}t          |t          ¦  «        rt          |dd¬¦  «        n|}|||d	œ| _        t          | j        ¦  «        | _	        d
| _
         t          ¦   «         j        d|||d|dœ|¤Ž d S )Nr   c                 ó   — g | ]}d |› d�‘Œ	S )z
<extra_id_ú>© ©Ú.0Úis     úh/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/byt5/tokenization_byt5.pyú
<listcomp>z*ByT5Tokenizer.__init__.<locals>.<listcomp>G   s$   € Ð(UÐ(UÐ(U¸qÐ):°aÐ):Ð):Ð):Ð(UÐ(UÐ(Uó    c                 ó>   — t          dt          | ¦  «        v ¦  «        S )NÚextra_id)ÚboolÚstr)Úxs    r   ú<lambda>z(ByT5Tokenizer.__init__.<locals>.<lambda>J   s   € µD¸ÅsÈ1ÁvÄvÐ9MÑ4NÔ4N€ r   zBoth extra_ids (z!) and additional_special_tokens (zm) are provided to ByT5Tokenizer. In this case the additional_special_tokens must include the extra_ids tokensT)ÚlstripÚrstrip)r   é   é   é   )Ú	eos_tokenÚ	unk_tokenÚ	pad_tokenÚ	extra_idsÚadditional_special_tokensr   )ÚrangeÚlenÚsetÚfilterÚ
ValueErrorÚ
isinstancer   r   Ú_added_tokens_decoderÚoffsetÚ_utf_vocab_sizeÚsuperÚ__init__)	Úselfr%   r&   r'   r(   r)   ÚkwargsÚextra_tokensÚ	__class__s	           €r   r4   zByT5Tokenizer.__init__<   s³  ø€ ð �qŠ=ˆ=Ð6Ð>Ø(UÐ(UÅEÈ)ÑDTÔDTÐ(UÑ(UÔ(UÐ%Ð%Ø˜Š]ˆ]Ð8ÐDÍÐMfÑIgÔIgÐjkÒIkÐIkå�s¥6Ð*NÐ*NÐPiÑ#jÔ#jÑkÔkÑlÔlˆLØ˜yÒ(Ð(Ý ð( yð (ð (ÐSlð (ð (ð (ñô ð õ HRÐR[Õ]`ÑGaÔGaÐp•J˜y°¸dÐCÑCÔCÐCÐgpˆ	åGQÐR[Õ]`ÑGaÔGaÐp•J˜y°¸dÐCÑCÔCÐCÐgpˆ	ÝGQÐR[Õ]`ÑGaÔGaÐp•J˜y°¸dÐCÑCÔCÐCÐgpˆ	à)2°yÀYÐ%OÐ%OˆÔ"Ý˜$Ô4Ñ5Ô5ˆŒØ#ˆÔØ�‰ŒÔð 	
ØØØØØ&?ð	
ð 	
ð ð	
ð 	
ð 	
ð 	
ð 	
r   c                 ó   — | j         S ©N)r2   )r5   s    r   Ú
vocab_sizezByT5Tokenizer.vocab_sizec   s   € àÔ#Ð#r   c                 óŒ   ‡ — ˆ fd„t          ‰ j        ‰ j        z   ¦  «        D ¦   «         }|                     ‰ j        ¦  «         |S )Nc                 ó<   •— i | ]}‰                      |¦  «        |“ŒS r   )Úconvert_ids_to_tokens)r   r   r5   s     €r   ú
<dictcomp>z+ByT5Tokenizer.get_vocab.<locals>.<dictcomp>h   s)   ø€ Ð`Ð`Ð`°a�×+Ò+¨AÑ.Ô.°Ð`Ð`Ð`r   )r*   r;   r1   ÚupdateÚadded_tokens_encoder)r5   Úvocabs   ` r   Ú	get_vocabzByT5Tokenizer.get_vocabg   sI   ø€ Ø`Ð`Ð`Ð`½5ÀÄÐSWÔS^ÑA^Ñ;_Ô;_Ð`Ñ`Ô`ˆØ�Š�TÔ.Ñ/Ô/Ð/Øˆr   FÚtoken_ids_0Útoken_ids_1Úalready_has_special_tokensc                 óà   •— |r$t          ¦   «                              ||d¬¦  «        S |€dgt          |¦  «        z  dgz   S dgt          |¦  «        z  dgz   dgt          |¦  «        z  z   dgz   S )aÄ  
        Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
        special tokens using the tokenizer `prepare_for_model` method.

        Args:
            token_ids_0 (`list[int]`):
                List of IDs.
            token_ids_1 (`list[int]`, *optional*):
                Optional second list of IDs for sequence pairs.
            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not the token list is already formatted with special tokens for the model.

        Returns:
            `list[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
        T)rD   rE   rF   Nr   r"   )r3   Úget_special_tokens_maskr+   )r5   rD   rE   rF   r8   s       €r   rH   z%ByT5Tokenizer.get_special_tokens_maskl   s‘   ø€ ð$ &ð 	Ý‘7”7×2Ò2Ø'°[Ð]að 3ñ ô ð ð
 ÐØ�C�#˜kÑ*Ô*Ñ*¨q¨cÑ1Ð1Ø�•c˜+Ñ&Ô&Ñ&¨1¨#Ñ-°!°µs¸;Ñ7GÔ7GÑ1GÑHÈAÈ3ÑNÐNr   Ú	token_idsc                 óž   — t          |¦  «        dk    r0|d         | j        k    rt          j        d| j        › d�¦  «         |S || j        gz   S )z.Do not add eos again if user already added it.r   éÿÿÿÿzThis sequence already has zQ. In future versions this behavior may lead to duplicated eos tokens being added.)r+   Úeos_token_idÚwarningsÚwarnr%   )r5   rI   s     r   Ú_add_eos_if_not_presentz%ByT5Tokenizer._add_eos_if_not_presentˆ   si   € åˆy‰>Œ>˜AÒÐ )¨B¤-°4Ô3DÒ"DÐ"DÝŒMð+¨T¬^ð +ð +ð +ñô ð ð Ðà Ô 1Ð2Ñ2Ð2r   c                 óz   — | j         g}|€t          ||z   ¦  «        dgz  S t          ||z   |z   |z   ¦  «        dgz  S )aÉ  
        Create a mask from the two sequences passed to be used in a sequence-pair classification task. ByT5 does not
        make use of token type ids, therefore a list of zeros is returned.

        Args:
            token_ids_0 (`list[int]`):
                List of IDs.
            token_ids_1 (`list[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `list[int]`: List of zeros.
        Nr   )rL   r+   )r5   rD   rE   Úeoss       r   Ú$create_token_type_ids_from_sequencesz2ByT5Tokenizer.create_token_type_ids_from_sequences“   sS   € ð  Ô Ð!ˆàÐÝ�{ SÑ(Ñ)Ô)¨Q¨CÑ/Ð/Ý�; Ñ$ {Ñ2°SÑ8Ñ9Ô9¸Q¸CÑ?Ð?r   c                 óh   — |                       |¦  «        }|€|S |                       |¦  «        }||z   S )a‚  
        Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and
        adding special tokens. A sequence has the following format:

        - single sequence: `X </s>`
        - pair of sequences: `A </s> B </s>`

        Args:
            token_ids_0 (`list[int]`):
                List of IDs to which the special tokens will be added.
            token_ids_1 (`list[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `list[int]`: List of [input IDs](../glossary#input-ids) with the appropriate special tokens.
        )rO   )r5   rD   rE   s      r   Ú build_inputs_with_special_tokensz.ByT5Tokenizer.build_inputs_with_special_tokens©   sA   € ð& ×2Ò2°;Ñ?Ô?ˆØÐØÐà×6Ò6°{ÑCÔCˆKØ Ñ,Ð,r   Útextc                 óD   — d„ |                      d¦  «        D ¦   «         }|S )zPTake as input a string and return a list of strings (tokens) for words/sub-wordsc                 ó,   — g | ]}t          |¦  «        ‘ŒS r   )Úchrr   s     r   r   z+ByT5Tokenizer._tokenize.<locals>.<listcomp>Å   s   € Ð7Ð7Ð7˜Q•#�a‘&”&Ð7Ð7Ð7r   úutf-8)Úencode)r5   rU   Útokenss      r   Ú	_tokenizezByT5Tokenizer._tokenizeÃ   s&   € à7Ð7 $§+¢+¨gÑ"6Ô"6Ð7Ñ7Ô7ˆØˆr   c                 ó`   — t          |¦  «        dk    rd}nt          |¦  «        | j        z   }|S )z0Converts a token (str) in an id using the vocab.r"   N)r+   Úordr1   )r5   ÚtokenÚtoken_ids      r   Ú_convert_token_to_idz"ByT5Tokenizer._convert_token_to_idÈ   s1   € õ ˆu‰:Œ:˜Š?ˆ?ØˆHˆHå˜5‘z”z D¤KÑ/ˆHàˆr   c                 ó4   — t          || j        z
  ¦  «        }|S )z=Converts an index (integer) in a token (str) using the vocab.)rX   r1   )r5   Úindexr_   s      r   Ú_convert_id_to_tokenz"ByT5Tokenizer._convert_id_to_tokenÒ   s   € å�E˜DœKÑ'Ñ(Ô(ˆØˆr   c                 ó  — d}|D ]m}|| j         v r!| j         |                              d¦  «        }n<|| j        v r|                     d¦  «        }nt          t	          |¦  «        g¦  «        }||z  }Œn|                     dd¬¦  «        }|S )z:Converts a sequence of tokens (string) in a single string.r   rY   Úignore)Úerrors)r0   rZ   Ú_added_tokens_encoderÚbytesr^   Údecode)r5   r[   Úbstringr_   Ú
tok_stringÚstrings         r   Úconvert_tokens_to_stringz&ByT5Tokenizer.convert_tokens_to_string×   s�   € àˆØð 	"ð 	"ˆEØ˜Ô2Ð2Ð2Ø!Ô7¸Ô>×EÒEÀgÑNÔN�
�
Ø˜$Ô4Ð4Ð4Ø"Ÿ\š\¨'Ñ2Ô2�
�
å"¥C¨¡J¤J <Ñ0Ô0�
Ø�zÑ!ˆGˆGØ—’ °�Ñ9Ô9ˆØˆr   Úsave_directoryÚfilename_prefixc                 ó   — dS )Nr   r   )r5   ro   rp   s      r   Úsave_vocabularyzByT5Tokenizer.save_vocabularyæ   s   € Øˆrr   )r   r   r   r   N)r   N)NFr:   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesr4   Úpropertyr;   rC   ÚlistÚintr   rH   rO   rR   rT   r   r\   ra   rd   rn   Útuplerr   Ú__classcell__)r8   s   @r   r   r      sR  ø€ € € € € ðð ð@ %Ð&6Ð7Ðð ØØØØ"&ð%
ð 
ð%
ð %
ð %
ð %
ð %
ð %
ðN ð$ð $ñ „Xð$ðð ð ð puðOð OØ œ9ðOØ37¸´9¸tÑ3CðOØhlðOà	ˆcŒðOð Oð Oð Oð Oð Oð8	3°°c´ð 	3¸tÀC¼yð 	3ð 	3ð 	3ð 	3ð GKð@ð @Ø œ9ð@Ø37¸´9¸tÑ3Cð@à	ˆcŒð@ð @ð @ð @ð. GKð-ð -Ø œ9ð-Ø37¸´9¸tÑ3Cð-à	ˆcŒð-ð -ð -ð -ð4˜cð  d¨3¤ið ð ð ð ð
ð ð ðð ð ð
ð ð ðð ¨cð ÀCÈ$ÁJð ÐZ_Ð`cÔZdð ð ð ð ð ð ð ð r   r   )rv   rM   Útokenization_pythonr   r   Úutilsr   Ú
get_loggerrs   Úloggerr   Ú__all__r   r   r   ú<module>r‚      s–   ðð )Ð (à €€€à BÐ BÐ BÐ BÐ BÐ BÐ BÐ BØ Ð Ð Ð Ð Ð ð 
ˆÔ	˜HÑ	%Ô	%€ðNð Nð Nð Nð NÐ'ñ Nô Nð Nðb Ð
€€€r   