§
    ‚Štjò  ã                   óš   — d Z ddlmZmZmZmZmZmZ ddlm	Z	 ddl
mZ ddlmZ  ej        e¦  «        Zddd	œZ G d
„ de¦  «        ZdgZdS )z'Tokenization classes for RemBert model.é    )ÚRegexÚ	TokenizerÚdecodersÚnormalizersÚpre_tokenizersÚ
processors)ÚUnigramé   )ÚTokenizersBackend)Úloggingzsentencepiece.modelztokenizer.json)Ú
vocab_fileÚtokenizer_filec                   ó²   ‡ — e Zd ZdZeZddgZeZ	 	 	 	 	 	 	 	 	 	 	 	 	 dde	e
ee	ef                  z  dz  dedede	de	de	de	de	de	de	de	dz  dedefˆ fd„Zˆ xZS )ÚRemBertTokenizeraÞ
  
    Construct a "fast" RemBert tokenizer (backed by HuggingFace's *tokenizers* library). Based on
    [Unigram](https://huggingface.co/docs/tokenizers/python/latest/components.html?highlight=unigram#models). This
    tokenizer inherits from [`AlbertTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods

    Args:
        do_lower_case (`bool`, *optional*, defaults to `True`):
            Whether or not to lowercase the input when tokenizing.
        remove_space (`bool`, *optional*, defaults to `True`):
            Whether or not to strip the text when tokenizing (removing excess spaces before and after the string).
        keep_accents (`bool`, *optional*, defaults to `True`):
            Whether or not to keep accents when tokenizing.
        bos_token (`str`, *optional*, defaults to `"[CLS]"`):
            The beginning of sequence token that was used during pretraining. Can be used a sequence classifier token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the beginning of
            sequence. The token used is the `cls_token`.

            </Tip>

        eos_token (`str`, *optional*, defaults to `"[SEP]"`):
            The end of sequence token. .. note:: When building a sequence using special tokens, this is not the token
            that is used for the end of sequence. The token used is the `sep_token`.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        sep_token (`str`, *optional*, defaults to `"[SEP]"`):
            The separator token, which is used when building a sequence from multiple sequences, e.g. two sequences for
            sequence classification or for a text and a question for question answering. It is also used as the last
            token of a sequence built with special tokens.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        cls_token (`str`, *optional*, defaults to `"[CLS]"`):
            The classifier token which is used when doing sequence classification (classification of the whole sequence
            instead of per-token classification). It is the first token of the sequence when built with special tokens.
        mask_token (`str`, *optional*, defaults to `"[MASK]"`):
            The token used for masking values. This is the token used when training this model with masked language
            modeling. This is the token which the model will try to predict.
    Ú	input_idsÚattention_maskNFTú[CLS]ú[SEP]ú<unk>ú<pad>ú[MASK]ÚvocabÚdo_lower_caseÚkeep_accentsÚ	bos_tokenÚ	eos_tokenÚ	unk_tokenÚ	sep_tokenÚ	pad_tokenÚ	cls_tokenÚ
mask_tokenÚ_spm_precompiled_charsmapÚadd_prefix_spaceÚremove_spacec                 ó~  •— || _         || _        || _        |�|| _        nWt	          |¦  «        dft	          |¦  «        dft	          |	¦  «        dft	          |¦  «        dft	          |
¦  «        dfg| _        t          t          | j        dd¬¦  «        ¦  «        | _        t          j	        dd¦  «        t          j	        dd¦  «        t          j	        t          d¦  «        d	¦  «        g}| j        sL|                     t          j        ¦   «         ¦  «         |                     t          j        ¦   «         ¦  «         | j        r&|                     t          j        ¦   «         ¦  «         |�(|                     t          j        |¦  «        g¦  «         t          j        |¦  «        | j        _        |rd
nd}t'          j        d|¬¦  «        | j        _        t-          j        d|¬¦  «        | j        _         t1          ¦   «         j        d|||||||	|||
|dœ|¤Ž t	          |	¦  «        }t	          |¦  «        }|                      |¦  «        }|                      |¦  «        }t7          j        |› d|› d�|› d|› d|› d�||f||fg¬¦  «        | j        _        t1          ¦   «                              ¦   «          d S )Ng        é   F)Úunk_idÚbyte_fallbackz``ú"z''z {2,}ú ÚalwaysÚneveru   â–�)ÚreplacementÚprepend_scheme)r#   r   r   r   r   r   r    r   r   r!   r$   z:0 $A:0 z:0z:0 $B:1 z:1)ÚsingleÚpairÚspecial_tokens© )r$   r   r   Ú_vocab_scoresÚstrr   r	   Ú
_tokenizerr   ÚReplacer   ÚappendÚNFKDÚStripAccentsÚ	LowercaseÚextendÚPrecompiledÚSequenceÚ
normalizerr   Ú	MetaspaceÚpre_tokenizerr   ÚdecoderÚsuperÚ__init__Úconvert_tokens_to_idsr   ÚTemplateProcessingÚpost_processorÚ
_post_init)Úselfr   r   r   r   r   r   r   r   r    r!   r"   r#   r$   ÚkwargsÚlist_normalizersr.   Úcls_token_strÚsep_token_strÚcls_token_idÚsep_token_idÚ	__class__s                        €ún/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/rembert/tokenization_rembert.pyrC   zRemBertTokenizer.__init__L   sñ  ø€ ð" )ˆÔØ*ˆÔØ(ˆÔàÐØ!&ˆDÔÐõ �Y‘” Ð%Ý�Y‘” Ð%Ý�Y‘” Ð%Ý�Y‘” Ð%Ý�Z‘” #Ð&ð"ˆDÔõ $ÝØÔ"ØØ#ðñ ô ñ
ô 
ˆŒõ Ô  cÑ*Ô*ÝÔ  cÑ*Ô*ÝÔ¥ g¡¤°Ñ4Ô4ð
Ðð
 Ô ð 	@Ø×#Ò#¥KÔ$4Ñ$6Ô$6Ñ7Ô7Ð7Ø×#Ò#¥KÔ$<Ñ$>Ô$>Ñ?Ô?Ð?ØÔð 	=Ø×#Ò#¥KÔ$9Ñ$;Ô$;Ñ<Ô<Ð<à$Ð0Ø×#Ò#¥[Ô%<Ð=VÑ%WÔ%WÐ$XÑYÔYÐYå%0Ô%9Ð:JÑ%KÔ%KˆŒÔ"à%5ÐB˜˜¸7ˆå(6Ô(@ÈUÐcqÐ(rÑ(rÔ(rˆŒÔ%å"*Ô"4ÀÐWeÐ"fÑ"fÔ"fˆŒÔØ�‰ŒÔð 	
Ø-Ø'Ø%ØØØØØØØ!Ø%ð	
ð 	
ð ð	
ð 	
ð 	
õ" ˜I™œˆÝ˜I™œˆØ×1Ò1°-Ñ@Ô@ˆØ×1Ò1°-Ñ@Ô@ˆå)3Ô)FØ#Ð>Ð>¨]Ð>Ð>Ð>Ø!ÐSÐS¨=ÐSÐSÀ-ÐSÐSÐSà Ð-Ø Ð-ðð*
ñ *
ô *
ˆŒÔ&õ 	‰Œ×ÒÑÔÐÐÐó    )NFTr   r   r   r   r   r   r   NTT)Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚVOCAB_FILES_NAMESÚvocab_files_namesÚmodel_input_namesr	   Úmodelr4   ÚlistÚtupleÚfloatÚboolrC   Ú__classcell__)rO   s   @rP   r   r      sC  ø€ € € € € ð)ð )ðV *ÐØ$Ð&6Ð7ÐØ€Eð 7;Ø#Ø!Ø Ø Ø Ø Ø Ø Ø"Ø04Ø!%Ø!ð`ð `à�T˜%  U 
Ô+Ô,Ñ,¨tÑ3ð`ð ð`ð ð	`ð
 ð`ð ð`ð ð`ð ð`ð ð`ð ð`ð ð`ð $'¨¡:ð`ð ð`ð ð`ð `ð `ð `ð `ð `ð `ð `ð `ð `rQ   r   N)rU   Ú
tokenizersr   r   r   r   r   r   Útokenizers.modelsr	   Útokenization_utils_tokenizersr   Úutilsr   Ú
get_loggerrR   ÚloggerrV   r   Ú__all__r2   rQ   rP   ú<module>rf      sØ   ðð .Ð -à ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZÐ ZØ %Ð %Ð %Ð %Ð %Ð %à >Ð >Ð >Ð >Ð >Ð >Ø Ð Ð Ð Ð Ð ð 
ˆÔ	˜HÑ	%Ô	%€à#8ÐL\Ð]Ð]Ð ðPð Pð Pð Pð PÐ(ñ Pô Pð Pðf Ð
€€€rQ   