§
    ‚Štj	  ã                   ó”   — d Z ddlmZmZmZmZ ddlmZ ddlm	Z	 ddl
mZ  ej        e¦  «        Zddd	d
œZ G d„ de	¦  «        ZdgZdS )z$Tokenization classes for OpenAI GPT.é    )Ú	TokenizerÚdecodersÚnormalizersÚpre_tokenizers)ÚBPEé   )ÚTokenizersBackend)Úloggingz
vocab.jsonz
merges.txtztokenizer.json)Ú
vocab_fileÚmerges_fileÚtokenizer_filec                   ó’   ‡ — e Zd ZdZeZddgZeZ	 	 	 dde	e
e	ef         z  dz  de	ee	         z  dz  de	fˆ fd	„Zed
„ ¦   «         Zˆ xZS )ÚOpenAIGPTTokenizera´  
    Construct a GPT Tokenizer (backed by HuggingFace's *tokenizers* library). Based on Byte-Pair-Encoding with
    the following peculiarities:

    - lower case all inputs
    - uses BERT's BasicTokenizer for pre-BPE tokenization

    This tokenizer inherits from [`TokenizersBackend`] which contains most of the main methods. Users should
    refer to this superclass for more information regarding those methods.

    Args:
        vocab_file (`str`, *optional*):
            Path to the vocabulary file.
        merges_file (`str`, *optional*):
            Path to the merges file.
        tokenizer_file (`str`, *optional*):
            Path to a tokenizers JSON file containing the serialization of a tokenizer.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        vocab (`str` or `dict[str, int]`, *optional*):
            Custom vocabulary dictionary. If not provided, a blank vocabulary is initialized.
        merges (`str` or `list[str]`, *optional*):
            Custom merges list. If not provided, an empty list is used.
    Ú	input_idsÚattention_maskNú<unk>ÚvocabÚmergesÚ	unk_tokenc                 ó¸  •— |�|nt          |¦  «        di| _        |pg | _        t          t	          | j        | j        d dddt          |¦  «        ¬¦  «        ¦  «        | _        t          j        d¬¦  «        | j        _        t          j
        ¦   «         | j        _        t          j        d¬¦  «        | j        _         t          ¦   «         j        d
d	|i|¤Ž d S )Nr   Ú z</w>F)r   r   ÚdropoutÚcontinuing_subword_prefixÚend_of_word_suffixÚfuse_unkr   T)Ú	lowercase)Úsuffixr   © )ÚstrÚ_vocabÚ_mergesr   r   Ú
_tokenizerr   ÚBertNormalizerÚ
normalizerr   ÚBertPreTokenizerÚpre_tokenizerr   Ú
BPEDecoderÚdecoderÚsuperÚ__init__)Úselfr   r   r   ÚkwargsÚ	__class__s        €úl/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/openai/tokenization_openai.pyr*   zOpenAIGPTTokenizer.__init__;   sì   ø€ ð  %Ð0�e�eµs¸9±~´~ÀqÐ6IˆŒØ�| ˆŒå#ÝØ”kØ”|ØØ*,Ø#)ØÝ˜i™.œ.ðñ ô ñ

ô 

ˆŒõ &1Ô%?È$Ð%OÑ%OÔ%OˆŒÔ"å(6Ô(GÑ(IÔ(IˆŒÔ%Ý"*Ô"5¸VÐ"DÑ"DÔ"DˆŒÔà�‰ŒÔð 	
ð 	
Øð	
àð	
ð 	
ð 	
ð 	
ð 	
ó    c                 ó   — dS )NTr   )r+   s    r.   Údo_lower_casez OpenAIGPTTokenizer.do_lower_case]   s   € àˆtr/   )NNr   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚVOCAB_FILES_NAMESÚvocab_files_namesÚmodel_input_namesr   Úmodelr   ÚdictÚintÚlistr*   Úpropertyr1   Ú__classcell__)r-   s   @r.   r   r      sÇ   ø€ € € € € ðð ð4 *ÐØ$Ð&6Ð7ÐØ€Eð .2Ø)-Ø ð	 
ð  
à�T˜#˜s˜(”^Ñ# dÑ*ð 
ð �d˜3”i‘ $Ñ&ð 
ð ð	 
ð  
ð  
ð  
ð  
ð  
ðD ðð ñ „Xðð ð ð ð r/   r   N)r5   Ú
tokenizersr   r   r   r   Útokenizers.modelsr   Útokenization_utils_tokenizersr	   Úutilsr
   Ú
get_loggerr2   Úloggerr6   r   Ú__all__r   r/   r.   ú<module>rF      sÎ   ðð +Ð *à GÐ GÐ GÐ GÐ GÐ GÐ GÐ GÐ GÐ GÐ GÐ GØ !Ð !Ð !Ð !Ð !Ð !à >Ð >Ð >Ð >Ð >Ð >Ø Ð Ð Ð Ð Ð ð 
ˆÔ	˜HÑ	%Ô	%€à#/ÀÐ`pÐqÐqÐ ðCð Cð Cð Cð CÐ*ñ Cô Cð CðL  Ð
 €€€r/   