§
    ‚ŠtjQ5  ã                   ó  — d Z ddlZddlmZ 	 ddlZn# e$ r dZY nw xY wddlmZ ddl	m
Z
 ddlmZmZmZ ddlmZmZmZ  ej        e¦  «        Zd	d
iZdZ ee¦  «         G d„ de
¦  «        ¦   «         Z G d„ d¦  «        ZdS )zT
SentencePiece-based tokenization class for loading from sentencepiece.model files.
é    N)Úcopyfileé   )Úimport_protobuf)ÚPreTrainedTokenizer)ÚINIT_TOKENIZER_DOCSTRINGÚ
AddedTokenÚgenerate_merges)Úadd_end_docstringsÚloggingÚrequires_backendsÚ
vocab_fileztokenizer.modelu   â–�c                   ó2  ‡ — e Zd ZdZeZˆ fd„Zedefd„¦   «         Z	d„ Z
ddee         ee         z  dedefd	„Zddee         d
z  fd„Zd„ Zd„ Zd„ Zdee         defd„Zddeded
z  dee         fd„Z	 	 	 ddeee         z  deded
z  dedef
ˆ fd„Zˆ xZS )ÚSentencePieceBackendaJ  
    Base class for SentencePiece-based tokenizers that load from sentencepiece.model files.

    Inherits from [`~tokenization_utils.PreTrainedTokenizer`].

    Handle all the shared methods for tokenization and special tokens as well as methods downloading/caching/loading
    pretrained tokenizers as well as adding tokens to the vocabulary.

    This class also contain the added tokens in a unified way on top of all tokenizers so we don't have to handle the
    specific vocabulary augmentation methods of the various underlying dictionary structures (BPE, sentencepiece...).
    c                 óò  •— t          | d¦  «         |                     d¦  «        | _        |                     dd¦  «        | _        |                     di ¦  «        | _        d|vrd|d<   t          j        di | j        ¤Ž}|                     | j        ¦  «         | j        syt          ¦   «         }|j
                             |                     ¦   «         ¦  «        }|j        j        r3d|j        _        |                     |                     ¦   «         ¦  «         || _        | j                             ¦   «         | _        | j        |d<    t)          ¦   «         j        di |¤Ž |                      ¦   «          d S )	NÚsentencepiecer   ÚlegacyTÚsp_model_kwargsÚbackendF© )r   Úgetr   r   Úpopr   ÚspmÚSentencePieceProcessorÚLoadr   Ú
ModelProtoÚ
FromStringÚserialized_model_protoÚnormalizer_specÚadd_dummy_prefixÚLoadFromSerializedProtoÚSerializeToStringÚsp_modelÚget_piece_sizeÚtotal_vocab_sizeÚsuperÚ__init__Ú_update_trie)ÚselfÚkwargsÚ	tokenizerÚ	model_pb2ÚprotoÚ	__class__s        €úk/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/tokenization_utils_sentencepiece.pyr&   zSentencePieceBackend.__init__<   sr  ø€ å˜$ Ñ0Ô0Ð0ð !Ÿ*š* \Ñ2Ô2ˆŒØ—j’j ¨4Ñ0Ô0ˆŒØ%ŸzšzÐ*;¸RÑ@Ô@ˆÔð ˜FÐ"Ð"Ø /ˆF�9Ñõ Ô.ÐFÐF°Ô1EÐFÐFˆ	Ø�Š�t”Ñ'Ô'Ð'àŒ{ð 	MÝ'Ñ)Ô)ˆIØÔ(×3Ò3°I×4TÒ4TÑ4VÔ4VÑWÔWˆEØÔ$Ô5ð MØ9>�Ô%Ô6Ø×1Ò1°%×2IÒ2IÑ2KÔ2KÑLÔLÐLà!ˆŒð !%¤× <Ò <Ñ >Ô >ˆÔð %)Ô$8ˆÐ Ñ!ð
 	�‰ŒÔÐ"Ð"˜6Ð"Ð"Ð"Ø×ÒÑÔÐÐÐó    Úreturnc                 ó4   — | j                              ¦   «         S )zReturns vocab size)r"   r#   )r(   s    r.   Ú
vocab_sizezSentencePieceBackend.vocab_sizec   s   € ð Œ}×+Ò+Ñ-Ô-Ð-r/   c                 ó|   ‡ — ˆ fd„t          ‰ j        ¦  «        D ¦   «         }|                     ‰ j        ¦  «         |S )zReturns vocab as a dictc                 ó<   •— i | ]}‰                      |¦  «        |“ŒS r   )Úconvert_ids_to_tokens)Ú.0Úir(   s     €r.   ú
<dictcomp>z2SentencePieceBackend.get_vocab.<locals>.<dictcomp>j   s)   ø€ ÐRÐRÐR°a�×+Ò+¨AÑ.Ô.°ÐRÐRÐRr/   )Úranger2   ÚupdateÚadded_tokens_encoder)r(   Úvocabs   ` r.   Ú	get_vocabzSentencePieceBackend.get_vocabh   s@   ø€ àRÐRÐRÐR½5ÀÄÑ;QÔ;QÐRÑRÔRˆØ�Š�TÔ.Ñ/Ô/Ð/Øˆr/   FÚ
new_tokensÚspecial_tokensc           	      ón  — |sdS t          | ¦  «        }d}|D �]ó}t          |t          t          f¦  «        s#t	          d|› dt          |¦  «        › d�¦  «        ‚t          |¦  «        dk    rŒVt          |t          ¦  «        r+|| j        v rŒu|| j        v p|}t          |dd| |¬¦  «        }n|r|                     d|j	        d	œ¦  «         || j
                             ¦   «         v rŒÑ|j        s6|j	        r/t          | d
d¦  «        r|j                             ¦   «         |_        | j                             |j        ¦  «        }|| j                             ¦   «         k     o"| j                             |¦  «        |j        k    }|r|}	n|}	|dz  }|dz  }|j        r0t          |¦  «        | j        vr| j                             |¦  «         || j
        |	<   |	| j        |j        <   | j        rt.                               d|› d�¦  «         �Œõ|                      ¦   «          |                      ¦   «          |S )a…  
        Add a list of new tokens to the tokenizer class. If the new tokens are not in the vocabulary, they are added to
        it with indices starting from length of the current vocabulary. Special tokens are sometimes already in the
        vocab which is why they have to be handled specifically.

        Args:
            new_tokens (`list[str]`or `list[tokenizers.AddedToken]`):
                Token(s) to add in vocabulary. A token is counted as added if it's not already in the vocabulary
                (tested by checking if the tokenizer assign the index of the `unk_token` to them). If a token is part
                of the vocabulary then we simply mark this token as an `AddedToken` which allows to control the
                stripping and normalization of this token. This is NOT possible in `tokenizers`.
            special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not the tokens should be added as special tokens.

        Returns:
            `int`: The number of tokens actually added to the vocabulary.

        Examples:

        ```python
        # Let's see how to increase the vocabulary of Bert model and tokenizer
        tokenizer = BertTokenizer.from_pretrained("google-bert/bert-base-uncased")
        model = BertModel.from_pretrained("google-bert/bert-base-uncased")

        num_added_toks = tokenizer.add_tokens(["new_tok1", "my_new-tok2"])
        print("We have added", num_added_toks, "tokens")
        # Note: resize_token_embeddings expects to receive the full size of the new vocabulary, i.e. the length of the tokenizer.
        model.resize_token_embeddings(len(tokenizer))
        ```r   zToken z is not a string but a ú.Ú F)ÚrstripÚlstripÚ
normalizedÚspecialT)rF   rE   Údo_lower_caser   zAdding z to the vocabulary)ÚlenÚ
isinstanceÚstrr   Ú	TypeErrorÚtypeÚ_added_tokens_encoderÚall_special_tokensÚ__setstate__rE   Ú_added_tokens_decoderÚvaluesrF   ÚgetattrÚcontentÚlowerr"   Úpiece_to_idr#   Ú	IdToPieceÚ_extra_special_tokensÚappendÚverboseÚloggerÚinfor'   Ú_update_total_vocab_size)
r(   r>   r?   Ú
next_indexÚ	num_addedÚtokenÚ
is_specialÚtok_idÚin_base_vocabÚtoken_indexs
             r.   Ú_add_tokensz SentencePieceBackend._add_tokensn   s~  € ð< ð 	Ø�1å˜‘Y”Yˆ
Øˆ	Øð '	Añ '	AˆEÝ˜e¥c­:Ð%6Ñ7Ô7ð WÝÐ U¨Ð UÐ UÅtÈEÁ{Ä{Ð UÐ UÐ UÑVÔVÐVÝ�5‰zŒz˜RÒÐØÝ˜%¥Ñ%Ô%ð VØ˜DÔ6Ð6Ð6ØØ" dÔ&=Ð=ÐOÀ�
Ý" 5°¸uÐU_ÐQ_ÐisÐtÑtÔt��Øð Vð ×"Ò"¨tÀ5ÔCSÐ#TÐ#TÑUÔUÐUà˜Ô2×9Ò9Ñ;Ô;Ð;Ð;ØØ”=ð 6 UÔ%5ð 6½'À$ÈÐY^Ñ:_Ô:_ð 6Ø %¤× 3Ò 3Ñ 5Ô 5�”ð ”]×.Ò.¨u¬}Ñ=Ô=ˆFà˜œ×5Ò5Ñ7Ô7Ò7Ðl¸D¼M×<SÒ<SÐTZÑ<[Ô<[Ð_dÔ_lÒ<lð ð ð Ø$��à(�Ø˜a‘�
Ø˜Q‘�	àŒ}ð 9¥ U¡¤°4Ô3JÐ!JÐ!JØÔ*×1Ò1°%Ñ8Ô8Ð8à6;ˆDÔ& {Ñ3Ø8CˆDÔ& u¤}Ñ5ØŒ|ð AÝ—’Ð? eÐ?Ð?Ð?Ñ@Ô@Ð@ùà×ÒÑÔÐØ×%Ò%Ñ'Ô'Ð'ØÐr/   NÚunique_no_split_tokensc                 ód  — | j                              ¦   «         D ]4}|j        | j        j        vr| j                             |j        ¦  «         Œ5| j        D ]*}|| j        j        vr| j                             |¦  «         Œ+|pg D ]*}|| j        j        vr| j                             |¦  «         Œ+d S ©N)rP   rQ   rS   Útokens_trieÚ_tokensÚaddrN   )r(   re   r_   s      r.   r'   z!SentencePieceBackend._update_trie¾   sÏ   € àÔ/×6Ò6Ñ8Ô8ð 	4ð 	4ˆEØŒ} DÔ$4Ô$<Ð<Ð<ØÔ ×$Ò$ U¤]Ñ3Ô3Ð3øàÔ,ð 	,ð 	,ˆEØ˜DÔ,Ô4Ð4Ð4ØÔ ×$Ò$ UÑ+Ô+Ð+øà+Ð1¨rð 	,ð 	,ˆEØ˜DÔ,Ô4Ð4Ð4ØÔ ×$Ò$ UÑ+Ô+Ð+øð	,ð 	,r/   c                 óŒ  — | j         s|                     t          df¦  «        s!| j                             |t
          ¬¦  «        S | j                             | j        |z   t
          ¬¦  «        }t          | j                             t          | j        ¦  «        ¦  «        ¦  «        }t          |¦  «        |k    r
||d…         n|S )u(  
        Returns a tokenized string.

        We de-activated the `add_dummy_prefix` option, thus the sentencepiece internals will always strip any
        SPIECE_UNDERLINE. For example: `self.sp_model.encode(f"{SPIECE_UNDERLINE}Hey", out_type = str)` will give
        `['H', 'e', 'y']` instead of `['â–�He', 'y']`. Thus we always encode `f"{unk_token}text"` and strip the
        `unk_token`. Here is an example with `unk_token = "<unk>"` and `unk_token_length = 4`.
        `self.tokenizer.sp_model.encode("<unk> Hey", out_type = str)[4:]`.
        ú )Úout_typeN)r   Ú
startswithÚSPIECE_UNDERLINEr"   ÚencoderJ   Ú	unk_tokenrH   )r(   Útextr)   ÚtokensÚunk_token_lengths        r.   Ú	_tokenizezSentencePieceBackend._tokenizeÌ   s¶   € ð Œ;ð 	<˜dŸošoÕ/?ÀÐ.EÑFÔFð 	<Ø”=×'Ò'¨µsÐ'Ñ;Ô;Ð;ð ”×%Ò% d¤n°tÑ&;ÅcÐ%ÑJÔJˆå˜tœ}×3Ò3µC¸¼Ñ4GÔ4GÑHÔHÑIÔIÐÝ,/°©K¬KÐ;KÒ,KÐ,KˆvÐ&Ð'Ð'Ô(Ð(ÐQWÐWr/   c                 ó6   — | j                              |¦  «        S )z0Converts a token (str) to an id using the vocab.)r"   rU   )r(   r_   s     r.   Ú_convert_token_to_idz)SentencePieceBackend._convert_token_to_idß   s   € àŒ}×(Ò(¨Ñ/Ô/Ð/r/   c                 ó:   — | j                              |¦  «        }|S )z=Converts an index (integer) in a token (str) using the vocab.)r"   rV   )r(   Úindexr_   s      r.   Ú_convert_id_to_tokenz)SentencePieceBackend._convert_id_to_tokenã   s   € à”×'Ò'¨Ñ.Ô.ˆØˆr/   rs   c                 ó†   — d                      |¦  «                             t          d¦  «                             ¦   «         }|S )z:Converts a sequence of tokens (string) in a single string.rB   rl   )ÚjoinÚreplacero   Ústrip)r(   rs   Ú
out_strings      r.   Úconvert_tokens_to_stringz-SentencePieceBackend.convert_tokens_to_stringè   s4   € à—W’W˜V‘_”_×,Ò,Õ-=¸sÑCÔC×IÒIÑKÔKˆ
ØÐr/   Úsave_directoryÚfilename_prefixc                 óâ  — t           j                             |¦  «        s t                               d|› d�¦  «         dS t           j                             ||r|dz   nd| j        d         z   ¦  «        }t           j                             | j        ¦  «        t           j                             |¦  «        k    r:t           j         	                    | j        ¦  «        rt          | j        |¦  «         nzt           j         	                    | j        ¦  «        sVt          |d¦  «        5 }| j                             ¦   «         }|                     |¦  «         ddd¦  «         n# 1 swxY w Y   |fS )aŒ  
        Save the sentencepiece vocabulary (copy original file) to a directory.

        Args:
            save_directory (`str`):
                The directory in which to save the vocabulary.
            filename_prefix (`str`, *optional*):
                An optional prefix to add to the named of the saved files.

        Returns:
            `tuple(str)`: Paths to the files saved.
        zVocabulary path (z) should be a directoryNú-rB   r   Úwb)ÚosÚpathÚisdirrZ   Úerrorr|   Úvocab_files_namesÚabspathr   Úisfiler   Úopenr"   r   Úwrite)r(   r�   r‚   Úout_vocab_fileÚfiÚcontent_spiece_models         r.   Úsave_vocabularyz$SentencePieceBackend.save_vocabularyí   s|  € õ Œw�}Š}˜^Ñ,Ô,ð 	Ý�LŠLÐT¨^ÐTÐTÐTÑUÔUÐUØˆFÝœŸšØ°oÐM˜_¨sÑ2Ð2È2ÐQUÔQgÐhtÔQuÑuñ
ô 
ˆõ Œ7�?Š?˜4œ?Ñ+Ô+­r¬w¯ª¸~Ñ/NÔ/NÒNÐNÕSUÔSZ×SaÒSaÐbfÔbqÑSrÔSrÐNÝ�T”_ nÑ5Ô5Ð5Ð5Ý”—’ ¤Ñ0Ô0ð 	/Ý�n dÑ+Ô+ð /¨rØ'+¤}×'KÒ'KÑ'MÔ'MÐ$Ø—’Ð-Ñ.Ô.Ð.ð/ð /ð /ñ /ô /ð /ð /ð /ð /ð /ð /øøøð /ð /ð /ð /ð Ð Ð s   Ä(/E#Å#E'Å*E'Ú	token_idsÚskip_special_tokensÚclean_up_tokenization_spacesÚspaces_between_special_tokensc                 ó>   •—  t          ¦   «         j        d|||dœ|¤ŽS )zØ
        Decode token ids to string.

        Uses the generic decode path from PreTrainedTokenizer which works for all vocabularies,
        including custom vocabularies that override _convert_id_to_token.
        )r“   r”   r•   r   )r%   Ú_decode)r(   r“   r”   r•   r–   r)   r-   s         €r.   r˜   zSentencePieceBackend._decode
  s<   ø€ ð �u‰wŒwŒð 
ØØ 3Ø)Eð
ð 
ð ð	
ð 
ð 	
r/   )Frg   )FNF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚVOCAB_FILES_NAMESrŠ   r&   ÚpropertyÚintr2   r=   ÚlistrJ   r   Úboolrd   r'   ru   rw   rz   r€   Útupler’   r˜   Ú__classcell__)r-   s   @r.   r   r   ,   së  ø€ € € € € ð
ð 
ð *Ðð%ð %ð %ð %ð %ðN ð.˜Cð .ð .ð .ñ „Xð.ðð ð ðNð N d¨3¤i°$°zÔ2BÑ&Bð NÐTXð NÐehð Nð Nð Nð Nð`,ð ,°4¸´9¸tÑ3Cð ,ð ,ð ,ð ,ðXð Xð Xð&0ð 0ð 0ðð ð ð
¨t°C¬yð ¸Sð ð ð ð ð
!ð !¨cð !ÀCÈ$ÁJð !ÐZ_Ð`cÔZdð !ð !ð !ð !ð@ %*Ø48Ø.3ð
ð 
à˜˜cœ‘?ð
ð "ð
ð '+¨T¡kð	
ð
 (,ð
ð 
ð
ð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
r/   r   c                   óv   — e Zd ZdZdefd„Zddeeeef         e	eee
f                  e	e         f         fd„ZdS )ÚSentencePieceExtractorzl
    Extractor implementation for SentencePiece trained models. https://github.com/google/sentencepiece
    Úmodelc                 ó„   — t          | d¦  «         ddlm}  |¦   «         | _        | j                             |¦  «         d S )Nr   r   )r   )r   r   r   Úspr   )r(   r¦   r   s      r.   r&   zSentencePieceExtractor.__init__&  sN   € Ý˜$ Ñ0Ô0Ð0Ø8Ð8Ð8Ð8Ð8Ð8à(Ð(Ñ*Ô*ˆŒØŒ�Š�UÑÔÐÐÐr/   Nr0   c                 óJ  ‡— | j         Šˆfd„t          ‰                     ¦   «         ¦  «        D ¦   «         }ˆfd„t          ‰                     ¦   «         ¦  «        D ¦   «         }t          ||¦  «        }ˆfd„t          ‰                     ¦   «         ¦  «        D ¦   «         }|||fS )zÂ
        By default will return vocab and merges with respect to their order, by sending `vocab_scores` we're going to
        order the merges with respect to the piece scores instead.
        c                 ó<   •— i | ]}‰                      |¦  «        |“ŒS r   )Úid_to_piece)r6   ry   r¨   s     €r.   r8   z2SentencePieceExtractor.extract.<locals>.<dictcomp>3  s'   ø€ ÐXÐXÐX°e�R—^’^ EÑ*Ô*¨EÐXÐXÐXr/   c                 ób   •— i | ]+}‰                      |¦  «        ‰                     |¦  «        “Œ,S r   ©r«   Ú	get_score©r6   r7   r¨   s     €r.   r8   z2SentencePieceExtractor.extract.<locals>.<dictcomp>5  s1   ø€ ÐbÐbÐbÀA˜RŸ^š^¨AÑ.Ô.°·²¸Q±´ÐbÐbÐbr/   c                 ód   •— g | ],}‰                      |¦  «        ‰                     |¦  «        f‘Œ-S r   r­   r¯   s     €r.   ú
<listcomp>z2SentencePieceExtractor.extract.<locals>.<listcomp>9  s4   ø€ ÐdÐdÐdÀa˜bŸnšn¨QÑ/Ô/°·²¸a±´ÐAÐdÐdÐdr/   )r¨   r9   ÚGetPieceSizer	   )r(   Úvocab_scoresÚ	vocab_idsÚvocab_scores_dictÚmergesÚvocab_scores_listr¨   s         @r.   ÚextractzSentencePieceExtractor.extract-  s°   ø€ ð
 ŒWˆØXÐXÐXÐX½uÀRÇ_Â_ÑEVÔEVÑ?WÔ?WÐXÑXÔXˆ	àbÐbÐbÐbÍÈrÏÊÑO`ÔO`ÑIaÔIaÐbÑbÔbÐå  Ð,=Ñ>Ô>ˆàdÐdÐdÐdÍ5ÐQS×Q`ÒQ`ÑQbÔQbÑKcÔKcÐdÑdÔdÐàÐ+¨VÐ3Ð3r/   rg   )r™   rš   r›   rœ   rJ   r&   r¢   ÚdictrŸ   r    Úfloatr¸   r   r/   r.   r¥   r¥   !  sƒ   € € € € € ðð ð˜cð ð ð ð ð4ð 4¨E°$°s¸C°x´.À$ÀuÈSÐRWÈZÔGXÔBYÐ[_Ð`eÔ[fÐ2fÔ,gð 4ð 4ð 4ð 4ð 4ð 4r/   r¥   )rœ   r†   Úshutilr   r   r   ÚImportErrorÚconvert_slow_tokenizerr   Útokenization_pythonr   Útokenization_utils_baser   r   r	   Úutilsr
   r   r   Ú
get_loggerr™   rZ   r�   ro   r   r¥   r   r/   r.   ú<module>rÂ      s‚  ððð ð 
€	€	€	Ø Ð Ð Ð Ð Ð ðØÐÐÐÐøØð ð ð Ø
€C€C€Cðøøøð 4Ð 3Ð 3Ð 3Ð 3Ð 3Ø 4Ð 4Ð 4Ð 4Ð 4Ð 4ðð ð ð ð ð ð ð ð ð ð
 BÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ Að 
ˆÔ	˜HÑ	%Ô	%€à!Ð#4Ð5Ð àÐ ð ÐÐ,Ñ-Ô-ðq
ð q
ð q
ð q
ð q
Ð.ñ q
ô q
ñ .Ô-ðq
ðh4ð 4ð 4ð 4ð 4ñ 4ô 4ð 4ð 4ð 4s   Ž “œ