§
    ‚Štj3  ã                   ó�   — d Z ddlZddlZddlmZ ddlmZ ddlmZ  ej	        e
¦  «        Zddd	œZd
„ Z G d„ de¦  «        ZdgZdS )z Tokenization classes for PhoBERTé    N)Úcopyfileé   )ÚPreTrainedTokenizer)Úloggingz	vocab.txtz	bpe.codes)Ú
vocab_fileÚmerges_filec                 óœ   — t          ¦   «         }| d         }| dd…         D ]}|                     ||f¦  «         |}Œt          |¦  «        }|S )z…
    Return set of symbol pairs in a word.

    Word is represented as tuple of symbols (symbols being variable-length strings).
    r   é   N)ÚsetÚadd)ÚwordÚpairsÚ	prev_charÚchars       ún/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/phobert/tokenization_phobert.pyÚ	get_pairsr   !   s[   € õ ‰EŒE€EØ�Q”€IØ�Q�R�R”ð ð ˆØ�	Š	�9˜dÐ#Ñ$Ô$Ð$Øˆ	ˆ	å�‰JŒJ€EØ€Ló    c            
       ól  ‡ — e Zd ZdZeZ	 	 	 	 	 	 	 dˆ fd„	Z	 dd	ee         d
ee         dz  dee         fd„Z		 dd	ee         d
ee         dz  de
dee         fˆ fd„Z	 dd	ee         d
ee         dz  dee         fd„Zed„ ¦   «         Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zddededz  dee         fd„Zd„ Zˆ xZS )ÚPhobertTokenizeraO	  
    Construct a PhoBERT tokenizer. Based on Byte-Pair-Encoding.

    This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods.

    Args:
        vocab_file (`str`):
            Path to the vocabulary file.
        merges_file (`str`):
            Path to the merges file.
        bos_token (`st`, *optional*, defaults to `"<s>"`):
            The beginning of sequence token that was used during pretraining. Can be used a sequence classifier token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the beginning of
            sequence. The token used is the `cls_token`.

            </Tip>

        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the end of sequence.
            The token used is the `sep_token`.

            </Tip>

        sep_token (`str`, *optional*, defaults to `"</s>"`):
            The separator token, which is used when building a sequence from multiple sequences, e.g. two sequences for
            sequence classification or for a text and a question for question answering. It is also used as the last
            token of a sequence built with special tokens.
        cls_token (`str`, *optional*, defaults to `"<s>"`):
            The classifier token which is used when doing sequence classification (classification of the whole sequence
            instead of per-token classification). It is the first token of the sequence when built with special tokens.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        mask_token (`str`, *optional*, defaults to `"<mask>"`):
            The token used for masking values. This is the token used when training this model with masked language
            modeling. This is the token which the model will try to predict.
    ú<s>ú</s>ú<unk>ú<pad>ú<mask>c
                 óô  •— || _         || _        i | _        d| j        t          |¦  «        <   d| j        t          |¦  «        <   d| j        t          |¦  «        <   d| j        t          |¦  «        <   |                      |¦  «         d„ | j                             ¦   «         D ¦   «         | _        t          |d¬¦  «        5 }|                     ¦   «          	                    d¦  «        d d	…         }d d d ¦  «         n# 1 swxY w Y   d
„ |D ¦   «         }t          t          |t          t          |¦  «        ¦  «        ¦  «        ¦  «        | _        i | _         t!          ¦   «         j        d|||||||	dœ|
¤Ž d S )Nr   r
   é   r   c                 ó   — i | ]\  }}||“Œ	S © r   )Ú.0ÚkÚvs      r   ú
<dictcomp>z-PhobertTokenizer.__init__.<locals>.<dictcomp>|   s   € Ð>Ð>Ð>¡  A˜˜1Ð>Ð>Ð>r   úutf-8©Úencodingú
éÿÿÿÿc                 ó`   — g | ]+}t          |                     ¦   «         d d…         ¦  «        ‘Œ,S )Nr'   )ÚtupleÚsplit)r   Úmerges     r   ú
<listcomp>z-PhobertTokenizer.__init__.<locals>.<listcomp>€   s1   € Ð@Ð@Ð@°•%˜Ÿš™œ c r cÔ*Ñ+Ô+Ð@Ð@Ð@r   )Ú	bos_tokenÚ	eos_tokenÚ	unk_tokenÚ	sep_tokenÚ	cls_tokenÚ	pad_tokenÚ
mask_tokenr   )r   r   ÚencoderÚstrÚadd_from_fileÚitemsÚdecoderÚopenÚreadr*   ÚdictÚzipÚrangeÚlenÚ	bpe_ranksÚcacheÚsuperÚ__init__)Úselfr   r   r-   r.   r0   r1   r/   r2   r3   ÚkwargsÚmerges_handleÚmergesÚ	__class__s                €r   rB   zPhobertTokenizer.__init__d   s¶  ø€ ð %ˆŒØ&ˆÔàˆŒØ'(ˆŒ•S˜‘^”^Ñ$Ø'(ˆŒ•S˜‘^”^Ñ$Ø'(ˆŒ•S˜‘^”^Ñ$Ø'(ˆŒ•S˜‘^”^Ñ$à×Ò˜:Ñ&Ô&Ð&à>Ð>¨¬×);Ò);Ñ)=Ô)=Ð>Ñ>Ô>ˆŒå�+¨Ð0Ñ0Ô0ð 	;°MØ"×'Ò'Ñ)Ô)×/Ò/°Ñ5Ô5°c°r°cÔ:ˆFð	;ð 	;ð 	;ñ 	;ô 	;ð 	;ð 	;ð 	;ð 	;ð 	;ð 	;øøøð 	;ð 	;ð 	;ð 	;à@Ð@¸Ð@Ñ@Ô@ˆå�c &­%µ°F±´Ñ*<Ô*<Ñ=Ô=Ñ>Ô>ˆŒØˆŒ
à�‰ŒÔð 		
ØØØØØØØ!ð		
ð 		
ð ð		
ð 		
ð 		
ð 		
ð 		
s   Ã0C=Ã=DÄDNÚtoken_ids_0Útoken_ids_1Úreturnc                 óp   — |€| j         g|z   | j        gz   S | j         g}| j        g}||z   |z   |z   |z   |z   S )a–  
        Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and
        adding special tokens. A PhoBERT sequence has the following format:

        - single sequence: `<s> X </s>`
        - pair of sequences: `<s> A </s></s> B </s>`

        Args:
            token_ids_0 (`list[int]`):
                List of IDs to which the special tokens will be added.
            token_ids_1 (`list[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `list[int]`: List of [input IDs](../glossary#input-ids) with the appropriate special tokens.
        )Úcls_token_idÚsep_token_id)rC   rH   rI   ÚclsÚseps        r   Ú build_inputs_with_special_tokensz1PhobertTokenizer.build_inputs_with_special_tokens�   s[   € ð( ÐØÔ%Ð&¨Ñ4¸Ô8IÐ7JÑJÐJØÔ Ð!ˆØÔ Ð!ˆØ�[Ñ  3Ñ&¨Ñ,¨{Ñ:¸SÑ@Ð@r   FÚalready_has_special_tokensc                 óò   •— |r$t          ¦   «                              ||d¬¦  «        S |€dgdgt          |¦  «        z  z   dgz   S dgdgt          |¦  «        z  z   ddgz   dgt          |¦  «        z  z   dgz   S )aÄ  
        Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
        special tokens using the tokenizer `prepare_for_model` method.

        Args:
            token_ids_0 (`list[int]`):
                List of IDs.
            token_ids_1 (`list[int]`, *optional*):
                Optional second list of IDs for sequence pairs.
            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not the token list is already formatted with special tokens for the model.

        Returns:
            `list[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
        T)rH   rI   rQ   Nr
   r   )rA   Úget_special_tokens_maskr>   )rC   rH   rI   rQ   rG   s       €r   rS   z(PhobertTokenizer.get_special_tokens_maskª   s£   ø€ ð& &ð 	Ý‘7”7×2Ò2Ø'°[Ð]að 3ñ ô ð ð ÐØ�3˜1˜#¥ KÑ 0Ô 0Ñ0Ñ1°Q°CÑ7Ð7Øˆs�q�c�C Ñ,Ô,Ñ,Ñ-°°A°Ñ6¸1¸#ÅÀKÑ@PÔ@PÑ:PÑQÐUVÐTWÑWÐWr   c                 óœ   — | j         g}| j        g}|€t          ||z   |z   ¦  «        dgz  S t          ||z   |z   |z   |z   |z   ¦  «        dgz  S )aÌ  
        Create a mask from the two sequences passed to be used in a sequence-pair classification task. PhoBERT does not
        make use of token type ids, therefore a list of zeros is returned.

        Args:
            token_ids_0 (`list[int]`):
                List of IDs.
            token_ids_1 (`list[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `list[int]`: List of zeros.
        Nr   )rM   rL   r>   )rC   rH   rI   rO   rN   s        r   Ú$create_token_type_ids_from_sequencesz5PhobertTokenizer.create_token_type_ids_from_sequencesÆ   sm   € ð" Ô Ð!ˆØÔ Ð!ˆàÐÝ�s˜[Ñ(¨3Ñ.Ñ/Ô/°1°#Ñ5Ð5Ý�3˜Ñ$ sÑ*¨SÑ0°;Ñ>ÀÑDÑEÔEÈÈÑKÐKr   c                 ó*   — t          | j        ¦  «        S ©N)r>   r4   ©rC   s    r   Ú
vocab_sizezPhobertTokenizer.vocab_sizeÞ   s   € å�4”<Ñ Ô Ð r   c                 ó0   — t          | j        fi | j        ¤ŽS rW   )r;   r4   Úadded_tokens_encoderrX   s    r   Ú	get_vocabzPhobertTokenizer.get_vocabâ   s   € Ý�D”LÐ>Ð> DÔ$=Ð>Ð>Ð>r   c                 óÜ  ‡ — |‰ j         v r‰ j         |         S t          |¦  «        }t          t          |d d…         ¦  «        |d         dz   gz   ¦  «        }t          |¦  «        }|s|S 	 t	          |ˆ fd„¬¦  «        }|‰ j        vr�n8|\  }}g }d}|t          |¦  «        k     ræ	 |                     ||¦  «        }	|                     |||	…         ¦  «         |	}n-# t          $ r  |                     ||d …         ¦  «         Y n†w xY w||         |k    rC|t          |¦  «        dz
  k     r-||dz            |k    r| 
                    ||z   ¦  «         |dz  }n | 
                    ||         ¦  «         |dz  }|t          |¦  «        k     °æt          |¦  «        }|}t          |¦  «        dk    rnt          |¦  «        }�ŒWd	                     |¦  «        }|d d
…         }|‰ j         |<   |S )Nr'   z</w>Tc                 óT   •— ‰j                              | t          d¦  «        ¦  «        S )NÚinf)r?   ÚgetÚfloat)ÚpairrC   s    €r   ú<lambda>z&PhobertTokenizer.bpe.<locals>.<lambda>ð   s    ø€ °´×1CÒ1CÀDÍ%ÐPUÉ,Ì,Ñ1WÔ1W€ r   )Úkeyr   r
   r   ú@@ éüÿÿÿ)r@   r)   Úlistr   Úminr?   r>   ÚindexÚextendÚ
ValueErrorÚappendÚjoin)
rC   Útokenr   r   ÚbigramÚfirstÚsecondÚnew_wordÚiÚjs
   `         r   ÚbpezPhobertTokenizer.bpeå   s#  ø€ Ø�D”JÐÐØ”:˜eÔ$Ð$Ý�U‰|Œ|ˆÝ•T˜$˜s ˜sœ)‘_”_¨¨R¬°6Ñ(9Ð':Ñ:Ñ;Ô;ˆÝ˜$‘”ˆàð 	ØˆLð	(Ý˜Ð$WÐ$WÐ$WÐ$WÐXÑXÔXˆFØ˜Tœ^Ð+Ð+ÙØ"‰MˆE�6ØˆHØˆAØ•c˜$‘i”i’-�-ðØŸ
š
 5¨!Ñ,Ô,�Að
 —O’O D¨¨1¨¤IÑ.Ô.Ð.Ø�A�Aøõ "ð ð ð Ø—O’O D¨¨¨¤HÑ-Ô-Ð-Ø�Eðøøøð ˜”7˜eÒ#Ð#¨­C°©I¬I¸©MÒ(9Ð(9¸dÀ1ÀqÁ5¼kÈVÒ>SÐ>SØ—O’O E¨F¡NÑ3Ô3Ð3Ø˜‘F�A�Aà—O’O D¨¤GÑ,Ô,Ð,Ø˜‘F�Að •c˜$‘i”i’-�-õ  ˜X‘”ˆHØˆDÝ�4‰yŒy˜AŠ~ˆ~Øå! $™œ�ñ9	(ð: �zŠz˜$ÑÔˆØ�C�R�CŒyˆØ ˆŒ
�5ÑØˆs   Â(C Ã'DÄDc                 óÎ   — g }t          j        d|¦  «        }|D ]J}|                     t          |                      |¦  «                             d¦  «        ¦  «        ¦  «         ŒK|S )zTokenize a string.z\S+\n?ú )ÚreÚfindallrj   rg   ru   r*   )rC   ÚtextÚsplit_tokensÚwordsrn   s        r   Ú	_tokenizezPhobertTokenizer._tokenize  sf   € àˆå”
˜9 dÑ+Ô+ˆàð 	Bð 	BˆEØ×Ò¥ T§X¢X¨e¡_¤_×%:Ò%:¸3Ñ%?Ô%?Ñ @Ô @ÑAÔAÐAÐAØÐr   c                 ór   — | j                              || j                              | j        ¦  «        ¦  «        S )z0Converts a token (str) in an id using the vocab.)r4   r`   r/   )rC   rn   s     r   Ú_convert_token_to_idz%PhobertTokenizer._convert_token_to_id  s,   € àŒ|×Ò  t¤|×'7Ò'7¸¼Ñ'GÔ'GÑHÔHÐHr   c                 óB   — | j                              || j        ¦  «        S )z=Converts an index (integer) in a token (str) using the vocab.)r8   r`   r/   )rC   ri   s     r   Ú_convert_id_to_tokenz%PhobertTokenizer._convert_id_to_token  s   € àŒ|×Ò  t¤~Ñ6Ô6Ð6r   c                 ó|   — d                      |¦  «                             dd¦  «                             ¦   «         }|S )z:Converts a sequence of tokens (string) in a single string.rw   re   Ú )rm   ÚreplaceÚstrip)rC   ÚtokensÚ
out_strings      r   Úconvert_tokens_to_stringz)PhobertTokenizer.convert_tokens_to_string#  s5   € à—X’X˜fÑ%Ô%×-Ò-¨e°RÑ8Ô8×>Ò>Ñ@Ô@ˆ
ØÐr   Úsave_directoryÚfilename_prefixc                 ó  — t           j                             |¦  «        s t                               d|› d�¦  «         d S t           j                             ||r|dz   ndt          d         z   ¦  «        }t           j                             ||r|dz   ndt          d         z   ¦  «        }t           j                             | j        ¦  «        t           j                             |¦  «        k    r:t           j         	                    | j        ¦  «        rt          | j        |¦  «         nzt           j         	                    | j        ¦  «        sVt          |d¦  «        5 }| j                             ¦   «         }|                     |¦  «         d d d ¦  «         n# 1 swxY w Y   t           j                             | j        ¦  «        t           j                             |¦  «        k    rt          | j        |¦  «         ||fS )NzVocabulary path (z) should be a directoryú-rƒ   r   r   Úwb)ÚosÚpathÚisdirÚloggerÚerrorrm   ÚVOCAB_FILES_NAMESÚabspathr   Úisfiler   r9   Úsp_modelÚserialized_model_protoÚwriter   )rC   r‰   rŠ   Úout_vocab_fileÚout_merge_fileÚfiÚcontent_spiece_models          r   Úsave_vocabularyz PhobertTokenizer.save_vocabulary(  sí  € ÝŒw�}Š}˜^Ñ,Ô,ð 	Ý�LŠLÐT¨^ÐTÐTÐTÑUÔUÐUØˆFÝœŸšØ°oÐM˜_¨sÑ2Ð2È2ÕQbÐcoÔQpÑpñ
ô 
ˆõ œŸšØ°oÐM˜_¨sÑ2Ð2È2ÕQbÐcpÔQqÑqñ
ô 
ˆõ Œ7�?Š?˜4œ?Ñ+Ô+­r¬w¯ª¸~Ñ/NÔ/NÒNÐNÕSUÔSZ×SaÒSaÐbfÔbqÑSrÔSrÐNÝ�T”_ nÑ5Ô5Ð5Ð5Ý”—’ ¤Ñ0Ô0ð 	/Ý�n dÑ+Ô+ð /¨rØ'+¤}×'KÒ'KÑ'MÔ'MÐ$Ø—’Ð-Ñ.Ô.Ð.ð/ð /ð /ñ /ô /ð /ð /ð /ð /ð /ð /øøøð /ð /ð /ð /õ Œ7�?Š?˜4Ô+Ñ,Ô,µ´·²ÀÑ0OÔ0OÒOÐOÝ�TÔ% ~Ñ6Ô6Ð6à˜~Ð-Ð-s   Å/FÆFÆFc                 ó  — t          |t          ¦  «        rs	 t          |dd¬¦  «        5 }|                      |¦  «         ddd¦  «         n# 1 swxY w Y   n0# t          $ r}|‚d}~wt
          $ r t          d|› d�¦  «        ‚w xY wdS |                     ¦   «         }|D ]f}|                     ¦   «         }| 	                    d¦  «        }|dk    rt          d	¦  «        ‚|d|…         }t          | j        ¦  «        | j        |<   ŒgdS )
zi
        Loads a pre-existing dictionary from a text file and adds its symbols to this instance.
        Úrr#   r$   NzIncorrect encoding detected in z, please rebuild the datasetrw   r'   z5Incorrect dictionary format, expected '<token> <cnt>')Ú
isinstancer5   r9   r6   ÚFileNotFoundErrorÚUnicodeErrorÚ	ExceptionÚ	readlinesr…   Úrfindrk   r>   r4   )	rC   ÚfÚfdÚfnfeÚlinesÚlineTmpÚlineÚidxr   s	            r   r6   zPhobertTokenizer.add_from_fileE  sq  € õ �a�ÑÔð 	ðcÝ˜!˜S¨7Ð3Ñ3Ô3ð +°rØ×&Ò& rÑ*Ô*Ð*ð+ð +ð +ñ +ô +ð +ð +ð +ð +ð +ð +øøøð +ð +ð +ð +øøå$ð ð ð Ø�
øøøøÝð cð cð cÝÐ aÀ!Ð aÐ aÐ aÑbÔbÐbðcøøøàˆFà—’‘”ˆØð 	3ð 	3ˆGØ—=’=‘?”?ˆDØ—*’*˜S‘/”/ˆCØ�bŠyˆyÝ Ð!XÑYÔYÐYØ˜˜˜”:ˆDÝ!$ T¤\Ñ!2Ô!2ˆDŒL˜ÑÐð	3ð 	3s9   —A ©A¿A ÁAÁA ÁAÁA Á
BÁ!A#Á#!B)r   r   r   r   r   r   r   rW   )NF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r“   Úvocab_files_namesrB   rg   ÚintrP   ÚboolrS   rU   ÚpropertyrY   r\   ru   r}   r   r�   rˆ   r5   r)   r�   r6   Ú__classcell__)rG   s   @r   r   r   1   s1  ø€ € € € € ð.ð .ð` *Ðð ØØØØØØð*
ð *
ð *
ð *
ð *
ð *
ðZ GKðAð AØ œ9ðAØ37¸´9¸tÑ3CðAà	ˆcŒðAð Að Að Að6 puðXð XØ œ9ðXØ37¸´9¸tÑ3CðXØhlðXà	ˆcŒðXð Xð Xð Xð Xð Xð: GKðLð LØ œ9ðLØ37¸´9¸tÑ3CðLà	ˆcŒðLð Lð Lð Lð0 ð!ð !ñ „Xð!ð?ð ?ð ?ð*ð *ð *ðXð ð ðIð Ið Ið7ð 7ð 7ðð ð ð
.ð .¨cð .ÀCÈ$ÁJð .ÐZ_Ð`cÔZdð .ð .ð .ð .ð:3ð 3ð 3ð 3ð 3ð 3ð 3r   r   )r°   rŽ   rx   Úshutilr   Útokenization_pythonr   Úutilsr   Ú
get_loggerr­   r‘   r“   r   r   Ú__all__r   r   r   ú<module>r»      sÏ   ðð 'Ð &à 	€	€	€	Ø 	€	€	€	Ø Ð Ð Ð Ð Ð à 6Ð 6Ð 6Ð 6Ð 6Ð 6Ø Ð Ð Ð Ð Ð ð 
ˆÔ	˜HÑ	%Ô	%€ð Øðð Ð ðð ð ð i3ð i3ð i3ð i3ð i3Ð*ñ i3ô i3ð i3ðX	 Ð
€€€r   