§
    ‚ŠtjJ/  ã                   ó„   — d Z ddlZddlZddlmZ ddlmZ  ej        e¦  «        Z	dddœZ
d	„ Z G d
„ de¦  «        ZdgZdS )z Tokenization classes for BioGPT.é    Né   )ÚPreTrainedTokenizer)Úloggingz
vocab.jsonz
merges.txt)Ú
vocab_fileÚmerges_filec                 ó~   — t          ¦   «         }| d         }| dd…         D ]}|                     ||f¦  «         |}Œ|S )zƒ
    Return set of symbol pairs in a word. word is represented as tuple of symbols (symbols being variable-length
    strings)
    r   é   N)ÚsetÚadd)ÚwordÚpairsÚ	prev_charÚchars       úl/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/biogpt/tokenization_biogpt.pyÚ	get_pairsr      sP   € õ
 ‰EŒE€EØ�Q”€IØ�Q�R�R”ð ð ˆØ�	Š	�9˜dÐ#Ñ$Ô$Ð$Øˆ	ˆ	Ø€Ló    c            
       óB  ‡ — e Zd ZdZeZddgZ	 	 	 	 	 dˆ fd„	Zed	„ ¦   «         Z	d
„ Z
d„ Zd„ Zd„ Zd d„Zd„ Zd„ Zd„ Z	 d!dee         dee         dz  dee         fd„Z	 d"dee         dee         dz  dedee         fˆ fd„Zd!dededz  dee         fd„Zd„ Zd„ Zˆ xZS )#ÚBioGptTokenizera:  
    Construct an FAIRSEQ Transformer tokenizer. Moses tokenization followed by Byte-Pair Encoding.

    This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods.

    Args:
        vocab_file (`str`):
            Path to the vocabulary file.
        merges_file (`str`):
            Merges file.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        bos_token (`str`, *optional*, defaults to `"<s>"`):
            The beginning of sequence token that was used during pretraining. Can be used a sequence classifier token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the beginning of
            sequence. The token used is the `cls_token`.

            </Tip>

        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.

            <Tip>

            When building a sequence using special tokens, this is not the token that is used for the end of sequence.
            The token used is the `sep_token`.

            </Tip>

        sep_token (`str`, *optional*, defaults to `"</s>"`):
            The separator token, which is used when building a sequence from multiple sequences, e.g. two sequences for
            sequence classification or for a text and a question for question answering. It is also used as the last
            token of a sequence built with special tokens.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
    Ú	input_idsÚattention_maskú<unk>ú<s>ú</s>ú<pad>c           
      óè  •— 	 dd l }	n# t          $ r t          d¦  «        ‚w xY wd| _        |	| _        i | _        i | _        	 t          |d¬¦  «        5 }
t          j        |
¦  «        | _	        d d d ¦  «         n# 1 swxY w Y   d„ | j	         
                    ¦   «         D ¦   «         | _        t          |d¬¦  «        5 }|                     ¦   «                              d¦  «        d d…         }d d d ¦  «         n# 1 swxY w Y   d	„ |D ¦   «         }t          t          |t!          t#          |¦  «        ¦  «        ¦  «        ¦  «        | _        i | _         t)          ¦   «         j        d|||||d
œ|¤Ž d S )Nr   zqYou need to install sacremoses to use BioGptTokenizer. See https://pypi.org/project/sacremoses/ for installation.Úenúutf-8©Úencodingc                 ó   — i | ]\  }}||“Œ	S © r!   )Ú.0ÚkÚvs      r   ú
<dictcomp>z,BioGptTokenizer.__init__.<locals>.<dictcomp>v   s   € Ð>Ð>Ð>¡  A˜˜1Ð>Ð>Ð>r   ú
éÿÿÿÿc                 ó`   — g | ]+}t          |                     ¦   «         d d…         ¦  «        ‘Œ,S )Né   )ÚtupleÚsplit)r"   Úmerges     r   ú
<listcomp>z,BioGptTokenizer.__init__.<locals>.<listcomp>y   s1   € Ð?Ð?Ð?¨u•%˜Ÿš™œ b q bÔ)Ñ*Ô*Ð?Ð?Ð?r   )Ú	bos_tokenÚ	eos_tokenÚ	sep_tokenÚ	unk_tokenÚ	pad_tokenr!   )Ú
sacremosesÚImportErrorÚlangÚsmÚcache_moses_tokenizerÚcache_moses_detokenizerÚopenÚjsonÚloadÚencoderÚitemsÚdecoderÚreadr+   ÚdictÚzipÚrangeÚlenÚ	bpe_ranksÚcacheÚsuperÚ__init__)Úselfr   r   r1   r.   r/   r0   r2   Úkwargsr3   Úvocab_handleÚmerges_handleÚmergesÚ	__class__s                €r   rG   zBioGptTokenizer.__init__Z   s  ø€ ð	ØÐÐÐÐøÝð 	ð 	ð 	ÝðMñô ð ð	øøøð ˆŒ	ØˆŒà%'ˆÔ"Ø')ˆÔ$àÝ�* wÐ/Ñ/Ô/ð 	3°<Ýœ9 \Ñ2Ô2ˆDŒLð	3ð 	3ð 	3ñ 	3ô 	3ð 	3ð 	3ð 	3ð 	3ð 	3ð 	3øøøð 	3ð 	3ð 	3ð 	3à>Ð>¨¬×);Ò);Ñ)=Ô)=Ð>Ñ>Ô>ˆŒÝ�+¨Ð0Ñ0Ô0ð 	;°MØ"×'Ò'Ñ)Ô)×/Ò/°Ñ5Ô5°c°r°cÔ:ˆFð	;ð 	;ð 	;ñ 	;ô 	;ð 	;ð 	;ð 	;ð 	;ð 	;ð 	;øøøð 	;ð 	;ð 	;ð 	;à?Ð?¸Ð?Ñ?Ô?ˆÝ�c &­%µ°F±´Ñ*<Ô*<Ñ=Ô=Ñ>Ô>ˆŒØˆŒ
à�‰ŒÔð 	
ØØØØØð	
ð 	
ð ð	
ð 	
ð 	
ð 	
ð 	
s,   ƒ ˆ"ÁA9Á9A=Â A=Â=0C9Ã9C=Ä C=c                 ó*   — t          | j        ¦  «        S )zReturns vocab size)rC   r<   ©rH   s    r   Ú
vocab_sizezBioGptTokenizer.vocab_size†   s   € õ �4”<Ñ Ô Ð r   c                 ó0   — t          | j        fi | j        ¤ŽS ©N)r@   r<   Úadded_tokens_encoderrO   s    r   Ú	get_vocabzBioGptTokenizer.get_vocab‹   s   € Ý�D”LÐ>Ð> DÔ$=Ð>Ð>Ð>r   c                 ó¦   — || j         vr%| j                             |¬¦  «        }|| j         |<   | j         |                              |ddd¬¦  «        S )N©r5   TF)Úaggressive_dash_splitsÚ
return_strÚescape)r7   r6   ÚMosesTokenizerÚtokenize)rH   Útextr5   Úmoses_tokenizers       r   Úmoses_tokenizezBioGptTokenizer.moses_tokenizeŽ   sc   € Ø�tÔ1Ð1Ð1Ø"œg×4Ò4¸$Ð4Ñ?Ô?ˆOØ/>ˆDÔ& tÑ,ØÔ)¨$Ô/×8Ò8Ø¨¸%Èð 9ñ 
ô 
ð 	
r   c                 óž   — || j         vr%| j                             |¬¦  «        }|| j         |<   | j         |                              |¦  «        S )NrV   )r8   r6   ÚMosesDetokenizerÚ
detokenize)rH   Útokensr5   Úmoses_detokenizers       r   Úmoses_detokenizez BioGptTokenizer.moses_detokenize–   sR   € Ø�tÔ3Ð3Ð3Ø $¤× 8Ò 8¸dÐ 8Ñ CÔ CÐØ1BˆDÔ(¨Ñ.ØÔ+¨DÔ1×<Ò<¸VÑDÔDÐDr   c                 ó¦  ‡ — t          |d d…         ¦  «        |d         dz   fz   }|‰ j        v r‰ j        |         S t          |¦  «        }|s|dz   S 	 t          |ˆ fd„¬¦  «        }|‰ j        vr�n8|\  }}g }d}|t          |¦  «        k     ræ	 |                     ||¦  «        }	|                     |||	…         ¦  «         |	}n-# t          $ r  |                     ||d …         ¦  «         Y n†w xY w||         |k    rC|t          |¦  «        dz
  k     r-||dz            |k    r| 	                    ||z   ¦  «         |dz  }n | 	                    ||         ¦  «         |dz  }|t          |¦  «        k     °æt          |¦  «        }|}t          |¦  «        dk    rnt          |¦  «        }�ŒWd	 
                    |¦  «        }|d
k    rd}|‰ j        |<   |S )Nr'   ú</w>Tc                 óT   •— ‰j                              | t          d¦  «        ¦  «        S )NÚinf)rD   ÚgetÚfloat)ÚpairrH   s    €r   ú<lambda>z%BioGptTokenizer.bpe.<locals>.<lambda>¦   s    ø€ °´×1CÒ1CÀDÍ%ÐPUÉ,Ì,Ñ1WÔ1W€ r   ©Úkeyr   r	   r)   ú z
  </w>z
</w>)r*   rE   r   ÚminrD   rC   ÚindexÚextendÚ
ValueErrorÚappendÚjoin)
rH   Útokenr   r   ÚbigramÚfirstÚsecondÚnew_wordÚiÚjs
   `         r   ÚbpezBioGptTokenizer.bpeœ   s  ø€ Ý�U˜3˜B˜3”ZÑ Ô  E¨"¤I°Ñ$6Ð#8Ñ8ˆØ�D”JÐÐØ”:˜eÔ$Ð$Ý˜$‘”ˆàð 	"Ø˜6‘>Ð!ð	(Ý˜Ð$WÐ$WÐ$WÐ$WÐXÑXÔXˆFØ˜Tœ^Ð+Ð+ÙØ"‰MˆE�6ØˆHØˆAØ•c˜$‘i”i’-�-ðØŸ
š
 5¨!Ñ,Ô,�Að
 —O’O D¨¨1¨¤IÑ.Ô.Ð.Ø�A�Aøõ "ð ð ð Ø—O’O D¨¨¨¤HÑ-Ô-Ð-Ø�Eðøøøð ˜”7˜eÒ#Ð#¨­C°©I¬I¸©MÒ(9Ð(9¸dÀ1ÀqÁ5¼kÈVÒ>SÐ>SØ—O’O E¨F¡NÑ3Ô3Ð3Ø˜‘F�A�Aà—O’O D¨¤GÑ,Ô,Ð,Ø˜‘F�Að •c˜$‘i”i’-�-õ  ˜X‘”ˆHØˆDÝ�4‰yŒy˜AŠ~ˆ~Øå! $™œ�ñ9	(ð: �xŠx˜‰~Œ~ˆØ�:ÒÐØˆDØ ˆŒ
�5ÑØˆs   ÂC Ã'C/Ã.C/Fc                 ó  — |r|                      ¦   «         }n|                      || j        ¦  «        }g }|D ]L}|rH|                     t	          |                      |¦  «                              d¦  «        ¦  «        ¦  «         ŒM|S )zReturns a tokenized string.ro   )r+   r^   r5   rr   Úlistr}   )rH   r\   Úbypass_tokenizerÚsplit_tokensrv   s        r   Ú	_tokenizezBioGptTokenizer._tokenizeÈ   sŠ   € àð 	8Ø—:’:‘<”<ˆDˆDà×&Ò& t¨T¬YÑ7Ô7ˆDàˆØð 	Fð 	FˆEØð FØ×#Ò#¥D¨¯ª°%©¬×)>Ò)>¸sÑ)CÔ)CÑ$DÔ$DÑEÔEÐEøàÐr   c                 ór   — | j                              || j                              | j        ¦  «        ¦  «        S )z0Converts a token (str) in an id using the vocab.)r<   ri   r1   )rH   rv   s     r   Ú_convert_token_to_idz$BioGptTokenizer._convert_token_to_idÖ   s,   € àŒ|×Ò  t¤|×'7Ò'7¸¼Ñ'GÔ'GÑHÔHÐHr   c                 óB   — | j                              || j        ¦  «        S )z=Converts an index (integer) in a token (str) using the vocab.)r>   ri   r1   )rH   rq   s     r   Ú_convert_id_to_tokenz$BioGptTokenizer._convert_id_to_tokenÚ   s   € àŒ|×Ò  t¤~Ñ6Ô6Ð6r   c                 ó¢   — d„ |D ¦   «         }d                      |¦  «                             ¦   «         }|                      || j        ¦  «        }|S )z:Converts a sequence of tokens (string) in a single string.c                 ób   — g | ],}|                      d d¦  «                              dd ¦  «        ‘Œ-S )ro   Ú rf   )Úreplace)r"   Úts     r   r-   z<BioGptTokenizer.convert_tokens_to_string.<locals>.<listcomp>á   s6   € ÐJÐJÐJ¸a�!—)’)˜C Ñ$Ô$×,Ò,¨V°SÑ9Ô9ÐJÐJÐJr   r‰   )ru   r+   rd   r5   )rH   rb   r\   s      r   Úconvert_tokens_to_stringz(BioGptTokenizer.convert_tokens_to_stringÞ   sO   € ð KÐJÀ6ÐJÑJÔJˆØ—’˜‘”×&Ò&Ñ(Ô(ˆà×$Ò$ V¨T¬YÑ7Ô7ˆØˆr   NÚtoken_ids_0Útoken_ids_1Úreturnc                 óB   — |€| j         g|z   S | j         g}||z   |z   |z   S )a‹  
        Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and
        adding special tokens. A BioGPT sequence has the following format:

        - single sequence: `</s> X `
        - pair of sequences: `</s> A </s> B `

        Args:
            token_ids_0 (`List[int]`):
                List of IDs to which the special tokens will be added.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `List[int]`: List of [input IDs](../glossary#input-ids) with the appropriate special tokens.
        )Úsep_token_id)rH   r�   rŽ   Úseps       r   Ú build_inputs_with_special_tokensz0BioGptTokenizer.build_inputs_with_special_tokensç   s;   € ð& ÐØÔ%Ð&¨Ñ4Ð4ØÔ Ð!ˆØ�[Ñ  3Ñ&¨Ñ4Ð4r   Úalready_has_special_tokensc                 óà   •— |r$t          ¦   «                              ||d¬¦  «        S |�/dgdgt          |¦  «        z  z   dgz   dgt          |¦  «        z  z   S dgdgt          |¦  «        z  z   S )aÄ  
        Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
        special tokens using the tokenizer `prepare_for_model` method.

        Args:
            token_ids_0 (`List[int]`):
                List of IDs.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.
            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not the token list is already formatted with special tokens for the model.

        Returns:
            `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
        T)r�   rŽ   r”   Nr	   r   )rF   Úget_special_tokens_maskrC   )rH   r�   rŽ   r”   rM   s       €r   r–   z'BioGptTokenizer.get_special_tokens_maskÿ   s‘   ø€ ð$ &ð 	Ý‘7”7×2Ò2Ø'°[Ð]að 3ñ ô ð ð Ð"Ø�3˜1˜#¥ KÑ 0Ô 0Ñ0Ñ1°Q°CÑ7¸A¸3ÅÀ[ÑAQÔAQÑ;QÑRÐRØˆs�q�c�C Ñ,Ô,Ñ,Ñ-Ð-r   Úsave_directoryÚfilename_prefixc           	      óz  — t           j                             |¦  «        s t                               d|› d�¦  «         d S t           j                             ||r|dz   ndt          d         z   ¦  «        }t           j                             ||r|dz   ndt          d         z   ¦  «        }t          |dd¬	¦  «        5 }|                     t          j
        | j        d
dd¬¦  «        dz   ¦  «         d d d ¦  «         n# 1 swxY w Y   d}t          |dd¬	¦  «        5 }t          | j                             ¦   «         d„ ¬¦  «        D ][\  }}	||	k    r t                               d|› d�¦  «         |	}|                     d                     |¦  «        dz   ¦  «         |dz  }Œ\	 d d d ¦  «         n# 1 swxY w Y   ||fS )NzVocabulary path (z) should be a directoryú-r‰   r   r   Úwr   r   r)   TF)ÚindentÚ	sort_keysÚensure_asciir&   r   c                 ó   — | d         S )Nr	   r!   )Úkvs    r   rl   z1BioGptTokenizer.save_vocabulary.<locals>.<lambda>*  s   € ÐY[Ð\]ÔY^€ r   rm   zSaving vocabulary to zZ: BPE merge indices are not consecutive. Please check that the tokenizer is not corrupted!ro   r	   )ÚosÚpathÚisdirÚloggerÚerrorru   ÚVOCAB_FILES_NAMESr9   Úwriter:   Údumpsr<   ÚsortedrD   r=   Úwarning)
rH   r—   r˜   r   Ú
merge_fileÚfrq   ÚwriterÚ
bpe_tokensÚtoken_indexs
             r   Úsave_vocabularyzBioGptTokenizer.save_vocabulary  sq  € ÝŒw�}Š}˜^Ñ,Ô,ð 	Ý�LŠLÐT¨^ÐTÐTÐTÑUÔUÐUØˆFÝ”W—\’\Ø°oÐM˜_¨sÑ2Ð2È2ÕQbÐcoÔQpÑpñ
ô 
ˆ
õ ”W—\’\Ø°oÐM˜_¨sÑ2Ð2È2ÕQbÐcpÔQqÑqñ
ô 
ˆ
õ �*˜c¨GÐ4Ñ4Ô4ð 	c¸Ø�GŠG•D”J˜tœ|°AÀÐTYÐZÑZÔZÐ]aÑaÑbÔbÐbð	cð 	cð 	cñ 	cô 	cð 	cð 	cð 	cð 	cð 	cð 	cøøøð 	cð 	cð 	cð 	cð ˆÝ�*˜c¨GÐ4Ñ4Ô4ð 		¸Ý+1°$´.×2FÒ2FÑ2HÔ2HÐN^ÐN^Ð+_Ñ+_Ô+_ð ð Ñ'�
˜KØ˜KÒ'Ð'Ý—N’NðM°
ð Mð Mð Mñô ð ð (�EØ—’˜SŸXšX jÑ1Ô1°DÑ8Ñ9Ô9Ð9Ø˜‘
��ðð		ð 		ð 		ñ 		ô 		ð 		ð 		ð 		ð 		ð 		ð 		øøøð 		ð 		ð 		ð 		ð ˜:Ð%Ð%s%   Â<4C<Ã<D ÄD ÄBF.Æ.F2Æ5F2c                 óB   — | j                              ¦   «         }d |d<   |S )Nr6   )Ú__dict__Úcopy)rH   Ústates     r   Ú__getstate__zBioGptTokenizer.__getstate__6  s#   € Ø”×"Ò"Ñ$Ô$ˆØˆˆd‰Øˆr   c                 óh   — || _         	 dd l}n# t          $ r t          d¦  «        ‚w xY w|| _        d S )Nr   znYou need to install sacremoses to use XLMTokenizer. See https://pypi.org/project/sacremoses/ for installation.)r²   r3   r4   r6   )rH   Údr3   s      r   Ú__setstate__zBioGptTokenizer.__setstate__;  s]   € ØˆŒð	ØÐÐÐÐøÝð 	ð 	ð 	ÝðMñô ð ð	øøøð ˆŒˆˆs   ‰ Ž()r   r   r   r   r   )FrR   )NF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r¦   Úvocab_files_namesÚmodel_input_namesrG   ÚpropertyrP   rT   r^   rd   r}   r‚   r„   r†   rŒ   r   Úintr“   Úboolr–   Ústrr*   r°   rµ   r¸   Ú__classcell__)rM   s   @r   r   r   ,   s  ø€ € € € € ð(ð (ðT *ÐØ$Ð&6Ð7Ðð ØØØØð*
ð *
ð *
ð *
ð *
ð *
ðX ð!ð !ñ „Xð!ð?ð ?ð ?ð
ð 
ð 
ðEð Eð Eð*ð *ð *ðXð ð ð ðIð Ið Ið7ð 7ð 7ðð ð ð GKð5ð 5Ø œ9ð5Ø37¸´9¸tÑ3Cð5à	ˆcŒð5ð 5ð 5ð 5ð2 puð.ð .Ø œ9ð.Ø37¸´9¸tÑ3Cð.Øhlð.à	ˆcŒð.ð .ð .ð .ð .ð .ð6&ð &¨cð &ÀCÈ$ÁJð &ÐZ_Ð`cÔZdð &ð &ð &ð &ð8ð ð ð
ð ð ð ð ð ð r   r   )r¼   r:   r¡   Útokenization_pythonr   Úutilsr   Ú
get_loggerr¹   r¤   r¦   r   r   Ú__all__r!   r   r   ú<module>rÈ      s½   ðð 'Ð &à €€€Ø 	€	€	€	à 6Ð 6Ð 6Ð 6Ð 6Ð 6Ø Ð Ð Ð Ð Ð ð 
ˆÔ	˜HÑ	%Ô	%€ð Øðð Ð ð
ð 
ð 
ðZð Zð Zð Zð ZÐ)ñ Zô Zð Zðz Ð
€€€r   