§
    ‚Štj¦F  ã                   ó<  — d dl Z d dlZd dlZd dlmZ d dlmZ d dlmZ d dl	Z	ddl
mZ ddlmZ ddlmZ  ej        e¦  «        Zd	d
ddddœZdZ ed¬¦  «         G d„ de¦  «        ¦   «         Zdedeeef         de	j        fd„Zdeddfd„Zdedeez  fd„ZdgZdS )é    N)ÚPath)Úcopyfile)ÚAnyé   )ÚPreTrainedTokenizer)Úlogging)Úrequiresz
source.spmz
target.spmz
vocab.jsonztarget_vocab.jsonztokenizer_config.json)Ú
source_spmÚ
target_spmÚvocabÚtarget_vocab_fileÚtokenizer_config_fileu   â–�)Úsentencepiece)Úbackendsc            
       óâ  ‡ — e Zd ZdZeZddgZ	 	 	 	 	 	 	 	 	 d0d
eee	f         dz  ddfˆ fd„Z
d„ Zdedefd„Zd„ Zdefd„Zdedee         fd„Zdedefd„Zˆ fd„Zˆ fd„Z	 	 d1dededz  defˆ fd„Zdee         defd„Zd2dee         fd„Zd„ Zd„ Zedefd „¦   «         Zd2d!ed"edz  dee         fd#„Zdefd$„Zd%„ Z d&„ Z!defd'„Z"d(eddfd)„Z#d*„ Z$d+„ Z%	 d3d,ed-edz  d.edee         fd/„Z&ˆ xZ'S )4ÚMarianTokenizeraB  
    Construct a Marian tokenizer. Based on [SentencePiece](https://github.com/google/sentencepiece).

    This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods.

    Args:
        source_spm (`str`):
            [SentencePiece](https://github.com/google/sentencepiece) file (generally has a .spm extension) that
            contains the vocabulary for the source language.
        target_spm (`str`):
            [SentencePiece](https://github.com/google/sentencepiece) file (generally has a .spm extension) that
            contains the vocabulary for the target language.
        source_lang (`str`, *optional*):
            A string representing the source language.
        target_lang (`str`, *optional*):
            A string representing the target language.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        model_max_length (`int`, *optional*, defaults to 512):
            The maximum sentence length the model accepts.
        additional_special_tokens (`list[str]`, *optional*, defaults to `["<eop>", "<eod>"]`):
            Additional special tokens used by the tokenizer.
        sp_model_kwargs (`dict`, *optional*):
            Will be passed to the `SentencePieceProcessor.__init__()` method. The [Python wrapper for
            SentencePiece](https://github.com/google/sentencepiece/tree/master/python) can be used, among other things,
            to set:

            - `enable_sampling`: Enable subword regularization.
            - `nbest_size`: Sampling parameters for unigram. Invalid for BPE-Dropout.

              - `nbest_size = {0,1}`: No sampling is performed.
              - `nbest_size > 1`: samples from the nbest_size results.
              - `nbest_size < 0`: assuming that nbest_size is infinite and samples from the all hypothesis (lattice)
                using forward-filtering-and-backward-sampling algorithm.

            - `alpha`: Smoothing parameter for unigram sampling, and dropout probability of merge operations for
              BPE-dropout.

    Examples:

    ```python
    >>> from transformers import MarianForCausalLM, MarianTokenizer

    >>> model = MarianForCausalLM.from_pretrained("Helsinki-NLP/opus-mt-en-de")
    >>> tokenizer = MarianTokenizer.from_pretrained("Helsinki-NLP/opus-mt-en-de")
    >>> src_texts = ["I am a small frog.", "Tom asked his teacher for advice."]
    >>> tgt_texts = ["Ich bin ein kleiner Frosch.", "Tom bat seinen Lehrer um Rat."]  # optional
    >>> inputs = tokenizer(src_texts, text_target=tgt_texts, return_tensors="pt", padding=True)

    >>> outputs = model(**inputs)  # should work
    ```Ú	input_idsÚattention_maskNú<unk>ú</s>ú<pad>é   FÚsp_model_kwargsÚreturnc                 óN  •— |€i n|| _         t          |¦  «                             ¦   «         sJ d|› �¦   «         ‚|| _        t	          |¦  «        | _        t          |¦  «        | j        vrt          d¦  «        ‚|rDt	          |¦  «        | _        d„ | j         	                    ¦   «         D ¦   «         | _
        g | _        n>d„ | j         	                    ¦   «         D ¦   «         | _
        d„ | j        D ¦   «         | _        || _        || _        ||g| _        t          || j         ¦  «        | _        t          || j         ¦  «        | _        | j        | _        | j        | _        |                      ¦   «          d| _         t-          ¦   «         j        d|||||	|
| j         ||dœ	|¤Ž d S )	Nzcannot find spm source z <unk> token must be in the vocabc                 ó   — i | ]\  }}||“Œ	S © r   ©Ú.0ÚkÚvs      úl/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/marian/tokenization_marian.pyú
<dictcomp>z,MarianTokenizer.__init__.<locals>.<dictcomp>†   s   € ÐIÐIÐI¡T Q¨˜A˜qÐIÐIÐIó    c                 ó   — i | ]\  }}||“Œ	S r   r   r   s      r"   r#   z,MarianTokenizer.__init__.<locals>.<dictcomp>‰   s   € ÐBÐBÐB¡T Q¨˜A˜qÐBÐBÐBr$   c                 óf   — g | ].}|                      d ¦  «        ¯|                     d¦  «        ¯,|‘Œ/S )ú>>ú<<)Ú
startswithÚendswith)r   r    s     r"   ú
<listcomp>z,MarianTokenizer.__init__.<locals>.<listcomp>Š   s?   € Ð2vÐ2vÐ2v¸ÈaÏlÊlÐ[_ÑN`ÔN`Ð2vÐef×eoÒeoÐptÑeuÔeuÐ2v°1Ð2vÐ2vÐ2vr$   F)	Úsource_langÚtarget_langÚ	unk_tokenÚ	eos_tokenÚ	pad_tokenÚmodel_max_lengthr   r   Úseparate_vocabsr   )r   r   Úexistsr2   Ú	load_jsonÚencoderÚstrÚKeyErrorÚtarget_encoderÚitemsÚdecoderÚsupported_language_codesr,   r-   Ú	spm_filesÚload_spmÚ
spm_sourceÚ
spm_targetÚcurrent_spmÚcurrent_encoderÚ_setup_normalizerÚ_decode_use_source_tokenizerÚsuperÚ__init__)Úselfr
   r   r   r   r,   r-   r.   r/   r0   r1   r   r2   ÚkwargsÚ	__class__s                 €r"   rE   zMarianTokenizer.__init__k   sÑ  ø€ ð  &5Ð%<˜r˜rÀ/ˆÔå�JÑÔ×&Ò&Ñ(Ô(ÐPÐPÐ*PÀJÐ*PÐ*PÑPÔPÐ(à.ˆÔÝ  Ñ'Ô'ˆŒÝˆy‰>Œ> ¤Ð-Ð-ÝÐ=Ñ>Ô>Ð>àð 	wÝ"+Ð,=Ñ">Ô">ˆDÔØIÐI¨TÔ-@×-FÒ-FÑ-HÔ-HÐIÑIÔIˆDŒLØ,.ˆDÔ)Ð)àBÐB¨T¬\×-?Ò-?Ñ-AÔ-AÐBÑBÔBˆDŒLØ2vÐ2v¸d¼lÐ2vÑ2vÔ2vˆDÔ)à&ˆÔØ&ˆÔØ$ jÐ1ˆŒõ # :¨tÔ/CÑDÔDˆŒÝ" :¨tÔ/CÑDÔDˆŒØœ?ˆÔØ#œ|ˆÔð 	×ÒÑ Ô Ð à,1ˆÔ)à�‰ŒÔð 	
à#Ø#ØØØØ-Ø Ô0Ø/Ø+ð	
ð 	
ð ð	
ð 	
ð 	
ð 	
ð 	
r$   c                 ó°   — 	 ddl m}  || j        ¦  «        j        | _        d S # t
          t          f$ r  t          j        d¦  «         d„ | _        Y d S w xY w)Nr   )ÚMosesPunctNormalizerz$Recommended: pip install sacremoses.c                 ó   — | S ©Nr   )Úxs    r"   ú<lambda>z3MarianTokenizer._setup_normalizer.<locals>.<lambda>±   s   € ¨Q€ r$   )	Ú
sacremosesrJ   r,   Ú	normalizeÚpunc_normalizerÚImportErrorÚFileNotFoundErrorÚwarningsÚwarn)rF   rJ   s     r"   rB   z!MarianTokenizer._setup_normalizerª   s}   € ð	/Ø7Ð7Ð7Ð7Ð7Ð7à#7Ð#7¸Ô8HÑ#IÔ#IÔ#SˆDÔ Ð Ð øÝÕ.Ð/ð 	/ð 	/ð 	/ÝŒMÐ@ÑAÔAÐAØ#. ;ˆDÔ Ð Ð Ð ð	/øøøs   ‚ $ ¤-AÁArM   c                 ó4   — |r|                       |¦  «        ndS )zHCover moses empty string edge case. They return empty list for '' input!Ú )rQ   )rF   rM   s     r"   rP   zMarianTokenizer.normalize³   s    € à*+Ð3ˆt×#Ò# AÑ&Ô&Ð&°Ð3r$   c                 óR   — || j         v r| j         |         S | j         | j                 S rL   )rA   r.   )rF   Útokens     r"   Ú_convert_token_to_idz$MarianTokenizer._convert_token_to_id·   s0   € Ø�DÔ(Ð(Ð(ØÔ'¨Ô.Ð.ð Ô# D¤NÔ3Ð3r$   Útextc                 óÈ   — g }|                      d¦  «        rH|                     d¦  «        x}dk    r-|                     |d|dz   …         ¦  «         ||dz   d…         }||fS )z6Remove language codes like >>fr<< before sentencepiecer'   r(   éÿÿÿÿNé   )r)   ÚfindÚappend)rF   r[   ÚcodeÚend_locs       r"   Úremove_language_codez$MarianTokenizer.remove_language_code¾   so   € àˆØ�?Š?˜4Ñ Ô ð 	'°·²¸4±´Ð&@ gÀRÒ%GÐ%GØ�KŠK˜˜]˜w¨™{˜]Ô+Ñ,Ô,Ð,Ø˜ !™˜˜Ô&ˆDØ�TˆzÐr$   c                 ó~   — |                       |¦  «        \  }}| j                             |t          ¬¦  «        }||z   S )N)Úout_type)rc   r@   Úencoder6   )rF   r[   ra   Úpiecess       r"   Ú	_tokenizezMarianTokenizer._tokenizeÆ   s>   € Ø×.Ò.¨tÑ4Ô4‰
ˆˆdØÔ!×(Ò(¨½Ð(Ñ<Ô<ˆØ�f‰}Ðr$   Úindexc                 ó˜   — || j         v r| j         |         S | j        r| j        n| j        }|                     |¦  «        }|r|n| j        S )z?Converts an index (integer) in a token (str) using the decoder.)r:   rC   r>   r?   Ú	IdToPiecer.   )rF   ri   Ú	spm_modelÚpieces       r"   Ú_convert_id_to_tokenz$MarianTokenizer._convert_id_to_tokenË   sU   € à�D”LÐ Ð Ø”< Ô&Ð&à'+Ô'HÐ]�D”O�OÈdÌoˆ	Ø×#Ò# EÑ*Ô*ˆØÐ1ˆuˆu 4¤>Ð1r$   c                 ó8   •—  t          ¦   «         j        |fi |¤ŽS )ad  
        Convert a list of lists of token ids into a list of strings by calling decode.

        Args:
            sequences (`Union[list[int], list[list[int]], np.ndarray, torch.Tensor]`):
                List of tokenized input ids. Can be obtained using the `__call__` method.
            skip_special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not to remove special tokens in the decoding.
            clean_up_tokenization_spaces (`bool`, *optional*):
                Whether or not to clean up the tokenization spaces. If `None`, will default to
                `self.clean_up_tokenization_spaces` (available in the `tokenizer_config`).
            use_source_tokenizer (`bool`, *optional*, defaults to `False`):
                Whether or not to use the source tokenizer to decode sequences (only applicable in sequence-to-sequence
                problems).
            kwargs (additional keyword arguments, *optional*):
                Will be passed to the underlying model specific decode method.

        Returns:
            `list[str]`: The list of decoded sentences.
        )rD   Úbatch_decode)rF   Ú	sequencesrG   rH   s      €r"   rp   zMarianTokenizer.batch_decodeÔ   s$   ø€ ð* $�u‰wŒwÔ# IÐ8Ð8°Ð8Ð8Ð8r$   c                 ó8   •—  t          ¦   «         j        |fi |¤ŽS )a÷  
        Converts a sequence of ids in a string, using the tokenizer and vocabulary with options to remove special
        tokens and clean up tokenization spaces.

        Similar to doing `self.convert_tokens_to_string(self.convert_ids_to_tokens(token_ids))`.

        Args:
            token_ids (`Union[int, list[int], np.ndarray, torch.Tensor]`):
                List of tokenized input ids. Can be obtained using the `__call__` method.
            skip_special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not to remove special tokens in the decoding.
            clean_up_tokenization_spaces (`bool`, *optional*):
                Whether or not to clean up the tokenization spaces. If `None`, will default to
                `self.clean_up_tokenization_spaces` (available in the `tokenizer_config`).
            use_source_tokenizer (`bool`, *optional*, defaults to `False`):
                Whether or not to use the source tokenizer to decode sequences (only applicable in sequence-to-sequence
                problems).
            kwargs (additional keyword arguments, *optional*):
                Will be passed to the underlying model specific decode method.

        Returns:
            `str`: The decoded sentence.
        )rD   Údecode)rF   Ú	token_idsrG   rH   s      €r"   rs   zMarianTokenizer.decodeë   s#   ø€ ð0 �u‰wŒwŒ~˜iÐ2Ð2¨6Ð2Ð2Ð2r$   Úskip_special_tokensÚclean_up_tokenization_spacesc                 ó„   •— | j          }|                     d|¦  «        | _         t          ¦   «         j        d|||dœ|¤ŽS )zCInternal decode method that handles use_source_tokenizer parameter.Úuse_source_tokenizer)rt   ru   rv   r   )r2   ÚpoprC   rD   Ú_decode)rF   rt   ru   rv   rG   Údefault_use_sourcerH   s         €r"   rz   zMarianTokenizer._decode  s`   ø€ ð "&Ô!5Ð5ÐØ,2¯JªJÐ7MÐOaÑ,bÔ,bˆÔ)Ø�u‰wŒwŒð 
ØØ 3Ø)Eð
ð 
ð ð	
ð 
ð 	
r$   Útokensc                 óJ  — | j         r| j        n| j        }g }d}|D ]A}|| j        v r!||                     |¦  «        |z   dz   z  }g }Œ,|                     |¦  «         ŒB||                     |¦  «        z  }|                     t          d¦  «        }|                     ¦   «         S )zQUses source spm if _decode_use_source_tokenizer is True, and target spm otherwiserW   ú )	rC   r>   r?   Úall_special_tokensÚdecode_piecesr`   ÚreplaceÚSPIECE_UNDERLINEÚstrip)rF   r|   Úsp_modelÚcurrent_sub_tokensÚ
out_stringrY   s         r"   Úconvert_tokens_to_stringz(MarianTokenizer.convert_tokens_to_string  sÄ   € à&*Ô&GÐ\�4”?�?ÈTÌ_ˆØÐØˆ
Øð 	1ð 	1ˆEà˜Ô/Ð/Ð/Ø˜h×4Ò4Ð5GÑHÔHÈ5ÑPÐSVÑVÑV�
Ø%'Ð"Ð"à"×)Ò)¨%Ñ0Ô0Ð0Ð0Ø�h×,Ò,Ð-?Ñ@Ô@Ñ@ˆ
Ø×'Ò'Õ(8¸#Ñ>Ô>ˆ
Ø×ÒÑ!Ô!Ð!r$   c                 ó8   — |€|| j         gz   S ||z   | j         gz   S )z=Build model inputs from a sequence by appending eos_token_id.)Úeos_token_id)rF   Útoken_ids_0Útoken_ids_1s      r"   Ú build_inputs_with_special_tokensz0MarianTokenizer.build_inputs_with_special_tokens&  s/   € àÐØ $Ô"3Ð!4Ñ4Ð4à˜[Ñ(¨DÔ,=Ð+>Ñ>Ð>r$   c                 ó6   — | j         | _        | j        | _        d S rL   )r>   r@   r5   rA   ©rF   s    r"   Ú_switch_to_input_modez%MarianTokenizer._switch_to_input_mode-  s   € Øœ?ˆÔØ#œ|ˆÔÐÐr$   c                 óH   — | j         | _        | j        r| j        | _        d S d S rL   )r?   r@   r2   r8   rA   rŽ   s    r"   Ú_switch_to_target_modez&MarianTokenizer._switch_to_target_mode1  s2   € Øœ?ˆÔØÔð 	7Ø#'Ô#6ˆDÔ Ð Ð ð	7ð 	7r$   c                 ó*   — t          | j        ¦  «        S rL   )Úlenr5   rŽ   s    r"   Ú
vocab_sizezMarianTokenizer.vocab_size6  s   € å�4”<Ñ Ô Ð r$   Úsave_directoryÚfilename_prefixc                 óÚ  — t           j                             |¦  «        s t                               d|› d�¦  «         d S g }| j        r¿t           j                             ||r|dz   ndt          d         z   ¦  «        }t           j                             ||r|dz   ndt          d         z   ¦  «        }t          | j	        |¦  «         t          | j
        |¦  «         |                     |¦  «         |                     |¦  «         n_t           j                             ||r|dz   ndt          d         z   ¦  «        }t          | j	        |¦  «         |                     |¦  «         t          t          d         t          d         g| j        | j        | j        g¦  «        D �];\  }}}	t           j                             ||r|dz   nd|z   ¦  «        }
t           j                             |¦  «        t           j                             |
¦  «        k    rEt           j                             |¦  «        r&t%          ||
¦  «         |                     |
¦  «         Œ¶t           j                             |¦  «        sft'          |
d	¦  «        5 }|	                     ¦   «         }|                     |¦  «         d d d ¦  «         n# 1 swxY w Y   |                     |
¦  «         �Œ=t-          |¦  «        S )
NzVocabulary path (z) should be a directoryú-rW   r   r   r
   r   Úwb)ÚosÚpathÚisdirÚloggerÚerrorr2   ÚjoinÚVOCAB_FILES_NAMESÚ	save_jsonr5   r8   r`   Úzipr<   r>   r?   ÚabspathÚisfiler   ÚopenÚserialized_model_protoÚwriteÚtuple)rF   r•   r–   Úsaved_filesÚout_src_vocab_fileÚout_tgt_vocab_fileÚout_vocab_fileÚspm_save_filenameÚspm_orig_pathrl   Úspm_save_pathÚfiÚcontent_spiece_models                r"   Úsave_vocabularyzMarianTokenizer.save_vocabulary:  s  € ÝŒw�}Š}˜^Ñ,Ô,ð 	Ý�LŠLÐT¨^ÐTÐTÐTÑUÔUÐUØˆFØˆàÔð 	/Ý!#¤§¢ØØ*9ÐA� 3Ñ&Ð&¸rÕEVÐW^ÔE_Ñ_ñ"ô "Ðõ "$¤§¢ØØ*9ÐA� 3Ñ&Ð&¸rÕEVÐWjÔEkÑkñ"ô "Ðõ �d”lÐ$6Ñ7Ô7Ð7Ý�dÔ)Ð+=Ñ>Ô>Ð>Ø×ÒÐ1Ñ2Ô2Ð2Ø×ÒÐ1Ñ2Ô2Ð2Ð2åœWŸ\š\Ø¸/Ð!Q °3Ñ!6Ð!6ÈrÕUfÐgnÔUoÑ oñô ˆNõ �d”l NÑ3Ô3Ð3Ø×Ò˜~Ñ.Ô.Ð.å;>Ý˜|Ô,Õ.?ÀÔ.MÐNØŒNØŒ_˜dœoÐ.ñ<
ô <
ð 	2ñ 	2Ñ7Ð˜}¨iõ
 œGŸLšLØ¸/Ð!Q °3Ñ!6Ð!6ÈrÐUfÑ fñô ˆMõ Œw�Š˜}Ñ-Ô-µ´·²ÀÑ1OÔ1OÒOÐOÕTVÔT[×TbÒTbÐcpÑTqÔTqÐOÝ˜¨Ñ6Ô6Ð6Ø×"Ò" =Ñ1Ô1Ð1Ð1Ý”W—^’^ MÑ2Ô2ð 2Ý˜-¨Ñ.Ô.ð 3°"Ø+4×+KÒ+KÑ+MÔ+MÐ(Ø—H’HÐ1Ñ2Ô2Ð2ð3ð 3ð 3ñ 3ô 3ð 3ð 3ð 3ð 3ð 3ð 3øøøð 3ð 3ð 3ð 3ð ×"Ò" =Ñ1Ô1Ð1ùå�[Ñ!Ô!Ð!s   Ê*J<Ê<K 	ËK 	c                 ó*   — |                       ¦   «         S rL   )Úget_src_vocabrŽ   s    r"   Ú	get_vocabzMarianTokenizer.get_vocabg  s   € Ø×!Ò!Ñ#Ô#Ð#r$   c                 ó0   — t          | j        fi | j        ¤ŽS rL   )Údictr5   Úadded_tokens_encoderrŽ   s    r"   r´   zMarianTokenizer.get_src_vocabj  s   € Ý�D”LÐ>Ð> DÔ$=Ð>Ð>Ð>r$   c                 ó0   — t          | j        fi | j        ¤ŽS rL   )r·   r8   Úadded_tokens_decoderrŽ   s    r"   Úget_tgt_vocabzMarianTokenizer.get_tgt_vocabm  s   € Ý�DÔ'ÐEÐE¨4Ô+DÐEÐEÐEr$   c                 ó–   — | j                              ¦   «         }|                     t                               g d¢¦  «        ¦  «         |S )N)r>   r?   r@   rQ   r   )Ú__dict__ÚcopyÚupdater·   Úfromkeys)rF   Ústates     r"   Ú__getstate__zMarianTokenizer.__getstate__p  sH   € Ø”×"Ò"Ñ$Ô$ˆØ�ŠÝ�MŠMÐmÐmÐmÑnÔnñ	
ô 	
ð 	
ð ˆr$   Údc                 óò   ‡ — |‰ _         t          ‰ d¦  «        si ‰ _        t          ‰ d¦  «        sd‰ _        ˆ fd„‰ j        D ¦   «         \  ‰ _        ‰ _        ‰ j        ‰ _        ‰                      ¦   «          d S )Nr   rC   Fc              3   óB   •K  — | ]}t          |‰j        ¦  «        V — Œd S rL   )r=   r   )r   ÚfrF   s     €r"   ú	<genexpr>z/MarianTokenizer.__setstate__.<locals>.<genexpr>€  s1   øè è € Ð+fÐ+fÐRS­H°Q¸Ô8LÑ,MÔ,MÐ+fÐ+fÐ+fÐ+fÐ+fÐ+fr$   )	r½   Úhasattrr   rC   r<   r>   r?   r@   rB   )rF   rÃ   s   ` r"   Ú__setstate__zMarianTokenizer.__setstate__w  sŠ   ø€ ØˆŒõ �tÐ.Ñ/Ô/ð 	&Ø#%ˆDÔ Ý�tÐ;Ñ<Ô<ð 	6Ø05ˆDÔ-à+fÐ+fÐ+fÐ+fÐW[ÔWeÐ+fÑ+fÔ+fÑ(ˆŒ˜œØœ?ˆÔØ×ÒÑ Ô Ð Ð Ð r$   c                 ó   — dS )zJust EOSé   r   )rF   ÚargsrG   s      r"   Únum_special_tokens_to_addz)MarianTokenizer.num_special_tokens_to_add„  s   € àˆqr$   c                 ó|   ‡— t          | j        ¦  «        Š‰                     | j        ¦  «         ˆfd„|D ¦   «         S )Nc                 ó    •— g | ]
}|‰v rd nd‘ŒS )rË   r   r   )r   rM   Úall_special_idss     €r"   r+   z7MarianTokenizer._special_token_mask.<locals>.<listcomp>‹  s'   ø€ Ð>Ð>Ð>°Q�Q˜/Ð)Ð)��¨qÐ>Ð>Ð>r$   )ÚsetrÐ   ÚremoveÚunk_token_id)rF   ÚseqrÐ   s     @r"   Ú_special_token_maskz#MarianTokenizer._special_token_maskˆ  sD   ø€ Ý˜dÔ2Ñ3Ô3ˆØ×Ò˜tÔ0Ñ1Ô1Ð1Ø>Ð>Ð>Ð>¸#Ð>Ñ>Ô>Ð>r$   rŠ   r‹   Úalready_has_special_tokensc                 óž   — |r|                       |¦  «        S |€|                       |¦  «        dgz   S |                       ||z   ¦  «        dgz   S )zCGet list where entries are [1] if a token is [eos] or [pad] else 0.NrË   )rÕ   )rF   rŠ   r‹   rÖ   s       r"   Úget_special_tokens_maskz'MarianTokenizer.get_special_tokens_mask�  sb   € ð &ð 	MØ×+Ò+¨KÑ8Ô8Ð8ØÐ Ø×+Ò+¨KÑ8Ô8¸A¸3Ñ>Ð>à×+Ò+¨K¸+Ñ,EÑFÔFÈ!ÈÑLÐLr$   )	NNNr   r   r   r   NF)FNrL   )NF)(Ú__name__Ú
__module__Ú__qualname__Ú__doc__r    Úvocab_files_namesÚmodel_input_namesr·   r6   r   rE   rB   rP   rZ   rc   Úlistrh   Úintrn   rp   rs   Úboolrz   r‡   rŒ   r�   r‘   Úpropertyr”   r¨   r²   rµ   r´   r»   rÂ   rÉ   rÍ   rÕ   rØ   Ú__classcell__)rH   s   @r"   r   r   ,   sz  ø€ € € € € ð8ð 8ðt *ÐØ$Ð&6Ð7Ðð ØØØØØØØ15Øð=
ð =
ð ˜c 3˜hœ¨$Ñ.ð=
ð 
ð=
ð =
ð =
ð =
ð =
ð =
ð~/ð /ð /ð4˜3ð 4 3ð 4ð 4ð 4ð 4ð4ð 4ð 4ð¨ð ð ð ð ð˜cð  d¨3¤ið ð ð ð ð
2¨#ð 2°#ð 2ð 2ð 2ð 2ð9ð 9ð 9ð 9ð 9ð.3ð 3ð 3ð 3ð 3ð: %*Ø48ð	
ð 
ð "ð
ð '+¨T¡kð	
ð 
ð
ð 
ð 
ð 
ð 
ð 
ð""¨t°C¬yð "¸Sð "ð "ð "ð "ð ?ð ?ÐQUÐVYÔQZð ?ð ?ð ?ð ?ð,ð ,ð ,ð7ð 7ð 7ð
 ð!˜Cð !ð !ð !ñ „Xð!ð+"ð +"¨cð +"ÀCÈ$ÁJð +"ÐZ_Ð`cÔZdð +"ð +"ð +"ð +"ðZ$˜4ð $ð $ð $ð $ð?ð ?ð ?ðFð Fð Fð˜dð ð ð ð ð!˜dð ! tð !ð !ð !ð !ðð ð ð?ð ?ð ?ð fkð	Mð 	MØð	MØ.2°T©kð	MØ^bð	Mà	ˆcŒð	Mð 	Mð 	Mð 	Mð 	Mð 	Mð 	Mð 	Mr$   r   r›   r   r   c                 óR   — t          j        di |¤Ž}|                     | ¦  «         |S )Nr   )r   ÚSentencePieceProcessorÚLoad)r›   r   Úspms      r"   r=   r=   ™  s,   € Ý
Ô
.Ð
AÐ
A°Ð
AÐ
A€CØ‡H‚HˆT�N„N€NØ€Jr$   c                 ó†   — t          |d¦  «        5 }t          j        | |d¬¦  «         d d d ¦  «         d S # 1 swxY w Y   d S )NÚwr^   )Úindent)r¥   ÚjsonÚdump)Údatar›   rÆ   s      r"   r¡   r¡   Ÿ  sˆ   € Ý	ˆd�C‰Œð %˜AÝŒ	�$˜ !Ð$Ñ$Ô$Ð$ð%ð %ð %ñ %ô %ð %ð %ð %ð %ð %ð %ð %øøøð %ð %ð %ð %ð %ð %s   ‘6¶:½:c                 ó~   — t          | d¦  «        5 }t          j        |¦  «        cd d d ¦  «         S # 1 swxY w Y   d S )NÚr)r¥   rë   Úload)r›   rÆ   s     r"   r4   r4   ¤  s|   € Ý	ˆd�C‰Œð ˜AÝŒy˜‰|Œ|ðð ð ð ñ ô ð ð ð ð ð ð øøøð ð ð ð ð ð s   ‘2²6¹6)rë   rš   rT   Úpathlibr   Úshutilr   Útypingr   r   Útokenization_pythonr   Úutilsr   Úutils.import_utilsr	   Ú
get_loggerrÙ   r�   r    r‚   r   r6   r·   rå   r=   r¡   rß   r4   Ú__all__r   r$   r"   ú<module>rù      sµ  ðð €€€Ø 	€	€	€	Ø €€€Ø Ð Ð Ð Ð Ð Ø Ð Ð Ð Ð Ð Ø Ð Ð Ð Ð Ð à Ð Ð Ð à 6Ð 6Ð 6Ð 6Ð 6Ð 6Ø Ð Ð Ð Ð Ð Ø *Ð *Ð *Ð *Ð *Ð *ð 
ˆÔ	˜HÑ	%Ô	%€ð ØØØ,Ø4ðð Ð ð Ð ð
 
€Ð%Ð&Ñ&Ô&ðiMð iMð iMð iMð iMÐ)ñ iMô iMñ 'Ô&ðiMðX�3ð ¨¨c°3¨h¬ð ¸MÔ<`ð ð ð ð ð%˜#ð % $ð %ð %ð %ð %ð
�Cð ˜D 4™Kð ð ð ð ð
 Ð
€€€r$   