§
    ‚Štj   ã                   ó´   — d Z ddlmZ ddlmZ ddlmZ ddlmZ ddl	m
Z
  ej        e¦  «        Zd	d
iZ ed¬¦  «         G d„ de¦  «        ¦   «         ZdgZdS )z Tokenization class for SpeechT5.é    )ÚAnyé   )ÚSentencePieceBackend)Úlogging)Úrequiresé   )ÚEnglishNumberNormalizerÚ
vocab_filezspm_char.model)Úsentencepiece)Úbackendsc            
       óD  ‡ — e Zd ZdZeZddgZdZ	 	 	 	 	 	 dd
ee	e
f         d	z  dd	fˆ fd„Zdd„Zed„ ¦   «         Zej        d„ ¦   «         Zddee         fd„Z	 ddee         dee         d	z  dedee         fˆ fd„Z	 ddee         dee         d	z  dee         fd„Zˆ xZS )ÚSpeechT5Tokenizera	  
    Construct a SpeechT5 tokenizer. Based on [SentencePiece](https://github.com/google/sentencepiece).

    This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods.

    Args:
        vocab_file (`str`):
            [SentencePiece](https://github.com/google/sentencepiece) file (generally has a *.spm* extension) that
            contains the vocabulary necessary to instantiate a tokenizer.
        bos_token (`str`, *optional*, defaults to `"<s>"`):
            The begin of sequence token.
        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        normalize (`bool`, *optional*, defaults to `False`):
            Whether to convert numeric quantities in the text to their spelt-out english counterparts.
        sp_model_kwargs (`dict`, *optional*):
            Will be passed to the `SentencePieceProcessor.__init__()` method. The [Python wrapper for
            SentencePiece](https://github.com/google/sentencepiece/tree/master/python) can be used, among other things,
            to set:

            - `enable_sampling`: Enable subword regularization.
            - `nbest_size`: Sampling parameters for unigram. Invalid for BPE-Dropout.

              - `nbest_size = {0,1}`: No sampling is performed.
              - `nbest_size > 1`: samples from the nbest_size results.
              - `nbest_size < 0`: assuming that nbest_size is infinite and samples from the all hypothesis (lattice)
                using forward-filtering-and-backward-sampling algorithm.

            - `alpha`: Smoothing parameter for unigram sampling, and dropout probability of merge operations for
              BPE-dropout.

    Attributes:
        sp_model (`SentencePieceProcessor`):
            The *SentencePiece* processor that is used for every conversion (string, tokens and IDs).
    Ú	input_idsÚattention_maskFú<s>ú</s>ú<unk>ú<pad>NÚsp_model_kwargsÚreturnc           
      ór   •— || _         d | _        |�||d<    t          ¦   «         j        d||||||dœ|¤Ž d S )Nr   )r
   Ú	bos_tokenÚ	eos_tokenÚ	unk_tokenÚ	pad_tokenÚ	normalize© )r   Ú_normalizerÚsuperÚ__init__)
Úselfr
   r   r   r   r   r   r   ÚkwargsÚ	__class__s
            €úp/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/speecht5/tokenization_speecht5.pyr    zSpeechT5Tokenizer.__init__M   st   ø€ ð #ˆŒØˆÔð Ð&Ø(7ˆFÐ$Ñ%ð 	�‰ŒÔð 	
Ø!ØØØØØð	
ð 	
ð ð	
ð 	
ð 	
ð 	
ð 	
ó    c                 ó|   — |                      d| j        ¦  «        }|rd|z   }|r|                      |¦  «        }||fS )Nr   ú )Úpopr   Ú
normalizer)r!   ÚtextÚis_split_into_wordsr"   r   s        r$   Úprepare_for_tokenizationz*SpeechT5Tokenizer.prepare_for_tokenizationj   sK   € Ø—J’J˜{¨D¬NÑ;Ô;ˆ	Øð 	Ø˜‘:ˆDØð 	)Ø—?’? 4Ñ(Ô(ˆDØ�fˆ~Ðr%   c                 óD   — | j         €t          ¦   «         | _         | j         S ©N)r   r	   )r!   s    r$   r)   zSpeechT5Tokenizer.normalizerr   s"   € àÔÐ#Ý6Ñ8Ô8ˆDÔØÔÐr%   c                 ó   — || _         d S r.   )r   )r!   Úvalues     r$   r)   zSpeechT5Tokenizer.normalizerx   s   € à ˆÔÐÐr%   c                 ó8   — |€|| j         gz   S ||z   | j         gz   S )z=Build model inputs from a sequence by appending eos_token_id.)Úeos_token_id)r!   Útoken_ids_0Útoken_ids_1s      r$   Ú build_inputs_with_special_tokensz2SpeechT5Tokenizer.build_inputs_with_special_tokens|   s/   € àÐØ $Ô"3Ð!4Ñ4Ð4à˜[Ñ(¨DÔ,=Ð+>Ñ>Ð>r%   r3   r4   Úalready_has_special_tokensc                 óÚ   •— |r$t          ¦   «                              ||d¬¦  «        S dg}|€dgt          |¦  «        z  |z   S dgt          |¦  «        z  dgt          |¦  «        z  z   |z   S )NT)r3   r4   r6   r   r   )r   Úget_special_tokens_maskÚlen)r!   r3   r4   r6   Úsuffix_onesr#   s        €r$   r8   z)SpeechT5Tokenizer.get_special_tokens_maskƒ   s�   ø€ ð &ð 	Ý‘7”7×2Ò2Ø'°[Ð]að 3ñ ô ð ð �cˆØÐØ�C�#˜kÑ*Ô*Ñ*¨kÑ9Ð9Ø�•c˜+Ñ&Ô&Ñ&¨A¨3µ°[Ñ1AÔ1AÑ+AÑBÀ[ÑPÐPr%   c                 ót   — | j         g}|€t          ||z   ¦  «        dgz  S t          ||z   |z   ¦  «        dgz  S )aÍ  
        Create a mask from the two sequences passed to be used in a sequence-pair classification task. SpeechT5 does not
        make use of token type ids, therefore a list of zeros is returned.

        Args:
            token_ids_0 (`list[int]`):
                List of IDs.
            token_ids_1 (`list[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `list[int]`: List of zeros.
        Nr   )r2   r9   )r!   r3   r4   Úeoss       r$   Ú$create_token_type_ids_from_sequencesz6SpeechT5Tokenizer.create_token_type_ids_from_sequences�   sN   € ð  Ô Ð!ˆØÐÝ�{ SÑ(Ñ)Ô)¨Q¨CÑ/Ð/Ý�; Ñ,¨sÑ2Ñ3Ô3°q°cÑ9Ð9r%   )r   r   r   r   FN)Fr.   )NF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚVOCAB_FILES_NAMESÚvocab_files_namesÚmodel_input_namesÚis_fastÚdictÚstrr   r    r,   Úpropertyr)   ÚsetterÚlistÚintr5   Úboolr8   r=   Ú__classcell__)r#   s   @r$   r   r      s¿  ø€ € € € € ð(ð (ðT *ÐØ$Ð&6Ð7ÐØ€Gð
 ØØØØØ15ð
ð 
ð ˜c 3˜hœ¨$Ñ.ð
ð 
ð
ð 
ð 
ð 
ð 
ð 
ð:ð ð ð ð ð ð  ñ „Xð ð
 Ôð!ð !ñ Ôð!ð?ð ?ÐQUÐVYÔQZð ?ð ?ð ?ð ?ð puðQð QØ œ9ðQØ37¸´9¸tÑ3CðQØhlðQà	ˆcŒðQð Qð Qð Qð Qð Qð GKð:ð :Ø œ9ð:Ø37¸´9¸tÑ3Cð:à	ˆcŒð:ð :ð :ð :ð :ð :ð :ð :r%   r   N)rA   Útypingr   Ú tokenization_utils_sentencepiecer   Úutilsr   Úutils.import_utilsr   Únumber_normalizerr	   Ú
get_loggerr>   ÚloggerrB   r   Ú__all__r   r%   r$   ú<module>rV      sä   ðð 'Ð &à Ð Ð Ð Ð Ð à DÐ DÐ DÐ DÐ DÐ DØ Ð Ð Ð Ð Ð Ø *Ð *Ð *Ð *Ð *Ð *Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6ð 
ˆÔ	˜HÑ	%Ô	%€à!Ð#3Ð4Ð ð 
€Ð%Ð&Ñ&Ô&ðE:ð E:ð E:ð E:ð E:Ð,ñ E:ô E:ñ 'Ô&ðE:ðP Ð
€€€r%   