§
    ‚Štjp;  ã                   óæ   — d Z ddlZddlZddlmZ ddlmZ ddlmZ  ed¬¦  «        e G d	„ d
e¦  «        ¦   «         ¦   «         Z	 ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z
d
dgZdS )zSpeechT5 model configurationé    N)Ústricté   )ÚPreTrainedConfig)Úauto_docstringzmicrosoft/speecht5_asr)Ú
checkpointc                   ó<  ‡ — e Zd ZU dZdZdddœZdZeed<   dZ	eed	<   d
Z
eed<   d
Zeed<   dZeed<   dZeez  ed<   dZeed<   dZeed<   d
Zeed<   dZeez  ed<   dZeed<   dZeez  ed<   dZeez  ed<   dZeez  ed<   dZeez  ed<   dZeed<   dZeed<   dZeed<   d Zeed!<   d"Zeez  ed#<   dZeed$<   d%Z e!e         e"ed&f         z  ed'<   d(Z#e!e         e"ed&f         z  ed)<   d*Z$e!e         e"ed&f         z  ed+<   dZ%eed,<   d-Z&eed.<   d/Z'eed0<   d1Z(eed2<   d3Z)eez  ed4<   d5Z*eed6<   d7Z+eed8<   d"Z,eez  ed9<   d5Z-eed:<   d;Z.eed<<   d=Z/ed>z  ed?<   d;Z0ed>z  ed@<   d7Z1ee!e         z  d>z  edA<   d7Z2ed>z  edB<   dCZ3eedD<   d7Z4eedE<   dFZ5eedG<   dHZ6eez  edI<   dJZ7eedK<   dLZ8eedM<   dFZ9eedN<   dLZ:eedO<   dHZ;eez  edP<   d7Z<eedQ<   dRZ=eedS<   dTZ>eedU<   dVZ?eedW<   d1Z@eedX<   d7ZAeedY<   dZZBeed[<   d\ZCeed]<   d1ZDeed^<   d1ZEeed_<   d1ZFeed`<   ˆ fda„ZGdb„ ZHdc„ ZIˆ xZJS )dÚSpeechT5Configat  
    positional_dropout (`float`, *optional*, defaults to 0.1):
        The dropout probability for the text position encoding layers.
    feat_extract_norm (`str`, *optional*, defaults to `"group"`):
        The norm to be applied to 1D convolutional layers in the speech encoder pre-net. One of `"group"` for group
        normalization of only the first 1D convolutional layer or `"layer"` for layer normalization of all 1D
        convolutional layers.
    feat_proj_dropout (`float`, *optional*, defaults to 0.0):
        The dropout probability for output of the speech encoder pre-net.
    feat_extract_activation (`str, `optional`, defaults to `"gelu"`):
        The non-linear activation function (function or string) in the 1D convolutional layers of the feature
        extractor. If string, `"gelu"`, `"relu"`, `"selu"` and `"gelu_new"` are supported.
    conv_dim (`tuple[int]` or `list[int]`, *optional*, defaults to `(512, 512, 512, 512, 512, 512, 512)`):
        A tuple of integers defining the number of input and output channels of each 1D convolutional layer in the
        speech encoder pre-net. The length of *conv_dim* defines the number of 1D convolutional layers.
    conv_stride (`tuple[int]` or `list[int]`, *optional*, defaults to `(5, 2, 2, 2, 2, 2, 2)`):
        A tuple of integers defining the stride of each 1D convolutional layer in the speech encoder pre-net. The
        length of *conv_stride* defines the number of convolutional layers and has to match the length of
        *conv_dim*.
    conv_kernel (`tuple[int]` or `list[int]`, *optional*, defaults to `(10, 3, 3, 3, 3, 3, 3)`):
        A tuple of integers defining the kernel size of each 1D convolutional layer in the speech encoder pre-net.
        The length of *conv_kernel* defines the number of convolutional layers and has to match the length of
        *conv_dim*.
    conv_bias (`bool`, *optional*, defaults to `False`):
        Whether the 1D convolutional layers have a bias.
    num_conv_pos_embeddings (`int`, *optional*, defaults to 128):
        Number of convolutional positional embeddings. Defines the kernel size of 1D convolutional positional
        embeddings layer.
    num_conv_pos_embedding_groups (`int`, *optional*, defaults to 16):
        Number of groups of 1D convolutional positional embeddings layer.
    apply_spec_augment (`bool`, *optional*, defaults to `True`):
        Whether to apply *SpecAugment* data augmentation to the outputs of the speech encoder pre-net. For
        reference see [SpecAugment: A Simple Data Augmentation Method for Automatic Speech
        Recognition](https://huggingface.co/papers/1904.08779).
    mask_time_prob (`float`, *optional*, defaults to 0.05):
        Percentage (between 0 and 1) of all feature vectors along the time axis which will be masked. The masking
        procedure generates ''mask_time_prob*len(time_axis)/mask_time_length'' independent masks over the axis. If
        reasoning from the probability of each feature vector to be chosen as the start of the vector span to be
        masked, *mask_time_prob* should be `prob_vector_start*mask_time_length`. Note that overlap may decrease the
        actual percentage of masked vectors. This is only relevant if `apply_spec_augment is True`.
    mask_time_length (`int`, *optional*, defaults to 10):
        Length of vector span along the time axis.
    mask_time_min_masks (`int`, *optional*, defaults to 2),:
        The minimum number of masks of length `mask_feature_length` generated along the time axis, each time step,
        irrespectively of `mask_feature_prob`. Only relevant if ''mask_time_prob*len(time_axis)/mask_time_length <
        mask_time_min_masks''
    mask_feature_prob (`float`, *optional*, defaults to 0.0):
        Percentage (between 0 and 1) of all feature vectors along the feature axis which will be masked. The
        masking procedure generates ''mask_feature_prob*len(feature_axis)/mask_time_length'' independent masks over
        the axis. If reasoning from the probability of each feature vector to be chosen as the start of the vector
        span to be masked, *mask_feature_prob* should be `prob_vector_start*mask_feature_length`. Note that overlap
        may decrease the actual percentage of masked vectors. This is only relevant if `apply_spec_augment is
        True`.
    mask_feature_length (`int`, *optional*, defaults to 10):
        Length of vector span along the feature axis.
    mask_feature_min_masks (`int`, *optional*, defaults to 0),:
        The minimum number of masks of length `mask_feature_length` generated along the feature axis, each time
        step, irrespectively of `mask_feature_prob`. Only relevant if
        ''mask_feature_prob*len(feature_axis)/mask_feature_length < mask_feature_min_masks''
    num_mel_bins (`int`, *optional*, defaults to 80):
        Number of mel features used per input features. Used by the speech decoder pre-net. Should correspond to
        the value used in the [`SpeechT5Processor`] class.
    speech_decoder_prenet_layers (`int`, *optional*, defaults to 2):
        Number of layers in the speech decoder pre-net.
    speech_decoder_prenet_units (`int`, *optional*, defaults to 256):
        Dimensionality of the layers in the speech decoder pre-net.
    speech_decoder_prenet_dropout (`float`, *optional*, defaults to 0.5):
        The dropout probability for the speech decoder pre-net layers.
    speaker_embedding_dim (`int`, *optional*, defaults to 512):
        Dimensionality of the *XVector* embedding vectors.
    speech_decoder_postnet_layers (`int`, *optional*, defaults to 5):
        Number of layers in the speech decoder post-net.
    speech_decoder_postnet_units (`int`, *optional*, defaults to 256):
        Dimensionality of the layers in the speech decoder post-net.
    speech_decoder_postnet_kernel (`int`, *optional*, defaults to 5):
        Number of convolutional filter channels in the speech decoder post-net.
    speech_decoder_postnet_dropout (`float`, *optional*, defaults to 0.5):
        The dropout probability for the speech decoder post-net layers.
    reduction_factor (`int`, *optional*, defaults to 2):
        Spectrogram length reduction factor for the speech decoder inputs.
    max_speech_positions (`int`, *optional*, defaults to 4000):
        The maximum sequence length of speech features that this model might ever be used with.
    max_text_positions (`int`, *optional*, defaults to 450):
        The maximum sequence length of text features that this model might ever be used with.
    encoder_max_relative_position (`int`, *optional*, defaults to 160):
        Maximum distance for relative position embedding in the encoder.
    use_guided_attention_loss (`bool`, *optional*, defaults to `True`):
        Whether to apply guided attention loss while training the TTS model.
    guided_attention_loss_num_heads (`int`, *optional*, defaults to 2):
        Number of attention heads the guided attention loss will be applied to. Use -1 to apply this loss to all
        attention heads.
    guided_attention_loss_sigma (`float`, *optional*, defaults to 0.4):
        Standard deviation for guided attention loss.
    guided_attention_loss_scale (`float`, *optional*, defaults to 10.0):
        Scaling coefficient for guided attention loss (also known as lambda).

    Example:

    ```python
    >>> from transformers import SpeechT5Model, SpeechT5Config

    >>> # Initializing a "microsoft/speecht5_asr" style configuration
    >>> configuration = SpeechT5Config()

    >>> # Initializing a model (with random weights) from the "microsoft/speecht5_asr" style configuration
    >>> model = SpeechT5Model(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úspeecht5Úencoder_attention_headsÚencoder_layers)Únum_attention_headsÚnum_hidden_layerséQ   Ú
vocab_sizei   Úhidden_sizeé   i   Úencoder_ffn_dimçš™™™™™¹?Úencoder_layerdropé   Údecoder_layersÚdecoder_ffn_dimÚdecoder_attention_headsÚdecoder_layerdropÚgeluÚ
hidden_actÚpositional_dropoutÚhidden_dropoutÚattention_dropoutÚactivation_dropoutg{®Gáz”?Úinitializer_rangegñhãˆµøä>Úlayer_norm_epsFÚscale_embeddingÚgroupÚfeat_extract_normg        Úfeat_proj_dropoutÚfeat_extract_activation)é   r(   r(   r(   r(   r(   r(   .Úconv_dim)é   é   r+   r+   r+   r+   r+   Úconv_stride)é
   r   r   r   r   r+   r+   Úconv_kernelÚ	conv_biasé€   Únum_conv_pos_embeddingsé   Únum_conv_pos_embedding_groupsTÚapply_spec_augmentgš™™™™™©?Úmask_time_probr-   Úmask_time_lengthr+   Úmask_time_min_masksÚmask_feature_probÚmask_feature_lengthr   Úmask_feature_min_masksé   NÚpad_token_idÚbos_token_idÚeos_token_idÚdecoder_start_token_idéP   Únum_mel_binsÚspeech_decoder_prenet_layersé   Úspeech_decoder_prenet_unitsg      à?Úspeech_decoder_prenet_dropoutr(   Úspeaker_embedding_dimr*   Úspeech_decoder_postnet_layersÚspeech_decoder_postnet_unitsÚspeech_decoder_postnet_kernelÚspeech_decoder_postnet_dropoutÚreduction_factori   Úmax_speech_positionsiÂ  Úmax_text_positionsé    Úencoder_max_relative_positionÚuse_guided_attention_lossÚguided_attention_loss_num_headsgš™™™™™Ù?Úguided_attention_loss_sigmag      $@Úguided_attention_loss_scaleÚ	use_cacheÚis_encoder_decoderÚtie_word_embeddingsc                 ól   •— t          | j        ¦  «        | _         t          ¦   «         j        di |¤Ž d S )N© )Úlenr)   Únum_feat_extract_layersÚsuperÚ__post_init__)ÚselfÚkwargsÚ	__class__s     €úq/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/speecht5/configuration_speecht5.pyr\   zSpeechT5Config.__post_init__É   s8   ø€ Ý'*¨4¬=Ñ'9Ô'9ˆÔ$Ø�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'ó    c           
      óR  — t          | j        ¦  «        | j        k    s:t          | j        ¦  «        | j        k    st          | j        ¦  «        | j        k    rOt          dt          | j        ¦  «        › dt          | j        ¦  «        › dt          | j        ¦  «        › d�¦  «        ‚dS )zOPart of `@strict`-powered validation. Validates the architecture of the config.zºConfiguration for convolutional layers is incorrect. It is required that `len(config.conv_dim)` == `len(config.conv_stride)` == `len(config.conv_kernel)`, but is `len(config.conv_dim) = z`, `len(config.conv_stride) = z`, `len(config.conv_kernel) = z`.N)rY   r,   rZ   r.   r)   Ú
ValueError©r]   s    r`   Úvalidate_architecturez$SpeechT5Config.validate_architectureÍ   sÄ   € õ �Ô!Ñ"Ô" dÔ&BÒBÐBÝ�DÔ$Ñ%Ô%¨Ô)EÒEÐEÝ�D”MÑ"Ô" dÔ&BÒBÐBåðIå˜œÑ&Ô&ðIð IåFIÈ$ÔJZÑF[ÔF[ðIð Iõ 03°4Ô3CÑ/DÔ/DðIð Ið Iñô ð ð CÐBra   c                 óL   — t          j        t          j        | j        d¦  «        S )Nr;   )Ú	functoolsÚreduceÚoperatorÚmulr,   rd   s    r`   Úinputs_to_logits_ratioz%SpeechT5Config.inputs_to_logits_ratioÛ   s   € ÝÔ¥¤¨dÔ.>ÀÑBÔBÐBra   )KÚ__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚattribute_mapr   ÚintÚ__annotations__r   r   r   r   r   Úfloatr   r   r   r   r   Ústrr   r   r   r    r!   r"   r#   Úboolr%   r&   r'   r)   ÚlistÚtupler,   r.   r/   r1   r3   r4   r5   r6   r7   r8   r9   r:   r<   r=   r>   r?   rA   rB   rD   rE   rF   rG   rH   rI   rJ   rK   rL   rM   rO   rP   rQ   rR   rS   rT   rU   rV   r\   re   rk   Ú__classcell__)r_   s   @r`   r	   r	      s¿  ø€ € € € € € ðmð mð^ €JØ,EÐ\lÐmÐm€Mà€J�ÐÐÑØ€K�ÐÐÑØ€N�CÐÐÑØ#%Ð˜SÐ%Ð%Ñ%Ø€O�SÐÐÑØ%(Ð�u˜s‘{Ð(Ð(Ñ(Ø€N�CÐÐÑØ€O�SÐÐÑØ#%Ð˜SÐ%Ð%Ñ%Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø€J�ÐÐÑØ&)Ð˜ ™Ð)Ð)Ñ)Ø"%€N�E˜C‘KÐ%Ð%Ñ%Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø&)Ð˜ ™Ð)Ð)Ñ)Ø#Ð�uÐ#Ð#Ñ#Ø €N�EÐ Ð Ñ Ø!€O�TÐ!Ð!Ñ!Ø$Ð�sÐ$Ð$Ñ$Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø#)Ð˜SÐ)Ð)Ñ)Ø,O€Hˆd�3Œi˜%  S œ/Ñ)ÐOÐOÑOØ/D€K��c”˜U 3¨ 8œ_Ñ,ÐDÐDÑDØ/E€K��c”˜U 3¨ 8œ_Ñ,ÐEÐEÑEØ€IˆtÐÐÑØ#&Ð˜SÐ&Ð&Ñ&Ø)+Ð! 3Ð+Ð+Ñ+Ø#Ð˜Ð#Ð#Ñ#Ø"&€N�E˜C‘KÐ&Ð&Ñ&ØÐ�cÐÐÑØ Ð˜Ð Ð Ñ Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø!Ð˜Ð!Ð!Ñ!Ø"#Ð˜CÐ#Ð#Ñ#Ø €L�#˜‘*Ð Ð Ñ Ø €L�#˜‘*Ð Ð Ñ Ø+,€L�#˜˜Sœ	‘/ DÑ(Ð,Ð,Ñ,Ø)*Ð˜C $™JÐ*Ð*Ñ*Ø€L�#ÐÐÑØ()Ð  #Ð)Ð)Ñ)Ø'*Ð Ð*Ð*Ñ*Ø14Ð! 5¨3¡;Ð4Ð4Ñ4Ø!$Ð˜3Ð$Ð$Ñ$Ø)*Ð! 3Ð*Ð*Ñ*Ø(+Ð  #Ð+Ð+Ñ+Ø)*Ð! 3Ð*Ð*Ñ*Ø25Ð" E¨C¡KÐ5Ð5Ñ5ØÐ�cÐÐÑØ $Ð˜#Ð$Ð$Ñ$Ø!Ð˜Ð!Ð!Ñ!Ø),Ð! 3Ð,Ð,Ñ,Ø&*Ð˜tÐ*Ð*Ñ*Ø+,Ð# SÐ,Ð,Ñ,Ø),Ð Ð,Ð,Ñ,Ø)-Ð Ð-Ð-Ñ-Ø€IˆtÐÐÑØ#Ð˜Ð#Ð#Ñ#Ø $Ð˜Ð$Ð$Ñ$ð(ð (ð (ð (ð (ðð ð ðCð Cð Cð Cð Cð Cð Cra   r	   c                   ó  — e Zd ZU dZdZdZeed<   dZeed<   dZ	eed<   d	Z
ee         eed
f         z  ed<   dZee         eed
f         z  ed<   dZee         eed
f         z  ed<   dZeez  ed<   dZeed<   dZeed<   dZeed<   dS )ÚSpeechT5HifiGanConfigaà  
    model_in_dim (`int`, *optional*, defaults to 80):
        The number of frequency bins in the input log-mel spectrogram.
    upsample_initial_channel (`int`, *optional*, defaults to 512):
        The number of input channels into the upsampling network.
    upsample_rates (`tuple[int]` or `list[int]`, *optional*, defaults to `[4, 4, 4, 4]`):
        A tuple of integers defining the stride of each 1D convolutional layer in the upsampling network. The
        length of *upsample_rates* defines the number of convolutional layers and has to match the length of
        *upsample_kernel_sizes*.
    upsample_kernel_sizes (`tuple[int]` or `list[int]`, *optional*, defaults to `[8, 8, 8, 8]`):
        A tuple of integers defining the kernel size of each 1D convolutional layer in the upsampling network. The
        length of *upsample_kernel_sizes* defines the number of convolutional layers and has to match the length of
        *upsample_rates*.
    resblock_kernel_sizes (`tuple[int]` or `list[int]`, *optional*, defaults to `[3, 7, 11]`):
        A tuple of integers defining the kernel sizes of the 1D convolutional layers in the multi-receptive field
        fusion (MRF) module.
    resblock_dilation_sizes (`tuple[tuple[int]]` or `list[list[int]]`, *optional*, defaults to `[[1, 3, 5], [1, 3, 5], [1, 3, 5]]`):
        A nested tuple of integers defining the dilation rates of the dilated 1D convolutional layers in the
        multi-receptive field fusion (MRF) module.
    leaky_relu_slope (`float`, *optional*, defaults to 0.1):
        The angle of the negative slope used by the leaky ReLU activation.
    normalize_before (`bool`, *optional*, defaults to `True`):
        Whether or not to normalize the spectrogram before vocoding using the vocoder's learned mean and variance.

    Example:

    ```python
    >>> from transformers import SpeechT5HifiGan, SpeechT5HifiGanConfig

    >>> # Initializing a "microsoft/speecht5_hifigan" style configuration
    >>> configuration = SpeechT5HifiGanConfig()

    >>> # Initializing a model (with random weights) from the "microsoft/speecht5_hifigan" style configuration
    >>> model = SpeechT5HifiGan(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úspeecht5_hifiganr@   Úmodel_in_dimi€>  Úsampling_rater(   Úupsample_initial_channel)é   r€   r€   r€   .Úupsample_rates)é   r‚   r‚   r‚   Úupsample_kernel_sizes)r   é   é   Úresblock_kernel_sizes)©r;   r   r*   r‡   r‡   Úresblock_dilation_sizesg{®Gáz„?r!   r   Úleaky_relu_slopeTÚnormalize_beforeN)rl   rm   rn   ro   rp   r}   rr   rs   r~   r   r�   rw   rx   rƒ   r†   rˆ   r!   rt   r‰   rŠ   rv   rX   ra   r`   r{   r{   ß   s  € € € € € € ð%ð %ðN $€Jà€L�#ÐÐÑØ€M�3ÐÐÑØ$'Ð˜cÐ'Ð'Ñ'Ø2>€N�D˜”I  c¨3 h¤Ñ/Ð>Ð>Ñ>Ø9EÐ˜4 œ9 u¨S°#¨X¤Ñ6ÐEÐEÑEØ9CÐ˜4 œ9 u¨S°#¨X¤Ñ6ÐCÐCÑCØ,MÐ˜T E™\ÐMÐMÑMØ#Ð�uÐ#Ð#Ñ#Ø!Ð�eÐ!Ð!Ñ!Ø!Ð�dÐ!Ð!Ñ!Ð!Ð!ra   r{   )ro   rg   ri   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   r	   r{   Ú__all__rX   ra   r`   ú<module>r�      s  ðð #Ð "à Ð Ð Ð Ø €€€à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø #Ð #Ð #Ð #Ð #Ð #ð €Ð3Ð4Ñ4Ô4ØðACð ACð ACð ACð ACÐ%ñ ACô ACñ „ñ 5Ô4ðACðH €Ð3Ð4Ñ4Ô4Øð3"ð 3"ð 3"ð 3"ð 3"Ð,ñ 3"ô 3"ñ „ñ 5Ô4ð3"ðl Ð4Ð
5€€€ra   