§
    ‚Štj{"  ã                   ó„   — d Z ddlmZ ddlmZ ddlmZ  ed¬¦  «        e G d„ d	e¦  «        ¦   «         ¦   «         Zd	gZd
S )zVITS model configurationé    )Ústricté   )ÚPreTrainedConfig)Úauto_docstringzfacebook/mms-tts-eng)Ú
checkpointc                   ó0  — e Zd ZU dZdZdZeed<   dZeed<   dZ	eed<   d	Z
eed
<   dZeed<   dZeed<   dZeed<   dZeez  ed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeez  ed<   dZeez  ed<   dZeez  ed<   dZeed<   dZeed <   dZeed!<   d"Zeed#<   d$Zeed%<   d&Zeed'<   d(Ze e         e!ed)f         z  ed*<   d+Z"e e         e!ed)f         z  ed,<   d-Z#e e         e!ed)f         z  ed.<   d/Z$e e!z  ed0<   dZ%eed1<   d	Z&eed2<   dZ'eed3<   d4Z(eed5<   d6Z)eed7<   dZ*eed8<   d9Z+eez  ed:<   dZ,eed;<   d<Z-eed=<   dZ.eed><   dZ/eed?<   d@Z0eedA<   dBZ1eedC<   d"Z2eedD<   dEZ3eez  edF<   dGZ4eez  edH<   dIZ5eedJ<   dKZ6eedL<   dMZ7eedN<   dOZ8edOz  edP<   dQ„ Z9dOS )RÚ
VitsConfiga  
    window_size (`int`, *optional*, defaults to 4):
        Window size for the relative positional embeddings in the attention layers of the Transformer encoder.
    use_bias (`bool`, *optional*, defaults to `True`):
        Whether to use bias in the key, query, value projection layers in the Transformer encoder.
    ffn_kernel_size (`int`, *optional*, defaults to 3):
        Kernel size of the 1D convolution layers used by the feed-forward network in the Transformer encoder.
    flow_size (`int`, *optional*, defaults to 192):
        Dimensionality of the flow layers.
    spectrogram_bins (`int`, *optional*, defaults to 513):
        Number of frequency bins in the target spectrogram.
    use_stochastic_duration_prediction (`bool`, *optional*, defaults to `True`):
        Whether to use the stochastic duration prediction module or the regular duration predictor.
    num_speakers (`int`, *optional*, defaults to 1):
        Number of speakers if this is a multi-speaker model.
    speaker_embedding_size (`int`, *optional*, defaults to 0):
        Number of channels used by the speaker embeddings. Is zero for single-speaker models.
    upsample_initial_channel (`int`, *optional*, defaults to 512):
        The number of input channels into the HiFi-GAN upsampling network.
    upsample_rates (`tuple[int]` or `list[int]`, *optional*, defaults to `[8, 8, 2, 2]`):
        A tuple of integers defining the stride of each 1D convolutional layer in the HiFi-GAN upsampling network.
        The length of `upsample_rates` defines the number of convolutional layers and has to match the length of
        `upsample_kernel_sizes`.
    upsample_kernel_sizes (`tuple[int]` or `list[int]`, *optional*, defaults to `[16, 16, 4, 4]`):
        A tuple of integers defining the kernel size of each 1D convolutional layer in the HiFi-GAN upsampling
        network. The length of `upsample_kernel_sizes` defines the number of convolutional layers and has to match
        the length of `upsample_rates`.
    resblock_kernel_sizes (`tuple[int]` or `list[int]`, *optional*, defaults to `[3, 7, 11]`):
        A tuple of integers defining the kernel sizes of the 1D convolutional layers in the HiFi-GAN
        multi-receptive field fusion (MRF) module.
    resblock_dilation_sizes (`tuple[tuple[int]]` or `list[list[int]]`, *optional*, defaults to `[[1, 3, 5], [1, 3, 5], [1, 3, 5]]`):
        A nested tuple of integers defining the dilation rates of the dilated 1D convolutional layers in the
        HiFi-GAN multi-receptive field fusion (MRF) module.
    leaky_relu_slope (`float`, *optional*, defaults to 0.1):
        The angle of the negative slope used by the leaky ReLU activation.
    depth_separable_channels (`int`, *optional*, defaults to 2):
        Number of channels to use in each depth-separable block.
    depth_separable_num_layers (`int`, *optional*, defaults to 3):
        Number of convolutional layers to use in each depth-separable block.
    duration_predictor_flow_bins (`int`, *optional*, defaults to 10):
        Number of channels to map using the unonstrained rational spline in the duration predictor model.
    duration_predictor_tail_bound (`float`, *optional*, defaults to 5.0):
        Value of the tail bin boundary when computing the unconstrained rational spline in the duration predictor
        model.
    duration_predictor_kernel_size (`int`, *optional*, defaults to 3):
        Kernel size of the 1D convolution layers used in the duration predictor model.
    duration_predictor_dropout (`float`, *optional*, defaults to 0.5):
        The dropout ratio for the duration predictor model.
    duration_predictor_num_flows (`int`, *optional*, defaults to 4):
        Number of flow stages used by the duration predictor model.
    duration_predictor_filter_channels (`int`, *optional*, defaults to 256):
        Number of channels for the convolution layers used in the duration predictor model.
    prior_encoder_num_flows (`int`, *optional*, defaults to 4):
        Number of flow stages used by the prior encoder flow model.
    prior_encoder_num_wavenet_layers (`int`, *optional*, defaults to 4):
        Number of WaveNet layers used by the prior encoder flow model.
    posterior_encoder_num_wavenet_layers (`int`, *optional*, defaults to 16):
        Number of WaveNet layers used by the posterior encoder model.
    wavenet_kernel_size (`int`, *optional*, defaults to 5):
        Kernel size of the 1D convolution layers used in the WaveNet model.
    wavenet_dilation_rate (`int`, *optional*, defaults to 1):
        Dilation rates of the dilated 1D convolutional layers used in the WaveNet model.
    wavenet_dropout (`float`, *optional*, defaults to 0.0):
        The dropout ratio for the WaveNet layers.
    speaking_rate (`float`, *optional*, defaults to 1.0):
        Speaking rate. Larger values give faster synthesised speech.
    noise_scale (`float`, *optional*, defaults to 0.667):
        How random the speech prediction is. Larger values create more variation in the predicted speech.
    noise_scale_duration (`float`, *optional*, defaults to 0.8):
        How random the duration prediction is. Larger values create more variation in the predicted durations.

    Example:

    ```python
    >>> from transformers import VitsModel, VitsConfig

    >>> # Initializing a "facebook/mms-tts-eng" style configuration
    >>> configuration = VitsConfig()

    >>> # Initializing a model (with random weights) from the "facebook/mms-tts-eng" style configuration
    >>> model = VitsModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úvitsé&   Ú
vocab_sizeéÀ   Úhidden_sizeé   Únum_hidden_layersé   Únum_attention_headsé   Úwindow_sizeTÚuse_biasi   Úffn_dimgš™™™™™¹?Ú	layerdropr   Úffn_kernel_sizeÚ	flow_sizei  Úspectrogram_binsÚreluÚ
hidden_actÚhidden_dropoutÚattention_dropoutÚactivation_dropoutg{®Gáz”?Úinitializer_rangegñhãˆµøä>Úlayer_norm_epsÚ"use_stochastic_duration_predictioné   Únum_speakersr   Úspeaker_embedding_sizei   Úupsample_initial_channel)é   r'   r   r   .Úupsample_rates)é   r)   r   r   Úupsample_kernel_sizes)r   é   é   Úresblock_kernel_sizes)©r#   r   é   r.   r.   Úresblock_dilation_sizesÚleaky_relu_slopeÚdepth_separable_channelsÚdepth_separable_num_layersé
   Úduration_predictor_flow_binsg      @Úduration_predictor_tail_boundÚduration_predictor_kernel_sizeg      à?Úduration_predictor_dropoutÚduration_predictor_num_flowsé   Ú"duration_predictor_filter_channelsÚprior_encoder_num_flowsÚ prior_encoder_num_wavenet_layersr)   Ú$posterior_encoder_num_wavenet_layersr/   Úwavenet_kernel_sizeÚwavenet_dilation_rateg        Úwavenet_dropoutg      ð?Úspeaking_rategòÒMbXå?Únoise_scalegš™™™™™é?Únoise_scale_durationi€>  Úsampling_rateNÚpad_token_idc                 óÎ   — t          | j        ¦  «        t          | j        ¦  «        k    r:t          dt          | j        ¦  «        › dt          | j        ¦  «        › d�¦  «        ‚dS )zOPart of `@strict`-powered validation. Validates the architecture of the config.z'The length of `upsample_kernel_sizes` (z-) must match the length of `upsample_rates` (ú)N)Úlenr*   r(   Ú
ValueError)Úselfs    úi/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/vits/configuration_vits.pyÚvalidate_architecturez VitsConfig.validate_architectureŸ   s}   € åˆtÔ)Ñ*Ô*­c°$Ô2EÑ.FÔ.FÒFÐFÝðA½#¸dÔ>XÑ:YÔ:Yð Að AÝ%(¨Ô)<Ñ%=Ô%=ðAð Að Añô ð ð GÐFó    ):Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typer   ÚintÚ__annotations__r   r   r   r   r   Úboolr   r   Úfloatr   r   r   r   Ústrr   r   r   r    r!   r"   r$   r%   r&   r(   ÚlistÚtupler*   r-   r0   r1   r2   r3   r5   r6   r7   r8   r9   r;   r<   r=   r>   r?   r@   rA   rB   rC   rD   rE   rF   rM   © rN   rL   r	   r	      s{  € € € € € € ðTð Tðl €Jà€J�ÐÐÑØ€K�ÐÐÑØÐ�sÐÐÑØ Ð˜Ð Ð Ñ Ø€K�ÐÐÑØ€HˆdÐÐÑØ€GˆSÐÐÑØ €Iˆu�s‰{Ð Ð Ñ Ø€O�SÐÐÑØ€IˆsÐÐÑØÐ�cÐÐÑØ€J�ÐÐÑØ"%€N�E˜C‘KÐ%Ð%Ñ%Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø&)Ð˜ ™Ð)Ð)Ñ)Ø#Ð�uÐ#Ð#Ñ#Ø €N�EÐ Ð Ñ Ø/3Ð&¨Ð3Ð3Ñ3Ø€L�#ÐÐÑØ"#Ð˜CÐ#Ð#Ñ#Ø$'Ð˜cÐ'Ð'Ñ'Ø2>€N�D˜”I  c¨3 h¤Ñ/Ð>Ð>Ñ>Ø9GÐ˜4 œ9 u¨S°#¨X¤Ñ6ÐGÐGÑGØ9CÐ˜4 œ9 u¨S°#¨X¤Ñ6ÐCÐCÑCØ,MÐ˜T E™\ÐMÐMÑMØ!Ð�eÐ!Ð!Ñ!Ø$%Ð˜cÐ%Ð%Ñ%Ø&'Ð Ð'Ð'Ñ'Ø(*Ð  #Ð*Ð*Ñ*Ø+.Ð! 5Ð.Ð.Ñ.Ø*+Ð" CÐ+Ð+Ñ+Ø.1Ð ¨¡Ð1Ð1Ñ1Ø()Ð  #Ð)Ð)Ñ)Ø.1Ð&¨Ð1Ð1Ñ1Ø#$Ð˜SÐ$Ð$Ñ$Ø,-Ð$ cÐ-Ð-Ñ-Ø02Ð(¨#Ð2Ð2Ñ2Ø Ð˜Ð Ð Ñ Ø!"Ð˜3Ð"Ð"Ñ"Ø#&€O�U˜S‘[Ð&Ð&Ñ&Ø!$€M�5˜3‘;Ð$Ð$Ñ$Ø€K�ÐÐÑØ"%Ð˜%Ð%Ð%Ñ%Ø€M�3ÐÐÑØ#€L�#˜‘*Ð#Ð#Ñ#ðð ð ð ð rN   r	   N)	rR   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   r	   Ú__all__r[   rN   rL   ú<module>r`      s©   ðð Ð à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø #Ð #Ð #Ð #Ð #Ð #ð €Ð1Ð2Ñ2Ô2ØðMð Mð Mð Mð MÐ!ñ Mô Mñ „ñ 3Ô2ðMð` ˆ.€€€rN   