§
    ‚Štj!  ã                   óŒ   — d dl mZ ddlmZ ddlmZ ddlmZ  ed¬¦  «        e G d„ d	e¦  «        ¦   «         ¦   «         Zd	gZ	d
S )é    )Ústricté   )ÚPreTrainedConfig)ÚRopeParameters)Úauto_docstringzUsefulSensors/moonshine-tiny)Ú
checkpointc                   óÔ  ‡ — e Zd ZU dZdZdgZdddddœZd	Zee	d
<   dZ
ee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZedz  e	d<   dZedz  e	d<   dZedz  e	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d <   dZeez  dz  e	d!<   dZee	d"<   d#Z ee	d$<   d%Z!eez  e	d&<   dZ"edz  e	d'<   d(Z#ee$e         z  dz  e	d)<   dZ%edz  e	d*<   dZ&ee	d+<   ˆ fd,„Z'ˆ xZ(S )-ÚMoonshineConfiga‹	  
    encoder_num_key_value_heads (`int`, *optional*):
        This is the number of key_value heads that should be used to implement Grouped Query Attention. If
        `encoder_num_key_value_heads=encoder_num_attention_heads`, the model will use Multi Head Attention (MHA), if
        `encoder_num_key_value_heads=1` the model will use Multi Query Attention (MQA) otherwise GQA is used. When
        converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
        by meanpooling all the original heads within that group. For more details, check out [this
        paper](https://huggingface.co/papers/2305.13245). If it is not specified, will default to
        `num_attention_heads`.
    decoder_num_key_value_heads (`int`, *optional*):
        This is the number of key_value heads that should be used to implement Grouped Query Attention. If
        `decoder_num_key_value_heads=decoder_num_attention_heads`, the model will use Multi Head Attention (MHA), if
        `decoder_num_key_value_heads=1` the model will use Multi Query Attention (MQA) otherwise GQA is used. When
        converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
        by meanpooling all the original heads within that group. For more details, check out [this
        paper](https://huggingface.co/papers/2305.13245). If it is not specified, will default to
        `decoder_num_attention_heads`.
    pad_head_dim_to_multiple_of (`int`, *optional*):
        Pad head dimension in encoder and decoder to the next multiple of this value. Necessary for using certain
        optimized attention implementations.
    encoder_hidden_act (`str` or `function`, *optional*, defaults to `"gelu"`):
        The non-linear activation function (function or string) in the encoder.
    decoder_hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
        The non-linear activation function (function or string) in the decoder.

    Example:

    ```python
    >>> from transformers import MoonshineModel, MoonshineConfig

    >>> # Initializing a Moonshine style configuration
    >>> configuration = MoonshineConfig().from_pretrained("UsefulSensors/moonshine-tiny")

    >>> # Initializing a model from the configuration
    >>> model = MoonshineModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Ú	moonshineÚpast_key_valuesÚdecoder_num_key_value_headsÚdecoder_num_attention_headsÚdecoder_num_hidden_layersÚdecoder_hidden_act)Únum_key_value_headsÚnum_attention_headsÚnum_hidden_layersÚ
hidden_acti €  Ú
vocab_sizei   Úhidden_sizei€  Úintermediate_sizeé   Úencoder_num_hidden_layersé   Úencoder_num_attention_headsNÚencoder_num_key_value_headsÚpad_head_dim_to_multiple_ofÚgeluÚencoder_hidden_actÚsilui   Úmax_position_embeddingsg{®Gáz”?Úinitializer_rangeé   Údecoder_start_token_idTÚ	use_cacheÚrope_parametersÚis_encoder_decoderFÚattention_biasg        Úattention_dropoutÚbos_token_idé   Úeos_token_idÚpad_token_idÚtie_word_embeddingsc                 ó²   •— | j         €| j        | _         | j        €| j        | _        |                     dd¦  «          t          ¦   «         j        di |¤Ž d S )NÚpartial_rotary_factorgÍÌÌÌÌÌì?© )r   r   r   r   Ú
setdefaultÚsuperÚ__post_init__)ÚselfÚkwargsÚ	__class__s     €ús/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/moonshine/configuration_moonshine.pyr4   zMoonshineConfig.__post_init__i   se   ø€ ØÔ+Ð3Ø/3Ô/OˆDÔ,àÔ+Ð3Ø/3Ô/OˆDÔ,à×ÒÐ1°3Ñ7Ô7Ð7Ø�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'ó    ))Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚattribute_mapr   ÚintÚ__annotations__r   r   r   r   r   r   r   r   r   r   Ústrr   r!   r"   Úfloatr$   r%   Úboolr&   r   Údictr'   r(   r)   r*   r,   Úlistr-   r.   r4   Ú__classcell__)r7   s   @r8   r
   r
      s.  ø€ € € € € € ð&ð &ðP €JØ#4Ð"5Ðà<Ø<Ø8Ø*ð	ð €Mð €J�ÐÐÑØ€K�ÐÐÑØ!Ð�sÐ!Ð!Ñ!Ø%&Ð˜sÐ&Ð&Ñ&Ø%&Ð˜sÐ&Ð&Ñ&Ø'(Ð Ð(Ð(Ñ(Ø'(Ð Ð(Ð(Ñ(Ø.2Ð  t¡Ð2Ð2Ñ2Ø.2Ð  t¡Ð2Ð2Ñ2Ø.2Ð  t¡Ð2Ð2Ñ2Ø$Ð˜Ð$Ð$Ñ$Ø$Ð˜Ð$Ð$Ñ$Ø#&Ð˜SÐ&Ð&Ñ&Ø#Ð�uÐ#Ð#Ñ#Ø"#Ð˜CÐ#Ð#Ñ#Ø€IˆtÐÐÑØ48€O�^ dÑ*¨TÑ1Ð8Ð8Ñ8Ø#Ð˜Ð#Ð#Ñ#Ø €N�DÐ Ð Ñ Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø €L�#˜‘*Ð Ð Ñ Ø+,€L�#˜˜Sœ	‘/ DÑ(Ð,Ð,Ñ,Ø#€L�#˜‘*Ð#Ð#Ñ#Ø $Ð˜Ð$Ð$Ñ$ð(ð (ð (ð (ð (ð (ð (ð (ð (r9   r
   N)
Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úmodeling_rope_utilsr   Úutilsr   r
   Ú__all__r1   r9   r8   ú<module>rN      s¶   ðð* /Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø #Ð #Ð #Ð #Ð #Ð #ð €Ð9Ð:Ñ:Ô:ØðS(ð S(ð S(ð S(ð S(Ð&ñ S(ô S(ñ „ñ ;Ô:ðS(ðl Ð
€€€r9   