§
    ‚Štjà  ã                   ó4  — d Z ddlmZ ddlmZ ddlmZ ddlmZ e edd¬	¦  «         G d
„ de¦  «        ¦   «         ¦   «         Z	e edd¬	¦  «         G d„ de¦  «        ¦   «         ¦   «         Z
 ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         ZdgZdS )zDBRX model configurationé    )Ústricté   )ÚPreTrainedConfig)ÚRopeParameters)Úauto_docstringz4This config is used to instantiate attention layers.z$transformers-community/dbrx-instruct)Úcustom_introÚ
checkpointc                   óT   — e Zd ZU dZdZdZeez  ed<   dZ	eez  dz  ed<   dZ
eed<   dS )	ÚDbrxAttentionConfigaz  
    attn_pdrop (`float`, *optional*, defaults to 0.0):
        The dropout probability for the attention layers.
    clip_qkv (`float`, *optional*):
        If set, clip the queries, keys, and values in the attention layer to this value.
    kv_n_heads (`int`, *optional*, defaults to 1):
        For grouped_query_attention only, allow user to specify number of kv heads.
    Úattn_configç        Ú
attn_pdropNÚclip_qkvé   Ú
kv_n_heads)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úbase_config_keyr   ÚfloatÚintÚ__annotations__r   r   © ó    úi/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/dbrx/configuration_dbrx.pyr   r      s`   € € € € € € ðð ð $€Oà!€J�˜‘Ð!Ð!Ñ!Ø#'€Hˆc�E‰k˜DÑ Ð'Ð'Ñ'Ø€J�ÐÐÑÐÐr   r   z6This config is used to instantiate feedforward layers.c                   óª   ‡ — e Zd ZU dZdZdZeed<   dZe	dz  ed<   dZ
eed<   d	Zeed
<   dZeed<   dZedz  ed<   dZeed<   dZedz  ed<   ˆ fd„Zˆ xZS )ÚDbrxFFNConfiga  
    ffn_act_fn (`dict`, *optional*, defaults to `None`):
        A dict specifying activation function for the FFN.
        The dict should have a key 'name' with the value being the name of the activation function along with
        any additional keyword arguments. If `None`, then set to `{"name": "silu"}`.
    ffn_hidden_size (`int`, *optional*, defaults to 3584):
        The hidden size of the feedforward network.
    moe_num_experts (`int`, *optional*, defaults to 4):
        The number of experts in the mixture of experts layer.
    moe_top_k (`int`, *optional*, defaults to 1):
        The number of experts to use in the mixture of experts layer.
    moe_jitter_eps (`float`, *optional*, defaults to `None`):
        If not `None`, the jitter epsilon for the mixture of experts layer.
    moe_loss_weight (`float`, *optional*, defaults to 0.01):
        The loss weight for the mixture of experts layer.
    moe_normalize_expert_weights (`float`, *optional*, defaults to 1.0):
        The normalization factor for the expert weights.
    Ú
ffn_configi   Úhidden_sizeNÚ
ffn_act_fni   Úffn_hidden_sizeé   Úmoe_num_expertsr   Ú	moe_top_kÚmoe_jitter_epsg{®Gáz„?Úmoe_loss_weightg      ð?Úmoe_normalize_expert_weightsc                 óà   •— | j         €	ddi| _         dD ]}||v r|                     |¦  «         Œt          |¦  «        dk    rt          d|›�¦  «        ‚ t	          ¦   «         j        di |¤Ž d S )NÚnameÚsilu)Ú
model_typeÚattn_implementationÚexperts_implementationÚtransformers_versionÚ_commit_hashÚtorch_dtypeÚdtyper   zFound unknown kwargs=r   )r!   ÚpopÚlenÚ
ValueErrorÚsuperÚ__post_init__)ÚselfÚkwargsÚkÚ	__class__s      €r   r7   zDbrxFFNConfig.__post_init__Q   sŽ   ø€ ØŒ?Ð"Ø% vÐ.ˆDŒOð
ð 
	ð 
	ˆAð �Fˆ{ˆ{Ø—
’
˜1‘”�øÝˆv‰;Œ;˜!ÒÐÝÐ7¨fÐ7Ð7Ñ8Ô8Ð8à�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'r   )r   r   r   r   r   r    r   r   r!   Údictr"   r$   r%   r&   r   r'   r(   r7   Ú__classcell__©r;   s   @r   r   r   -   sØ   ø€ € € € € € ðð ð& #€Oà€K�ÐÐÑØ"€J��t‘Ð"Ð"Ñ"Ø€O�SÐÐÑØ€O�SÐÐÑØ€IˆsÐÐÑØ#'€N�E˜D‘LÐ'Ð'Ñ'Ø!€O�UÐ!Ð!Ñ!Ø14Ð  %¨$¡,Ð4Ð4Ñ4ð(ð (ð (ð (ð (ð (ð (ð (ð (r   r   )r	   c                   ó¦  ‡ — e Zd ZU dZdZeedœZdddddœZd	Z	e
d
z  ed<   dZe
d
z  ed<   dZe
d
z  ed<   d	Ze
d
z  ed<   dZe
ed<   dZed
z  ed<   dZed
z  ed<   d
Zeez  d
z  ed<   d
Zeez  d
z  ed<   dZeed<   dZeed<   dZed
z  ed<   d
Zeez  d
z  ed<   d
Ze
d
z  ed<   d
Ze
d
z  ed<   d
Ze
ee
         z  d
z  ed<   dZ eed<   ˆ fd„Z!d „ Z"ˆ xZ#S )!Ú
DbrxConfigaä  
    max_seq_len (`int`, *optional*, defaults to 2048):
        The maximum sequence length of the model.
    attn_config (`dict`, *optional*):
        A dictionary used to configure the model's attention module.
    ffn_config (`dict`, *optional*):
        A dictionary used to configure the model's FFN module.

    Example:
    ```python
    >>> from transformers import DbrxConfig, DbrxModel

    >>> # Initializing a Dbrx configuration
    >>> configuration = DbrxConfig(n_layers=2, d_model=256, n_heads=8, vocab_size=128)

    >>> # Initializing a model (with random weights) from the configuration
    >>> model = DbrxModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```
    Údbrx)r   r   Ún_headsÚd_modelÚn_layersÚmax_seq_len)Únum_attention_headsr    Únum_hidden_layersÚmax_position_embeddingsi   Né   é   i }  Ú
vocab_sizer   Úresid_pdropÚ	emb_pdropr   r   TÚ	use_cacheg{®Gáz”?Úinitializer_rangeFÚoutput_router_logitsÚrope_parametersÚpad_token_idÚbos_token_idÚeos_token_idÚtie_word_embeddingsc                 óˆ  •— | j         €t          ¦   «         | _         n0t          | j         t          ¦  «        rt          di | j         ¤Ž| _         | j        €t          ¦   «         | _        n0t          | j        t          ¦  «        rt          di | j        ¤Ž| _        | j         j        | _         t          ¦   «         j	        di |¤Ž d S )Nr   )
r   r   Ú
isinstancer<   r   r   r   Únum_key_value_headsr6   r7   )r8   r9   r;   s     €r   r7   zDbrxConfig.__post_init__›   s½   ø€ ØÔÐ#Ý2Ñ4Ô4ˆDÔÐÝ˜Ô(­$Ñ/Ô/ð 	GÝ2ÐFÐF°TÔ5EÐFÐFˆDÔàŒ?Ð"Ý+™oœoˆDŒOˆOÝ˜œ­Ñ.Ô.ð 	?Ý+Ð>Ð>¨d¬oÐ>Ð>ˆDŒOà#'Ô#3Ô#>ˆÔ Ø�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'r   c                 ó2   — | j         rt          d¦  «        ‚dS )zOPart of `@strict`-powered validation. Validates the architecture of the config.z5tie_word_embeddings is not supported for DBRX models.N)rU   r5   )r8   s    r   Úvalidate_architecturez DbrxConfig.validate_architecture©   s)   € àÔ#ð 	VÝÐTÑUÔUÐUð	Vð 	Vr   )$r   r   r   r   r,   r   r   Úsub_configsÚattribute_maprC   r   r   rB   rD   rE   rK   rL   r   rM   r   r<   r   rN   ÚboolrO   rP   rQ   r   rR   rS   rT   ÚlistrU   r7   rZ   r=   r>   s   @r   r@   r@   f   së  ø€ € € € € € ðð ð. €JØ"5À]ÐSÐS€Kà(Ø Ø'Ø#0ð	ð €Mð €GˆS�4‰ZÐÐÑØ€GˆS�4‰ZÐÐÑØ€Hˆc�D‰jÐÐÑØ"€K��t‘Ð"Ð"Ñ"Ø€J�ÐÐÑØ #€K�˜‘Ð#Ð#Ñ#Ø!€Iˆu�t‰|Ð!Ð!Ñ!Ø59€KÐ$ tÑ+¨dÑ2Ð9Ð9Ñ9Ø.2€J� Ñ$ tÑ+Ð2Ð2Ñ2Ø€IˆtÐÐÑØ#Ð�uÐ#Ð#Ñ#Ø(-Ð˜$ ™+Ð-Ð-Ñ-Ø48€O�^ dÑ*¨TÑ1Ð8Ð8Ñ8Ø#€L�#˜‘*Ð#Ð#Ñ#Ø#€L�#˜‘*Ð#Ð#Ñ#Ø+/€L�#˜˜Sœ	‘/ DÑ(Ð/Ð/Ñ/Ø %Ð˜Ð%Ð%Ñ%ð(ð (ð (ð (ð (ðVð Vð Vð Vð Vð Vð Vr   r@   N)r   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úmodeling_rope_utilsr   Úutilsr   r   r   r@   Ú__all__r   r   r   ú<module>rd      s{  ðð Ð à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø #Ð #Ð #Ð #Ð #Ð #ð Ø€ØGØ5ðñ ô ðð ð ð ð Ð*ñ ô ñ	ô ñ „ð
ð" Ø€ØIØ5ðñ ô ð1(ð 1(ð 1(ð 1(ð 1(Ð$ñ 1(ô 1(ñ	ô ñ „ð
1(ðh €ÐAÐBÑBÔBØðDVð DVð DVð DVð DVÐ!ñ DVô DVñ „ñ CÔBðDVðN ˆ.€€€r   