§
    ‚Štj+  ã                   óà   — d Z ddlmZ ddlmZ ddlmZ ddlmZ  ed¬¦  «        e G d	„ d
e¦  «        ¦   «         ¦   «         Z	 ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z
dgZdS )zMpt configurationé    )ÚLiteral)Ústricté   )ÚPreTrainedConfig)Úauto_docstringzmosaicml/mpt-7b)Ú
checkpointc                   ó¼   — e Zd ZU dZdZdZed         ed<   dZe	ed<   dZ
eed	<   d
Zed
z  ed<   d
Zed
z  ed<   dZeed<   dZeed<   dZeed<   dZeed<   dZe	ed<   d
S )ÚMptAttentionConfigaD  
    attn_type (`str`, *optional*, defaults to `"multihead_attention"`):
        type of attention to use. Options: `"multihead_attention"`, `"multiquery_attention"`.
    attn_pdrop (`float`, *optional*, defaults to `0.0`):
        The dropout probability for the attention layers.
    attn_impl (`str`, *optional*, defaults to `"torch"`):
        The attention implementation to use. One of `"torch"`, `"flash"`, or `"triton"`.
    clip_qkv (`float`, *optional*):
        If not `None`, clip the queries, keys, and values in the attention layer to this value.
    softmax_scale (`float`, *optional*):
        If not `None`, scale the softmax in the attention layer by this value. If `None`, will default to
        `1/sqrt(hidden_size)`.
    prefix_lm (`bool`, *optional*, defaults to `False`):
        Whether the model should operate as a Prefix LM. This requires passing an extra `prefix_mask` argument
        which indicates which tokens belong to the prefix. Tokens in the prefix can attend to one another
        bi-directionally. Tokens outside the prefix use causal attention.
    qk_ln (`bool`, *optional*, defaults to `False`):
        Whether to apply layer normalization to the queries and keys in the attention layer.
    attn_uses_sequence_id (`bool`, *optional*, defaults to `False`):
        Whether to restrict attention to tokens that have the same token_type_ids. When the model is in `train`
        mode, this requires passing an extra *token_type_ids* argument which indicates which sub-sequence each
        token belongs to. Defaults to `False` meaning any provided *token_type_ids* will be ignored.
    alibi (`bool`, *optional*, defaults to `True`):
        Whether or not to use the alibi bias instead of positional embedding.
    alibi_bias_max (`int`, *optional*, defaults to 8):
        The maximum value of the alibi bias.
    Úattn_configÚmultihead_attention)r   Úmultiquery_attentionÚ	attn_typer   Ú
attn_pdropÚtorchÚ	attn_implNÚclip_qkvÚsoftmax_scaleFÚ	prefix_lmÚqk_lnÚattn_uses_sequence_idTÚalibié   Úalibi_bias_max)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úbase_config_keyr   r   Ú__annotations__r   Úintr   Ústrr   Úfloatr   r   Úboolr   r   r   r   © ó    úg/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/mpt/configuration_mpt.pyr
   r
      sÒ   € € € € € € ðð ð8 $€OàH]€IˆwÐDÔEÐ]Ð]Ñ]Ø€J�ÐÐÑØ€IˆsÐÐÑØ!€Hˆe�d‰lÐ!Ð!Ñ!Ø"&€M�5˜4‘<Ð&Ð&Ñ&Ø€IˆtÐÐÑØ€Eˆ4ÐÐÑØ"'Ð˜4Ð'Ð'Ñ'Ø€Eˆ4ÐÐÑØ€N�CÐÐÑÐÐr%   r
   c                   ó¸  ‡ — e Zd ZU dZdZdeiZddddœZdZe	e
d<   d	Ze	e
d<   d
Ze	e
d<   dZe	e
d<   dZe	e
d<   dZe	e
d<   dZee	z  e
d<   dZee
d<   dZee	z  e
d<   dZee
d<   dZeez  dz  e
d<   dZee
d<   dZeez  dz  e
d<   dZee
d<   dZee
d<   dZee
d<   d Zee
d!<   d"Zee
d#<   dZ ee
d$<   dZ!e	dz  e
d%<   dZ"e	dz  e
d&<   dZ#e	e$e	         z  dz  e
d'<   ˆ fd(„Z%ˆ xZ&S ))Ú	MptConfigaZ  
    expansion_ratio (`int`, *optional*, defaults to 4):
        The ratio of the up/down scale in the MLP.
    max_seq_len (`int`, *optional*, defaults to 2048):
        The maximum sequence length of the model.
    layer_norm_epsilon (`float`, *optional*, defaults to 1e-05):
        The epsilon to use in the layer normalization layers.
    learned_pos_emb (`bool`, *optional*, defaults to `True`):
        Whether to use learned positional embeddings.
    attn_config (`dict`, *optional*):
        A dictionary used to configure the model's attention module.
    init_device (`str`, *optional*, defaults to `"cpu"`):
        The device to use for parameter initialization. Defined for backward compatibility
    logit_scale (`float`, *optional*):
        If not None, scale the logits by this value.
    no_bias (`bool`, *optional*, defaults to `True`):
        Whether to use bias in all linear layers.
    embedding_fraction (`float`, *optional*, defaults to 1.0):
        The fraction to scale the gradients of the embedding layer by.
    norm_type (`str`, *optional*, defaults to `"low_precision_layernorm"`):
        Type of layer norm to use. All MPT models uses the same layer norm implementation. Defined for backward
        compatibility.

    Example:

    ```python
    >>> from transformers import MptConfig, MptModel

    >>> # Initializing a Mpt configuration
    >>> configuration = MptConfig()

    >>> # Initializing a model (with random weights) from the configuration
    >>> model = MptModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```
    Úmptr   Ún_headsÚd_modelÚn_layers)Únum_attention_headsÚhidden_sizeÚnum_hidden_layersi   é   é   é   Úexpansion_ratioÚmax_seq_leniÀÄ  Ú
vocab_sizeg        Úresid_pdropgñhãˆµøä>Úlayer_norm_epsilonÚ	emb_pdropTÚlearned_pos_embNÚcpuÚinit_deviceÚlogit_scaleÚno_biasg      ð?Úembedding_fractionÚlow_precision_layernormÚ	norm_typeFÚ	use_cacheg{®Gáz”?Úinitializer_rangeÚtie_word_embeddingsÚpad_token_idÚbos_token_idÚeos_token_idc                 óÐ   •— | j         €t          ¦   «         | _         n0t          | j         t          ¦  «        rt          di | j         ¤Ž| _          t	          ¦   «         j        di |¤Ž d S )Nr$   )r   r
   Ú
isinstanceÚdictÚsuperÚ__post_init__)ÚselfÚkwargsÚ	__class__s     €r&   rK   zMptConfig.__post_init__Ž   so   ø€ ØÔÐ#Ý1Ñ3Ô3ˆDÔÐÝ˜Ô(­$Ñ/Ô/ð 	FÝ1ÐEÐE°DÔ4DÐEÐEˆDÔØ�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'r%   )'r   r   r   r   Ú
model_typer
   Úsub_configsÚattribute_mapr+   r    r   r*   r,   r3   r4   r5   r6   r"   r7   r8   r9   r#   r   rI   r;   r!   r<   r=   r>   r@   rA   rB   rC   rD   rE   rF   ÚlistrK   Ú__classcell__)rN   s   @r&   r(   r(   E   s  ø€ € € € € € ð%ð %ðN €JØ Ð"4Ð5€Kà(Ø Ø'ðð €Mð €GˆSÐÐÑØ€GˆSÐÐÑØ€HˆcÐÐÑØ€O�SÐÐÑØ€K�ÐÐÑØ€J�ÐÐÑØ"€K�˜‘Ð"Ð"Ñ"Ø $Ð˜Ð$Ð$Ñ$Ø €Iˆu�s‰{Ð Ð Ñ Ø €O�TÐ Ð Ñ Ø48€K�Ð*Ñ*¨TÑ1Ð8Ð8Ñ8Ø€K�ÐÐÑØ&*€K�˜‘˜tÑ#Ð*Ð*Ñ*Ø€GˆTÐÐÑØ #Ð˜Ð#Ð#Ñ#Ø.€IˆsÐ.Ð.Ñ.Ø€IˆtÐÐÑØ#Ð�uÐ#Ð#Ñ#Ø $Ð˜Ð$Ð$Ñ$Ø#€L�#˜‘*Ð#Ð#Ñ#Ø#€L�#˜‘*Ð#Ð#Ñ#Ø+/€L�#˜˜Sœ	‘/ DÑ(Ð/Ð/Ñ/ð(ð (ð (ð (ð (ð (ð (ð (ð (r%   r(   N)r   Útypingr   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   r
   r(   Ú__all__r$   r%   r&   ú<module>rY      s  ðð Ð à Ð Ð Ð Ð Ð à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø #Ð #Ð #Ð #Ð #Ð #ð €Ð,Ð-Ñ-Ô-Øð(ð (ð (ð (ð (Ð)ñ (ô (ñ „ñ .Ô-ð(ðV €Ð,Ð-Ñ-Ô-ØðL(ð L(ð L(ð L(ð L(Ð ñ L(ô L(ñ „ñ .Ô-ðL(ð^ ˆ-€€€r%   