§
    ‚ŠtjÃ  ã                   óŒ   — d dl mZ ddlmZ ddlmZ ddlmZ  ed¬¦  «        e G d„ d	e¦  «        ¦   «         ¦   «         Zd	gZ	d
S )é    )Ústricté   )ÚPreTrainedConfig)ÚRopeParameters)Úauto_docstringz,naver-hyperclovax/HyperCLOVAX-SEED-Think-14B)Ú
checkpointc                   ó&  ‡ — e Zd ZU dZdZdgZddddddddœZdgdgfd	d
gd	gfd	gd	gfdœZdZe	e
d<   dZe	e
d<   dZe	e
d<   dZe	e
d<   dZe	e
d<   dZe	dz  e
d<   dZee
d<   dZe	e
d<   dZee
d<   dZee
d<   dZee
d <   dZe	dz  e
d!<   d"Ze	dz  e
d#<   d$Ze	ee	         z  dz  e
d%<   d&Zee
d'<   dZeez  dz  e
d(<   d&Z ee
d)<   d*Z!ee	z  e
d+<   d&Z"ee
d,<   d-Z#ee	z  e
d.<   d-Z$ee	z  e
d/<   d-Z%ee	z  e
d0<   dZ&edz  e
d1<   dZ'e	dz  e
d2<   dZ(ee
d3<   ˆ fd4„Z)d5„ Z*ˆ xZ+S )6ÚHyperCLOVAXConfiga@  
    embedding_multiplier (`float`, *optional*, defaults to `1.0`):
        Scaling factor applied to the token embedding outputs. Used in MuP to control the
        scale of the embedding activations.
    logits_scaling (`float`, *optional*, defaults to `1.0`):
        Scaling factor **multiplied** to the final logits before loss computation or sampling.
        Used in MuP to ensure consistent output scale across model sizes. Note: unlike
        [`GraniteConfig`], this is a multiplier, not a divisor.
    residual_multiplier (`float`, *optional*, defaults to `1.0`):
        Scaling factor applied to each sub-layer output before adding to the residual stream.
        Used in Maximal Update Parametrization (MuP) to stabilize training across model sizes.
    attention_multiplier (`float`, *optional*, defaults to `head_dim ** -0.5`):
        Scaling factor applied to attention logits before softmax, replacing the standard
        `1 / sqrt(head_dim)` scaling. Set explicitly for MuP-based training; when `None`,
        defaults to the standard value.
    use_post_norm (`bool`, *optional*, defaults to `True`):
        Whether to apply an extra RMSNorm after each sub-layer output (Peri-Layer Normalization).

    ```python
    >>> from transformers import HyperCLOVAXModel, HyperCLOVAXConfig

    >>> # Initializing a HyperCLOVAX style configuration
    >>> configuration = HyperCLOVAXConfig()

    >>> # Initializing a model from the configuration
    >>> model = HyperCLOVAXModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```ÚhyperclovaxÚpast_key_valuesÚcolwiseÚrowwise)zlayers.*.self_attn.q_projzlayers.*.self_attn.k_projzlayers.*.self_attn.v_projzlayers.*.self_attn.o_projzlayers.*.mlp.gate_projzlayers.*.mlp.up_projzlayers.*.mlp.down_projÚ	input_idsÚinputs_embedsÚhidden_statesÚattention_mask)Úembed_tokensÚlayersÚnormi }  Ú
vocab_sizei   Úhidden_sizei +  Úintermediate_sizeé    Únum_hidden_layersÚnum_attention_headsNÚnum_key_value_headsÚsiluÚ
hidden_acti   Úmax_position_embeddingsg{®Gáz”?Úinitializer_rangeg�íµ ÷Æ°>Úrms_norm_epsTÚ	use_cacheÚpad_token_idé   Úbos_token_idé   Úeos_token_idFÚtie_word_embeddingsÚrope_parametersÚattention_biasg        Úattention_dropoutÚmlp_biasg      ð?Úembedding_multiplierÚlogits_scalingÚresidual_multiplierÚattention_multiplierÚhead_dimÚuse_post_normc                 óÆ   •— | j         €| j        | j        z  | _         | j        €| j        | _         t	          ¦   «         j        di |¤Ž | j        €| j         dz  | _        d S d S )Ng      à¿© )r1   r   r   r   ÚsuperÚ__post_init__r0   )ÚselfÚkwargsÚ	__class__s     €úw/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/hyperclovax/configuration_hyperclovax.pyr6   zHyperCLOVAXConfig.__post_init__n   sx   ø€ ð Œ=Ð Ø Ô,°Ô0HÑHˆDŒMØÔ#Ð+Ø'+Ô'?ˆDÔ$à�‰ŒÔÐ'Ð' Ð'Ð'Ð'ð Ô$Ð,Ø(,¬°tÑ(;ˆDÔ%Ð%Ð%ð -Ð,ó    c                 ól   — | j         | j        z  dk    r t          d| j         › d| j        › d�¦  «        ‚dS )zCValidates that `hidden_size` is divisible by `num_attention_heads`.r   zThe hidden size (z6) is not a multiple of the number of attention heads (z).N)r   r   Ú
ValueError)r7   s    r:   Úvalidate_architecturez'HyperCLOVAXConfig.validate_architecture}   s[   € àÔ˜dÔ6Ñ6¸!Ò;Ð;Ýð7 DÔ$4ð 7ð 7ØÔ2ð7ð 7ð 7ñô ð ð <Ð;r;   ),Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚbase_model_tp_planÚbase_model_pp_planr   ÚintÚ__annotations__r   r   r   r   r   r   Ústrr   r    Úfloatr!   r"   Úboolr#   r%   r'   Úlistr(   r)   r   Údictr*   r+   r,   r-   r.   r/   r0   r1   r2   r6   r>   Ú__classcell__)r9   s   @r:   r
   r
      sž  ø€ € € € € € ðð ð> €JØ#4Ð"5Ðð &/Ø%.Ø%.Ø%.Ø"+Ø )Ø"+ðð Ðð &˜¨Ð(9Ð:Ø#Ð%5Ð6¸Ð8IÐJØ!Ð" _Ð$5Ð6ðð Ðð €J�ÐÐÑØ€K�ÐÐÑØ"Ð�sÐ"Ð"Ñ"ØÐ�sÐÐÑØ!Ð˜Ð!Ð!Ñ!Ø&*Ð˜˜t™Ð*Ð*Ñ*Ø€J�ÐÐÑØ#'Ð˜SÐ'Ð'Ñ'Ø#Ð�uÐ#Ð#Ñ#Ø€L�%ÐÐÑØ€IˆtÐÐÑØ#€L�#˜‘*Ð#Ð#Ñ#Ø €L�#˜‘*Ð Ð Ñ Ø+,€L�#˜˜Sœ	‘/ DÑ(Ð,Ð,Ñ,Ø %Ð˜Ð%Ð%Ñ%Ø48€O�^ dÑ*¨TÑ1Ð8Ð8Ñ8Ø €N�DÐ Ð Ñ Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø€HˆdÐÐÑØ(+Ð˜% #™+Ð+Ð+Ñ+Ø"%€N�E˜C‘KÐ%Ð%Ñ%Ø'*Ð˜ ™Ð*Ð*Ñ*ð *.Ð˜% $™,Ð-Ð-Ñ-à€Hˆc�D‰jÐÐÑð €M�4ÐÐÑð<ð <ð <ð <ð <ðð ð ð ð ð ð r;   r
   N)
Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úmodeling_rope_utilsr   Úutilsr   r
   Ú__all__r4   r;   r:   ú<module>rT      s·   ðð( /Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø #Ð #Ð #Ð #Ð #Ð #ð €ÐIÐJÑJÔJØðfð fð fð fð fÐ(ñ fô fñ „ñ KÔJðfðR Ð
€€€r;   