§
    ‚ŠtjP  ã                   ó  — d dl mZ ddlmZ ddlmZ ddlmZ ddlm	Z	 ddl
mZmZ dd	lmZmZmZmZmZ dd
lmZmZmZmZmZmZ  ej        e¦  «        Z ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z  G d„ de¦  «        Z! G d„ de¦  «        Z" G d„ de¦  «        Z# G d„ de¦  «        Z$ G d„ de¦  «        Z% G d„ d e¦  «        Z& G d!„ d"e¦  «        Z'g d#¢Z(d$S )%é    )Ústricté   )ÚPreTrainedConfig)ÚCausalLMOutputWithPast)ÚRopeParameters)ÚUnpack)Úauto_docstringÚloggingé   )ÚDeepseekV3DecoderLayerÚDeepseekV3MLPÚDeepseekV3MoEÚDeepseekV3PreTrainedModelÚDeepseekV3TopkRouter)ÚQwen3AttentionÚQwen3ForCausalLMÚ
Qwen3ModelÚQwen3RMSNormÚQwen3RotaryEmbeddingÚTransformersKwargszrednote-hilab/dots.llm1.base)Ú
checkpointc                   ó¼  ‡ — e Zd ZU dZdZdgZdddddddddddddddd	œZd
gdgfddgdgfdgdgfdœZdddddœZddiZ	dZ
eed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZed z  ed!<   d Zed z  ed"<   d Zed z  ed<   d#Zed z  ed$<   d#Zed z  ed%<   d Zed z  ed&<   d'Zed z  ed(<   d)Zed z  ed*<   d+Zeed,<   d-Zeed.<   d/Zeed0<   d1Z eed2<   d3Z!eed4<   d)Z"eed5<   d Z#e$e%z  d z  ed6<   d)Z&eed7<   d8Z'eez  d z  ed9<   d:Z(eed;<   d<Z)ed z  ed=<   dZ*ed z  ed><   d Z+e,e         d z  ed?<   d Z-ed z  ed@<   d Z.ed z  edA<   d Z/ee,e         z  d z  edB<   ˆ fdC„Z0ˆ xZ1S )DÚDots1Configa  
    n_group (`int`, *optional*, defaults to 1):
        Number of groups for routed experts.
    first_k_dense_replace (`int`, *optional*, defaults to 0):
        Number of dense layers at the beginning of the model before the first MoE layer.

    Examples:

    ```python
    >>> from transformers import Dots1Model, Dots1Config
    >>> # Initializing a Dots1 style configuration
    >>> configuration = Dots1Config()
    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```
    Údots1Úpast_key_valuesÚcolwiseÚrowwiseÚreplicated_with_grad_allreduceÚpacked_colwiseÚmoe_tp_experts)zlayers.*.self_attn.q_projzlayers.*.self_attn.k_projzlayers.*.self_attn.v_projzlayers.*.self_attn.o_projzlayers.*.self_attn.q_normzlayers.*.self_attn.k_normú!layers.*.mlp.experts.gate_up_projúlayers.*.mlp.experts.down_projúlayers.*.mlp.expertsz%layers.*.mlp.shared_experts.gate_projz#layers.*.mlp.shared_experts.up_projz%layers.*.mlp.shared_experts.down_projzlayers.*.mlp.gate_projzlayers.*.mlp.up_projzlayers.*.mlp.down_projÚ	input_idsÚinputs_embedsÚhidden_statesÚattention_mask)Úembed_tokensÚlayersÚnormÚ	ep_routerÚgrouped_gemm)zlayers.*.mlp.gater!   r"   r#   Únum_local_expertsÚn_routed_expertsi R Ú
vocab_sizei   Úhidden_sizeiÀ*  Úintermediate_sizei€  Úmoe_intermediate_sizeé>   Únum_hidden_layersé    Únum_attention_headsNÚnum_key_value_headsÚn_shared_expertsé   Ún_groupÚ
topk_groupÚnum_experts_per_tokr   Úfirst_k_dense_replaceFÚnorm_topk_probÚsiluÚ
hidden_acti   Úmax_position_embeddingsg{®Gáz”?Úinitializer_rangeg�íµ ÷Æ°>Úrms_norm_epsTÚ	use_cacheÚtie_word_embeddingsÚrope_parametersÚattention_biasg        Úattention_dropoutg      ð?Úrouted_scaling_factori   Úsliding_windowÚmax_window_layersÚlayer_typesÚpad_token_idÚbos_token_idÚeos_token_idc                 óº   •‡ — ‰ j         €‰ j        ‰ _         ‰ j        €%ˆ fd„t          ‰ j        ¦  «        D ¦   «         ‰ _         t          ¦   «         j        di |¤Ž d S )Nc                 ó<   •— g | ]}‰j         �|‰j        k    rdnd‘ŒS )NÚsliding_attentionÚfull_attention)rJ   rK   )Ú.0ÚiÚselfs     €úe/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/dots1/modular_dots1.pyú
<listcomp>z-Dots1Config.__post_init__.<locals>.<listcomp>…   sI   ø€ ð  ð  ð  ð ð Ô&Ð2°q¸DÔ<RÒ7RÐ7Rð $Ð#à%ð ð  ð  ó    © )r7   r6   rL   Úranger4   ÚsuperÚ__post_init__)rV   ÚkwargsÚ	__class__s   ` €rW   r]   zDots1Config.__post_init__€   s~   øø€ ØÔ#Ð+Ø'+Ô'?ˆDÔ$àÔÐ#ð ð  ð  ð  õ ˜tÔ5Ñ6Ô6ð	 ñ  ô  ˆDÔð 	�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'rY   )2Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚbase_model_tp_planÚbase_model_pp_planÚbase_model_ep_planÚattribute_mapr/   ÚintÚ__annotations__r0   r1   r2   r4   r6   r7   r8   r.   r:   r;   r<   r=   r>   Úboolr@   ÚstrrA   rB   ÚfloatrC   rD   rE   rF   r   ÚdictrG   rH   rI   rJ   rK   rL   ÚlistrM   rN   rO   r]   Ú__classcell__©r_   s   @rW   r   r   )   s?  ø€ € € € € € ðð ð" €JØ#4Ð"5Ðð &/Ø%.Ø%.Ø%.Ø%EØ%EØ-=Ø*3Ø 0Ø1:Ø/8Ø1:Ø"+Ø )Ø"+ðð Ðð& &˜¨Ð(9Ð:Ø#Ð%5Ð6¸Ð8IÐJØ!Ð" _Ð$5Ð6ðð Ðð )Ø-;Ø*8Ø 0ð	ð Ðð 	Ð/ð€Mð €J�ÐÐÑØ€K�ÐÐÑØ"Ð�sÐ"Ð"Ñ"Ø!%Ð˜3Ð%Ð%Ñ%ØÐ�sÐÐÑØ!Ð˜Ð!Ð!Ñ!Ø&(Ð˜˜t™Ð(Ð(Ñ(Ø#'Ð�c˜D‘jÐ'Ð'Ñ'Ø#'Ð�c˜D‘jÐ'Ð'Ñ'Ø€GˆS�4‰ZÐÐÑØ€J��d‘
ÐÐÑØ&*Ð˜˜t™Ð*Ð*Ñ*Ø()Ð˜3 ™:Ð)Ð)Ñ)Ø"'€N�D˜4‘KÐ'Ð'Ñ'Ø€J�ÐÐÑØ#'Ð˜SÐ'Ð'Ñ'Ø#Ð�uÐ#Ð#Ñ#Ø€L�%ÐÐÑØ€IˆtÐÐÑØ %Ð˜Ð%Ð%Ñ%Ø48€O�^ dÑ*¨TÑ1Ð8Ð8Ñ8Ø €N�DÐ Ð Ñ Ø,/Ð�u˜s‘{ TÑ)Ð/Ð/Ñ/Ø#&Ð˜5Ð&Ð&Ñ&Ø!%€N�C˜$‘JÐ%Ð%Ñ%Ø$&Ð�s˜T‘zÐ&Ð&Ñ&Ø$(€K��c”˜TÑ!Ð(Ð(Ñ(Ø#€L�#˜‘*Ð#Ð#Ñ#Ø#€L�#˜‘*Ð#Ð#Ñ#Ø+/€L�#˜˜Sœ	‘/ DÑ(Ð/Ð/Ñ/ð(ð (ð (ð (ð (ð (ð (ð (ð (rY   r   c                   ó   — e Zd ZdS )ÚDots1RMSNormN©r`   ra   rb   rZ   rY   rW   rt   rt   �   ó   € € € € € Ø€DrY   rt   c                   ó   — e Zd ZdS )ÚDots1RotaryEmbeddingNru   rZ   rY   rW   rx   rx   “   rv   rY   rx   c                   ó   — e Zd ZdS )ÚDots1AttentionNru   rZ   rY   rW   rz   rz   —   rv   rY   rz   c                   ó   — e Zd ZdS )ÚDots1MLPNru   rZ   rY   rW   r|   r|   ›   rv   rY   r|   c                   ó   — e Zd ZdS )ÚDots1TopkRouterNru   rZ   rY   rW   r~   r~   Ÿ   rv   rY   r~   c                   ó   — e Zd ZdS )ÚDots1MoENru   rZ   rY   rW   r€   r€   £   rv   rY   r€   c                   ó   — e Zd ZdS )ÚDots1DecoderLayerNru   rZ   rY   rW   r‚   r‚   §   rv   rY   r‚   c                   ó   — e Zd ZdZdS )ÚDots1PreTrainedModelN)r`   ra   rb   Ú"_keys_to_ignore_on_load_unexpectedrZ   rY   rW   r„   r„   «   s   € € € € € Ø)-Ð&Ð&Ð&rY   r„   c                   ó   — e Zd ZdS )Ú
Dots1ModelNru   rZ   rY   rW   r‡   r‡   ¯   rv   rY   r‡   c                   ó4   ‡ — e Zd Zdee         defˆ fd„Zˆ xZS )ÚDots1ForCausalLMÚsuper_kwargsÚreturnc                 ó6   •—  t          ¦   «         j        di |¤ŽS )a~  
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.

        Example:

        ```python
        >>> from transformers import AutoTokenizer, Dots1ForCausalLM

        >>> model = Dots1ForCausalLM.from_pretrained("rednote-hilab/dots1.llm1.inst")
        >>> tokenizer = AutoTokenizer.from_pretrained("rednote-hilab/dots1.llm1.inst")

        >>> prompt = "Hey, are you conscious? Can you talk to me?"
        >>> inputs = tokenizer(prompt, return_tensors="pt")

        >>> # Generate
        >>> generate_ids = model.generate(inputs.input_ids, max_length=30)
        >>> tokenizer.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
        "Hey, are you conscious? Can you talk to me?\nI'm not conscious, but I can talk to you."
        ```rZ   )r\   Úforward)rV   rŠ   r_   s     €rW   r�   zDots1ForCausalLM.forward´   s!   ø€ ð4 �u‰wŒwŒÐ.Ð. Ð.Ð.Ð.rY   )r`   ra   rb   r   r   r   r�   rq   rr   s   @rW   r‰   r‰   ³   sU   ø€ € € € € ð/àÐ1Ô2ð/ð 
 ð/ð /ð /ð /ð /ð /ð /ð /ð /ð /rY   r‰   )r   r„   r‡   r‰   N))Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úmodeling_outputsr   Úmodeling_rope_utilsr   Úprocessing_utilsr   Úutilsr	   r
   Ú deepseek_v3.modeling_deepseek_v3r   r   r   r   r   Úqwen3.modeling_qwen3r   r   r   r   r   r   Ú
get_loggerr`   Úloggerr   rt   rx   rz   r|   r~   r€   r‚   r„   r‡   r‰   Ú__all__rZ   rY   rW   ú<module>r™      s:  ðð /Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6Ø 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø &Ð &Ð &Ð &Ð &Ð &Ø ,Ð ,Ð ,Ð ,Ð ,Ð ,Ð ,Ð ,ðð ð ð ð ð ð ð ð ð ð ð ð ð ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð 
ˆÔ	˜HÑ	%Ô	%€ð €Ð9Ð:Ñ:Ô:Øða(ð a(ð a(ð a(ð a(Ð"ñ a(ô a(ñ „ñ ;Ô:ða(ðH	ð 	ð 	ð 	ð 	�<ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð/ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	�^ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	ˆ}ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð*ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	ˆ}ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð.ñ 	ô 	ð 	ð.ð .ð .ð .ð .Ð4ñ .ô .ð .ð	ð 	ð 	ð 	ð 	�ñ 	ô 	ð 	ð/ð /ð /ð /ð /Ð'ñ /ô /ð /ð<ð ð €€€rY   