§
    ‚ŠtjºX  ã                   óÚ  — d dl mZ d dlmZ d dlmZ ddlmZ ddlm	Z	m
Z
mZmZ  e
¦   «         rd dlmZmZ  ej        e¦  «        Z e	d¬	¦  «        e G d
„ de¦  «        ¦   «         ¦   «         Z e	d¬	¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z e	d¬	¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z e	d¬	¦  «        e G d„ de¦  «        ¦   «         ¦   «         Zg d¢ZdS )é    )ÚSequence)ÚAny)Ústricté   )ÚPreTrainedConfig)Úauto_docstringÚis_timm_availableÚloggingÚrequires_backends)ÚImageNetInfoÚinfer_imagenet_subsetzgoogle/gemma-3n-E4B)Ú
checkpointc                   ó¨  ‡ — e Zd ZU dZdZdgZdddddddddddœ
Zdgd	gfd
dgd
gfd
gd
gfdœZdZe	e
d<   dZe	e
d<   dZe	ee	         z  e
d<   dZe	e
d<   dZe	e
d<   dZe	e
d<   dZe	e
d<   dZee
d<   dZe	e
d<   dZee
d <   d!Zee
d"<   d#Zee
d$<   d%Ze	d&z  e
d'<   d(Ze	ee	         z  d&z  e
d)<   dZe	d&z  e
d*<   d#Zee
d+<   d&Zed&z  e
d,<   d-Z ee
d.<   d/Z!e	ez  d&z  e
d0<   d1Z"e	e
d2<   d&Z#ee         d&z  e
d3<   d4Z$ee
d5<   d6d7d8œZ%d9Z&e	e
d:<   dZ'e	e
d;<   d%Z(e	e
d<<   d=Z)ee
d><   d#Z*ee
d?<   d@Z+e	e
dA<   dBZ,e	e
dC<   dDZ-e	e
dE<   d&Z.eee         z  d&z  e
dF<   ˆ fdG„Z/dH„ Z0dI„ Z1ˆ xZ2S )JÚGemma3nTextConfigaL	  
    vocab_size_per_layer_input (`int`, *optional*, defaults to 262144):
        Vocabulary size of the per-layer text embeddings that augment the standard embeddings.
    hidden_size_per_layer_input (`int`, *optional*, defaults to 256):
        Dimension of the hidden representations for per-layer embeddings.
    altup_active_idx (`int`, *optional*, defaults to 0):
        The index of the prediction from which AltUp will compute additional predictions or correct the active prediction.
    altup_coef_clip (`float`, *optional*, defaults to 120.0):
        The maximum amplitude of an AltUp prediction or correction coefficient weight.
    altup_correct_scale (`bool`, *optional*, defaults to `True`):
        If True, apply the `AltUp.correct_output_scale` to the corrected prediction at `altup_active_idx`.
    altup_num_inputs (`int`, *optional*, defaults to 4):
        The number of predictions that AltUp should make given the input sequence.
    num_kv_shared_layers (`int`, *optional*, defaults to 15):
        The number of layers that share KV cache values. During the forward pass, the last `num_kv_shared_layers`
        layers in the model "share" the KV values in that each local and global layer in this range uses the KV
        cache values computed for the last local or global layer, respectively, before entering this range. The
        value should be a multiple of the attention pattern size (see `layer_types` parameter).
    laurel_rank (`int`, *optional*, defaults to 64):
        The intermediate size for the linear projections in the Learned Augmented Residual Layer.
    activation_sparsity_pattern (`Sequence[float]`, *optional*):
        The sparsity factor used to extract the top-k activations for a given layer. The provided Sequence must
        explicitly provide a sparsity value for each layer in the model. By default, the first 10 layers are
        sparse with a sparsity factor of 0.95 and the rest are dense.

    ```python
    >>> from transformers import Gemma3nTextModel, Gemma3nTextConfig

    >>> # Initializing a Gemma3nText gemma3n_text-E4B style configuration
    >>> configuration = Gemma3nTextConfig()

    >>> # Initializing a model from the gemma3n_text-E4B style configuration
    >>> model = Gemma3nTextModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```
    Úgemma3n_textÚpast_key_valuesÚcolwiseÚreplicated_with_grad_allreduceÚrowwise)
zlayers.*.self_attn.q_projzlayers.*.self_attn.k_projzlayers.*.self_attn.v_projzlayers.*.self_attn.q_normzlayers.*.self_attn.k_normzlayers.*.self_attn.v_normzlayers.*.self_attn.o_projzlayers.*.mlp.gate_projzlayers.*.mlp.up_projzlayers.*.mlp.down_projÚ	input_idsÚinputs_embedsÚhidden_statesÚattention_mask)Úembed_tokensÚlayersÚnormi  Ú
vocab_sizeé   Úhidden_sizei @  Úintermediate_sizeé#   Únum_hidden_layersé   Únum_attention_headsé   Únum_key_value_headsé   Úhead_dimÚgelu_pytorch_tanhÚhidden_activationi €  Úmax_position_embeddingsç{®Gáz”?Úinitializer_rangeç�íµ ÷Æ°>Úrms_norm_epsTÚ	use_cacher   NÚpad_token_idé   Úeos_token_idÚbos_token_idÚtie_word_embeddingsÚrope_parametersFÚattention_biasç        Úattention_dropouti   Úsliding_windowÚlayer_typesg      >@Úfinal_logit_softcappingg    €„.Ag     ˆÃ@)ÚglobalÚlocalé   Úvocab_size_per_layer_inputÚhidden_size_per_layer_inputÚaltup_active_idxg      ^@Úaltup_coef_clipÚaltup_correct_scaleé   Úaltup_num_inputsé   Únum_kv_shared_layersé@   Úlaurel_rankÚactivation_sparsity_patternc                 óh  •— t          | j        t          ¦  «        r:t          | j        ¦  «        x}| j        k    rt          d| j        › d|› d�¦  «        ‚t          | j        t          ¦  «        s| j        g| j        z  | _        | j        €#d„ t          | j        ¦  «        D ¦   «         | _        | j        €)| j        dk    rdnd}dg|z  dg| j        |z
  z  z   | _        t          | j        ¦  «        x}| j        k    rt          d	| j        › d|› d�¦  «        ‚ t          ¦   «         j
        d
i |¤Ž d S )Nzjintermediate_size must have an explicit intermediate size for every layer or one for all layers. Expected z values but got ú.c                 ó.   — g | ]}|d z   dz  dk    rdnd‘ŒS )r2   é   r   Úfull_attentionÚsliding_attention© )Ú.0Úis     úo/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/gemma3n/configuration_gemma3n.pyú
<listcomp>z3Gemma3nTextConfig.__post_init__.<locals>.<listcomp>�   s>   € ð  ð  ð  ØRS Q¨¡U¨a¡K°1Ò$4Ð$4Ð Ð Ð:Mð ð  ð  ó    é
   r   gffffffî?r8   zeactivation_sparsity_pattern must have an explicit activation sparsity value for every layer.Expected rR   )Ú
isinstancer    r   Úlenr"   Ú
ValueErrorr;   ÚrangerK   ÚsuperÚ__post_init__)ÚselfÚkwargsÚintsize_lenÚnum_sparse_layersÚlen_aspÚ	__class__s        €rU   r^   zGemma3nTextConfig.__post_init__ƒ   s«  ø€ å�tÔ-­xÑ8Ô8ð		Wå # DÔ$:Ñ ;Ô ;Ð;�ÀÔ@VÒVÐVåðSØ Ô2ðSð SØDOðSð Sð Sñô ð õ ˜DÔ2µHÑ=Ô=ð 	WØ&*Ô&<Ð%=ÀÔ@VÑ%VˆDÔ"àÔÐ#ð ð  ÝW\Ð]aÔ]sÑWtÔWtð ñ  ô  ˆDÔð Ô+Ð3Ø&*Ô&<¸rÒ&AÐ&A  ÀqÐØ04¨vÐ8IÑ/IÈSÈEØÔ&Ð):Ñ:ñMñ 0ˆDÔ,õ ˜4Ô;Ñ<Ô<Ð<ˆGÀÔAWÒWÐWÝðOØ Ô2ðOð OØDKðOð Oð Oñô ð ð
 	�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'rW   c                 ól   — | j         | j        z  dk    r t          d| j         › d| j        › d�¦  «        ‚dS )zOPart of `@strict`-powered validation. Validates the architecture of the config.r   zThe hidden size (z6) is not a multiple of the number of attention heads (z).N)r   r$   r[   )r_   s    rU   Úvalidate_architecturez'Gemma3nTextConfig.validate_architecture¢   s[   € àÔ˜dÔ6Ñ6¸!Ò;Ð;Ýð7 DÔ$4ð 7ð 7ØÔ2ð7ð 7ð 7ñô ð ð <Ð;rW   c                 ór  — |                      dd ¦  «        }ddiddidœ}| j        �| j        n|| _        |� | j        d                              |¦  «         | j                             d¦  «        €ddi| j        d<   | j        d                              d|                      d| j        d         ¦  «        ¦  «         | j                             d¦  «        €ddi| j        d<   | j        d                              d|                      d	| j        d
         ¦  «        ¦  «         |                      ¦   «          |S )NÚrope_scalingÚ	rope_typeÚdefault)rQ   rP   rP   Ú
rope_thetar=   rQ   Úrope_local_base_freqr>   )Úpopr6   ÚupdateÚgetÚ
setdefaultÚdefault_thetaÚstandardize_rope_params)r_   r`   rh   Údefault_rope_paramss       rU   Úconvert_rope_params_to_dictz-Gemma3nTextConfig.convert_rope_params_to_dictª   s_  € Ø—z’z .°$Ñ7Ô7ˆð
 #.¨yÐ!9Ø*¨IÐ6ð
ð 
Ðð 8<Ô7KÐ7W˜tÔ3Ð3Ð]pˆÔØÐ#ØÔ Ð!1Ô2×9Ò9¸,ÑGÔGÐGð Ô×#Ò#Ð$4Ñ5Ô5Ð=Ø6AÀ9Ð5MˆDÔ Ð!1Ñ2ØÔÐ-Ô.×9Ò9Ø˜&Ÿ*š* \°4Ô3EÀhÔ3OÑPÔPñ	
ô 	
ð 	
ð Ô×#Ò#Ð$7Ñ8Ô8Ð@Ø9DÀiÐ8PˆDÔ Ð!4Ñ5ØÔÐ0Ô1×<Ò<Ø˜&Ÿ*š*Ð%;¸TÔ=OÐPWÔ=XÑYÔYñ	
ô 	
ð 	
ð
 	×$Ò$Ñ&Ô&Ð&ØˆrW   )3Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚbase_model_tp_planÚbase_model_pp_planr   ÚintÚ__annotations__r   r    Úlistr"   r$   r&   r(   r*   Ústrr+   r-   Úfloatr/   r0   Úboolr1   r3   r4   r5   r6   Údictr7   r9   r:   r;   r<   rq   r@   rA   rB   rC   rD   rF   rH   rJ   rK   r^   rf   rt   Ú__classcell__©rd   s   @rU   r   r   $   s*  ø€ € € € € € ð%ð %ðN  €JØ#4Ð"5Ðà%.Ø%.Ø%.Ø%EØ%EØ%EØ%.Ø"+Ø )Ø"+ðð Ðð &˜¨Ð(9Ð:Ø#Ð%5Ð6¸Ð8IÐJØ!Ð" _Ð$5Ð6ðð Ðð €J�ÐÐÑØ€K�ÐÐÑØ)/Ð�s˜T #œY‘Ð/Ð/Ñ/ØÐ�sÐÐÑØ Ð˜Ð Ð Ñ Ø Ð˜Ð Ð Ñ Ø€HˆcÐÐÑØ0Ð�sÐ0Ð0Ñ0Ø#)Ð˜SÐ)Ð)Ñ)Ø#Ð�uÐ#Ð#Ñ#Ø€L�%ÐÐÑØ€IˆtÐÐÑØ €L�#˜‘*Ð Ð Ñ Ø+,€L�#˜˜Sœ	‘/ DÑ(Ð,Ð,Ñ,Ø €L�#˜‘*Ð Ð Ñ Ø $Ð˜Ð$Ð$Ñ$Ø#'€O�T˜D‘[Ð'Ð'Ñ'Ø €N�DÐ Ð Ñ Ø,/Ð�s˜U‘{ TÑ)Ð/Ð/Ñ/Ø€N�CÐÐÑØ$(€K��c”˜TÑ!Ð(Ð(Ñ(Ø%)Ð˜UÐ)Ð)Ñ)Ø*°XÐ>Ð>€MØ&-Ð Ð-Ð-Ñ-Ø'*Ð Ð*Ð*Ñ*ØÐ�cÐÐÑØ"€O�UÐ"Ð"Ñ"Ø $Ð˜Ð$Ð$Ñ$ØÐ�cÐÐÑØ "Ð˜#Ð"Ð"Ñ"Ø€K�ÐÐÑØ>BÐ ¨¨e¬Ñ!4°tÑ!;ÐBÐBÑBð(ð (ð (ð (ð (ð>ð ð ðð ð ð ð ð ð rW   r   c                   ó°  — e Zd ZU dZdZdZeed<   dZeed<   dZ	eed<   dZ
eed	<   d
Zeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZee         eeef         z  ed <   d!Zeed"<   d#Zeeeeef         eeef         f         z  ed$<   d%Zeeeeef         eeef         f         z  ed&<   d'S )(ÚGemma3nAudioConfiga‡  
    vocab_offset (`int`, *optional*, defaults to 262272):
        Offset between the tokenizer vocab index for the token ids embedded by `Gemma3nMultimodalEmbedder` and the
        0-indexed `Gemma3nMultimodalEmbedder.embedding` table.
    input_feat_size (`int`, *optional*, defaults to 128):
        The number of channels in each mel-spectrogram frame.
    gradient_clipping (`float`, *optional*, defaults to 10000000000.0):
        Clipping value used to stabilize extremely large gradient values.
    conf_attention_chunk_size (`int`, *optional*, defaults to 12):
        The sub-sequence size for local attention processing inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_attention_context_left (`int`, *optional*, defaults to 13):
        The left context size of the local attention inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_attention_context_right (`int`, *optional*, defaults to 0):
        The right context size of the local attention inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_attention_logit_cap (`float`, *optional*, defaults to 50.0):
        Logit cap applied during local attention inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_num_attention_heads (`int`, *optional*, defaults to 8):
        The number of attention heads in local attention inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_num_hidden_layers (`int`, *optional*, defaults to 12):
        The number of layers that use local attention inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_conv_kernel_size (`int`, *optional*, defaults to 5):
        Convolution kernel size for the conformer block inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_reduction_factor (`int`, *optional*, defaults to 4):
        Reduction factor used in the conformer block inside the Conformer ("conf") section of the
        Universal Speech Model.
    conf_residual_weight (`float`, *optional*, defaults to 0.5):
        Residual connection weight inside the Conformer ("conf") section of the
        Universal Speech Model.
    sscp_conv_channel_size (`tuple(int, int)`, *optional*, defaults to `(128, 32)`):
        The channel sizes for the first and second convolutional layers in the Sub-sample Convolution Projection
        ("sscp") section of the Universal Speech Model.
    sscp_conv_group_norm_eps (`float`, *optional*, defaults to 0.001):
        Epsilon used in group normalization in the subsample convolution projection in the Sub-sample Convolution
        Projection ("sscp") section of the Universal Speech Model.
    sscp_conv_kernel_size (`tuple(tuple(int, int), tuple(int, int))`, *optional*, defaults to `((3, 3), (3, 3))`):
        Kernel sizes of the two convolutional layers in the subsample convolution projection  in the Sub-sample
        Convolution Projection ("sscp") section of the Universal Speech Model. The kernel sizes are specified as a
        tuple of height and width for each layer, where the height corresponds to the time dimension and the width
        corresponds to the frequency dimension.
    sscp_conv_stride_size (`tuple(tuple(int, int), tuple(int, int))`, *optional*, defaults to `((2, 2), (2, 2))`):
        Stride sizes of the two convolutional layers in the subsample convolution projection in the Sub-sample
        Convolution Projection ("sscp") section of the Universal Speech Model. The stride sizes are specified as a
        tuple of height and width for each layer, where the height corresponds to the time dimension and the width
        corresponds to the frequency dimension.

    Example:

    ```python
    >>> from transformers import Gemma3nAudioConfig, Gemma3nAudioEncoder

    >>> # Initializing a Gemma3nAudioEncoder gemma3n_audio-E4B-style configuration
    >>> configuration = Gemma3nAudioConfig()

    >>> # Initializing a model from the gemma3n_audio-E4B style configuration
    >>> model = Gemma3nAudioEncoder(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```
    Úgemma3n_audioé€   r   é€  Úvocab_offsetÚinput_feat_sizei   r   r.   r/   g    _ BÚgradient_clippingé   Úconf_attention_chunk_sizeé   Úconf_attention_context_leftr   Úconf_attention_context_rightg      I@Úconf_attention_logit_capr#   Úconf_num_attention_headsÚconf_num_hidden_layersrO   Úconf_conv_kernel_sizerE   Úconf_reduction_factorg      à?Úconf_residual_weight)r‰   é    Ússcp_conv_channel_sizegü©ñÒMbP?Ússcp_conv_group_norm_eps)©r   r   rœ   Ússcp_conv_kernel_size)©r%   r%   rž   Ússcp_conv_stride_sizeN)ru   rv   rw   rx   ry   r   r}   r~   r‹   rŒ   r   r/   r�   r�   r�   r‘   r’   r“   r”   r•   r–   r—   r˜   rš   r   Útupler›   r�   rŸ   rR   rW   rU   r‡   r‡   È   sÉ  € € € € € € ðBð BðH !€Jà€J�ÐÐÑØ%€L�#Ð%Ð%Ñ%Ø€O�SÐÐÑØ€K�ÐÐÑØ€L�%ÐÐÑØ/Ð�uÐ/Ð/Ñ/Ø%'Ð˜sÐ'Ð'Ñ'Ø')Ð Ð)Ð)Ñ)Ø()Ð  #Ð)Ð)Ñ)Ø&*Ð˜eÐ*Ð*Ñ*Ø$%Ð˜cÐ%Ð%Ñ%Ø"$Ð˜CÐ$Ð$Ñ$Ø!"Ð˜3Ð"Ð"Ñ"Ø!"Ð˜3Ð"Ð"Ñ"Ø"%Ð˜%Ð%Ð%Ñ%Ø:CÐ˜D œI¨¨c°3¨h¬Ñ7ÐCÐCÑCØ&*Ð˜eÐ*Ð*Ñ*ðMÐ˜4 %¨¨c°3¨h¬¸¸sÀC¸x¼Ð(HÔ"IÑIð ð ñ ðMÐ˜4 %¨¨c°3¨h¬¸¸sÀC¸x¼Ð(HÔ"IÑIð ð ñ ð ð rW   r‡   c                   óä   ‡ — e Zd ZU dZdZdZeed<   dZe	ed<   dZ
eed<   d	Zed	z  ed
<   dZeed<   dZeed<   dZeed<   dZe	ed<   edeeef         fˆ fd„¦   «         Zdeeef         fˆ fd„Zˆ xZS )ÚGemma3nVisionConfiga³  
    architecture (`str`, *optional*, defaults to `"resnet50"`):
        The timm architecture to load.
    do_pooling (`bool`, *optional*, defaults to `True`):
        Whether to do pooling for the last_hidden_state in `TimmWrapperModel` or not.
    model_args (`dict[str, Any]`, *optional*):
        Additional keyword arguments to pass to the `timm.create_model` function. e.g. `model_args={"depth": 3}`
        for `timm/vit_base_patch32_clip_448.laion2b_ft_in12k_in1k` to create a model with 3 blocks. Defaults to `None`.
    vocab_offset (`int`, *optional*, defaults to 262144):
        Offset between the tokenizer vocab index for the token ids embedded by `Gemma3nMultimodalEmbedder` and the
        0-indexed `Gemma3nMultimodalEmbedder.embedding` table.

    Example:
    ```python
    >>> from transformers import Gemma3nVisionConfig, TimmWrapper

    >>> # Initializing a TimmWrapper gemma3n_vision-E4B-style configuration
    >>> configuration = Gemma3nVisionConfig()

    >>> # Initializing a gemma3n_vision-E4B-style TimmWrapper from the configuration
    >>> model = TimmWrapper(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```
    Úgemma3n_visionÚmobilenetv5_300m_encÚarchitecturer,   r-   FÚ
do_poolingNÚ
model_argsr   r   r‰   r   r?   r‹   r.   r/   Úconfig_dictc                 ó  •‡
— |                      ¦   «         }|                     d¦  «        }d|v pd|v }|€k|sit          | dg¦  «         t          |¦  «        }|rGt	          |¦  «        }|                     ¦   «         }|                     d¬¦  «        Š
ˆ
fd„|D ¦   «         }|�p|snt          t          |¦  «        ¦  «        |d<   t          t          |¦  «        ¦  «        t          |¦  «        k    rd„ t          |¦  «        D ¦   «         |d	<   nd |d	<   |                     dd ¦  «        }|                     d
d ¦  «        }	|p|	|d<   d|v r&d
|d         v r|d                              d
d ¦  «          t          ¦   «         j        |fi |¤ŽS )NÚlabel_namesÚ
num_labelsÚid2labelÚtimmT)Úas_dictc                 ó    •— g | ]
}‰|         ‘ŒS rR   rR   )rS   ÚsynsetÚlabel_descriptionss     €rU   rV   z1Gemma3nVisionConfig.from_dict.<locals>.<listcomp>e  s   ø€ ÐPÐPÐP¸fÐ1°&Ô9ÐPÐPÐPrW   c                 ó   — i | ]\  }}||“Œ	S rR   rR   )rS   rT   Únames      rU   ú
<dictcomp>z1Gemma3nVisionConfig.from_dict.<locals>.<dictcomp>l  s   € Ð%TÐ%TÐ%T±'°!°T d¨AÐ%TÐ%TÐ%TrW   Úlabel2idÚnum_classesÚpretrained_cfg)Úcopyro   r   r   r   rª   r±   rƒ   Ú	enumeraterZ   Úsetrm   r]   Ú	from_dict)Úclsr¨   r`   rª   Úis_custom_modelÚimagenet_subsetÚdataset_infoÚsynsetsÚnum_labels_in_kwargsÚnum_labels_in_dictr±   rd   s             @€rU   r»   zGemma3nVisionConfig.from_dictU  sÊ  øø€ ð "×&Ò&Ñ(Ô(ˆà!—o’o mÑ4Ô4ˆØ&¨&Ð0ÐH°JÀ&Ð4Hˆð Ð ÐÝ˜c F 8Ñ,Ô,Ð,Ý3°KÑ@Ô@ˆOØð QÝ+¨OÑ<Ô<�Ø&×2Ò2Ñ4Ô4�Ø%1×%DÒ%DÈTÐ%DÑ%RÔ%RÐ"ØPÐPÐPÐPÈÐPÑPÔP�àÐ"¨?Ð"Ý!%¥i°Ñ&<Ô&<Ñ!=Ô!=ˆF�:Ñõ •3�{Ñ#Ô#Ñ$Ô$­¨KÑ(8Ô(8Ò8Ð8Ø%TÐ%T½YÀ{Ñ=SÔ=SÐ%TÑ%TÔ%T��zÑ"Ð"à%)��zÑ"ð
  &Ÿzšz¨,¸Ñ=Ô=ÐØ(Ÿ_š_¨]¸DÑAÔAÐð  4ÐIÐ7Iˆˆ|Ñð ˜{Ð*Ð*¨}ÀÐL\Ô@]Ð/]Ð/]ØÐ(Ô)×-Ò-¨m¸TÑBÔBÐBà �u‰wŒwÔ  Ð7Ð7°Ð7Ð7Ð7rW   Úreturnc                 óJ  •— t          ¦   «                              ¦   «         }|                     d| j        ¦  «         |                     dt	          | j                             ¦   «         ¦  «        ¦  «         |                     dd ¦  «         |                     dd ¦  «         |S )Nr¶   rª   r¬   rµ   )r]   Úto_dictrp   r«   r   r¬   Úvaluesrm   )r_   Úoutputrd   s     €rU   rÅ   zGemma3nVisionConfig.to_dict€  s‡   ø€ Ý‘”—’Ñ"Ô"ˆØ×Ò˜-¨¬Ñ9Ô9Ð9Ø×Ò˜-­¨d¬m×.BÒ.BÑ.DÔ.DÑ)EÔ)EÑFÔFÐFØ�
Š
�:˜tÑ$Ô$Ð$Ø�
Š
�:˜tÑ$Ô$Ð$ØˆrW   )ru   rv   rw   rx   ry   r¥   r€   r~   r-   r�   r¦   r‚   r§   rƒ   r   r}   r   r‹   r/   Úclassmethodr   r»   rÅ   r„   r…   s   @rU   r¢   r¢   ,  s  ø€ € € € € € ðð ð6 "€JØ.€L�#Ð.Ð.Ñ.à#Ð�uÐ#Ð#Ñ#Ø€J�ÐÐÑØ"€J��t‘Ð"Ð"Ñ"Ø€K�ÐÐÑØ€J�ÐÐÑØ€L�#ÐÐÑØ€L�%ÐÐÑàð(8 D¨¨c¨¤Nð (8ð (8ð (8ð (8ð (8ñ „[ð(8ðT˜˜c 3˜hœð ð ð ð ð ð ð ð ð ð rW   r¢   c                   óˆ  ‡ — e Zd ZU dZdZeeedœZdZ	ee
eef         z  dz  ed<   dZee
eef         z  dz  ed<   dZee
eef         z  dz  ed<   dZedz  ed	<   d
Zedz  ed<   dZedz  ed<   dZedz  ed<   dZedz  ed<   dZedz  ed<   dZedz  ed<   dZedz  ed<   dZedz  ed<   dZedz  ed<   dZeed<   ˆ fd„Zˆ xZS )ÚGemma3nConfigaæ  
    audio_soft_tokens_per_image (`int`, *optional*, defaults to 188):
        The number of soft tokens per audio clip.
    vision_soft_tokens_per_image (`int`, *optional*, defaults to 256):
        The number of soft tokens per image.
    boi_token_id (`int`, *optional*, defaults to 255999):
        The begin-of-image token index to wrap the image prompt.
    eoi_token_id (`int`, *optional*, defaults to 262144):
        The end-of-image token index to wrap the image prompt.
    boa_token_id (`int`, *optional*, defaults to 256000):
        The begin-of-audio token index to wrap the audio prompt.
    eoa_token_id (`int`, *optional*, defaults to 262272):
        The end-of-audio token index to wrap the audio prompt.

    Example:

    ```python
    >>> from transformers import Gemma3nForConditionalGeneration, Gemma3nConfig, Gemma3nTextConfig

    >>> # Initializing a MobileNet vision config, which is loaded from TIMM
    >>> vision_config = Gemma3nVisionConfig()

    >>> # Initializing a Gemma3n Audio config
    >>> audio_config = Gemma3nAudioConfig()

    >>> # Initializing a Gemma3n Text config
    >>> text_config = Gemma3nTextConfig()

    >>> # Initializing a Gemma3n gemma-3-4b style configuration
    >>> configuration = Gemma3nConfig(text_config, vision_config, audio_config)

    >>> # Initializing a model from the gemma-3-4b style configuration
    >>> model = Gemma3nTextConfig(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úgemma3n)Útext_configÚvision_configÚaudio_configNrÌ   rÍ   rÎ   é¼   Úaudio_soft_tokens_per_imager'   Úvision_soft_tokens_per_imageiÿç Úboi_token_idr?   Úeoi_token_idi  Úimage_token_idi è Úboa_token_idrŠ   Úeoa_token_idi�  Úaudio_token_idr,   r-   Tr5   r0   c                 ó˜  •— | j         €.t          ¦   «         | _         t                               d¦  «         n0t	          | j         t
          ¦  «        rt          di | j         ¤Ž| _         t	          | j        t
          ¦  «        rt          di | j        ¤Ž| _        n4| j        €-t          ¦   «         | _        t                               d¦  «         t	          | j        t
          ¦  «        rt          di | j        ¤Ž| _        n4| j        €-t          ¦   «         | _        t                               d¦  «          t          ¦   «         j        di |¤Ž d S )NzAtext_config is None, using default Gemma3nTextConfig text config.zGvision_config is None, using default Gemma3nVisionConfig vision config.z7audio_config is None. Using default Gemma3nAudioConfig.rR   )rÌ   r   ÚloggerÚinforY   rƒ   rÍ   r¢   rÎ   r‡   r]   r^   )r_   r`   rd   s     €rU   r^   zGemma3nConfig.__post_init__È  s:  ø€ ØÔÐ#Ý0Ñ2Ô2ˆDÔÝ�KŠKÐ[Ñ\Ô\Ð\Ð\Ý˜Ô(­$Ñ/Ô/ð 	EÝ0ÐDÐD°4Ô3CÐDÐDˆDÔå�dÔ(­$Ñ/Ô/ð 	cÝ!4Ð!JÐ!J°tÔ7IÐ!JÐ!JˆDÔÐØÔÐ'Ý!4Ñ!6Ô!6ˆDÔÝ�KŠKÐaÑbÔbÐbå�dÔ'­Ñ.Ô.ð 	SÝ 2Ð GÐ G°TÔ5FÐ GÐ GˆDÔÐØÔÐ&Ý 2Ñ 4Ô 4ˆDÔÝ�KŠKÐQÑRÔRÐRà�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'rW   ) ru   rv   rw   rx   ry   r   r¢   r‡   Úsub_configsrÌ   rƒ   r€   r   r~   rÍ   rÎ   rÐ   r}   rÑ   rÒ   rÓ   rÔ   rÕ   rÖ   r×   r-   r�   r5   r‚   r0   r^   r„   r…   s   @rU   rÊ   rÊ   ‰  s©  ø€ € € € € € ð$ð $ðL €Jà(Ø,Ø*ðð €Kð >B€KÐ" T¨#¨s¨(¤^Ñ3°dÑ:ÐAÐAÑAØAE€MÐ&¨¨c°3¨h¬Ñ7¸$Ñ>ÐEÐEÑEØ?C€LÐ$ t¨C°¨H¤~Ñ5¸Ñ<ÐCÐCÑCØ.1Ð  t¡Ð1Ð1Ñ1Ø/2Ð  #¨¡*Ð2Ð2Ñ2Ø&€L�#˜‘*Ð&Ð&Ñ&Ø&€L�#˜‘*Ð&Ð&Ñ&Ø!(€N�C˜$‘JÐ(Ð(Ñ(Ø&€L�#˜‘*Ð&Ð&Ñ&Ø&€L�#˜‘*Ð&Ð&Ñ&Ø!(€N�C˜$‘JÐ(Ð(Ñ(Ø&*Ð�u˜t‘|Ð*Ð*Ñ*Ø'+Ð˜ ™Ð+Ð+Ñ+Ø€IˆtÐÐÑð(ð (ð (ð (ð (ð (ð (ð (ð (rW   rÊ   )r‡   rÊ   r   r¢   N)Úcollections.abcr   Útypingr   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   r	   r
   r   Ú	timm.datar   r   Ú
get_loggerru   rÙ   r   r‡   r¢   rÊ   Ú__all__rR   rW   rU   ú<module>rä      s&  ðð* %Ð $Ð $Ð $Ð $Ð $Ø Ð Ð Ð Ð Ð à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ Rð ÐÑÔð >Ø=Ð=Ð=Ð=Ð=Ð=Ð=Ð=à	ˆÔ	˜HÑ	%Ô	%€ð €Ð0Ð1Ñ1Ô1Øð_ð _ð _ð _ð _Ð(ñ _ô _ñ „ñ 2Ô1ð_ðD €Ð0Ð1Ñ1Ô1Øð_ð _ð _ð _ð _Ð)ñ _ô _ñ „ñ 2Ô1ð_ðD €Ð0Ð1Ñ1Ô1ØðXð Xð Xð Xð XÐ*ñ Xô Xñ „ñ 2Ô1ðXðv €Ð0Ð1Ñ1Ô1ØðP(ð P(ð P(ð P(ð P(Ð$ñ P(ô P(ñ „ñ 2Ô1ðP(ðf ^Ð
]Ð
]€€€rW   