§
    ‚ŠtjÒo  ã                   ó  — d dl Z d dlmZ d dlmZ d dlZd dlmZ d dlm	Z	 d dlm
Z
mZmZ ddlmZ ddlmZ dd	lmZmZmZmZ dd
lmZ ddlmZ ddlmZmZmZ ddlm Z  ddl!m"Z"m#Z#m$Z$ ddl%m&Z& ddl'm(Z(m)Z)  ej*        e+¦  «        Z, ed¬¦  «        e	 G d„ de¦  «        ¦   «         ¦   «         Z-e G d„ de¦  «        ¦   «         Z.e G d„ de¦  «        ¦   «         Z/ G d„ dej0        ¦  «        Z1e G d„ de)¦  «        ¦   «         Z2 ed¬¦  «         G d „ d!e(¦  «        ¦   «         Z3 G d"„ d#e&¦  «        Z4e G d$„ d%e2¦  «        ¦   «         Z5 ed&¬¦  «         G d'„ d(e2¦  «        ¦   «         Z6 ed)¬¦  «         G d*„ d+e2¦  «        ¦   «         Z7g d,¢Z8dS )-é    N)Ú	dataclass)ÚLiteral)Ústrict)ÚBCEWithLogitsLossÚCrossEntropyLossÚMSELossé   )Úinitialization)ÚPreTrainedConfig)ÚBaseModelOutputÚMaskedLMOutputÚSequenceClassifierOutputÚTokenClassifierOutput)ÚPreTrainedModel)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚlogging)Úcan_return_tupleé   )ÚCONFIG_MAPPINGÚ
AutoConfigÚ	AutoModel)ÚModernBertPredictionHead)ÚSmolVLMModelÚSmolVLMPreTrainedModelúModernVBERT/modernvbert)Ú
checkpointc                   óè   ‡ — e Zd ZU dZdZeedœZdZee	z  dz  e
d<   dZee	z  dz  e
d<   dZee
d<   d	Zee
d
<   dZee
d<   dZee
d<   dZed         e
d<   dZeez  e
d<   dZee
d<   dZee
d<   ˆ fd„Zˆ xZS )ÚModernVBertConfiga?  
    pixel_shuffle_factor (`int | None`, *optional*, defaults to 4):
        Scale factor used by any pixel-shuffle / upsampling operations in the vision head.
    initializer_cutoff_factor (`float | None`, *optional*, defaults to 2.0):
        The cutoff factor for the truncated_normal_initializer for initializing all weight matrices.
    classifier_pooling (`Literal["cls", "mean"]`, *optional*, defaults to `"cls"`):
        The pooling strategy to use for classification tasks.
    classifier_bias (`bool | None`, *optional*, defaults to `False`):
        Whether to add a bias term to the classification head

    Example:
    ```python
    >>> from transformers import ModernVBertConfig

    >>> # Initializing configuration
    >>> configuration = ModernVBertConfig()

    >>> # Initializing a model from the configuration (model class is implemented in
    >>> # `modernvbert.modeling_modernvbert`)

    >>> from transformers import ModernVBertModel
    >>> model = ModernVBertModel(configuration)

    >>> # Accessing the model configuration
    >>> cfg = model.config
    ```Úmodernvbert)Útext_configÚvision_configNr"   r#   içÄ  Úimage_token_idé   Úpixel_shuffle_factorg{®Gáz”?Úinitializer_rangeç       @Úinitializer_cutoff_factorÚcls)r*   ÚmeanÚclassifier_poolingç        Úclassifier_dropoutFÚclassifier_biasÚtie_word_embeddingsc                 ó–  •— | j         €t          d         ¦   «         | _         n6t          | j         t          ¦  «        rt          d         di | j         ¤Ž| _         | j        €t          d         ¦   «         | _        n6t          | j        t          ¦  «        rt          d         di | j        ¤Ž| _         t          ¦   «         j        di |¤Ž d S )NÚ
modernbertÚsiglip_vision_model© )r"   r   Ú
isinstanceÚdictr#   ÚsuperÚ__post_init__)ÚselfÚkwargsÚ	__class__s     €úq/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/modernvbert/modular_modernvbert.pyr8   zModernVBertConfig.__post_init__X   sÎ   ø€ ØÔÐ#Ý-¨lÔ;Ñ=Ô=ˆDÔÐÝ˜Ô(­$Ñ/Ô/ð 	PÝ-¨lÔ;ÐOÐO¸dÔ>NÐOÐOˆDÔàÔÐ%Ý!/Ð0EÔ!FÑ!HÔ!HˆDÔÐÝ˜Ô*­DÑ1Ô1ð 	]Ý!/Ð0EÔ!FÐ!\Ð!\ÈÔI[Ð!\Ð!\ˆDÔà�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'ó    )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typer   Úsub_configsr"   r   r6   Ú__annotations__r#   r$   Úintr&   r'   Úfloatr)   r,   r   r.   r/   Úboolr0   r8   Ú__classcell__©r;   s   @r<   r    r    ,   s  ø€ € € € € € ðð ð6 €JØ",¸zÐJÐJ€Kà26€KÐ! DÑ(¨4Ñ/Ð6Ð6Ñ6Ø48€MÐ# dÑ*¨TÑ1Ð8Ð8Ñ8Ø€N�CÐÐÑØ !Ð˜#Ð!Ð!Ñ!Ø#Ð�uÐ#Ð#Ñ#Ø'*Ð˜uÐ*Ð*Ñ*Ø16Ð˜ Ô.Ð6Ð6Ñ6Ø&)Ð˜ ™Ð)Ð)Ñ)Ø!€O�TÐ!Ð!Ñ!Ø %Ð˜Ð%Ð%Ñ%ð(ð (ð (ð (ð (ð (ð (ð (ð (r=   r    c                   óª   — e Zd ZU dZdZej        ed<   dZe	ej                 dz  ed<   dZ
e	ej                 dz  ed<   dZe	ej                 dz  ed<   dS )ÚModernVBertBaseModelOutputaY  
    Base class for ModernVBERT model's outputs.
    Args:
        last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
            Sequence of hidden-states at the output of the last layer of the model.
            If `past_key_values` is used only the last hidden-state of the sequences of shape `(batch_size, 1,
            hidden_size)` is output.
        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
            sequence_length)`.
            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
            heads.
        image_hidden_states (`tuple(torch.FloatTensor)`, *optional*):
            Tuple of `torch.FloatTensor` (one for the output of the image embeddings, `(batch_size, num_images,
            sequence_length, hidden_size)`.
            image_hidden_states of the model produced by the vision encoder
    NÚlast_hidden_stateÚhidden_statesÚ
attentionsÚimage_hidden_states)r>   r?   r@   rA   rL   ÚtorchÚFloatTensorrD   rM   ÚtuplerN   rO   r4   r=   r<   rK   rK   f   sŠ   € € € € € € ðð ð, ,0Ð�uÔ(Ð/Ð/Ñ/Ø59€M�5˜Ô*Ô+¨dÑ2Ð9Ð9Ñ9Ø26€J��eÔ'Ô(¨4Ñ/Ð6Ð6Ñ6Ø;?Ð˜˜uÔ0Ô1°DÑ8Ð?Ð?Ñ?Ð?Ð?r=   rK   c                   óÄ   — e Zd ZU dZdZej        dz  ed<   dZej        ed<   dZ	e
ej        df         dz  ed<   dZe
ej        df         dz  ed<   dZej        dz  ed<   dS )	ÚModernVBertMaskedLMOutputaG  
    Base class for ModernVBERT model's outputs with masked language modeling loss.
    Args:
        loss (`torch.FloatTensor`, *optional*, returned when `labels` is provided):
            Masked language modeling (MLM) loss.
        logits (`torch.FloatTensor`):
            Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
            sequence_length)`.
            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
            heads.
        image_hidden_states (`tuple(torch.FloatTensor)`, *optional*):
            Tuple of `torch.FloatTensor` (one for the output of the image embeddings, `(batch_size, num_images,
            sequence_length, hidden_size)`.
            image_hidden_states of the model produced by the vision encoder
    NÚlossÚlogits.rM   rN   rO   )r>   r?   r@   rA   rU   rP   rQ   rD   rV   rM   rR   rN   rO   r4   r=   r<   rT   rT   „   s¦   € € € € € € ðð ð, &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø $€FˆEÔÐ$Ð$Ñ$Ø:>€M�5˜Ô*¨CÐ/Ô0°4Ñ7Ð>Ð>Ñ>Ø7;€J��eÔ'¨Ð,Ô-°Ñ4Ð;Ð;Ñ;Ø48Ð˜Ô*¨TÑ1Ð8Ð8Ñ8Ð8Ð8r=   rT   c                   ó.   ‡ — e Zd ZdZˆ fd„Zd„ Zd„ Zˆ xZS )ÚModernVBertConnectorzê
    Connector module for ModernVBERT. It performs a pixel shuffle operation followed by a linear projection to match the text model's hidden size.
    Based on https://pytorch.org/docs/stable/generated/torch.nn.PixelShuffle.html
    c                 óÖ   •— t          ¦   «                              ¦   «          |j        | _        t          j        |j        j        |j        dz  z  |j        j        d¬¦  «        | _        d S )Nr   F©Úbias)	r7   Ú__init__r&   ÚnnÚLinearr#   Úhidden_sizer"   Úmodality_projection©r9   Úconfigr;   s     €r<   r\   zModernVBertConnector.__init__©   se   ø€ Ý‰Œ×ÒÑÔÐØ$*Ô$?ˆÔ!Ý#%¤9ØÔ Ô,°Ô0KÈQÑ0NÑOØÔÔ*Øð$
ñ $
ô $
ˆÔ Ð Ð r=   c                 ó  — |                      ¦   «         \  }}}t          |dz  ¦  «        x}}|                     ||||¦  «        }|                     ||t          ||z  ¦  «        ||z  ¦  «        }|                     dddd¦  «        }|                     |t          ||z  ¦  «        t          ||z  ¦  «        ||dz  z  ¦  «        }|                     dddd¦  «        }|                     |t          ||dz  z  ¦  «        ||dz  z  ¦  «        S )Ng      à?r   r   é   r	   )ÚsizerE   ÚviewÚpermuteÚreshape)r9   rO   r&   Ú
batch_sizeÚ
seq_lengthÚ	embed_dimÚheightÚwidths           r<   Úpixel_shufflez"ModernVBertConnector.pixel_shuffle²   s=  € Ø,?×,DÒ,DÑ,FÔ,FÑ)ˆ
�J 	Ý˜Z¨™_Ñ-Ô-Ð-ˆ�Ø1×6Ò6°zÀ6È5ÐR[Ñ\Ô\ÐØ1×6Ò6Ø˜¥ EÐ,@Ñ$@Ñ AÔ AÀ9ÐOcÑCcñ
ô 
Ðð 2×9Ò9¸!¸QÀÀ1ÑEÔEÐØ1×9Ò9ØÝ�Ð,Ñ,Ñ-Ô-Ý�Ð-Ñ-Ñ.Ô.ØÐ-¨qÑ0Ñ1ñ	
ô 
Ðð 2×9Ò9¸!¸QÀÀ1ÑEÔEÐØ"×*Ò*Ø�˜JÐ*>ÀÑ*AÑBÑCÔCÀYÐRfÐhiÑRiÑEjñ
ô 
ð 	
r=   c                 ób   — |                       || j        ¦  «        }|                      |¦  «        S ©N)rn   r&   r`   )r9   rO   s     r<   ÚforwardzModernVBertConnector.forwardÅ   s1   € Ø"×0Ò0Ð1DÀdÔF_Ñ`Ô`ÐØ×'Ò'Ð(;Ñ<Ô<Ð<r=   )r>   r?   r@   rA   r\   rn   rq   rH   rI   s   @r<   rX   rX   £   s`   ø€ € € € € ðð ð

ð 
ð 
ð 
ð 
ð
ð 
ð 
ð&=ð =ð =ð =ð =ð =ð =r=   rX   c                   óF   — e Zd ZeZg Z ej        ¦   «         d„ ¦   «         ZdS )ÚModernVBertPreTrainedModelc                 óŽ  ‡ — t          j        ‰ |¦  «         dt          j        dt          fˆ fd„}t          |t          ¦  «        rF‰ j        j        t          j
        d‰ j        j        j        z  ¦  «        z  } ||j        |¦  «         d S t          |t          ¦  «        rF‰ j        j        t          j
        d‰ j        j        j        z  ¦  «        z  } ||j        |¦  «         d S t          |t           t"          f¦  «        rC‰ j        j        t          j
        ‰ j        j        j        ¦  «        z  } ||j        |¦  «         d S d S )NÚmoduleÚstdc                 ó  •— t          ‰j        dd¦  «        }t          j        | j        d|| |z  ||z  ¬¦  «         t          | t          j        t          j        f¦  «        r"| j	        �t          j
        | j	        ¦  «         d S d S d S )Nr)   r(   r-   )r+   rv   ÚaÚb)Úgetattrrb   ÚinitÚtrunc_normal_Úweightr5   r]   r^   ÚConv2dr[   Úzeros_)ru   rv   Úcutoff_factorr9   s      €r<   Úinit_weightz=ModernVBertPreTrainedModel._init_weights.<locals>.init_weightÓ   s›   ø€ Ý# D¤KÐ1LÈcÑRÔRˆMÝÔØ”ØØØ �. 3Ñ&Ø #Ñ%ðñ ô ð õ ˜&¥2¤9­b¬iÐ"8Ñ9Ô9ð -Ø”;Ð*Ý”K ¤Ñ,Ô,Ð,Ð,Ð,ð-ð -Ø*Ð*r=   r(   )r   Ú_init_weightsr]   ÚModulerF   r5   rX   rb   r'   ÚmathÚsqrtr"   Únum_hidden_layersr`   ÚModernVBertForMaskedLMÚlm_headÚ$ModernVBertForSequenceClassificationÚ!ModernVBertForTokenClassificationr_   Ú
classifier)r9   ru   r�   Úout_stdÚfinal_out_stds   `    r<   r‚   z(ModernVBertPreTrainedModel._init_weightsÏ   sU  ø€ åÔ% d¨FÑ3Ô3Ð3ð	-¥¤	ð 	-µð 	-ð 	-ð 	-ð 	-ð 	-ð 	-õ �fÕ2Ñ3Ô3ð 	:Ø”kÔ3µd´iÀÀdÄkÔF]ÔFoÑ@oÑ6pÔ6pÑpˆGØˆK˜Ô2°GÑ<Ô<Ð<Ð<Ð<Ý˜Õ 6Ñ7Ô7ð 	:Ø”kÔ3µd´iÀÀdÄkÔF]ÔFoÑ@oÑ6pÔ6pÑpˆGØˆK˜œ¨Ñ0Ô0Ð0Ð0Ð0ÝØå4Ý1ðñ
ô 
ð 	:ð !œKÔ9½D¼IÀdÄkÔF]ÔFiÑ<jÔ<jÑjˆMØˆK˜Ô)¨=Ñ9Ô9Ð9Ð9Ð9ð	:ð 	:r=   N)	r>   r?   r@   r    Úconfig_classÚ_no_split_modulesrP   Úno_gradr‚   r4   r=   r<   rs   rs   Ê   s@   € € € € € à$€LØÐà€U„]�_„_ð:ð :ñ „_ð:ð :ð :r=   rs   aG  
    ModernVBertModel is a model that combines a vision encoder (SigLIP) and a text encoder (ModernBert).

    ModernVBert is the base model of the visual retriever ColModernVBert, and was introduced in the following paper:
    [*ModernVBERT: Towards Smaller Visual Document Retrievers*](https://arxiv.org/abs/2510.01149).
    )Úcustom_introc                   ó  ‡ — e Zd Zdefˆ fd„Ze edd¬¦  «        	 	 	 	 	 	 	 ddej        dej	        dz  d	ej        dz  d
ej
        dz  dej
        dz  dej        dz  dej
        dz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )ÚModernVBertModelrb   c                 ó„  •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          j        |j        ¦  «        | _        t	          j        |j        ¦  «        | _	        t          |j        j        |j        j        z  dz  |j        dz  z  ¦  «        | _        |                      ¦   «          d S )Nr   )r7   r\   rX   Ú	connectorr   Úfrom_configr"   Ú
text_modelr#   Úvision_modelrE   Ú
image_sizeÚ
patch_sizer&   Úimage_seq_lenÚ	post_initra   s     €r<   r\   zModernVBertModel.__init__û   s«   ø€ Ý‰Œ×Ò˜Ñ Ô Ð õ .¨fÑ5Ô5ˆŒÝ#Ô/°Ô0BÑCÔCˆŒÝ%Ô1°&Ô2FÑGÔGˆÔå ØÔ"Ô-°Ô1EÔ1PÑPÐUVÑVØÔ*¨AÑ-ñ/ñ
ô 
ˆÔð 	�ŠÑÔÐÐÐr=   áØ  
        Inputs fed to the model can have an arbitrary number of images. To account for this, pixel_values fed to
        the model have image padding -> (batch_size, max_num_images, 3, max_heights, max_widths) where
        max_num_images is the maximum number of images among the batch_size samples in the batch.
        Padding images are not needed beyond padding the pixel_values at the entrance of the model.
        For efficiency, we only pass through the vision_model's forward the real images by
        discarding the padding images i.e. pixel_values of size (image_batch_size, 3, height, width) where
        image_batch_size would be 7 when num_images_per_sample=[1, 3, 1, 2] and max_num_images would be 3.
        r   ©r‘   r   NÚ	input_idsÚattention_maskÚposition_idsÚinputs_embedsÚpixel_valuesÚpixel_attention_maskrO   r:   Úreturnc                 ó’  — |€: | j                              ¦   «         |¦  «                             |j        ¦  «        }|�|                      ||¬¦  «        j        }|�9|                     |j        |j        ¬¦  «        }|                      |||¬¦  «        } | j         d|||dœ|¤Ž}	t          |	j	        |	j
        |	j        |¬¦  «        S )a|  
        pixel_attention_mask (`torch.Tensor` of shape `(batch_size, image_size, image_size)`, *optional*):
            Mask to avoid performing attention on padding pixel indices.
        image_hidden_states (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The hidden states of the image encoder after modality projection.
        N)r£   r¤   )ÚdtypeÚdevice)rŸ   r¢   rO   )r¢   r    r¡   )rL   rM   rN   rO   r4   )r—   Úget_input_embeddingsÚtor¨   Úget_image_featuresÚpooler_outputr§   Úinputs_mergerrK   rL   rM   rN   )
r9   rŸ   r    r¡   r¢   r£   r¤   rO   r:   Úoutputss
             r<   rq   zModernVBertModel.forward  s  € ð> Ð ØB˜DœO×@Ò@ÑBÔBÀ9ÑMÔM×PÒPÐQZÔQaÑbÔbˆMð Ð#Ø"&×"9Ò"9Ø)Ð@Tð #:ñ #ô #äð  ð
 Ð*Ø"5×"8Ò"8¸}Ô?RÐ[hÔ[oÐ"8Ñ"pÔ"pÐØ ×.Ò.Ø#°=ÐVið /ñ ô ˆMð
 "�$”/ð 
Ø'Ø)Ø%ð
ð 
ð ð	
ð 
ˆõ *Ø%Ô7Ø!Ô/ØÔ)Ø 3ð	
ñ 
ô 
ð 	
r=   )NNNNNNN)r>   r?   r@   r    r\   r   r   rP   Ú
LongTensorÚTensorrQ   Ú
BoolTensorr   r   rR   rK   rq   rH   rI   s   @r<   r“   r“   ò   s@  ø€ € € € € ðÐ0ð ð ð ð ð ð ð  Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<ð/
ð /
àÔ#ð/
ð œ tÑ+ð/
ð Ô&¨Ñ-ð	/
ð
 Ô(¨4Ñ/ð/
ð Ô'¨$Ñ.ð/
ð $Ô.°Ñ5ð/
ð #Ô.°Ñ5ð/
ð Ð+Ô,ð/
ð 
Ð+Ñ	+ð/
ð /
ð /
ñô ñ Ôð/
ð /
ð /
ð /
ð /
r=   r“   c                   ó   — e Zd ZdS )ÚModernVBertPredictionHeadN)r>   r?   r@   r4   r=   r<   r³   r³   J  s   € € € € € Ø€Dr=   r³   c                   ó<  ‡ — e Zd ZddiZdefˆ fd„Zd„ Zd„ Ze e	dd¬	¦  «        	 	 	 	 	 	 	 	 dde
j        de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )r‡   zlm_head.weightz1model.text_model.embeddings.tok_embeddings.weightrb   c                 óX  •— t          ¦   «                              |¦  «         |j        j        | _        t	          |¦  «        | _        t          |j        ¦  «        | _        t          j	        |j        j
        | j        |j        j        ¬¦  «        | _        |                      ¦   «          d S )NrZ   )r7   r\   r"   Ú
vocab_sizer“   Úmodelr³   Úprojection_headr]   r^   r_   Údecoder_biasrˆ   rœ   ra   s     €r<   r\   zModernVBertForMaskedLM.__init__R  s‰   ø€ Ý‰Œ×Ò˜Ñ Ô Ð à Ô,Ô7ˆŒå% fÑ-Ô-ˆŒ
Ý8¸Ô9KÑLÔLˆÔÝ”y Ô!3Ô!?ÀÄÐW]ÔWiÔWvÐwÑwÔwˆŒð 	�ŠÑÔÐÐÐr=   c                 ó   — | j         S rp   ©rˆ   )r9   s    r<   Úget_output_embeddingsz,ModernVBertForMaskedLM.get_output_embeddings^  s
   € ØŒ|Ðr=   c                 ó   — || _         d S rp   r»   )r9   Únew_embeddingss     r<   Úset_output_embeddingsz,ModernVBertForMaskedLM.set_output_embeddingsa  s   € Ø%ˆŒˆˆr=   r�   r   rž   NrŸ   r    r¡   r¢   r£   r¤   rO   Úlabelsr:   r¥   c	                 óf  —  | j         d|||||||dœ|	¤Ž}
|
d         }|                      |                      |¦  «        ¦  «        }d}|�Ft          ¦   «         } ||                     d| j        ¦  «        |                     d¦  «        ¦  «        }t          |||
j        |
j        |
j	        ¬¦  «        S )á  
        pixel_attention_mask (`torch.Tensor` of shape `(batch_size, image_size, image_size)`, *optional*):
            Mask to avoid performing attention on padding pixel indices.
        image_hidden_states (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The hidden states of the image encoder after modality projection.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            text_config.]` or `model.image_token_id`. Tokens with indices set to `model.image_token_id` are
            ignored (masked), the loss is only computed for the tokens with labels in `[0, ..., text_config.]`.
        ©rŸ   r    r¡   r¢   r£   r¤   rO   r   Néÿÿÿÿ)rU   rV   rM   rN   rO   r4   )
r·   rˆ   r¸   r   rf   r¶   rT   rM   rN   rO   )r9   rŸ   r    r¡   r¢   r£   r¤   rO   rÀ   r:   r®   rM   rV   rU   Ú	criterions                  r<   rq   zModernVBertForMaskedLM.forwardd  sÛ   € ðH �$”*ð 	
ØØ)Ø%Ø'Ø%Ø!5Ø 3ð	
ð 	
ð ð	
ð 	
ˆð   œ
ˆà—’˜d×2Ò2°=ÑAÔAÑBÔBˆàˆØÐÝ(Ñ*Ô*ˆIØ�9˜VŸ[š[¨¨T¬_Ñ=Ô=¸v¿{º{È2¹¼ÑOÔOˆDå(ØØØ!Ô/ØÔ)Ø 'Ô ;ð
ñ 
ô 
ð 	
r=   ©NNNNNNNN)r>   r?   r@   Ú_tied_weights_keysr    r\   r¼   r¿   r   r   rP   r¯   r°   rQ   r±   r   r   rR   rT   rq   rH   rI   s   @r<   r‡   r‡   N  s  ø€ € € € € à*Ð,_Ð`Ðð
Ð0ð 
ð 
ð 
ð 
ð 
ð 
ðð ð ð&ð &ð &ð Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<Ø*.ð0
ð 0
àÔ#ð0
ð œ tÑ+ð0
ð Ô&¨Ñ-ð	0
ð
 Ô(¨4Ñ/ð0
ð Ô'¨$Ñ.ð0
ð $Ô.°Ñ5ð0
ð #Ô.°Ñ5ð0
ð Ô  4Ñ'ð0
ð Ð+Ô,ð0
ð 
Ð*Ñ	*ð0
ð 0
ð 0
ñô ñ Ôð0
ð 0
ð 0
ð 0
ð 0
r=   r‡   za
    The ModernVBert Model with a sequence classification head on top that performs pooling.
    c                   ó(  ‡ — e Zd Zdefˆ fd„Ze edd¬¦  «        	 	 	 	 	 	 	 	 ddej        dej	        dz  d	ej        dz  d
ej
        dz  dej
        dz  dej        dz  dej
        dz  dej        dz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )r‰   rb   c                 ó€  •— t          ¦   «                              |¦  «         |j        | _        || _        t	          |¦  «        | _        t          |j        ¦  «        | _        t          j
        |j        ¦  «        | _        t          j        |j        j        |j        ¦  «        | _        |                      ¦   «          d S rp   )r7   r\   Ú
num_labelsrb   r“   r·   r³   r"   Úheadr]   ÚDropoutr.   Údropr^   r_   r‹   rœ   ra   s     €r<   r\   z-ModernVBertForSequenceClassification.__init__ª  s•   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒØˆŒå% fÑ-Ô-ˆŒ
Ý-¨fÔ.@ÑAÔAˆŒ	Ý”J˜vÔ8Ñ9Ô9ˆŒ	Ýœ) FÔ$6Ô$BÀFÔDUÑVÔVˆŒð 	�ŠÑÔÐÐÐr=   r�   r   rž   NrŸ   r    r¡   r¢   r£   r¤   rO   rÀ   r:   r¥   c	                 óL  —  | j         d|||||||dœ|	¤Ž}
|
d         }| j        j        dk    r|dd…df         }n°| j        j        dk    r |�|j        dd…         \  }}n|j        dd…         \  }}|�|j        n|j        }|€#t          j        ||f|t
          j        ¬¦  «        }||                     d¦  «        z   	                    d	¬
¦  «        | 	                    d	d¬¦  «        z  }|  
                    |¦  «        }|                      |¦  «        }|                      |¦  «        }d}|��Z| j        j        €f| j        d	k    rd| j        _        nN| j        d	k    r7|j        t
          j        k    s|j        t
          j        k    rd| j        _        nd| j        _        | j        j        dk    rWt%          ¦   «         }| j        d	k    r1 ||                     ¦   «         |                     ¦   «         ¦  «        }nŽ |||¦  «        }n�| j        j        dk    rGt)          ¦   «         } ||                     d| j        ¦  «        |                     d¦  «        ¦  «        }n*| j        j        dk    rt-          ¦   «         } |||¦  «        }t/          |||
j        |
j        ¬¦  «        S )rÂ   rÃ   r   r*   Nr+   r   )r¨   r§   rÄ   rd   )ÚdimT)rÏ   ÚkeepdimÚ
regressionÚsingle_label_classificationÚmulti_label_classification©rU   rV   rM   rN   r4   )r·   rb   r,   Úshaper¨   rP   ÚonesrG   Ú	unsqueezeÚsumrË   rÍ   r‹   Úproblem_typerÊ   r§   ÚlongrE   r   Úsqueezer   rf   r   r   rM   rN   )r9   rŸ   r    r¡   r¢   r£   r¤   rO   rÀ   r:   r®   rL   ri   Úseq_lenr¨   Úpooled_outputrV   rU   Úloss_fcts                      r<   rq   z,ModernVBertForSequenceClassification.forward·  sè  € ðF �$”*ð 	
ØØ)Ø%Ø'Ø%Ø!5Ø 3ð	
ð 	
ð ð	
ð 	
ˆð $ AœJÐàŒ;Ô)¨UÒ2Ð2Ø 1°!°!°!°Q°$Ô 7ÐÐØŒ[Ô+¨vÒ5Ð5ØÐ(Ø&3Ô&9¸"¸1¸"Ô&=Ñ#�
˜G˜Gà&/¤o°b°q°bÔ&9Ñ#�
˜GØ)2Ð)>�YÔ%Ð%ÀMÔDXˆFàÐ%Ý!&¤¨Z¸Ð,AÈ&ÕX]ÔXbÐ!cÑ!cÔ!c�Ø!2°^×5MÒ5MÈbÑ5QÔ5QÑ!Q× VÒ VÐ[\Ð VÑ ]Ô ]Ð`n×`rÒ`rØ˜tð asñ aô añ !Ðð Ÿ	š	Ð"3Ñ4Ô4ˆØŸ	š	 -Ñ0Ô0ˆØ—’ Ñ/Ô/ˆàˆØÑØŒ{Ô'Ð/Ø”? aÒ'Ð'Ø/;�D”KÔ,Ð,Ø”_ qÒ(Ð(¨f¬l½e¼jÒ.HÐ.HÈFÌLÕ\aÔ\eÒLeÐLeØ/L�D”KÔ,Ð,à/K�D”KÔ,àŒ{Ô'¨<Ò7Ð7Ý"™9œ9�Ø”? aÒ'Ð'Ø#˜8 F§N¢NÑ$4Ô$4°f·n²nÑ6FÔ6FÑGÔG�D�Dà#˜8 F¨FÑ3Ô3�D�DØ”Ô)Ð-JÒJÐJÝ+Ñ-Ô-�Ø�x §¢¨B°´Ñ @Ô @À&Ç+Â+ÈbÁ/Ä/ÑRÔR��Ø”Ô)Ð-IÒIÐIÝ,Ñ.Ô.�Ø�x ¨Ñ/Ô/�å'ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r=   rÆ   )r>   r?   r@   r    r\   r   r   rP   r¯   r°   rQ   r±   r   r   rR   r   rq   rH   rI   s   @r<   r‰   r‰   ¤  sh  ø€ € € € € ðÐ0ð ð ð ð ð ð ð Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<Ø*.ðQ
ð Q
àÔ#ðQ
ð œ tÑ+ðQ
ð Ô&¨Ñ-ð	Q
ð
 Ô(¨4Ñ/ðQ
ð Ô'¨$Ñ.ðQ
ð $Ô.°Ñ5ðQ
ð #Ô.°Ñ5ðQ
ð Ô  4Ñ'ðQ
ð Ð+Ô,ðQ
ð 
Ð)Ñ	)ðQ
ð Q
ð Q
ñô ñ ÔðQ
ð Q
ð Q
ð Q
ð Q
r=   r‰   zw
    The ModernVBert Model with a token classification head on top, e.g. for Named Entity Recognition (NER) tasks.
    c                   ó(  ‡ — e Zd Zdefˆ fd„Ze edd¬¦  «        	 	 	 	 	 	 	 	 ddej        dej	        dz  d	ej        dz  d
ej
        dz  dej
        dz  dej        dz  dej
        dz  dej        dz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )rŠ   rb   c                 ór  •— t          ¦   «                              |¦  «         |j        | _        t          |¦  «        | _        t          |j        ¦  «        | _        t          j	        |j
        ¦  «        | _        t          j        |j        j        |j        ¦  «        | _        |                      ¦   «          d S rp   )r7   r\   rÊ   r“   r·   r³   r"   rË   r]   rÌ   r.   rÍ   r^   r_   r‹   rœ   ra   s     €r<   r\   z*ModernVBertForTokenClassification.__init__  sŽ   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒå% fÑ-Ô-ˆŒ
Ý-¨fÔ.@ÑAÔAˆŒ	Ý”J˜vÔ8Ñ9Ô9ˆŒ	Ýœ) FÔ$6Ô$BÀFÔDUÑVÔVˆŒð 	�ŠÑÔÐÐÐr=   r�   r   rž   NrŸ   r    r¡   r¢   r£   r¤   rO   rÀ   r:   r¥   c	                 óˆ  —  | j         d|||||||dœ|	¤Ž}
|
d         }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }d}|�Ft	          ¦   «         } ||                     d| j        ¦  «        |                     d¦  «        ¦  «        }t          |||
j        |
j	        ¬¦  «        S )rÂ   rÃ   r   NrÄ   rÔ   r4   )
r·   rË   rÍ   r‹   r   rf   rÊ   r   rM   rN   )r9   rŸ   r    r¡   r¢   r£   r¤   rO   rÀ   r:   r®   rL   rV   rU   rÞ   s                  r<   rq   z)ModernVBertForTokenClassification.forward*  sï   € ðH �$”*ð 	
ØØ)Ø%Ø'Ø%Ø!5Ø 3ð	
ð 	
ð ð	
ð 	
ˆð $ AœJÐà ŸIšIÐ&7Ñ8Ô8ÐØ ŸIšIÐ&7Ñ8Ô8ÐØ—’Ð!2Ñ3Ô3ˆàˆØÐÝ'Ñ)Ô)ˆHØ�8˜FŸKšK¨¨D¬OÑ<Ô<¸f¿kºkÈ"¹o¼oÑNÔNˆDå$ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r=   rÆ   )r>   r?   r@   r    r\   r   r   rP   r¯   r°   rQ   r±   r   r   rR   r   rq   rH   rI   s   @r<   rŠ   rŠ     sU  ø€ € € € € ð
Ð0ð 
ð 
ð 
ð 
ð 
ð 
ð Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<Ø*.ð1
ð 1
àÔ#ð1
ð œ tÑ+ð1
ð Ô&¨Ñ-ð	1
ð
 Ô(¨4Ñ/ð1
ð Ô'¨$Ñ.ð1
ð $Ô.°Ñ5ð1
ð #Ô.°Ñ5ð1
ð Ô  4Ñ'ð1
ð Ð+Ô,ð1
ð 
Ð&Ñ	&ð1
ð 1
ð 1
ñô ñ Ôð1
ð 1
ð 1
ð 1
ð 1
r=   rŠ   )r    rs   r“   r‡   r‰   rŠ   )9r„   Údataclassesr   Útypingr   rP   Útorch.nnr]   Úhuggingface_hub.dataclassesr   r   r   r   Ú r
   r{   Úconfiguration_utilsr   Úmodeling_outputsr   r   r   r   Úmodeling_utilsr   Úprocessing_utilsr   Úutilsr   r   r   Úutils.genericr   Úautor   r   r   Úmodernbert.modeling_modernbertr   Úsmolvlm.modeling_smolvlmr   r   Ú
get_loggerr>   Úloggerr    rK   rT   rƒ   rX   rs   r“   r³   r‡   r‰   rŠ   Ú__all__r4   r=   r<   ú<module>ró      sE  ðð €€€Ø !Ð !Ð !Ð !Ð !Ð !Ø Ð Ð Ð Ð Ð à €€€Ø Ð Ð Ð Ð Ð Ø .Ð .Ð .Ð .Ð .Ð .Ø AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ Aà &Ð &Ð &Ð &Ð &Ð &Ø 3Ð 3Ð 3Ð 3Ð 3Ð 3ðð ð ð ð ð ð ð ð ð ð ð ð .Ð -Ð -Ð -Ð -Ð -Ø &Ð &Ð &Ð &Ð &Ð &Ø @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ø -Ð -Ð -Ð -Ð -Ð -Ø 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ø EÐ EÐ EÐ EÐ EÐ EØ KÐ KÐ KÐ KÐ KÐ KÐ KÐ Kð 
ˆÔ	˜HÑ	%Ô	%€ð €Ð4Ð5Ñ5Ô5Øð5(ð 5(ð 5(ð 5(ð 5(Ð(ñ 5(ô 5(ñ „ñ 6Ô5ð5(ðp ð@ð @ð @ð @ð @ ñ @ô @ñ „ð@ð: ð9ð 9ð 9ð 9ð 9 ñ 9ô 9ñ „ð9ð<$=ð $=ð $=ð $=ð $=˜2œ9ñ $=ô $=ð $=ðN ð$:ð $:ð $:ð $:ð $:Ð!7ñ $:ô $:ñ „ð$:ðN €ððñ ô ðM
ð M
ð M
ð M
ð M
�|ñ M
ô M
ñô ðM
ð`	ð 	ð 	ð 	ð 	Ð 8ñ 	ô 	ð 	ð ðR
ð R
ð R
ð R
ð R
Ð7ñ R
ô R
ñ „ðR
ðj €ððñ ô ð
l
ð l
ð l
ð l
ð l
Ð+Eñ l
ô l
ñô ð
l
ð^ €ððñ ô ð
K
ð K
ð K
ð K
ð K
Ð(Bñ K
ô K
ñô ð
K
ð\ð ð €€€r=   