§
    ‚Štjú‚  ã                   óŠ  — d dl Z d dlmZ d dlZd dlmZ d dlmZmZmZ ddl	m
Z ddlmZ ddlmZmZmZmZmZ ddlmZ dd	lmZ dd
lmZmZmZ ddlmZ ddlmZ ddl m!Z! e G d„ de¦  «        ¦   «         Z"e G d„ de¦  «        ¦   «         Z# G d„ dej$        ¦  «        Z%e G d„ de¦  «        ¦   «         Z& ed¬¦  «         G d„ de&¦  «        ¦   «         Z' G d„ dej$        ¦  «        Z(e G d„ de&¦  «        ¦   «         Z) ed ¬¦  «         G d!„ d"e&¦  «        ¦   «         Z* ed#¬¦  «         G d$„ d%e&¦  «        ¦   «         Z+g d&¢Z,dS )'é    N)Ú	dataclass)ÚBCEWithLogitsLossÚCrossEntropyLossÚMSELossé   )Úinitialization)ÚACT2FN)ÚBaseModelOutputÚBaseModelOutputWithPoolingÚMaskedLMOutputÚSequenceClassifierOutputÚTokenClassifierOutput)ÚPreTrainedModel)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚtorch_compilable_check)Úcan_return_tupleé   )Ú	AutoModelé   )ÚModernVBertConfigc                   óª   — e Zd ZU dZdZej        ed<   dZe	ej                 dz  ed<   dZ
e	ej                 dz  ed<   dZe	ej                 dz  ed<   dS )ÚModernVBertBaseModelOutputaY  
    Base class for ModernVBERT model's outputs.
    Args:
        last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
            Sequence of hidden-states at the output of the last layer of the model.
            If `past_key_values` is used only the last hidden-state of the sequences of shape `(batch_size, 1,
            hidden_size)` is output.
        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
            sequence_length)`.
            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
            heads.
        image_hidden_states (`tuple(torch.FloatTensor)`, *optional*):
            Tuple of `torch.FloatTensor` (one for the output of the image embeddings, `(batch_size, num_images,
            sequence_length, hidden_size)`.
            image_hidden_states of the model produced by the vision encoder
    NÚlast_hidden_stateÚhidden_statesÚ
attentionsÚimage_hidden_states)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   ÚtorchÚFloatTensorÚ__annotations__r   Útupler   r   © ó    úr/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/modernvbert/modeling_modernvbert.pyr   r   -   sŠ   € € € € € € ðð ð, ,0Ð�uÔ(Ð/Ð/Ñ/Ø59€M�5˜Ô*Ô+¨dÑ2Ð9Ð9Ñ9Ø26€J��eÔ'Ô(¨4Ñ/Ð6Ð6Ñ6Ø;?Ð˜˜uÔ0Ô1°DÑ8Ð?Ð?Ñ?Ð?Ð?r(   r   c                   óÄ   — e Zd ZU dZdZej        dz  ed<   dZej        ed<   dZ	e
ej        df         dz  ed<   dZe
ej        df         dz  ed<   dZej        dz  ed<   dS )	ÚModernVBertMaskedLMOutputaG  
    Base class for ModernVBERT model's outputs with masked language modeling loss.
    Args:
        loss (`torch.FloatTensor`, *optional*, returned when `labels` is provided):
            Masked language modeling (MLM) loss.
        logits (`torch.FloatTensor`):
            Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
            sequence_length)`.
            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
            heads.
        image_hidden_states (`tuple(torch.FloatTensor)`, *optional*):
            Tuple of `torch.FloatTensor` (one for the output of the image embeddings, `(batch_size, num_images,
            sequence_length, hidden_size)`.
            image_hidden_states of the model produced by the vision encoder
    NÚlossÚlogits.r   r   r   )r   r    r!   r"   r,   r#   r$   r%   r-   r   r&   r   r   r'   r(   r)   r+   r+   K   s¦   € € € € € € ðð ð, &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø $€FˆEÔÐ$Ð$Ñ$Ø:>€M�5˜Ô*¨CÐ/Ô0°4Ñ7Ð>Ð>Ñ>Ø7;€J��eÔ'¨Ð,Ô-°Ñ4Ð;Ð;Ñ;Ø48Ð˜Ô*¨TÑ1Ð8Ð8Ñ8Ð8Ð8r(   r+   c                   ó.   ‡ — e Zd ZdZˆ fd„Zd„ Zd„ Zˆ xZS )ÚModernVBertConnectorzê
    Connector module for ModernVBERT. It performs a pixel shuffle operation followed by a linear projection to match the text model's hidden size.
    Based on https://pytorch.org/docs/stable/generated/torch.nn.PixelShuffle.html
    c                 óÖ   •— t          ¦   «                              ¦   «          |j        | _        t          j        |j        j        |j        dz  z  |j        j        d¬¦  «        | _        d S )Nr   F©Úbias)	ÚsuperÚ__init__Úpixel_shuffle_factorÚnnÚLinearÚvision_configÚhidden_sizeÚtext_configÚmodality_projection©ÚselfÚconfigÚ	__class__s     €r)   r4   zModernVBertConnector.__init__p   se   ø€ Ý‰Œ×ÒÑÔÐØ$*Ô$?ˆÔ!Ý#%¤9ØÔ Ô,°Ô0KÈQÑ0NÑOØÔÔ*Øð$
ñ $
ô $
ˆÔ Ð Ð r(   c                 ó  — |                      ¦   «         \  }}}t          |dz  ¦  «        x}}|                     ||||¦  «        }|                     ||t          ||z  ¦  «        ||z  ¦  «        }|                     dddd¦  «        }|                     |t          ||z  ¦  «        t          ||z  ¦  «        ||dz  z  ¦  «        }|                     dddd¦  «        }|                     |t          ||dz  z  ¦  «        ||dz  z  ¦  «        S )Ng      à?r   r   r   r   )ÚsizeÚintÚviewÚpermuteÚreshape)r=   r   r5   Ú
batch_sizeÚ
seq_lengthÚ	embed_dimÚheightÚwidths           r)   Úpixel_shufflez"ModernVBertConnector.pixel_shuffley   s=  € Ø,?×,DÒ,DÑ,FÔ,FÑ)ˆ
�J 	Ý˜Z¨™_Ñ-Ô-Ð-ˆ�Ø1×6Ò6°zÀ6È5ÐR[Ñ\Ô\ÐØ1×6Ò6Ø˜¥ EÐ,@Ñ$@Ñ AÔ AÀ9ÐOcÑCcñ
ô 
Ðð 2×9Ò9¸!¸QÀÀ1ÑEÔEÐØ1×9Ò9ØÝ�Ð,Ñ,Ñ-Ô-Ý�Ð-Ñ-Ñ.Ô.ØÐ-¨qÑ0Ñ1ñ	
ô 
Ðð 2×9Ò9¸!¸QÀÀ1ÑEÔEÐØ"×*Ò*Ø�˜JÐ*>ÀÑ*AÑBÑCÔCÀYÐRfÐhiÑRiÑEjñ
ô 
ð 	
r(   c                 ób   — |                       || j        ¦  «        }|                      |¦  «        S ©N)rK   r5   r;   )r=   r   s     r)   ÚforwardzModernVBertConnector.forwardŒ   s1   € Ø"×0Ò0Ð1DÀdÔF_Ñ`Ô`ÐØ×'Ò'Ð(;Ñ<Ô<Ð<r(   )r   r    r!   r"   r4   rK   rN   Ú__classcell__©r?   s   @r)   r/   r/   j   s`   ø€ € € € € ðð ð

ð 
ð 
ð 
ð 
ð
ð 
ð 
ð&=ð =ð =ð =ð =ð =ð =r(   r/   c                   ó~   ‡ — e Zd ZU eed<   dZdZdZg ZdgZ	dZ
dZdZdZeZ ej        ¦   «         ˆ fd„¦   «         Zˆ xZS )ÚModernVBertPreTrainedModelr>   Úmodel)ÚimageÚtextTÚpast_key_valuesc                 ó¨  •‡ — t          ¦   «                              |¦  «         dt          j        dt          fˆ fd„}t          |t          ¦  «        rF‰ j        j        t          j
        d‰ j        j        j        z  ¦  «        z  } ||j        |¦  «         d S t          |t          ¦  «        rF‰ j        j        t          j
        d‰ j        j        j        z  ¦  «        z  } ||j        |¦  «         d S t          |t           t"          f¦  «        rC‰ j        j        t          j
        ‰ j        j        j        ¦  «        z  } ||j        |¦  «         d S d S )NÚmoduleÚstdc                 ó  •— t          ‰j        dd¦  «        }t          j        | j        d|| |z  ||z  ¬¦  «         t          | t          j        t          j        f¦  «        r"| j	        �t          j
        | j	        ¦  «         d S d S d S )NÚinitializer_cutoff_factorç       @ç        )ÚmeanrY   ÚaÚb)Úgetattrr>   ÚinitÚtrunc_normal_ÚweightÚ
isinstancer6   r7   ÚConv2dr2   Úzeros_)rX   rY   Úcutoff_factorr=   s      €r)   Úinit_weightz=ModernVBertPreTrainedModel._init_weights.<locals>.init_weight£   s›   ø€ Ý# D¤KÐ1LÈcÑRÔRˆMÝÔØ”ØØØ �. 3Ñ&Ø #Ñ%ðñ ô ð õ ˜&¥2¤9­b¬iÐ"8Ñ9Ô9ð -Ø”;Ð*Ý”K ¤Ñ,Ô,Ð,Ð,Ð,ð-ð -Ø*Ð*r(   r\   )r3   Ú_init_weightsr6   ÚModuleÚfloatre   r/   r>   Úinitializer_rangeÚmathÚsqrtr:   Únum_hidden_layersr;   ÚModernVBertForMaskedLMÚlm_headÚ$ModernVBertForSequenceClassificationÚ!ModernVBertForTokenClassificationr9   Ú
classifier)r=   rX   ri   Úout_stdÚfinal_out_stdr?   s   `    €r)   rj   z(ModernVBertPreTrainedModel._init_weightsŸ   s[  øø€ å‰Œ×Ò˜fÑ%Ô%Ð%ð	-¥¤	ð 	-µð 	-ð 	-ð 	-ð 	-ð 	-ð 	-õ �fÕ2Ñ3Ô3ð 	:Ø”kÔ3µd´iÀÀdÄkÔF]ÔFoÑ@oÑ6pÔ6pÑpˆGØˆK˜Ô2°GÑ<Ô<Ð<Ð<Ð<Ý˜Õ 6Ñ7Ô7ð 	:Ø”kÔ3µd´iÀÀdÄkÔF]ÔFoÑ@oÑ6pÔ6pÑpˆGØˆK˜œ¨Ñ0Ô0Ð0Ð0Ð0ÝØå4Ý1ðñ
ô 
ð 	:ð !œKÔ9½D¼IÀdÄkÔF]ÔFiÑ<jÔ<jÑjˆMØˆK˜Ô)¨=Ñ9Ô9Ð9Ð9Ð9ð	:ð 	:r(   )r   r    r!   r   r%   Úbase_model_prefixÚinput_modalitiesÚsupports_gradient_checkpointingÚ_no_split_modulesÚ_skip_keys_device_placementÚ_supports_flash_attnÚ_supports_sdpaÚ_supports_flex_attnÚ_supports_attention_backendÚconfig_classr#   Úno_gradrj   rO   rP   s   @r)   rR   rR   ‘   s•   ø€ € € € € € àÐÐÑØÐØ(ÐØ&*Ð#ØÐØ#4Ð"5ÐØÐØ€NØÐØ"&ÐØ$€Là€U„]�_„_ð:ð :ð :ð :ñ „_ð:ð :ð :ð :ð :r(   rR   aG  
    ModernVBertModel is a model that combines a vision encoder (SigLIP) and a text encoder (ModernBert).

    ModernVBert is the base model of the visual retriever ColModernVBert, and was introduced in the following paper:
    [*ModernVBERT: Towards Smaller Visual Document Retrievers*](https://arxiv.org/abs/2510.01149).
    ©Úcustom_introc                   óÐ  ‡ — e Zd ZdZdefˆ fd„Zd„ Zd„ Zdej	        dej
        dej
        fd	„Ze ed
¬¦  «        	 ddej        dej	        dz  dee         deez  fd„¦   «         ¦   «         Ze edd¬¦  «        	 	 	 	 	 	 	 ddej	        dej
        dz  dej	        dz  dej        dz  dej        dz  dej        dz  dej        dz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )ÚModernVBertModelz§
    A subclass of Idefics3Model. We do *not* remove or block the call to inputs_merger
    in forward. Instead, we override inputs_merger here with custom logic.
    r>   c                 óþ  •— t          ¦   «                              |¦  «         | j        j        j        | _        | j        j        j        | _        t          j        |j	        ¦  «        | _
        t          |¦  «        | _        t          j        |j        ¦  «        | _        t          |j	        j        |j	        j        z  dz  |j        dz  z  ¦  «        | _        | j        j        | _        |                      ¦   «          d S )Nr   )r3   r4   r>   r:   Úpad_token_idÚpadding_idxÚ
vocab_sizer   Úfrom_configr8   Úvision_modelr/   Ú	connectorÚ
text_modelrB   Ú
image_sizeÚ
patch_sizer5   Úimage_seq_lenÚimage_token_idÚ	post_initr<   s     €r)   r4   zModernVBertModel.__init__Ð   s×   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Øœ;Ô2Ô?ˆÔØœ+Ô1Ô<ˆŒÝ%Ô1°&Ô2FÑGÔGˆÔõ .¨fÑ5Ô5ˆŒÝ#Ô/°Ô0BÑCÔCˆŒå ØÔ"Ô-°Ô1EÔ1PÑPÐUVÑVØÔ*¨AÑ-ñ/ñ
ô 
ˆÔð #œkÔ8ˆÔà�ŠÑÔÐÐÐr(   c                 ó4   — | j                              ¦   «         S rM   )rŽ   Úget_input_embeddings©r=   s    r)   r•   z%ModernVBertModel.get_input_embeddingsâ   s   € ØŒ×3Ò3Ñ5Ô5Ð5r(   c                 ó:   — | j                              |¦  «         d S rM   )rŽ   Úset_input_embeddings)r=   Úvalues     r)   r˜   z%ModernVBertModel.set_input_embeddingså   s   € ØŒ×,Ò,¨UÑ3Ô3Ð3Ð3Ð3r(   Ú	input_idsÚinputs_embedsr   c                 ó0  — |j         \  }}}|€X| |                      ¦   «         t          j        | j        j        t          j        |j        ¬¦  «        ¦  «        k    }|d         }n|| j        j        k    }|                     d¬¦  «        }t          t          j
        ||z  dk    ¦  «        d¦  «         ||z  }t          j        j                             |                     d¬¦  «        dd¬	¦  «        }	|	dd
…         }
|                     d
¬¦  «        }|dz
  |z  }|dz
  |z  }|
                     d¦  «        |z   }t          j        |¦  «        }|||         ||         dd…f         ||<   t          j        |                     d
¦  «        ||¦  «        }|S )as  
        This method aims at merging the token embeddings with the image hidden states into one single sequence of vectors that are fed to the transformer LM.
        The merging happens as follows:
        - The text token sequence is: `tok_1 tok_2 tok_3 <fake_token_around_image> <image> <image> ... <image> <fake_token_around_image> tok_4`.
        - We get the image hidden states for the image through the vision encoder and that hidden state, after a pixel shuffle operation, is then projected into the text embedding space.
        We thus have a sequence of image hidden states of size (1, image_seq_len, hidden_dim), where 1 is for batch_size of 1 image and hidden_dim is the hidden_dim of the LM transformer.
        - The merging happens so that we obtain the following sequence: `vector_tok_1 vector_tok_2 vector_tok_3 vector_fake_tok_around_image {sequence of image_seq_len image hidden states} vector_fake_toke_around_image vector_tok_4`. That sequence is fed to the LM.
        - To fit the format of that sequence, `input_ids`, `inputs_embeds`, `attention_mask` are all 3 adapted to insert the image hidden states.
        N©ÚdtypeÚdevice).r   r   ©Údimr   zCAt least one sample has <image> tokens not divisible by patch_size.)r   r   )r™   éÿÿÿÿ)Úshaper•   r#   Útensorr>   r’   ÚlongrŸ   Úsumr   Úallr6   Ú
functionalÚpadÚcumsumÚ	unsqueezeÚ
zeros_likeÚwhere)r=   rš   r›   r   Ú_r�   Ú
image_maskÚnum_image_tokensÚblocks_per_sampleÚoffsetsÚblock_offsetÚrow_cumÚ	chunk_idxÚ	local_idxÚ	block_idxÚimage_embedsÚmerged_embedss                    r)   Úinputs_mergerzModernVBertModel.inputs_mergerè   s°  € ð /Ô4Ñˆˆ:�qàÐØ&Ð*E¨$×*CÒ*CÑ*EÔ*EÝ”˜Tœ[Ô7½u¼zÐR_ÔRfÐgÑgÔgñ+ô +ò ˆJð $ FÔ+ˆJˆJà" d¤kÔ&@Ò@ˆJà%Ÿ>š>¨a˜>Ñ0Ô0ÐÝÝŒIÐ&¨Ñ3°qÒ8Ñ9Ô9ØQñ	
ô 	
ð 	
ð -°
Ñ:Ðå”(Ô%×)Ò)Ð*;×*BÒ*BÀqÐ*BÑ*IÔ*IÈ6ÐYZÐ)Ñ[Ô[ˆØ˜s ˜s”|ˆØ×#Ò#¨Ð#Ñ+Ô+ˆØ˜q‘[ ZÑ/ˆ	Ø˜q‘[ JÑ.ˆ	Ø ×*Ò*¨1Ñ-Ô-°	Ñ9ˆ	åÔ'¨Ñ6Ô6ˆØ#6°yÀÔ7LÈiÐXbÔNcÐefÐefÐefÐ7fÔ#gˆ�ZÑ åœ J×$8Ò$8¸Ñ$<Ô$<¸lÈMÑZÔZˆØÐr(   zVEncodes images into continuous embeddings that can be forwarded to the language model.rƒ   NÚpixel_valuesÚpixel_attention_maskÚkwargsÚreturnc                 ó¨  ‡— ‰j         \  }}}}}‰                     | j        ¬¦  «        Š ‰j        ||z  g‰j         dd…         ¢R Ž Š‰j         dd…                              ¦   «         }	‰dk                         d¬¦  «        |	k    }
|
dxx         t          j        |
¦  «         z  cc<   ‰|
                              ¦   «         Š|€3t          j	        ˆfd	„d
D ¦   «         t          j
        ‰j        ¬¦  «        }n8 |j        ||z  g|j         dd…         ¢R Ž }||
                              ¦   «         }| j        j        j        }|                     d||¬¦  «        }|                     d||¬¦  «        }|                     d¬¦  «        dk     
                    ¦   «         } | j        d‰|ddœ|¤Ž}|j        }|                      |¦  «        }||_        |S )a4  
        pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The tensors corresponding to the input images.
        pixel_attention_mask (`torch.LongTensor`, *optional*):
            The attention mask indicating padded regions in the image.
        )rž   r   Nr   r]   )r¢   éþÿÿÿéýÿÿÿr    r   c                 ó*   •— g | ]}‰j         |         ‘ŒS r'   )r£   )Ú.0Úir»   s     €r)   ú
<listcomp>z7ModernVBertModel.get_image_features.<locals>.<listcomp>1  s!   ø€ Ð?Ð?Ð?°�lÔ(¨Ô+Ð?Ð?Ð?r(   )r   r   r   )rA   rž   rŸ   )Ú	dimensionrA   Ústep)r¢   rÀ   T)r»   Úpatch_attention_maskÚreturn_dictr'   )r£   Útorž   rC   Únumelr¦   r#   ÚanyÚ
contiguousÚonesÚboolrŸ   r>   r8   r�   ÚunfoldrŒ   r   r�   Úpooler_output)r=   r»   r¼   r½   rF   Ú
num_imagesÚnum_channelsrI   rJ   Únb_values_per_imageÚreal_images_indsr�   Úpatches_subgridrÈ   Úimage_outputsr   Úimage_featuress    `               r)   Úget_image_featuresz#ModernVBertModel.get_image_features  s7  ø€ ð  ?KÔ>PÑ;ˆ
�J ¨f°eØ#—’¨T¬Z�Ñ8Ô8ˆØ(�|Ô(¨°jÑ)@ÐZÀ<ÔCUÐVWÐVXÐVXÔCYÐZÐZÐZˆð +Ô0°°°Ô4×:Ò:Ñ<Ô<ÐØ(¨CÒ/×4Ò4¸Ð4ÑFÔFÐJ]Ò]Ðð 	˜ÐÐÔ¥¤	Ð*:Ñ ;Ô ;Ð;Ñ;ÐÐÑà#Ð$4Ô5×@Ò@ÑBÔBˆàÐ'Ý#(¤:Ø?Ð?Ð?Ð?°YÐ?Ñ?Ô?Ý”jØ#Ô*ð$ñ $ô $Ð Ð ð $=Ð#7Ô#<¸ZÈ*Ñ=TÐ#vÐWkÔWqÐrsÐrtÐrtÔWuÐ#vÐ#vÐ#vÐ Ø#7Ð8HÔ#I×#TÒ#TÑ#VÔ#VÐ Ø”[Ô.Ô9ˆ
Ø.×5Ò5ÀÈ
ÐYcÐ5ÑdÔdˆØ)×0Ò0¸1À:ÐT^Ð0Ñ_Ô_ˆØ /× 3Ò 3¸Ð 3Ñ AÔ AÀAÒ E×KÒKÑMÔMÐð *˜Ô)ð 
Ø%Ð<PÐ^bð
ð 
Øflð
ð 
ˆð ,Ô=Ðð ŸšÐ(;Ñ<Ô<ˆØ&4ˆÔ#àÐr(   áØ  
        Inputs fed to the model can have an arbitrary number of images. To account for this, pixel_values fed to
        the model have image padding -> (batch_size, max_num_images, 3, max_heights, max_widths) where
        max_num_images is the maximum number of images among the batch_size samples in the batch.
        Padding images are not needed beyond padding the pixel_values at the entrance of the model.
        For efficiency, we only pass through the vision_model's forward the real images by
        discarding the padding images i.e. pixel_values of size (image_batch_size, 3, height, width) where
        image_batch_size would be 7 when num_images_per_sample=[1, 3, 1, 2] and max_num_images would be 3.
        úModernVBERT/modernvbert©r„   Ú
checkpointÚattention_maskÚposition_idsc                 ó’  — |€: | j                              ¦   «         |¦  «                             |j        ¦  «        }|�|                      ||¬¦  «        j        }|�9|                     |j        |j        ¬¦  «        }|                      |||¬¦  «        } | j         d|||dœ|¤Ž}	t          |	j	        |	j
        |	j        |¬¦  «        S )a|  
        pixel_attention_mask (`torch.Tensor` of shape `(batch_size, image_size, image_size)`, *optional*):
            Mask to avoid performing attention on padding pixel indices.
        image_hidden_states (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The hidden states of the image encoder after modality projection.
        N)r»   r¼   r�   )rš   r›   r   )r›   rÞ   rß   )r   r   r   r   r'   )rŽ   r•   rÊ   rŸ   rÙ   rÑ   rž   rº   r   r   r   r   )
r=   rš   rÞ   rß   r›   r»   r¼   r   r½   Úoutputss
             r)   rN   zModernVBertModel.forwardJ  s  € ð> Ð ØB˜DœO×@Ò@ÑBÔBÀ9ÑMÔM×PÒPÐQZÔQaÑbÔbˆMð Ð#Ø"&×"9Ò"9Ø)Ð@Tð #:ñ #ô #äð  ð
 Ð*Ø"5×"8Ò"8¸}Ô?RÐ[hÔ[oÐ"8Ñ"pÔ"pÐØ ×.Ò.Ø#°=ÐVið /ñ ô ˆMð
 "�$”/ð 
Ø'Ø)Ø%ð
ð 
ð ð	
ð 
ˆõ *Ø%Ô7Ø!Ô/ØÔ)Ø 3ð	
ñ 
ô 
ð 	
r(   rM   )NNNNNNN)r   r    r!   r"   r   r4   r•   r˜   r#   Ú
LongTensorÚTensorrº   r   r   r$   r   r   r&   r   rÙ   Ú
BoolTensorr   rN   rO   rP   s   @r)   r†   r†   Â   s'  ø€ € € € € ðð ð
Ð0ð ð ð ð ð ð ð$6ð 6ð 6ð4ð 4ð 4ð(ØÔ)ð(Ø:?¼,ð(Ø]bÔ]ið(ð (ð (ð (ðT Ø€^Ømðñ ô ð 9=ð2ð 2àÔ'ð2ð $Ô.°Ñ5ð2ð Ð+Ô,ð	2ð
 
Ð+Ñ	+ð2ð 2ð 2ñô ñ Ôð2ðh Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<ð/
ð /
àÔ#ð/
ð œ tÑ+ð/
ð Ô&¨Ñ-ð	/
ð
 Ô(¨4Ñ/ð/
ð Ô'¨$Ñ.ð/
ð $Ô.°Ñ5ð/
ð #Ô.°Ñ5ð/
ð Ð+Ô,ð/
ð 
Ð+Ñ	+ð/
ð /
ð /
ñô ñ Ôð/
ð /
ð /
ð /
ð /
r(   r†   c                   óH   ‡ — e Zd Zdefˆ fd„Zdej        dej        fd„Zˆ xZS )ÚModernVBertPredictionHeadr>   c                 ó.  •— t          ¦   «                              ¦   «          || _        t          j        |j        |j        |j        ¦  «        | _        t          |j	                 | _
        t          j        |j        |j        |j        ¬¦  «        | _        d S )N)Úepsr2   )r3   r4   r>   r6   r7   r9   Úclassifier_biasÚdenser	   Úclassifier_activationÚactÚ	LayerNormÚnorm_epsÚ	norm_biasÚnormr<   s     €r)   r4   z"ModernVBertPredictionHead.__init__Š  sq   ø€ Ý‰Œ×ÒÑÔÐØˆŒÝ”Y˜vÔ1°6Ô3EÀvÔG]Ñ^Ô^ˆŒ
Ý˜&Ô6Ô7ˆŒÝ”L Ô!3¸¼ÈvÔO_Ð`Ñ`Ô`ˆŒ	ˆ	ˆ	r(   r   r¾   c                 óx   — |                       |                      |                      |¦  «        ¦  «        ¦  «        S rM   )rð   rì   rê   )r=   r   s     r)   rN   z!ModernVBertPredictionHead.forward‘  s,   € Ø�yŠy˜Ÿš $§*¢*¨]Ñ";Ô";Ñ<Ô<Ñ=Ô=Ð=r(   )	r   r    r!   r   r4   r#   rã   rN   rO   rP   s   @r)   ræ   ræ   ‰  sr   ø€ € € € € ðaÐ0ð að að að að að að> U¤\ð >°e´lð >ð >ð >ð >ð >ð >ð >ð >r(   ræ   c                   ó<  ‡ — e Zd ZddiZdefˆ fd„Zd„ Zd„ Ze e	dd¬	¦  «        	 	 	 	 	 	 	 	 dde
j        de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  de
j        d
z  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )rq   zlm_head.weightz1model.text_model.embeddings.tok_embeddings.weightr>   c                 óX  •— t          ¦   «                              |¦  «         |j        j        | _        t	          |¦  «        | _        t          |j        ¦  «        | _        t          j	        |j        j
        | j        |j        j        ¬¦  «        | _        |                      ¦   «          d S )Nr1   )r3   r4   r:   rŠ   r†   rS   ræ   Úprojection_headr6   r7   r9   Údecoder_biasrr   r“   r<   s     €r)   r4   zModernVBertForMaskedLM.__init__™  s‰   ø€ Ý‰Œ×Ò˜Ñ Ô Ð à Ô,Ô7ˆŒå% fÑ-Ô-ˆŒ
Ý8¸Ô9KÑLÔLˆÔÝ”y Ô!3Ô!?ÀÄÐW]ÔWiÔWvÐwÑwÔwˆŒð 	�ŠÑÔÐÐÐr(   c                 ó   — | j         S rM   ©rr   r–   s    r)   Úget_output_embeddingsz,ModernVBertForMaskedLM.get_output_embeddings¥  s
   € ØŒ|Ðr(   c                 ó   — || _         d S rM   r÷   )r=   Únew_embeddingss     r)   Úset_output_embeddingsz,ModernVBertForMaskedLM.set_output_embeddings¨  s   € Ø%ˆŒˆˆr(   rÚ   rÛ   rÜ   Nrš   rÞ   rß   r›   r»   r¼   r   Úlabelsr½   r¾   c	                 óf  —  | j         d|||||||dœ|	¤Ž}
|
d         }|                      |                      |¦  «        ¦  «        }d}|�Ft          ¦   «         } ||                     d| j        ¦  «        |                     d¦  «        ¦  «        }t          |||
j        |
j        |
j	        ¬¦  «        S )á  
        pixel_attention_mask (`torch.Tensor` of shape `(batch_size, image_size, image_size)`, *optional*):
            Mask to avoid performing attention on padding pixel indices.
        image_hidden_states (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The hidden states of the image encoder after modality projection.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            text_config.]` or `model.image_token_id`. Tokens with indices set to `model.image_token_id` are
            ignored (masked), the loss is only computed for the tokens with labels in `[0, ..., text_config.]`.
        ©rš   rÞ   rß   r›   r»   r¼   r   r   Nr¢   )r,   r-   r   r   r   r'   )
rS   rr   rô   r   rC   rŠ   r+   r   r   r   )r=   rš   rÞ   rß   r›   r»   r¼   r   rü   r½   rá   r   r-   r,   Ú	criterions                  r)   rN   zModernVBertForMaskedLM.forward«  sÛ   € ðH �$”*ð 	
ØØ)Ø%Ø'Ø%Ø!5Ø 3ð	
ð 	
ð ð	
ð 	
ˆð   œ
ˆà—’˜d×2Ò2°=ÑAÔAÑBÔBˆàˆØÐÝ(Ñ*Ô*ˆIØ�9˜VŸ[š[¨¨T¬_Ñ=Ô=¸v¿{º{È2¹¼ÑOÔOˆDå(ØØØ!Ô/ØÔ)Ø 'Ô ;ð
ñ 
ô 
ð 	
r(   ©NNNNNNNN)r   r    r!   Ú_tied_weights_keysr   r4   rø   rû   r   r   r#   râ   rã   r$   rä   r   r   r&   r+   rN   rO   rP   s   @r)   rq   rq   •  s  ø€ € € € € à*Ð,_Ð`Ðð
Ð0ð 
ð 
ð 
ð 
ð 
ð 
ðð ð ð&ð &ð &ð Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<Ø*.ð0
ð 0
àÔ#ð0
ð œ tÑ+ð0
ð Ô&¨Ñ-ð	0
ð
 Ô(¨4Ñ/ð0
ð Ô'¨$Ñ.ð0
ð $Ô.°Ñ5ð0
ð #Ô.°Ñ5ð0
ð Ô  4Ñ'ð0
ð Ð+Ô,ð0
ð 
Ð*Ñ	*ð0
ð 0
ð 0
ñô ñ Ôð0
ð 0
ð 0
ð 0
ð 0
r(   rq   za
    The ModernVBert Model with a sequence classification head on top that performs pooling.
    c                   ó(  ‡ — e Zd Zdefˆ fd„Ze edd¬¦  «        	 	 	 	 	 	 	 	 ddej        dej	        dz  d	ej        dz  d
ej
        dz  dej
        dz  dej        dz  dej
        dz  dej        dz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )rs   r>   c                 ó€  •— t          ¦   «                              |¦  «         |j        | _        || _        t	          |¦  «        | _        t          |j        ¦  «        | _        t          j
        |j        ¦  «        | _        t          j        |j        j        |j        ¦  «        | _        |                      ¦   «          d S rM   )r3   r4   Ú
num_labelsr>   r†   rS   ræ   r:   Úheadr6   ÚDropoutÚclassifier_dropoutÚdropr7   r9   ru   r“   r<   s     €r)   r4   z-ModernVBertForSequenceClassification.__init__ñ  s•   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒØˆŒå% fÑ-Ô-ˆŒ
Ý-¨fÔ.@ÑAÔAˆŒ	Ý”J˜vÔ8Ñ9Ô9ˆŒ	Ýœ) FÔ$6Ô$BÀFÔDUÑVÔVˆŒð 	�ŠÑÔÐÐÐr(   rÚ   rÛ   rÜ   Nrš   rÞ   rß   r›   r»   r¼   r   rü   r½   r¾   c	                 óL  —  | j         d|||||||dœ|	¤Ž}
|
d         }| j        j        dk    r|dd…df         }n°| j        j        dk    r |�|j        dd…         \  }}n|j        dd…         \  }}|�|j        n|j        }|€#t          j        ||f|t
          j        ¬¦  «        }||                     d¦  «        z   	                    d	¬
¦  «        | 	                    d	d¬¦  «        z  }|  
                    |¦  «        }|                      |¦  «        }|                      |¦  «        }d}|��Z| j        j        €f| j        d	k    rd| j        _        nN| j        d	k    r7|j        t
          j        k    s|j        t
          j        k    rd| j        _        nd| j        _        | j        j        dk    rWt%          ¦   «         }| j        d	k    r1 ||                     ¦   «         |                     ¦   «         ¦  «        }nŽ |||¦  «        }n�| j        j        dk    rGt)          ¦   «         } ||                     d| j        ¦  «        |                     d¦  «        ¦  «        }n*| j        j        dk    rt-          ¦   «         } |||¦  «        }t/          |||
j        |
j        ¬¦  «        S )rþ   rÿ   r   ÚclsNr^   r   )rŸ   rž   r¢   r   r    T)r¡   ÚkeepdimÚ
regressionÚsingle_label_classificationÚmulti_label_classification©r,   r-   r   r   r'   )rS   r>   Úclassifier_poolingr£   rŸ   r#   rÎ   rÏ   r«   r¦   r  r	  ru   Úproblem_typer  rž   r¥   rB   r   Úsqueezer   rC   r   r   r   r   )r=   rš   rÞ   rß   r›   r»   r¼   r   rü   r½   rá   r   rF   Úseq_lenrŸ   Úpooled_outputr-   r,   Úloss_fcts                      r)   rN   z,ModernVBertForSequenceClassification.forwardþ  sè  € ðF �$”*ð 	
ØØ)Ø%Ø'Ø%Ø!5Ø 3ð	
ð 	
ð ð	
ð 	
ˆð $ AœJÐàŒ;Ô)¨UÒ2Ð2Ø 1°!°!°!°Q°$Ô 7ÐÐØŒ[Ô+¨vÒ5Ð5ØÐ(Ø&3Ô&9¸"¸1¸"Ô&=Ñ#�
˜G˜Gà&/¤o°b°q°bÔ&9Ñ#�
˜GØ)2Ð)>�YÔ%Ð%ÀMÔDXˆFàÐ%Ý!&¤¨Z¸Ð,AÈ&ÕX]ÔXbÐ!cÑ!cÔ!c�Ø!2°^×5MÒ5MÈbÑ5QÔ5QÑ!Q× VÒ VÐ[\Ð VÑ ]Ô ]Ð`n×`rÒ`rØ˜tð asñ aô añ !Ðð Ÿ	š	Ð"3Ñ4Ô4ˆØŸ	š	 -Ñ0Ô0ˆØ—’ Ñ/Ô/ˆàˆØÑØŒ{Ô'Ð/Ø”? aÒ'Ð'Ø/;�D”KÔ,Ð,Ø”_ qÒ(Ð(¨f¬l½e¼jÒ.HÐ.HÈFÌLÕ\aÔ\eÒLeÐLeØ/L�D”KÔ,Ð,à/K�D”KÔ,àŒ{Ô'¨<Ò7Ð7Ý"™9œ9�Ø”? aÒ'Ð'Ø#˜8 F§N¢NÑ$4Ô$4°f·n²nÑ6FÔ6FÑGÔG�D�Dà#˜8 F¨FÑ3Ô3�D�DØ”Ô)Ð-JÒJÐJÝ+Ñ-Ô-�Ø�x §¢¨B°´Ñ @Ô @À&Ç+Â+ÈbÁ/Ä/ÑRÔR��Ø”Ô)Ð-IÒIÐIÝ,Ñ.Ô.�Ø�x ¨Ñ/Ô/�å'ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r(   r  )r   r    r!   r   r4   r   r   r#   râ   rã   r$   rä   r   r   r&   r   rN   rO   rP   s   @r)   rs   rs   ë  sh  ø€ € € € € ðÐ0ð ð ð ð ð ð ð Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<Ø*.ðQ
ð Q
àÔ#ðQ
ð œ tÑ+ðQ
ð Ô&¨Ñ-ð	Q
ð
 Ô(¨4Ñ/ðQ
ð Ô'¨$Ñ.ðQ
ð $Ô.°Ñ5ðQ
ð #Ô.°Ñ5ðQ
ð Ô  4Ñ'ðQ
ð Ð+Ô,ðQ
ð 
Ð)Ñ	)ðQ
ð Q
ð Q
ñô ñ ÔðQ
ð Q
ð Q
ð Q
ð Q
r(   rs   zw
    The ModernVBert Model with a token classification head on top, e.g. for Named Entity Recognition (NER) tasks.
    c                   ó(  ‡ — e Zd Zdefˆ fd„Ze edd¬¦  «        	 	 	 	 	 	 	 	 ddej        dej	        dz  d	ej        dz  d
ej
        dz  dej
        dz  dej        dz  dej
        dz  dej        dz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )rt   r>   c                 ór  •— t          ¦   «                              |¦  «         |j        | _        t          |¦  «        | _        t          |j        ¦  «        | _        t          j	        |j
        ¦  «        | _        t          j        |j        j        |j        ¦  «        | _        |                      ¦   «          d S rM   )r3   r4   r  r†   rS   ræ   r:   r  r6   r  r  r	  r7   r9   ru   r“   r<   s     €r)   r4   z*ModernVBertForTokenClassification.__init__e  sŽ   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒå% fÑ-Ô-ˆŒ
Ý-¨fÔ.@ÑAÔAˆŒ	Ý”J˜vÔ8Ñ9Ô9ˆŒ	Ýœ) FÔ$6Ô$BÀFÔDUÑVÔVˆŒð 	�ŠÑÔÐÐÐr(   rÚ   rÛ   rÜ   Nrš   rÞ   rß   r›   r»   r¼   r   rü   r½   r¾   c	                 óˆ  —  | j         d|||||||dœ|	¤Ž}
|
d         }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }d}|�Ft	          ¦   «         } ||                     d| j        ¦  «        |                     d¦  «        ¦  «        }t          |||
j        |
j	        ¬¦  «        S )rþ   rÿ   r   Nr¢   r  r'   )
rS   r  r	  ru   r   rC   r  r   r   r   )r=   rš   rÞ   rß   r›   r»   r¼   r   rü   r½   rá   r   r-   r,   r  s                  r)   rN   z)ModernVBertForTokenClassification.forwardq  sï   € ðH �$”*ð 	
ØØ)Ø%Ø'Ø%Ø!5Ø 3ð	
ð 	
ð ð	
ð 	
ˆð $ AœJÐà ŸIšIÐ&7Ñ8Ô8ÐØ ŸIšIÐ&7Ñ8Ô8ÐØ—’Ð!2Ñ3Ô3ˆàˆØÐÝ'Ñ)Ô)ˆHØ�8˜FŸKšK¨¨D¬OÑ<Ô<¸f¿kºkÈ"¹o¼oÑNÔNˆDå$ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r(   r  )r   r    r!   r   r4   r   r   r#   râ   rã   r$   rä   r   r   r&   r   rN   rO   rP   s   @r)   rt   rt   _  sU  ø€ € € € € ð
Ð0ð 
ð 
ð 
ð 
ð 
ð 
ð Ø€^ðð -ðñ ô ð '+Ø.2Ø04Ø26Ø15Ø8<Ø8<Ø*.ð1
ð 1
àÔ#ð1
ð œ tÑ+ð1
ð Ô&¨Ñ-ð	1
ð
 Ô(¨4Ñ/ð1
ð Ô'¨$Ñ.ð1
ð $Ô.°Ñ5ð1
ð #Ô.°Ñ5ð1
ð Ô  4Ñ'ð1
ð Ð+Ô,ð1
ð 
Ð&Ñ	&ð1
ð 1
ð 1
ñô ñ Ôð1
ð 1
ð 1
ð 1
ð 1
r(   rt   )rR   r†   rq   rs   rt   )-rn   Údataclassesr   r#   Útorch.nnr6   r   r   r   Ú r   rb   Úactivationsr	   Úmodeling_outputsr
   r   r   r   r   Úmodeling_utilsr   Úprocessing_utilsr   Úutilsr   r   r   Úutils.genericr   Úautor   Úconfiguration_modernvbertr   r   r+   rk   r/   rR   r†   ræ   rq   rs   rt   Ú__all__r'   r(   r)   ú<module>r&     s£  ðð* €€€Ø !Ð !Ð !Ð !Ð !Ð !à €€€Ø Ð Ð Ð Ð Ð Ø AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ Aà &Ð &Ð &Ð &Ð &Ð &Ø !Ð !Ð !Ð !Ð !Ð !ðð ð ð ð ð ð ð ð ð ð ð ð ð ð .Ð -Ð -Ð -Ð -Ð -Ø &Ð &Ð &Ð &Ð &Ð &Ø OÐ OÐ OÐ OÐ OÐ OÐ OÐ OÐ OÐ OØ -Ð -Ð -Ð -Ð -Ð -Ø Ð Ð Ð Ð Ð Ø 8Ð 8Ð 8Ð 8Ð 8Ð 8ð ð@ð @ð @ð @ð @ ñ @ô @ñ „ð@ð: ð9ð 9ð 9ð 9ð 9 ñ 9ô 9ñ „ð9ð<$=ð $=ð $=ð $=ð $=˜2œ9ñ $=ô $=ð $=ðN ð-:ð -:ð -:ð -:ð -: ñ -:ô -:ñ „ð-:ð` €ððñ ô ð|
ð |
ð |
ð |
ð |
Ð1ñ |
ô |
ñô ð|
ð~	>ð 	>ð 	>ð 	>ð 	> ¤	ñ 	>ô 	>ð 	>ð ðR
ð R
ð R
ð R
ð R
Ð7ñ R
ô R
ñ „ðR
ðj €ððñ ô ð
l
ð l
ð l
ð l
ð l
Ð+Eñ l
ô l
ñô ð
l
ð^ €ððñ ô ð
K
ð K
ð K
ð K
ð K
Ð(Bñ K
ô K
ñô ð
K
ð\ð ð €€€r(   