§
    ‚Štj]œ  ã                   óÐ  — d dl mZ d dlmZ d dlmZ d dlZd dlmZ d dl	mc m
Z d dlmZ ddlmZ ddlmZ dd	lmZmZmZ dd
lmZmZ ddlmZmZmZ ddlmZm Z m!Z!m"Z"m#Z#m$Z$ ddl%m&Z& ddl'm(Z( ddl)m*Z* ddl+m,Z, ddl-m.Z. ddl/m0Z0m1Z1 ddl2m3Z3 ddl4m5Z5 ddl6m7Z7m8Z8m9Z9m:Z:m;Z;  e#j<        e=¦  «        Z>dZ? e!d¬¦  «        e G d„ de5¦  «        ¦   «         ¦   «         Z@ e!d¬¦  «        e G d„ de1¦  «        ¦   «         ¦   «         ZA e!d¬¦  «        e G d„ d e0¦  «        ¦   «         ¦   «         ZB G d!„ d"e3¦  «        ZC G d#„ d$ed%¬&¦  «        ZDe! G d'„ d(e¦  «        ¦   «         ZE e!d)¬*¦  «        e G d+„ d,e¦  «        ¦   «         ¦   «         ZF e!d-¬*¦  «        e G d.„ d/e¦  «        ¦   «         ¦   «         ZG G d0„ d1e;¦  «        ZH G d2„ d3e8¦  «        ZI G d4„ d5e8¦  «        ZJ G d6„ d7ejK        ¦  «        ZL G d8„ d9e7¦  «        ZM G d:„ d;ejN        ¦  «        ZO G d<„ d=e9¦  «        ZPe! G d>„ d?e:¦  «        ¦   «         ZQ e!d@¬*¦  «         G dA„ dBeQ¦  «        ¦   «         ZR G dC„ dDe7¦  «        ZS e!dE¬*¦  «         G dF„ dGeQ¦  «        ¦   «         ZT e!dH¬*¦  «         G dI„ dJeQ¦  «        ¦   «         ZU e!dK¬*¦  «         G dL„ dMeQ¦  «        ¦   «         ZV e!dN¬*¦  «         G dO„ dPeQ¦  «        ¦   «         ZWg dQ¢ZXdS )Ré    )ÚCallable)Ú	dataclass)ÚAnyN)Ústricté   )Úinitialization)Úcreate_causal_mask)ÚBaseModelOutputÚBaseModelOutputWithPoolingÚImageClassifierOutput)ÚALL_ATTENTION_FUNCTIONSÚPreTrainedModel)ÚProcessingKwargsÚProcessorMixinÚUnpack)ÚModelOutputÚTransformersKwargsÚauto_docstringÚcan_return_tupleÚloggingÚ	torch_int)Úmerge_with_config_defaults)Úcapture_outputsé   )Úcreate_sinusoidal_positions)Úeager_attention_forward)Úl2norm)ÚSiglipConfigÚSiglipTextConfig)ÚT5Tokenizer)ÚVivitConfig)ÚVivitAttentionÚVivitEmbeddingsÚ
VivitLayerÚVivitPreTrainedModelÚVivitTubeletEmbeddingsg^$3eG÷?zgoogle/videoprism-base-f16r288)Ú
checkpointc                   ó&  — e Zd ZU dZdZdZdZeee         z  e	eef         z  e
d<   dZee
d<   dZee         e	ed	f         z  e
d
<   dZee
d<   dZee
d<   dZee
d<   dZee
d<   dZee
d<   dZee
d<    e¦   «         Z e¦   «         Z e¦   «         Zd„ ZdS )ÚVideoPrismVisionConfiga™  
    num_frames (`int`, *optional*, defaults to 16):
        The number of frames in the input video.
    tubelet_size (`List[int]`, *optional*, defaults to `[1, 18, 18]`):
        The size of the tubelet patch.
    num_spatial_layers (`int`, *optional*, defaults to 12):
        Number of spatial transformer blocks.
    num_temporal_layers (`int`, *optional*, defaults to 4):
        Number of temporal transformer blocks.
    attn_logit_softcapping (`float`, *optional*, defaults to 50.0):
        Softcapping constant for attention logits.
    num_auxiliary_layers (`int`, *optional*, defaults to 2):
        Number of auxiliary layers. This is used in the VideoPrismVideoModel that is a part of VideoPrismClipModel.
    apply_l2norm (`bool`, *optional*, defaults to `True`):
        Whether to apply L2 normalization to the output. This is used in the VideoPrismVideoModel that is a part of VideoPrismClipModel.
    Úvideoprism_vision_modelÚvision_configé   Ú
image_sizeé   Ú
num_frames)é   é   r1   .Útubelet_sizeé   Únum_spatial_layersé   Únum_temporal_layersÚgelu_pythonÚ
hidden_actç      I@Úattn_logit_softcappingr   Únum_auxiliary_layersTÚapply_l2normc                 ó    — t          d¦  «        ‚©NzNot used here©ÚAttributeError©ÚselfÚkwargss     úo/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/videoprism/modular_videoprism.pyÚ__post_init__z$VideoPrismVisionConfig.__post_init__X   ó   € Ý˜_Ñ-Ô-Ð-ó    N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚbase_config_keyr-   ÚintÚlistÚtupleÚ__annotations__r/   r2   r4   r6   r8   Ústrr:   Úfloatr;   r<   Úboolr@   Únum_hidden_layersÚ
pooler_actÚpooler_output_sizerE   © rG   rD   r)   r)   5   s  € € € € € € ðð ð" +€JØ%€OØ47€J��d˜3”i‘ %¨¨S¨¤/Ñ1Ð7Ð7Ñ7Ø€J�ÐÐÑØ0;€L�$�s”)˜e C¨ HœoÑ-Ð;Ð;Ñ;Ø Ð˜Ð Ð Ñ Ø Ð˜Ð Ð Ñ Ø#€J�Ð#Ð#Ñ#Ø$(Ð˜EÐ(Ð(Ñ(Ø !Ð˜#Ð!Ð!Ñ!Ø€L�$ÐÐÑØ&˜Ñ(Ô(ÐØ�Ñ!Ô!€JØ'˜Ñ)Ô)Ðð.ð .ð .ð .ð .rG   r)   z"google/videoprism-lvt-base-f16r288c                   óø   — e Zd ZU dZdZeed<   dZedz  ed<   dZ	edz  ed<   dZ
eee         z  dz  ed<   d	Zeez  ed
<   dZeed<   dZeed<   d	Zeed<   dZeed<   dZeed<    e¦   «         Z e¦   «         Zd„ ZdS )ÚVideoPrismTextConfiga	  
    apply_l2norm (`bool`, *optional*, defaults to `True`):
        Whether to apply L2 normalization to the output of VideoPrismTextEncoder.
    attn_logit_softcapping (`float`, *optional*, defaults to 50.0):
        Softcapping constant for attention logits.
    Úrelur8   r   NÚpad_token_idÚbos_token_idÚeos_token_idç        Úattention_probs_dropout_probTr<   Úqkv_biasÚhidden_dropout_probg{®Gáz”?Úinitializer_ranger9   r:   c                 ó    — t          d¦  «        ‚r>   r?   rA   s     rD   rE   z"VideoPrismTextConfig.__post_init__s   rF   rG   )rH   rI   rJ   rK   r8   rR   rQ   r\   rN   r]   r^   rO   r`   rS   r<   rT   ra   rb   rc   r:   r@   Úattention_dropoutÚprojection_sizerE   rX   rG   rD   rZ   rZ   \   s	  € € € € € € ðð ð €J�ÐÐÑØ €L�#˜‘*Ð Ð Ñ Ø#€L�#˜‘*Ð#Ð#Ñ#Ø+/€L�#˜˜Sœ	‘/ DÑ(Ð/Ð/Ñ/Ø03Ð  %¨#¡+Ð3Ð3Ñ3Ø€L�$ÐÐÑØ€HˆdÐÐÑØ!$Ð˜Ð$Ð$Ñ$Ø#Ð�uÐ#Ð#Ñ#Ø$(Ð˜EÐ(Ð(Ñ(Ø&˜Ñ(Ô(ÐØ$�nÑ&Ô&€Oð.ð .ð .ð .ð .rG   rZ   c                   ó&   — e Zd ZdZ e¦   «         ZdS )ÚVideoPrismConfiga¤  
    Example:

    ```python
    >>> from transformers import VideoPrismClipModel, VideoPrismConfig

    >>> # Initializing a VideoPrismConfig with default values
    >>> configuration = VideoPrismConfig()

    >>> # Initializing a VideoPrismClipModel with the configuration
    >>> model = VideoPrismClipModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```
    N)rH   rI   rJ   rK   r@   Úinitializer_factorrX   rG   rD   rh   rh   w   s*   € € € € € ðð ð" (˜Ñ)Ô)ÐÐÐrG   rh   c                   ó`   ‡ — e Zd ZdZ	 	 	 	 	 	 	 d	deeeeef                  z  dz  fˆ fd„Zˆ xZ	S )
ÚVideoPrismTokenizeraV  
    Constructs a VideoPrism tokenizer, which is essentially a T5 tokenizer without its postprocessor
    (appending an EOS token at the end of the sequence).

    This tokenizer inherits from [`T5Tokenizer`] which contains most of the main methods. Users should refer to this
    superclass for more information regarding those methods.
    Nú</s>ú<unk>ú<pad>éd   Úvocabc                 óX   •—  t          ¦   «         j        d|||||||dœ|¤Ž | j        `d S )N)rp   Ú	eos_tokenÚ	unk_tokenÚ	pad_tokenÚ_spm_precompiled_charsmapÚ	extra_idsÚadditional_special_tokensrX   )ÚsuperÚ__init__Ú
_tokenizerÚpost_processor)
rB   rp   rr   rs   rt   ru   rv   rw   rC   Ú	__class__s
            €rD   ry   zVideoPrismTokenizer.__init__—   sY   ø€ ð 	�‰ŒÔð 		
ØØØØØ&?ØØ&?ð		
ð 		
ð ð		
ð 		
ð 		
ð ŒOÐ*Ð*Ð*rG   )Nrl   rm   rn   Nro   N)
rH   rI   rJ   rK   rR   rO   rP   rS   ry   Ú__classcell__©r|   s   @rD   rk   rk   Ž   sƒ   ø€ € € € € ðð ð 7;ØØØØ"&ØØ"&ð+ð +à�T˜%  U 
Ô+Ô,Ñ,¨tÑ3ð+ð +ð +ð +ð +ð +ð +ð +ð +ð +rG   rk   c                   ó.   — e Zd Zddddœdddœdddœd	œZd
S )ÚVideoPrismProcessorKwargsÚ
max_lengthTé@   )ÚpaddingÚ
truncationr�   r,   )ÚheightÚwidthF)ÚsizeÚdo_normalizeÚdo_sample_frames)Útext_kwargsÚvideo_kwargsN)rH   rI   rJ   Ú	_defaultsrX   rG   rD   r€   r€   °   sL   € € € € € ð $ØØð
ð 
ð  #¨SÐ1Ð1Ø!Ø $ð
ð 
ðð €I€I€IrG   r€   F)Útotalc                   ó$   ‡ — e Zd ZeZdˆ fd„	Zˆ xZS )ÚVideoPrismProcessorNc                 óL   •— t          ¦   «                              ||¦  «         d S ©N)rx   ry   )rB   Úvideo_processorÚ	tokenizerr|   s      €rD   ry   zVideoPrismProcessor.__init__Ã   s#   ø€ Ý‰Œ×Ò˜¨)Ñ4Ô4Ð4Ð4Ð4rG   )NN)rH   rI   rJ   r€   Úvalid_processor_kwargsry   r}   r~   s   @rD   r�   r�   ¿   sC   ø€ € € € € à6Ðð5ð 5ð 5ð 5ð 5ð 5ð 5ð 5ð 5ð 5rG   r�   zFBase class for model outputs that include spatial and temporal states.)Úcustom_introc                   óP   — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dS )Ú+BaseModelOutputWithSpatialAndTemporalStatesa™  
    last_temporal_hidden_state (`torch.FloatTensor`, *optional*):
        The last hidden state of the temporal encoder, typically of shape
        `(batch_size * num_patches, num_frames, hidden_size)`.
    last_spatial_hidden_state (`torch.FloatTensor`, *optional*):
        The last hidden state of the spatial encoder, typically of shape
        `(batch_size * num_frames, num_patches, hidden_size)`.
    NÚlast_temporal_hidden_stateÚlast_spatial_hidden_state)	rH   rI   rJ   rK   r˜   ÚtorchÚFloatTensorrQ   r™   rX   rG   rD   r—   r—   Ç   sQ   € € € € € € ðð ð <@Ð Ô 1°DÑ 8Ð?Ð?Ñ?Ø:>Ð˜uÔ0°4Ñ7Ð>Ð>Ñ>Ð>Ð>rG   r—   z+Base class for VideoPrismClipModel outputs.c                   óÞ   — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dZ	ej        dz  ed<   dZ
ej        dz  ed<   dZeed<   dZeed<   dZej        dz  ed	<   d
ee         fd„ZdS )ÚVideoPrismClipOutputa¼  
    logits_per_video (`torch.FloatTensor` of shape `(video_batch_size, text_batch_size)`):
        The scaled dot product scores between `video_embeds` and `text_embeds`. This represents the video-text
        similarity scores.
    logits_per_text (`torch.FloatTensor` of shape `(text_batch_size, video_batch_size)`):
        The scaled dot product scores between `text_embeds` and `video_embeds`. This represents the text-video
        similarity scores.
    video_embeds (`torch.FloatTensor` of shape `(batch_size, output_dim)`):
        The video embeddings obtained by applying the projection layer to the pooled output of [`VideoPrismVideoModel`].
    text_embeds (`torch.FloatTensor` of shape `(batch_size, output_dim)`):
        The text embeddings obtained by applying the projection layer to the pooled output of [`VideoPrismTextModel`].
    video_model_output (`BaseModelOutputWithPooling`):
        The output of [`VideoPrismVideoModel`].
    text_model_output (`BaseModelOutputWithPooling`):
        The output of the [`VideoPrismTextModel`].
    loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `return_loss` is `True`):
        Contrastive loss for video-text similarity.
    NÚlogits_per_videoÚlogits_per_textÚvideo_embedsÚtext_embedsÚvideo_model_outputÚtext_model_outputÚlossÚreturnc                 ó^   ‡ — t          ˆ fd„‰                      ¦   «         D ¦   «         ¦  «        S )Nc              3   ót   •K  — | ]2}|d vr‰|         n!t          ‰|¦  «                             ¦   «         V — Œ3dS ))r£   r¢   N)ÚgetattrÚto_tuple)Ú.0ÚkrB   s     €rD   ú	<genexpr>z0VideoPrismClipOutput.to_tuple.<locals>.<genexpr>ø   sc   øè è € ð 
ð 
àð Ð KÐKÐKˆD�ŒGˆGÕQXÐY]Ð_`ÑQaÔQa×QjÒQjÑQlÔQlð
ð 
ð 
ð 
ð 
ð 
rG   )rP   Úkeys©rB   s   `rD   r©   zVideoPrismClipOutput.to_tuple÷   sC   ø€ Ýð 
ð 
ð 
ð 
à—Y’Y‘[”[ð
ñ 
ô 
ñ 
ô 
ð 	
rG   )rH   rI   rJ   rK   rž   rš   r›   rQ   rŸ   r    r¡   r¢   r   r£   r¤   rP   r   r©   rX   rG   rD   r�   r�   ×   sÞ   € € € € € € ð
ð ð& 26Ð�eÔ'¨$Ñ.Ð5Ð5Ñ5Ø04€O�UÔ&¨Ñ-Ð4Ð4Ñ4Ø-1€L�%Ô# dÑ*Ð1Ð1Ñ1Ø,0€K�Ô" TÑ)Ð0Ð0Ñ0Ø59ÐÐ2Ð9Ð9Ñ9Ø48ÐÐ1Ð8Ð8Ñ8Ø%)€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)ð
˜% œ*ð 
ð 
ð 
ð 
ð 
ð 
rG   r�   c                   óR   ‡ — e Zd ZdZdefˆ fd„Zd	dej        dedej        fd„Z	ˆ xZ
S )
ÚVideoPrismTubeletEmbeddingsañ  
    VideoPrism Tubelet Embeddings.

    The authors of Videoprism use the Factorized Encoder architecture, i.e. "Model 2", introduced in the VIVIT paper (https://huggingface.co/papers/2103.15691).
    This differs from Vivit by using a convolution of `tubelet_size=(1, 18, 18)`, which is essentially a 2d convolution in the spatial dimension.
    The temporal dimension is also merged with the `batch_size` in order to make sure the image embeddings have no temporal component, unlike Vivit.
    Úconfigc                 ó  •— t          ¦   «                              |¦  «         | `| j        d         t          d         z  | j        d         t          d         z  g| _        | j        d         | j        d         z  | _        d S )Nr   r0   r   )rx   ry   Únum_patchesr-   r2   Úpos_emb_shape©rB   r±   r|   s     €rD   ry   z$VideoPrismTubeletEmbeddings.__init__  su   ø€ Ý‰Œ×Ò˜Ñ Ô Ð ØÐØ"œo¨aÔ0µLÀ´OÑCÀTÄ_ÐUVÔEWÕ[gÐhiÔ[jÑEjÐkˆÔØÔ-¨aÔ0°4Ô3EÀaÔ3HÑHˆÔÐÐrG   FÚpixel_values_videosÚinterpolate_pos_encodingr¥   c                 óÄ  — |j         \  }}}}}|sT|| j        d         k    s|| j        d         k    r2t          d|› d|› d| j        d         › d| j        d         › d�	¦  «        ‚|                     dd¦  «        }|                      |¦  «        }|                     d¦  «                             dddd¦  «        }|j         \  }}}	}
|                     ||z  |	|
¦  «        }|S )	Nr   r0   zImage size (Ú*z) doesn't match model (z[). Set interpolate_pos_encoding=True to automatically resize the model position embeddings.r   r   )Úshaper-   Ú
ValueErrorÚ	transposeÚ
projectionÚflattenÚpermuteÚreshape)rB   r¶   r·   Ú
batch_sizer/   Únum_channelsr…   r†   Úhidden_statesr³   Úhidden_sizes              rD   Úforwardz#VideoPrismTubeletEmbeddings.forward  s/  € Ø>QÔ>WÑ;ˆ
�J ¨f°eØ'ð 	¨V°t´ÀqÔ7IÒ-IÐ-IÈUÐVZÔVeÐfgÔVhÒMhÐMhÝð K˜vð  Kð  K¨ð  Kð  KÀdÄoÐVWÔFXð  Kð  KÐ[_Ô[jÐklÔ[mð  Kð  Kð  Kñô ð ð 2×;Ò;¸A¸qÑAÔAÐØŸšÐ(;Ñ<Ô<ˆà%×-Ò-¨aÑ0Ô0×8Ò8¸¸A¸qÀ!ÑDÔDˆà;HÔ;NÑ8ˆ
�J ¨[Ø%×-Ò-¨j¸:Ñ.EÀ{ÐT_Ñ`Ô`ˆàÐrG   ©F)rH   rI   rJ   rK   r)   ry   rš   ÚTensorrT   rÅ   r}   r~   s   @rD   r°   r°   þ   s‹   ø€ € € € € ðð ðIÐ5ð Ið Ið Ið Ið Ið Iðð ¨5¬<ð ÐSWð ÐdiÔdpð ð ð ð ð ð ð ð rG   r°   c                   ó‚   ‡ — e Zd Zdefˆ fd„Zdej        dededej        fd„Z	 dd	ej        d
e	dz  dej        fd„Z
ˆ xZS )ÚVideoPrismSpatialEmbeddingsr±   c                 óÀ   •— t          ¦   «                              |¦  «         | `| `t	          j        t          j        dt          |j	        ¦  «        ¦  «        | _
        d S ©Nr0   )rx   ry   Ú	cls_tokenr-   ÚnnÚ	Parameterrš   Úzerosr³   rÄ   Úposition_embeddingsrµ   s     €rD   ry   z$VideoPrismSpatialEmbeddings.__init__   sN   ø€ Ý‰Œ×Ò˜Ñ Ô Ð ØˆNØˆOÝ#%¤<µ´¸A½{ÈFÔL^Ñ0_Ô0_Ñ#`Ô#`ˆÔ Ð Ð rG   Ú
embeddingsr…   r†   r¥   c                 ó2  — |j         d         }| j        j         d         }t          j                             ¦   «         s||k    r||k    r| j        S |j         d         }|| j        d         z  }|| j        d         z  }t          |dz  ¦  «        }	| j                             d|	|	|¦  «        }
|
                     dddd¦  «        }
t          j
                             |
||fdd¬	¦  «        }
|
                     dddd¦  «                             dd|¦  «        }
|
S )
a   
        This method allows to interpolate the pre-trained position encodings, to be able to use the model on higher resolution
        images. This method is also adapted to support torch.jit tracing.

        Adapted from:
        - https://github.com/facebookresearch/dino/blob/de9ee3df6cf39fac952ab558447af1fa1365362a/vision_transformer.py#L174-L194, and
        - https://github.com/facebookresearch/dinov2/blob/e1277af2ba9496fbadf7aec6eba56e8d882d1e35/dinov2/models/vision_transformer.py#L179-L211
        r0   éÿÿÿÿr   ç      à?r   r   ÚbilinearT©r‡   ÚmodeÚ	antialias)rº   rÐ   rš   ÚjitÚ
is_tracingÚ
patch_sizer   rÀ   r¿   rÍ   Ú
functionalÚinterpolateÚview)rB   rÑ   r…   r†   r³   Únum_positionsÚdimÚnum_row_patchesÚnum_col_patchesÚsqrt_num_positionsÚpatch_pos_embeds              rD   r·   z4VideoPrismSpatialEmbeddings.interpolate_pos_encoding&  s1  € ð !Ô& qÔ)ˆØÔ0Ô6°qÔ9ˆõ Œy×#Ò#Ñ%Ô%ð 	,¨+¸Ò*FÐ*FÈ6ÐUZÊ?È?ØÔ+Ð+àÔ˜rÔ"ˆà  D¤O°AÔ$6Ñ6ˆØ 4¤?°1Ô#5Ñ5ˆå& }°cÑ'9Ñ:Ô:ÐØÔ2×:Ò:¸1Ð>PÐRdÐfiÑjÔjˆØ)×1Ò1°!°Q¸¸1Ñ=Ô=ˆõ œ-×3Ò3ØØ! ?Ð3ØØð	 4ñ 
ô 
ˆð *×1Ò1°!°Q¸¸1Ñ=Ô=×BÒBÀ1ÀbÈ#ÑNÔNˆØÐrG   Fr¶   r·   Nc                 óÄ   — |j         \  }}}}}|                      ||¦  «        }|r||                      |||¦  «        z   }n
|| j        z   }|                      |¦  «        }|S r‘   )rº   Úpatch_embeddingsr·   rÐ   Údropout)	rB   r¶   r·   ÚbatchÚframesÚchannelr…   r†   rÑ   s	            rD   rÅ   z#VideoPrismSpatialEmbeddings.forwardL  s}   € ð
 1DÔ0IÑ-ˆˆv�w ¨Ø×*Ò*Ð+>Ð@XÑYÔYˆ
ð $ð 	?Ø# d×&CÒ&CÀJÐPVÐX]Ñ&^Ô&^Ñ^ˆJˆJà# dÔ&>Ñ>ˆJà—\’\ *Ñ-Ô-ˆ
àÐrG   rÆ   )rH   rI   rJ   r)   ry   rš   rÇ   rN   r·   rT   rÅ   r}   r~   s   @rD   rÉ   rÉ     sÆ   ø€ € € € € ðaÐ5ð að að að að að að$°5´<ð $Èð $ÐUXð $Ð]bÔ]ið $ð $ð $ð $ðR 16ðð à"œ\ðð #'¨¡+ðð 
Œð	ð ð ð ð ð ð ð rG   rÉ   c            	       óŒ   ‡ — e Zd ZdZdefˆ fd„Zdej        dej        fd„Z	 ddej        d	ej	        d
e
dz  dej        fd„Zˆ xZS )ÚVideoPrismTemporalEmbeddingszÍ
    VideoPrism Temporal Embeddings.

    Receives embeddings from spatial encoder, reshapes the hidden state to
    (batch_size * num_patches, num_frames, hidden_size) and adds positional embeddings.
    r±   c                 óÊ   •— t          ¦   «                              |¦  «         | `| `| `~| `t          j        t          j	        d|j
        |j        ¦  «        ¦  «        | _        d S rË   )rx   ry   rÌ   ræ   rÛ   r-   rÍ   rÎ   rš   rÏ   r/   rÄ   rÐ   )rB   r±   r³   r|   s      €rD   ry   z%VideoPrismTemporalEmbeddings.__init__g  s`   ø€ Ý‰Œ×Ò˜Ñ Ô Ð ØˆNØÐ!ØˆOØØˆOÝ#%¤<µ´¸A¸vÔ?PÐRXÔRdÑ0eÔ0eÑ#fÔ#fˆÔ Ð Ð rG   rÑ   r¥   c                 ó\  — |j         d         }| j        j         d         }t          j                             ¦   «         s||k    r| j        S | j        }|j         d         }|                     d¦  «        }t          j                             |||fdd¬¦  «        }| 	                    d¦  «        S )Nr0   rÓ   rÕ   TrÖ   )
rº   rÐ   rš   rÙ   rÚ   Ú	unsqueezerÍ   rÜ   rÝ   Úsqueeze)rB   rÑ   Útarget_emb_lengthÚsource_emb_lengthÚ
source_embrà   s         rD   r·   z5VideoPrismTemporalEmbeddings.interpolate_pos_encodingp  s¹   € Ø&Ô,¨QÔ/ÐØ Ô4Ô:¸1Ô=Ðõ Œy×#Ò#Ñ%Ô%ð 	,Ð*;Ð?PÒ*PÐ*PØÔ+Ð+àÔ-ˆ
ØÔ˜rÔ"ˆØ×)Ò)¨!Ñ,Ô,ˆ
Ý”]×.Ò.ØØ# SÐ)ØØð	 /ñ 
ô 
ˆ
ð ×!Ò! !Ñ$Ô$Ð$rG   Fr¶   Úinput_shaper·   Nc                 ó4  — |�|\  }}}}}|j         \  }	}
}|                     |||
|¦  «        }|                     dd¦  «        }|                     ||
z  ||¦  «        }|r||                      |¦  «        z   }n
|| j        z   }|                      |¦  «        }|S )Nr   r0   )rº   rÞ   r¼   rÀ   r·   rÐ   rç   )rB   r¶   rô   r·   rè   ré   rê   r…   r†   Ú_Úfeaturesrà   rÃ   rÑ   s                 rD   rÅ   z$VideoPrismTemporalEmbeddings.forward„  s»   € ð Ð"Ø4?Ñ1ˆE�6˜7 F¨EØ.Ô4Ñˆˆ8�SØ+×0Ò0°¸ÀÈ#ÑNÔNˆØ%×/Ò/°°1Ñ5Ô5ˆØ"×*Ò*¨5°8Ñ+;¸VÀSÑIÔIˆ
ð $ð 	?Ø# d×&CÒ&CÀJÑ&OÔ&OÑOˆJˆJà# dÔ&>Ñ>ˆJØ—\’\ *Ñ-Ô-ˆ
ØÐrG   rÆ   )rH   rI   rJ   rK   r)   ry   rš   rÇ   r·   ÚSizerT   rÅ   r}   r~   s   @rD   rì   rì   _  sÊ   ø€ € € € € ðð ðgÐ5ð gð gð gð gð gð gð%°5´<ð %ÀEÄLð %ð %ð %ð %ð0 16ð	ð à"œ\ðð ”Zðð #'¨¡+ð	ð
 
Œðð ð ð ð ð ð ð rG   rì   c            	       ó~   ‡ — e Zd Zdefˆ fd„Z	 	 	 d	dej        dz  dej        dz  dej        dz  dej        fd„Z	ˆ xZ
S )
ÚVideoPrismTextEmbeddingsr±   c                 ó   •— t          ¦   «                              ¦   «          || _        |j        }t	          j        |j        |¦  «        | _        |                      dt          |j
        |j        ¦  «        ¦  «         |                      dt          j        |j
        ¦  «                             d¦  «        ¦  «         t	          j        t          j        dd|j        ¦  «        ¦  «        | _        |j        dz  | _        d S )NÚposition_embeddingÚposition_ids©r0   rÓ   r0   rÔ   )rx   ry   r±   rÄ   rÍ   Ú	EmbeddingÚ
vocab_sizeÚtoken_embeddingÚregister_bufferr   Úmax_position_embeddingsrš   ÚarangeÚexpandrÎ   rÏ   Úcls_embÚscaling)rB   r±   Ú	embed_dimr|   s      €rD   ry   z!VideoPrismTextEmbeddings.__init__›  s×   ø€ Ý‰Œ×ÒÑÔÐØˆŒØÔ&ˆ	Ý!œ|¨FÔ,=¸yÑIÔIˆÔØ×ÒØ Õ"=¸fÔ>\Ð^dÔ^pÑ"qÔ"qñ	
ô 	
ð 	
ð 	×Ò˜^­U¬\¸&Ô:XÑ-YÔ-Y×-`Ò-`ÐahÑ-iÔ-iÑjÔjÐjÝ”|¥E¤K°°1°fÔ6HÑ$IÔ$IÑJÔJˆŒØÔ)¨3Ñ.ˆŒˆˆrG   NÚ	input_idsrý   Úinputs_embedsr¥   c                 óp  — |€|                       |¦  «        }|€| j        d d …d |j        d         …f         }|| j        z  }| j        |                              |j        ¬¦  «        }||z   }| j        | j        z  }|                     |j        d         dd¦  «        }t          j
        ||fd¬¦  «        }|S )Nr0   )Údtyper   rÓ   ©rà   )r  rý   rº   r  rü   Útor  r  r  rš   Úcat)rB   r	  rý   r
  rÐ   rÑ   r  s          rD   rÅ   z VideoPrismTextEmbeddings.forward§  sÊ   € ð Ð Ø ×0Ò0°Ñ;Ô;ˆMàÐØÔ,¨Q¨Q¨QÐ0H°-Ô2EÀaÔ2HÐ0HÐ-HÔIˆLà%¨¬Ñ4ˆØ"Ô5°lÔC×FÒFÈ]ÔM`ÐFÑaÔaÐØ"Ð%8Ñ8ˆ
à”, ¤Ñ-ˆØ—.’. Ô!1°!Ô!4°b¸"Ñ=Ô=ˆÝ”Y 
¨GÐ4¸!Ð<Ñ<Ô<ˆ
ØÐrG   )NNN)rH   rI   rJ   rZ   ry   rš   Ú
LongTensorr›   rÇ   rÅ   r}   r~   s   @rD   rú   rú   š  s©   ø€ € € € € ð
/Ð3ð 
/ð 
/ð 
/ð 
/ð 
/ð 
/ð .2Ø04Ø26ð	ð àÔ# dÑ*ðð Ô&¨Ñ-ðð Ô(¨4Ñ/ð	ð
 
Œðð ð ð ð ð ð ð rG   rú   c                   ó�   ‡ — e Zd Zdeez  fˆ fd„Z	 d	dej        dej        dz  dee	         de
ej        ej        f         fd„Zˆ xZS )
ÚVideoPrismAttentionr±   c                 ót   •— t          ¦   «                              |¦  «         | `d| _        |j        | _        d S ©Nç      ð?)rx   ry   Únum_attention_headsÚnum_key_value_groupsr:   rµ   s     €rD   ry   zVideoPrismAttention.__init__¾  s:   ø€ Ý‰Œ×Ò˜Ñ Ô Ð ØÐ$Ø$'ˆÔ!Ø&,Ô&CˆÔ#Ð#Ð#rG   NrÃ   Úattention_maskrC   r¥   c                 ó¸  — |j         d d…         }g |¢d‘| j        ‘R }|                      |¦  «                             |¦  «                             dd¦  «        }|                      |¦  «                             |¦  «                             dd¦  «        }|                      |¦  «                             |¦  «                             dd¦  «        }t          j        | j	        j
        t          ¦  «        }	 |	| ||||f| j        sdn| j        | j        | j        dœ|¤Ž\  }
} |
j        g |¢d‘R Ž                      ¦   «         }
|                      |
¦  «        }
|
|fS )NrÓ   r0   r   r_   )rç   r  Úsoftcap)rº   Úhead_dimÚq_projrÞ   r¼   Úk_projÚv_projr   Úget_interfacer±   Ú_attn_implementationr   Útrainingre   r  r:   rÀ   Ú
contiguousÚo_proj)rB   rÃ   r  rC   rô   Úhidden_shapeÚquery_statesÚ
key_statesÚvalue_statesÚattention_interfaceÚattn_outputÚattn_weightss               rD   rÅ   zVideoPrismAttention.forwardÄ  s}  € ð $Ô)¨#¨2¨#Ô.ˆØ8˜Ð8 bÐ8¨$¬-Ð8Ð8ˆà—{’{ =Ñ1Ô1×6Ò6°|ÑDÔD×NÒNÈqÐRSÑTÔTˆØ—[’[ Ñ/Ô/×4Ò4°\ÑBÔB×LÒLÈQÐPQÑRÔRˆ
Ø—{’{ =Ñ1Ô1×6Ò6°|ÑDÔD×NÒNÈqÐRSÑTÔTˆå(?Ô(MØŒKÔ,Õ.Eñ)
ô )
Ðð %8Ð$7ØØØØØð
%
ð  $œ}ÐH�C�C°$Ô2HØ”LØÔ/ð
%
ð 
%
ð ð
%
ð 
%
Ñ!ˆ�\ð *�kÔ)Ð;¨;Ð;¸Ð;Ð;Ð;×FÒFÑHÔHˆØ—k’k +Ñ.Ô.ˆØ˜LÐ(Ð(rG   r‘   )rH   rI   rJ   r)   rZ   ry   rš   rÇ   r   r   rP   rÅ   r}   r~   s   @rD   r  r  ½  s³   ø€ € € € € ðDÐ5Ð8LÑLð Dð Dð Dð Dð Dð Dð /3ð)ð )à”|ð)ð œ tÑ+ð)ð Ð+Ô,ð	)ð
 
ˆuŒ|˜Uœ\Ð)Ô	*ð)ð )ð )ð )ð )ð )ð )ð )rG   r  c                   ó2   — e Zd Zdej        dej        fd„ZdS )ÚVideoPrismLayerNormrÃ   r¥   c                 ó`   — t          j        || j        | j        dz   | j        | j        ¦  «        S r  )ÚFÚ
layer_normÚnormalized_shapeÚweightÚbiasÚeps)rB   rÃ   s     rD   rÅ   zVideoPrismLayerNorm.forwardç  s/   € õ Œ|˜M¨4Ô+@À$Ä+ÐPSÑBSÐUYÔU^Ð`dÔ`hÑiÔiÐirG   N)rH   rI   rJ   rš   rÇ   rÅ   rX   rG   rD   r,  r,  æ  sB   € € € € € ðj U¤\ð j°e´lð jð jð jð jð jð jrG   r,  c                   ó*   ‡ — e Zd Zdeez  fˆ fd„Zˆ xZS )ÚVideoPrismLayerr±   c                 óò   •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          |j        |j        ¬¦  «        | _        t	          |j        |j        ¬¦  «        | _        d S ©N©r3  )	rx   ry   r  Ú	attentionr,  rÄ   Úlayer_norm_epsÚlayernorm_beforeÚlayernorm_afterrµ   s     €rD   ry   zVideoPrismLayer.__init__ï  sf   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý,¨VÑ4Ô4ˆŒÝ 3°FÔ4FÈFÔLaÐ bÑ bÔ bˆÔÝ2°6Ô3EÈ6ÔK`ÐaÑaÔaˆÔÐÐrG   )rH   rI   rJ   r)   rZ   ry   r}   r~   s   @rD   r5  r5  î  sV   ø€ € € € € ðbÐ5Ð8LÑLð bð bð bð bð bð bð bð bð bð brG   r5  c                   óv   — e Zd ZU eed<   dZdZdZg d¢ZdZ	 e
¦   «         Z ej        ¦   «         d„ ¦   «         ZdS )	ÚVideoPrismPreTrainedModelr±   Úmodelr¶   )ÚvideoÚtext)rÉ   rì   r5  rú   Ú'VideoPrismMultiheadAttentionPoolingHeadFc                 óê  — t          j        | |¦  «         t          |t          j        t          j        f¦  «        rt          j        |j        ¦  «         d S t          |t          ¦  «        rt          j        |j
        ¦  «         d S t          |t          ¦  «        rt          j        |j
        ¦  «         d S t          |t          ¦  «        r4t          j        |j        ¦  «         t          j        |j        ¦  «         d S t          |t           ¦  «        rt          j        |j        ¦  «         d S t          |t"          ¦  «        r¸t%          |j        j        |j        j        ¦  «                             |j        j        |j        j        ¬¦  «        }t          j        |j        |¦  «         t          j        |j        t9          j        |j        j        d         ¦  «                             d¦  «        ¦  «         d S t          |t@          ¦  «        rat          j!        |j"        j#        j        |j        j        dz  ¬¦  «         t          j!        |j"        j$        |j        j        dz  ¬¦  «         d S d S )N©Údevicer  rÓ   rþ   g      à¿)Ústd)%r   Ú_init_weightsÚ
isinstancerÍ   ÚLinearÚConv3dÚinitÚlecun_normal_r1  rÉ   rÐ   rì   rB  Úzeros_Úper_dim_scaleÚpooling_attention_queryr,  rú   r   r±   r  rÄ   r  rü   rE  r  Úcopy_rý   rš   r  rº   r  ÚVideoPrismTextModelÚnormal_rÑ   r  r  )rB   Úmodulerü   s      rD   rG  z'VideoPrismPreTrainedModel._init_weights  s2  € åÔ% d¨FÑ3Ô3Ð3Ý�f�rœy­"¬)Ð4Ñ5Ô5ð 	YÝÔ˜vœ}Ñ-Ô-Ð-Ð-Ð-å˜Õ ;Ñ<Ô<ð 	YÝÔ˜vÔ9Ñ:Ô:Ð:Ð:Ð:å˜Õ <Ñ=Ô=ð 	YÝÔ˜vÔ9Ñ:Ô:Ð:Ð:Ð:å˜Õ GÑHÔHð 	YÝŒK˜Ô,Ñ-Ô-Ð-ÝÔ˜vÔ=Ñ>Ô>Ð>Ð>Ð>å˜Õ 3Ñ4Ô4ð 	YÝŒK˜œÑ&Ô&Ð&Ð&Ð&å˜Õ 8Ñ9Ô9ð 		YÝ!<Ø”Ô5°v´}Ô7Pñ"ô "çŠb˜Ô1Ô8ÀÔ@YÔ@_ˆbÑ`Ô`ð õ ŒJ�vÔ0Ð2DÑEÔEÐEÝŒJ�vÔ*­E¬L¸Ô9LÔ9RÐSUÔ9VÑ,WÔ,W×,^Ò,^Ð_fÑ,gÔ,gÑhÔhÐhÐhÐhå˜Õ 3Ñ4Ô4ð 	YÝŒL˜Ô*Ô:ÔAÀvÄ}ÔG`ÐbfÑGfÐgÑgÔgÐgÝŒL˜Ô*Ô2¸¼Ô8QÐSWÑ8WÐXÑXÔXÐXÐXÐXð	Yð 	YrG   N)rH   rI   rJ   rh   rQ   Úbase_model_prefixÚmain_input_nameÚinput_modalitiesÚ_no_split_modulesÚ_supports_sdpar@   Ú_input_embed_layerrš   Úno_gradrG  rX   rG   rD   r>  r>  ö  s€   € € € € € € àÐÐÑØÐØ+€OØ(Ððð ð Ðð €NØ'˜Ñ)Ô)Ðà€U„]�_„_ðYð Yñ „_ðYð Yð YrG   r>  z¬
    The bare VideoPrism vision encoder outputting raw hidden-states without any specific head on top. This model is the backbone encoder used in VideoPrismVideoModel.
    c                   óÔ   ‡ — e Zd ZU eed<   dZdZdefˆ fd„Zdej	        fd„Z
dej	        fd„Zeee	 	 ddej        d	z  ded	z  dee         defd„¦   «         ¦   «         ¦   «         Zˆ xZS )ÚVideoPrismVisionModelr±   ©r@  r?  c                 ó   •‡— t          ¦   «                              ‰¦  «         t          ‰j        ‰j        ¬¦  «        | _        t          ‰j        ‰j        ¬¦  «        | _        t          ‰¦  «        | _        t          ‰¦  «        | _
        t          j        ˆfd„t          ‰j        ¦  «        D ¦   «         ¦  «        | _        t          j        ˆfd„t          ‰j        ¦  «        D ¦   «         ¦  «        | _        |                      ¦   «          d S )Nr8  c                 ó.   •— g | ]}t          ‰¦  «        ‘ŒS rX   ©r5  ©rª   rö   r±   s     €rD   ú
<listcomp>z2VideoPrismVisionModel.__init__.<locals>.<listcomp>7  s!   ø€ Ð,oÐ,oÐ,oÈ­_¸VÑ-DÔ-DÐ,oÐ,oÐ,orG   c                 ó.   •— g | ]}t          ‰¦  «        ‘ŒS rX   r`  ra  s     €rD   rb  z2VideoPrismVisionModel.__init__.<locals>.<listcomp>8  s!   ø€ Ð-qÐ-qÐ-qÈ!­o¸fÑ.EÔ.EÐ-qÐ-qÐ-qrG   )rx   ry   r,  rÄ   r:  Ú
layernorm1Ú
layernorm2rÉ   Úspatial_embeddingsrì   Útemporal_embeddingsrÍ   Ú
ModuleListÚranger4   Úspatial_layersr6   Útemporal_layersÚ	post_initrµ   s    `€rD   ry   zVideoPrismVisionModel.__init__1  sì   øø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý-¨fÔ.@ÀfÔF[Ð\Ñ\Ô\ˆŒÝ-¨fÔ.@ÀfÔF[Ð\Ñ\Ô\ˆŒÝ"=¸fÑ"EÔ"EˆÔÝ#?ÀÑ#GÔ#GˆÔ Ý œmÐ,oÐ,oÐ,oÐ,oÍeÐTZÔTmÑNnÔNnÐ,oÑ,oÔ,oÑpÔpˆÔÝ!œ}Ð-qÐ-qÐ-qÐ-qÍuÐU[ÔUoÑOpÔOpÐ-qÑ-qÔ-qÑrÔrˆÔØ�ŠÑÔÐÐÐrG   r¥   c                 ó   — | j         j        S r‘   ©rf  ræ   r®   s    rD   Úget_input_embeddingsz*VideoPrismVisionModel.get_input_embeddings;  s   € ØÔ&Ô7Ð7rG   Úvaluec                 ó   — || j         _        d S r‘   rn  ©rB   rp  s     rD   Úset_input_embeddingsz*VideoPrismVisionModel.set_input_embeddings>  s   € Ø38ˆÔÔ0Ð0Ð0rG   NFr¶   r·   rC   c                 óN  — |€t          d¦  «        ‚|j        }|                      ||¦  «        }|}| j        D ]} ||fi |¤Ž}Œ|                      |¦  «        }|                      |||¦  «        }	|	}
| j        D ]} ||
fi |¤Ž}
Œ|                      |
¦  «        }|j        \  }}}|                     |d         d||¦  «         	                    dd¦  «         
                    ¦   «         }|j        \  }}}}|                     |d         ||z  d¦  «        }t          ||
|¬¦  «        S )Nz'You have to specify pixel_values_videosr   rÓ   r0   r   )Úlast_hidden_stater˜   r™   )r»   rº   rf  rj  rd  rg  rk  re  rÞ   r¼   r"  r—   )rB   r¶   r·   rC   rô   Úspatial_embedsÚspatial_hidden_statesÚspatial_layerr÷   Útemporal_embedsÚtemporal_hidden_statesÚtemporal_layerrö   r/   rà   r³   s                   rD   rÅ   zVideoPrismVisionModel.forwardA  s{  € ð Ð&ÝÐFÑGÔGÐGà)Ô/ˆð ×0Ò0Ð1DÐF^Ñ_Ô_ˆØ .ÐØ!Ô0ð 	Sð 	SˆMØ$1 MÐ2GÐ$RÐ$RÈ6Ð$RÐ$RÐ!Ð!Ø—?’?Ð#8Ñ9Ô9ˆð ×2Ò2°8¸[ÐJbÑcÔcˆØ!0ÐØ"Ô2ð 	Vð 	VˆNØ%3 ^Ð4JÐ%UÐ%UÈfÐ%UÐ%UÐ"Ð"Ø—?’?Ð#9Ñ:Ô:ˆð &œ^Ñˆˆ:�sØ—=’= ¨Q¤°°ZÀÑEÔE×OÒOÐPQÐSTÑUÔU×`Ò`ÑbÔbˆØ*2¬.Ñ'ˆˆ:�{ CØ—=’= ¨Q¤°¸kÑ1IÈ2ÑNÔNˆå:Ø&Ø'=Ø&;ð
ñ 
ô 
ð 	
rG   ©NF)rH   rI   rJ   r)   rQ   rV  rT  ry   rÍ   ÚModulero  rs  r   r   r   rš   r›   rT   r   r   r—   rÅ   r}   r~   s   @rD   r\  r\  '  s  ø€ € € € € € ð #Ð"Ð"Ñ"Ø!ÐØÐðÐ5ð ð ð ð ð ð ð8 b¤ið 8ð 8ð 8ð 8ð9¨"¬)ð 9ð 9ð 9ð 9ð  ØØð 9=Ø05ð#
ð #
à"Ô.°Ñ5ð#
ð #'¨¡+ð#
ð Ð+Ô,ð	#
ð
 
5ð#
ð #
ð #
ñ „^ñ „_ñ  Ôð#
ð #
ð #
ð #
ð #
rG   r\  c                   óŠ   ‡ — e Zd Zdefˆ fd„Z	 d	dej        dej        dz  dee	         de
ej        ej        f         fd„Zˆ xZS )
rB  r±   c                 óv  •— t          ¦   «                              |¦  «         | `|j        |j        z  | _        d| _        t          j        t          j	        | j        ¦  «        ¦  «        | _
        t          | j        dz  z  | _        t          j        t          j	        dd|j        ¦  «        ¦  «        | _        d S )Nr  rÔ   r0   )rx   ry   r  Úintermediate_sizer  r  rÍ   rÎ   rš   rÏ   rN  Ú_R_SOFTPLUS_0r  rÄ   rO  rµ   s     €rD   ry   z0VideoPrismMultiheadAttentionPoolingHead.__init__k  s•   ø€ Ý‰Œ×Ò˜Ñ Ô Ð ØÐ$ØÔ0°FÔ4NÑNˆŒØ$'ˆÔ!Ýœ\­%¬+°d´mÑ*DÔ*DÑEÔEˆÔÝ$¨¬°sÑ(:Ñ;ˆŒÝ')¤|µE´KÀÀ1ÀfÔFXÑ4YÔ4YÑ'ZÔ'ZˆÔ$Ð$Ð$rG   NrÃ   r  rC   r¥   c                 ó|  — |j         d d…         }g |¢d‘| j        ‘R }| j                             |d         dd¦  «        } |                      |¦  «        j        g |j         d d…         ¢d‘| j        ‘R Ž                      dd¦  «        }|| j        z  t          j	         
                    | j        ¦  «        z  }|                      |¦  «                             |¦  «                             dd¦  «        }	|                      |¦  «                             |¦  «                             dd¦  «        }
t          j        | j        j        t$          ¦  «        } || ||	|
|fd| j        sdn| j        d dœ|¤Ž\  }} |j        g |j         d d…         ¢d‘R Ž                      ¦   «         }|                      |¦  «        }||fS )NrÓ   r   r0   r   r  r_   )r  rç   r  )rº   r  rO  r  r  rÞ   r¼   r  rÍ   rÜ   ÚsoftplusrN  r  r  r   r  r±   r   r   r!  re   rÀ   r"  r#  )rB   rÃ   r  rC   rô   r$  ÚqueryÚquery_layerr%  r&  r'  r(  r)  r*  s                 rD   rÅ   z/VideoPrismMultiheadAttentionPoolingHead.forwardt  sâ  € ð $Ô)¨#¨2¨#Ô.ˆØ8˜Ð8 bÐ8¨$¬-Ð8Ð8ˆàÔ,×3Ò3°KÀ´NÀBÈÑKÔKˆØ-�d—k’k %Ñ(Ô(Ô-ÐS¨u¬{¸3¸B¸3Ô/?ÐSÀÐSÀTÄ]ÐSÐSÐS×]Ò]Ð^_ÐabÑcÔcˆØ" T¤\Ñ1µB´M×4JÒ4JÈ4ÔK]Ñ4^Ô4^Ñ^ˆà—[’[ Ñ/Ô/×4Ò4°\ÑBÔB×LÒLÈQÐPQÑRÔRˆ
Ø—{’{ =Ñ1Ô1×6Ò6°|ÑDÔD×NÒNÈqÐRSÑTÔTˆå(?Ô(MØŒKÔ,Õ.Eñ)
ô )
Ðð %8Ð$7ØØØØØð
%
ð Ø#œ}ÐH�C�C°$Ô2HØð
%
ð 
%
ð ð
%
ð 
%
Ñ!ˆ�\ð *�kÔ)Ð@¨5¬;°s¸°sÔ+;Ð@¸RÐ@Ð@Ð@×KÒKÑMÔMˆØ—k’k +Ñ.Ô.ˆØ˜LÐ(Ð(rG   r‘   )rH   rI   rJ   r)   ry   rš   r›   r  r   r   rP   rÅ   r}   r~   s   @rD   rB  rB  j  s±   ø€ € € € € ð[Ð5ð [ð [ð [ð [ð [ð [ð 37ð")ð ")àÔ(ð")ð Ô(¨4Ñ/ð")ð Ð+Ô,ð	")ð
 
ˆuÔ  %Ô"3Ð3Ô	4ð")ð ")ð ")ð ")ð ")ð ")ð ")ð ")rG   rB  z•
    The bare VideoPrism text encoder outputting last hidden states without any specific head on top. This model is used in VideoPrismClipModel.
    c                   óî   ‡ — e Zd ZU eed<   dZdZdZddgZdZ	defˆ fd„Z
eee	 	 	 	 ddej        d	z  d
ej        d	z  dej        d	z  dej        d	z  dee         defd„¦   «         ¦   «         ¦   «         Zˆ xZS )rQ  r±   )rA  r?  r	  rú   r5  r  c                 óJ  •‡— t          ¦   «                              ‰¦  «         t          ‰¦  «        | _        t	          j        ˆfd„t          ‰j        ¦  «        D ¦   «         ¦  «        | _        t          ‰j
        ‰j        ¬¦  «        | _        |                      ¦   «          d S )Nc                 ó.   •— g | ]}t          ‰¦  «        ‘ŒS rX   r`  ra  s     €rD   rb  z0VideoPrismTextModel.__init__.<locals>.<listcomp>©  s!   ø€ Ð$fÐ$fÐ$fÀ¥_°VÑ%<Ô%<Ð$fÐ$fÐ$frG   r8  )rx   ry   rú   rÑ   rÍ   rh  ri  rU   Úlayersr,  rÄ   r:  Ú	layernormrl  rµ   s    `€rD   ry   zVideoPrismTextModel.__init__¦  sŒ   øø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý2°6Ñ:Ô:ˆŒÝ”mÐ$fÐ$fÐ$fÐ$fÅeÈFÔLdÑFeÔFeÐ$fÑ$fÔ$fÑgÔgˆŒÝ,¨VÔ-?ÀVÔEZÐ[Ñ[Ô[ˆŒØ�ŠÑÔÐÐÐrG   Nr  r
  rý   rC   r¥   c                 óæ  — |d u |d uz  rt          d¦  «        ‚|                      |||¬¦  «        }|�]t          j        |j        d         d|j        |j        ¬¦  «        }t          j        ||fd¬¦  «        }t          | j	        ||d ¬¦  «        }| j
        D ]} |||fi |¤Ž}Œ|                      |¦  «        }|d d …df         }	| j	        j        rt          |	d¬¦  «        }	t          ||	¬	¦  «        S )
Nz:You must specify exactly one of input_ids or inputs_embeds)r	  rý   r
  r   r0   rD  r  )r±   r
  r  Úpast_key_valuesrÓ   )ru  Úpooler_output)r»   rÑ   rš   Úonesrº   rE  r  r  r	   r±   r‰  rŠ  r<   r   r   )
rB   r	  r  r
  rý   rC   rÃ   Úcls_paddingÚlayerÚtext_embeddingss
             rD   rÅ   zVideoPrismTextModel.forward­  s9  € ð ˜Ð -°tÐ";Ñ<ð 	[ÝÐYÑZÔZÐZàŸš°)È,Ðfs˜ÑtÔtˆàÐ%Ýœ*ØÔ# AÔ&¨°.Ô2GÈ~ÔOcðñ ô ˆKõ #œY¨¸Ð'DÈ!ÐLÑLÔLˆNÝ/Ø”{Ø+Ø-Ø $ð	ñ ô ˆNð ”[ð 	Kð 	KˆEØ!˜E -°ÐJÐJÀ6ÐJÐJˆMˆMØŸš }Ñ5Ô5ˆà'¨¨¨¨2¨Ô.ˆØŒ;Ô#ð 	>Ý$ _¸"Ð=Ñ=Ô=ˆOå)¸MÐYhÐiÑiÔiÐirG   )NNNN)rH   rI   rJ   rZ   rQ   rV  rT  rU  rW  rY  ry   r   r   r   rš   r  rÇ   r   r   r   rÅ   r}   r~   s   @rD   rQ  rQ  ™  s,  ø€ € € € € € ð !Ð Ð Ñ Ø ÐØÐØ!€OØ3Ð5FÐGÐØ*ÐðÐ3ð ð ð ð ð ð ð  ØØð .2Ø.2Ø-1Ø,0ð!jð !jàÔ# dÑ*ð!jð œ tÑ+ð!jð ”| dÑ*ð	!jð
 ”l TÑ)ð!jð Ð+Ô,ð!jð 
$ð!jð !jð !jñ „^ñ „_ñ  Ôð!jð !jð !jð !jð !jrG   rQ  z¹
    VideoPrism video model consisting of the vision encoder backbone with auxiliary encoder layers and an attention pooling head on top. This model is used in VideoPrismClipModel.
    c                   ó´   ‡ — e Zd ZU eed<   defˆ fd„Zdej        fd„Zdej        fd„Z	e
e	 ddej        d	ed
z  dee         defd„¦   «         ¦   «         Zˆ xZS )ÚVideoPrismVideoModelr±   c                 óˆ  •‡— t          ¦   «                              ‰¦  «         t                               ‰¦  «        | _        t          j        ˆfd„t          ‰j        ¦  «        D ¦   «         ¦  «        | _	        t          ‰¦  «        | _        t          ‰j        ‰j        ¬¦  «        | _        |                      ¦   «          d S )Nc                 ó.   •— g | ]}t          ‰¦  «        ‘ŒS rX   r`  ra  s     €rD   rb  z1VideoPrismVideoModel.__init__.<locals>.<listcomp>ß  s!   ø€ Ð.sÐ.sÐ.sÈ1­¸vÑ/FÔ/FÐ.sÐ.sÐ.srG   r8  )rx   ry   r\  Ú_from_configÚvision_modelrÍ   rh  ri  r;   Úauxiliary_layersrB  Úheadr,  rÄ   r:  Úhead_layernormrl  rµ   s    `€rD   ry   zVideoPrismVideoModel.__init__Ü  s¦   øø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý1×>Ò>¸vÑFÔFˆÔÝ "¤Ð.sÐ.sÐ.sÐ.sÕPUÐV\ÔVqÑPrÔPrÐ.sÑ.sÔ.sÑ tÔ tˆÔÝ;¸FÑCÔCˆŒ	Ý1°&Ô2DÈ&ÔJ_Ð`Ñ`Ô`ˆÔØ�ŠÑÔÐÐÐrG   r¥   c                 ó4   — | j                              ¦   «         S r‘   ©r—  ro  r®   s    rD   ro  z)VideoPrismVideoModel.get_input_embeddingsä  ó   € ØÔ ×5Ò5Ñ7Ô7Ð7rG   rp  c                 ó:   — | j                              |¦  «         d S r‘   ©r—  rs  rr  s     rD   rs  z)VideoPrismVideoModel.set_input_embeddingsç  ó   € ØÔ×.Ò.¨uÑ5Ô5Ð5Ð5Ð5rG   Fr¶   r·   NrC   c                 ó  —  | j         d||dœ|¤Ž}|j        }| j        D ]} ||fi |¤Ž}Œ | j        |fi |¤Ž}|                      |d         ¦  «        }| j        j        rt          |d¬¦  «        }t          |||j	        |j
        ¬¦  «        S )N©r¶   r·   r   rÓ   r  )ru  r�  rÃ   Ú
attentionsrX   )r—  ru  r˜  r™  rš  r±   r<   r   r   rÃ   r£  )	rB   r¶   r·   rC   Úvision_model_outputsÚauxiliary_hidden_statesr�  Úhead_outputÚvideo_embeddingss	            rD   rÅ   zVideoPrismVideoModel.forwardê  sé   € ð  1˜tÔ0ð  
Ø 3ÐNfð 
ð  
Øjpð 
ð  
Ðð #7Ô"HÐØÔ*ð 	Oð 	OˆEØ&+ eÐ,CÐ&NÐ&NÀvÐ&NÐ&NÐ#Ð#à�d”iÐ 7ÐBÐB¸6ÐBÐBˆØ×.Ò.¨{¸1¬~Ñ>Ô>ÐØŒ;Ô#ð 	@Ý%Ð&6¸BÐ?Ñ?Ô?Ðå)Ø5Ø*Ø.Ô<Ø+Ô6ð	
ñ 
ô 
ð 	
rG   rÆ   )rH   rI   rJ   r)   rQ   ry   rÍ   r}  ro  rs  r   r   rš   r›   rT   r   r   r   rÅ   r}   r~   s   @rD   r“  r“  Ô  s÷   ø€ € € € € € ð #Ð"Ð"Ñ"ðÐ5ð ð ð ð ð ð ð8 b¤ið 8ð 8ð 8ð 8ð6¨"¬)ð 6ð 6ð 6ð 6ð Øð 16ð
ð 
à"Ô.ð
ð #'¨¡+ð
ð Ð+Ô,ð	
ð
 
$ð
ð 
ð 
ñ „^ñ Ôð
ð 
ð 
ð 
ð 
rG   r“  zÆ
    VideoPrism model for video-text contrastive learning. This model consists of a VideoPrismVideoModel and a VideoPrismTextModel, and computes similarity scores between video and text inputs.
    c                   óª  ‡ — e Zd Zdefˆ fd„Zdej        fd„Zdej        fd„Ze	e
	 ddej        d	ej        dz  d
ee         deez  fd„¦   «         ¦   «         Ze	e
	 ddej        dedz  d
ee         deez  fd„¦   «         ¦   «         Ze	e
	 	 	 	 ddej        dej        d	ej        dz  dedz  dedz  dedz  d
ee         defd„¦   «         ¦   «         Zˆ xZS )ÚVideoPrismClipModelr±   c                 ó  •— t          ¦   «                              |¦  «         t                               |j        ¦  «        | _        t                               |j        ¦  «        | _        |  	                    ¦   «          d S r‘   )
rx   ry   r“  r–  r+   Úvideo_modelrQ  Útext_configÚ
text_modelrl  rµ   s     €rD   ry   zVideoPrismClipModel.__init__  sb   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý/×<Ò<¸VÔ=QÑRÔRˆÔÝ-×:Ò:¸6Ô;MÑNÔNˆŒØ�ŠÑÔÐÐÐrG   r¥   c                 ó4   — | j                              ¦   «         S r‘   )r­  ro  r®   s    rD   ro  z(VideoPrismClipModel.get_input_embeddings  s   € ØŒ×3Ò3Ñ5Ô5Ð5rG   rp  c                 ó:   — | j                              |¦  «         d S r‘   )r­  rs  rr  s     rD   rs  z(VideoPrismClipModel.set_input_embeddings  s   € ØŒ×,Ò,¨UÑ3Ô3Ð3Ð3Ð3rG   Nr	  r  rC   c                 ó"   —  | j         d||dœ|¤ŽS )a  
        Examples:

        ```python
        >>> from transformers import AutoTokenizer, VideoPrismClipModel

        >>> model = VideoPrismClipModel.from_pretrained("google/videoprism-lvt-base-f16r288")
        >>> tokenizer = AutoTokenizer.from_pretrained("google/videoprism-lvt-base-f16r288")

        >>> inputs = tokenizer(["a video of a cat.", "a video of a dog."], padding="max_length", return_tensors="pt")
        >>> with torch.no_grad():
        ...     text_features = model.get_text_features(**inputs)
        ```©r	  r  rX   )r­  )rB   r	  r  rC   s       rD   Úget_text_featuresz%VideoPrismClipModel.get_text_features  s$   € ð* ˆtŒÐ\¨À>Ð\Ð\ÐU[Ð\Ð\Ð\rG   Fr¶   r·   c                 ó"   —  | j         d||dœ|¤ŽS )a÷  
        Examples:

        ```python
        >>> from transformers import VideoPrismProcessor, VideoPrismClipModel

        >>> model = VideoPrismClipModel.from_pretrained("google/videoprism-lvt-base-f16r288")
        >>> processor = VideoPrismProcessor.from_pretrained("google/videoprism-lvt-base-f16r288")

        >>> inputs = processor(videos="path/to/video.mp4", return_tensors="pt")
        >>> with torch.no_grad():
        ...     video_features = model.get_video_features(**inputs)
        ```r¢  rX   )r«  )rB   r¶   r·   rC   s       rD   Úget_video_featuresz&VideoPrismClipModel.get_video_features/  s5   € ð*  ˆtÔð 
Ø 3Ø%=ð
ð 
ð ð
ð 
ð 	
rG   ÚtemperatureÚreturn_lossc           	      ó4  —  | j         d||dœ|¤Ž} | j        d||dœ|¤Ž}	|j        }
|	j        }|
j        d         }|j        d         }|
                     d|¦  «        }|                     d|¦  «        }t          j        ||j        ¦  «        }|�||z  }t          j        |¦  «        }|j        }|t          j	        |dd¬¦  «        z  }|t          j	        |dd¬¦  «        z  }d}|r›t          j
        |                     d¦  «        |j        ¬¦  «        }t          j        |¦  «         d	|z  z   }t
          j        j                             ||z  ¦  «        }t          j	        |d¬
¦  «         }|                     ¦   «         }t%          ||||||	|¬¦  «        S )a  
        temperature (`float`, *optional*):
            A temperature scalar to scale the similarity scores. If not provided, no scaling is applied.
        return_loss (`bool`, *optional*):
            Whether or not to return the contrastive loss.
        r¢  r±  rÓ   Nr   T)rà   Úkeepdims)rE  r   r  )rž   rŸ   r    r¡   r¢   r£   r¤   rX   )r´  r²  r�  rº   rÀ   rš   ÚmatmulÚTÚexpÚsumÚeyer‡   rE  Ú	ones_likerÍ   rÜ   Ú
logsigmoidÚmeanr�   )rB   r¶   r	  r  r·   rµ  r¶  rC   Úvideo_model_outputsÚtext_model_outputsr§  r‘  Úvideo_emb_dimÚtext_emb_dimr    r¡   Úsimilarity_matrixrž   rŸ   r¤   r½  Úm1_diag1ÚloglikÚnlls                           rD   rÅ   zVideoPrismClipModel.forwardJ  sÞ  € ð& 6˜dÔ5ð 
Ø 3ÐNfð
ð 
Øjpð
ð 
Ðð 4˜TÔ3Ðq¸iÐXfÐqÐqÐjpÐqÐqÐà.Ô<ÐØ,Ô:ˆØ(Ô.¨rÔ2ˆØ&Ô,¨RÔ0ˆà'×/Ò/°°MÑBÔBˆØ%×-Ò-¨b°,Ñ?Ô?ˆÝ!œL¨°{´}ÑEÔEÐàÐ"Ø Ñ,Ðå œ9Ð%6Ñ7Ô7ÐØ*Ô,ˆØ+­e¬iÐ8HÈaÐZ^Ð._Ñ._Ô._Ñ_ÐØ)­E¬I°oÈ1ÐW[Ð,\Ñ,\Ô,\Ñ\ˆð ˆØð 	å”)˜O×0Ò0°Ñ3Ô3¸OÔ<RÐSÑSÔSˆCÝœ¨Ñ8Ô8Ð8¸1¸s¹7ÑBˆHÝ”XÔ(×3Ò3°H¸Ñ4NÑOÔOˆFÝ”9˜V¨Ð,Ñ,Ô,Ð,ˆCØ—8’8‘:”:ˆDå#Ø-Ø+Ø%Ø#Ø2Ø0Øð
ñ 
ô 
ð 	
rG   r‘   rÆ   )NFNN)rH   rI   rJ   rh   ry   rÍ   r}  ro  rs  r   r   rš   rÇ   r   r   rP   r   r²  r›   rT   r´  rS   r�   rÅ   r}   r~   s   @rD   r©  r©    s  ø€ € € € € ðÐ/ð ð ð ð ð ð ð6 b¤ið 6ð 6ð 6ð 6ð4¨"¬)ð 4ð 4ð 4ð 4ð Øð /3ð]ð ]à”<ð]ð œ tÑ+ð]ð Ð+Ô,ð	]ð
 
Ð+Ñ	+ð]ð ]ð ]ñ „^ñ Ôð]ð* Øð 16ð
ð 
à"Ô.ð
ð #'¨¡+ð
ð Ð+Ô,ð	
ð
 
Ð+Ñ	+ð
ð 
ð 
ñ „^ñ Ôð
ð2 Øð
 /3Ø05Ø$(Ø#'ð9
ð 9
à"Ô.ð9
ð ”<ð9
ð œ tÑ+ð	9
ð
 #'¨¡+ð9
ð ˜T‘\ð9
ð ˜D‘[ð9
ð Ð+Ô,ð9
ð 
ð9
ð 9
ð 9
ñ „^ñ Ôð9
ð 9
ð 9
ð 9
ð 9
rG   r©  z
    VideoPrism Model transformer with a video classification head on top (a linear layer on top of the attention pooler).
    c                   óÒ   ‡ — e Zd ZU eed<   dZdZdefˆ fd„Zdej	        fd„Z
dej	        fd„Zee	 	 ddej        dej        d	z  ded	z  dee         def
d„¦   «         ¦   «         Zˆ xZS )Ú VideoPrismForVideoClassificationr±   r]  r?  c                 ó`  •— t          ¦   «                              |¦  «         t                               |¦  «        | _        t          |¦  «        | _        t          |j        |j	        ¬¦  «        | _
        t          j        |j        |j        ¦  «        | _        |                      ¦   «          d S r7  )rx   ry   r\  r–  r—  rB  r™  r,  rÄ   r:  rš  rÍ   rI  Ú
num_labelsÚ
classifierrl  rµ   s     €rD   ry   z)VideoPrismForVideoClassification.__init__’  sŠ   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý1×>Ò>¸vÑFÔFˆÔÝ;¸FÑCÔCˆŒ	Ý1°&Ô2DÈ&ÔJ_Ð`Ñ`Ô`ˆÔÝœ) FÔ$6¸Ô8IÑJÔJˆŒØ�ŠÑÔÐÐÐrG   r¥   c                 ó4   — | j                              ¦   «         S r‘   rœ  r®   s    rD   ro  z5VideoPrismForVideoClassification.get_input_embeddingsš  r�  rG   rp  c                 ó:   — | j                              |¦  «         d S r‘   rŸ  rr  s     rD   rs  z5VideoPrismForVideoClassification.set_input_embeddings�  r   rG   NFr¶   Úlabelsr·   rC   c                 ó  —  | j         d||dœ|¤Ž}|j        }|                       | j        |fi |¤Žd         ¦  «        }|                      |¦  «        }d }	|� | j        ||| j        fi |¤Ž}	t          |	||j        |j	        ¬¦  «        S )Nr¢  r   )r¤   ÚlogitsrÃ   r£  rX   )
r—  ru  rš  r™  rÍ  Úloss_functionr±   r   rÃ   r£  )
rB   r¶   rÐ  r·   rC   r¤  Úsequence_outputÚpooled_outputrÒ  r¤   s
             rD   rÅ   z(VideoPrismForVideoClassification.forward   sÍ   € ð  1˜tÔ0ð  
Ø 3ÐNfð 
ð  
Øjpð 
ð  
Ðð /Ô@ˆØ×+Ò+¨I¨D¬I°oÐ,PÐ,PÈÐ,PÐ,PÐQRÔ,SÑTÔTˆØ—’ Ñ/Ô/ˆØˆØÐØ%�4Ô% f¨f°d´kÐLÐLÀVÐLÐLˆDå$ØØØ.Ô<Ø+Ô6ð	
ñ 
ô 
ð 	
rG   r|  )rH   rI   rJ   r)   rQ   rV  rT  ry   rÍ   r}  ro  rs  r   r   rš   r›   r  rT   r   r   r   rÅ   r}   r~   s   @rD   rÊ  rÊ  ˆ  s  ø€ € € € € € ð #Ð"Ð"Ñ"Ø!ÐØÐðÐ5ð ð ð ð ð ð ð8 b¤ið 8ð 8ð 8ð 8ð6¨"¬)ð 6ð 6ð 6ð 6ð Øð +/Ø05ð	
ð 
à"Ô.ð
ð Ô  4Ñ'ð
ð #'¨¡+ð	
ð
 Ð+Ô,ð
ð 
ð
ð 
ð 
ñ „^ñ Ôð
ð 
ð 
ð 
ð 
rG   rÊ  )r)   rZ   rh   r\  r>  r“  rQ  r©  rÊ  rk   r�   )YÚcollections.abcr   Údataclassesr   Útypingr   rš   Útorch.nnrÍ   Útorch.nn.functionalrÜ   r.  Úhuggingface_hub.dataclassesr   Ú r   rK  Úmasking_utilsr	   Úmodeling_outputsr
   r   r   Úmodeling_utilsr   r   Úprocessing_utilsr   r   r   Úutilsr   r   r   r   r   r   Úutils.genericr   Úutils.output_capturingr   Úcodegen.modeling_codegenr   Úgemma2.modeling_gemma2r   Úqwen3_next.modeling_qwen3_nextr   Úsiglip.configuration_siglipr   r   Út5.tokenization_t5r    Úvivit.configuration_vivitr!   Úvivit.modeling_vivitr"   r#   r$   r%   r&   Ú
get_loggerrH   Úloggerr�  r)   rZ   rh   rk   r€   r�   r—   r�   r°   rÉ   rì   r}  rú   r  Ú	LayerNormr,  r5  r>  r\  rB  rQ  r“  r©  rÊ  Ú__all__rX   rG   rD   ú<module>rï     sµ  ðð  %Ð $Ð $Ð $Ð $Ð $Ø !Ð !Ð !Ð !Ð !Ð !Ø Ð Ð Ð Ð Ð à €€€Ø Ð Ð Ð Ð Ð Ø Ð Ð Ð Ð Ð Ð Ð Ð Ø .Ð .Ð .Ð .Ð .Ð .à &Ð &Ð &Ð &Ð &Ð &Ø /Ð /Ð /Ð /Ð /Ð /Ø bÐ bÐ bÐ bÐ bÐ bÐ bÐ bÐ bÐ bØ FÐ FÐ FÐ FÐ FÐ FÐ FÐ FØ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HØ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jØ 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø 5Ð 5Ð 5Ð 5Ð 5Ð 5Ø BÐ BÐ BÐ BÐ BÐ BØ <Ð <Ð <Ð <Ð <Ð <Ø 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø HÐ HÐ HÐ HÐ HÐ HÐ HÐ HØ ,Ð ,Ð ,Ð ,Ð ,Ð ,Ø 3Ð 3Ð 3Ð 3Ð 3Ð 3ðð ð ð ð ð ð ð ð ð ð ð ð ð ð 
ˆÔ	˜HÑ	%Ô	%€à€ð €Ð;Ð<Ñ<Ô<Øð".ð ".ð ".ð ".ð ".˜[ñ ".ô ".ñ „ñ =Ô<ð".ðJ €Ð?Ð@Ñ@Ô@Øð.ð .ð .ð .ð .Ð+ñ .ô .ñ „ñ AÔ@ð.ð2 €Ð?Ð@Ñ@Ô@Øð*ð *ð *ð *ð *�|ñ *ô *ñ „ñ AÔ@ð*ð*+ð +ð +ð +ð +˜+ñ +ô +ð +ðDð ð ð ð Ð 0¸ð ñ ô ð ð ð5ð 5ð 5ð 5ð 5˜.ñ 5ô 5ñ „ð5ð €ÐiÐjÑjÔjØ
ð?ð ?ð ?ð ?ð ?°/ñ ?ô ?ñ „ñ kÔjð?ð €ØBðñ ô ð ð 
ð  
ð  
ð  
ð  
˜;ñ  
ô  
ñ „ñô ð 
ðFð ð ð ð Ð"8ñ ô ð ðB=ð =ð =ð =ð = /ñ =ô =ð =ð@8ð 8ð 8ð 8ð 8 ?ñ 8ô 8ð 8ðv ð  ð  ð  ð  ˜rœyñ  ô  ð  ðF&)ð &)ð &)ð &)ð &)˜.ñ &)ô &)ð &)ðRjð jð jð jð j˜"œ,ñ jô jð jðbð bð bð bð b�jñ bô bð bð ð-Yð -Yð -Yð -Yð -YÐ 4ñ -Yô -Yñ „ð-Yð` €ððñ ô ð
;
ð ;
ð ;
ð ;
ð ;
Ð5ñ ;
ô ;
ñô ð
;
ð|,)ð ,)ð ,)ð ,)ð ,)¨nñ ,)ô ,)ð ,)ð^ €ððñ ô ð
3jð 3jð 3jð 3jð 3jÐ3ñ 3jô 3jñô ð
3jðl €ððñ ô ð
*
ð *
ð *
ð *
ð *
Ð4ñ *
ô *
ñô ð
*
ðZ €ððñ ô ð
z
ð z
ð z
ð z
ð z
Ð3ñ z
ô z
ñô ð
z
ðz €ððñ ô ð
+
ð +
ð +
ð +
ð +
Ð'@ñ +
ô +
ñô ð
+
ð\ð ð €€€rG   