§
    ‚ŠtjÈ8  ã                   ó^  — d dl Z d dlmZ d dl mZ ddlmZmZ ddlmZ ddl	m
Z
 ddlmZ dd	lmZ dd
lmZmZmZmZmZ ddlmZ ddlmZmZ ddlmZ ddlmZ ddlmZm Z m!Z!m"Z"m#Z#  ej$        e%¦  «        Z& ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z' G d„ de"¦  «        Z( G d„ de#¦  «        Z) ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z* G d„ de¦  «        Z+ G d„ de¦  «        Z, G d„ d e¦  «        Z- G d!„ d"e!¦  «        Z. G d#„ d$e ¦  «        Z/g d%¢Z0dS )&é    N)Ústrict)Únné   )ÚCacheÚDynamicCache)ÚGenerationConfig)ÚFlashAttentionKwargs)ÚBaseModelOutputWithPooling)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚcan_return_tupleÚloggingÚtorch_compilable_check)Úmerge_with_config_defaultsé   )ÚIdefics3ConfigÚIdefics3VisionConfig)ÚIdefics3ImageProcessor)ÚIdefics3ImageProcessorPil)ÚIdefics3BaseModelOutputWithPastÚ Idefics3ForConditionalGenerationÚIdefics3ModelÚIdefics3PreTrainedModelÚIdefics3VisionTransformerz$HuggingFaceTB/SmolVLM2-2.2B-Instruct)Ú
checkpointc                   ó   — e Zd ZdZdZdS )ÚSmolVLMVisionConfiga  
    Example:

    ```python
    >>> from transformers.models.smolvlm.modeling_smolvlm import SmolVLMVisionTransformer
    >>> from transformers.models.smolvlm.configuration_smolvlm import SmolVLMVisionConfig

    >>> # Initializing a SmolVLMVisionConfig with google/siglip-so400m-patch14-384 style configuration
    >>> configuration = SmolVLMVisionConfig()

    >>> # Initializing a SmolVLMVisionTransformer (with random weights) from the google/siglip-so400m-patch14-384 style configuration
    >>> model = SmolVLMVisionTransformer(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úsmolvlm_visionN©Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_type© ó    úi/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/smolvlm/modular_smolvlm.pyr   r   +   s   € € € € € ðð ð" "€J€J€Jr'   r   c                   ó   — e Zd ZdS )ÚSmolVLMPreTrainedModelN©r!   r"   r#   r&   r'   r(   r*   r*   B   ó   € € € € € Ø€Dr'   r*   c                   ó   — e Zd ZdS )ÚSmolVLMVisionTransformerNr+   r&   r'   r(   r.   r.   F   r,   r'   r.   c                   ó   — e Zd ZdZdZdS )ÚSmolVLMConfigaÆ  
    scale_factor (`int`, *optional*, defaults to 2):
        The scale factor for the image encoder.

    Example:
    ```python
    >>> from transformers import SmolVLMModel, SmolVLMConfig
    >>> # Initializing configuration
    >>> configuration = SmolVLMConfig()
    >>> # Initializing a model from the configuration
    >>> model = SmolVLMModel(configuration)
    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```ÚsmolvlmNr    r&   r'   r(   r0   r0   J   s   € € € € € ðð ð €J€J€Jr'   r0   c                   ó   — e Zd ZdS )ÚSmolVLMImageProcessorNr+   r&   r'   r(   r3   r3   _   r,   r'   r3   c                   ó   — e Zd ZdS )ÚSmolVLMImageProcessorPilNr+   r&   r'   r(   r5   r5   c   r,   r'   r5   c                   ó   — e Zd ZdS )ÚSmolVLMBaseModelOutputWithPastNr+   r&   r'   r(   r7   r7   g   r,   r'   r7   c                   óÚ  — e Zd ZdZdej        dej        dej        fd„Ze e	d¬¦  «        	 dd	ej
        d
ej        dz  dee         deez  fd„¦   «         ¦   «         Zee e	d¬¦  «        	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dedz  dej
        dz  d	ej
        dz  d
ej        dz  dej
        dz  dedz  dee         deez  fd„¦   «         ¦   «         ¦   «         ZdS )ÚSmolVLMModelz§
    A subclass of Idefics3Model. We do *not* remove or block the call to inputs_merger
    in forward. Instead, we override inputs_merger here with custom logic.
    Ú	input_idsÚinputs_embedsÚimage_hidden_statesc                 ó0  — |j         \  }}}|€X| |                      ¦   «         t          j        | j        j        t          j        |j        ¬¦  «        ¦  «        k    }|d         }n|| j        j        k    }|                     d¬¦  «        }t          t          j
        ||z  dk    ¦  «        d¦  «         ||z  }t          j        j                             |                     d¬¦  «        dd¬¦  «        }	|	d d	…         }
|                     d	¬¦  «        }|dz
  |z  }|dz
  |z  }|
                     d¦  «        |z   }t          j        |¦  «        }|||         ||         d d …f         ||<   t          j        |                     d	¦  «        ||¦  «        }|S )
N©ÚdtypeÚdevice).r   é   ©Údimr   zCAt least one sample has <image> tokens not divisible by patch_size.)rA   r   )Úvalueéÿÿÿÿ)ÚshapeÚget_input_embeddingsÚtorchÚtensorÚconfigÚimage_token_idÚlongr@   Úsumr   Úallr   Ú
functionalÚpadÚcumsumÚ	unsqueezeÚ
zeros_likeÚwhere)Úselfr:   r;   r<   Ú_Ú
patch_sizeÚ
image_maskÚnum_image_tokensÚblocks_per_sampleÚoffsetsÚblock_offsetÚrow_cumÚ	chunk_idxÚ	local_idxÚ	block_idxÚimage_embedsÚmerged_embedss                    r(   Úinputs_mergerzSmolVLMModel.inputs_mergerq   s°  € ð /Ô4Ñˆˆ:�qàÐØ&Ð*E¨$×*CÒ*CÑ*EÔ*EÝ”˜Tœ[Ô7½u¼zÐR_ÔRfÐgÑgÔgñ+ô +ò ˆJð $ FÔ+ˆJˆJà" d¤kÔ&@Ò@ˆJà%Ÿ>š>¨a˜>Ñ0Ô0ÐÝÝŒIÐ&¨Ñ3°qÒ8Ñ9Ô9ØQñ	
ô 	
ð 	
ð -°
Ñ:Ðå”(Ô%×)Ò)Ð*;×*BÒ*BÀqÐ*BÑ*IÔ*IÈ6ÐYZÐ)Ñ[Ô[ˆØ˜s ˜s”|ˆØ×#Ò#¨Ð#Ñ+Ô+ˆØ˜q‘[ ZÑ/ˆ	Ø˜q‘[ JÑ.ˆ	Ø ×*Ò*¨1Ñ-Ô-°	Ñ9ˆ	åÔ'¨Ñ6Ô6ˆØ#6°yÀÔ7LÈiÐXbÔNcÐefÐefÐefÐ7fÔ#gˆ�ZÑ åœ J×$8Ò$8¸Ñ$<Ô$<¸lÈMÑZÔZˆØÐr'   zVEncodes images into continuous embeddings that can be forwarded to the language model.)Úcustom_introNÚpixel_valuesÚpixel_attention_maskÚkwargsÚreturnc                 ó¨  ‡— ‰j         \  }}}}}‰                     | j        ¬¦  «        Š ‰j        ||z  g‰j         dd…         ¢R Ž Š‰j         dd…                              ¦   «         }	‰dk                         d¬¦  «        |	k    }
|
dxx         t          j        |
¦  «         z  cc<   ‰|
                              ¦   «         Š|€3t          j	        ˆfd	„d
D ¦   «         t          j
        ‰j        ¬¦  «        }n8 |j        ||z  g|j         dd…         ¢R Ž }||
                              ¦   «         }| j        j        j        }|                     d||¬¦  «        }|                     d||¬¦  «        }|                     d¬¦  «        dk     
                    ¦   «         } | j        d‰|ddœ|¤Ž}|j        }|                      |¦  «        }||_        |S )a4  
        pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The tensors corresponding to the input images.
        pixel_attention_mask (`torch.LongTensor`, *optional*):
            The attention mask indicating padded regions in the image.
        )r?   r   NrA   g        )rE   éþÿÿÿéýÿÿÿrB   r   c                 ó*   •— g | ]}‰j         |         ‘ŒS r&   )rF   )Ú.0Úire   s     €r(   ú
<listcomp>z3SmolVLMModel.get_image_features.<locals>.<listcomp>±   s!   ø€ Ð?Ð?Ð?°�lÔ(¨Ô+Ð?Ð?Ð?r'   )r   r   r   )Úsizer?   r@   )Ú	dimensionrp   Ústep)rE   rj   T)re   Úpatch_attention_maskÚreturn_dictr&   )rF   Útor?   ÚviewÚnumelrM   rH   ÚanyÚ
contiguousÚonesÚboolr@   rJ   Úvision_configrW   ÚunfoldÚvision_modelÚlast_hidden_stateÚ	connectorÚpooler_output)rU   re   rf   rg   Ú
batch_sizeÚ
num_imagesÚnum_channelsÚheightÚwidthÚnb_values_per_imageÚreal_images_indsrW   Úpatches_subgridrs   Úimage_outputsr<   Úimage_featuress    `               r(   Úget_image_featureszSmolVLMModel.get_image_features’   s7  ø€ ð  ?KÔ>PÑ;ˆ
�J ¨f°eØ#—’¨T¬Z�Ñ8Ô8ˆØ(�|Ô(¨°jÑ)@ÐZÀ<ÔCUÐVWÐVXÐVXÔCYÐZÐZÐZˆð +Ô0°°°Ô4×:Ò:Ñ<Ô<ÐØ(¨CÒ/×4Ò4¸Ð4ÑFÔFÐJ]Ò]Ðð 	˜ÐÐÔ¥¤	Ð*:Ñ ;Ô ;Ð;Ñ;ÐÐÑà#Ð$4Ô5×@Ò@ÑBÔBˆàÐ'Ý#(¤:Ø?Ð?Ð?Ð?°YÐ?Ñ?Ô?Ý”jØ#Ô*ð$ñ $ô $Ð Ð ð $=Ð#7Ô#<¸ZÈ*Ñ=TÐ#vÐWkÔWqÐrsÐrtÐrtÔWuÐ#vÐ#vÐ#vÐ Ø#7Ð8HÔ#I×#TÒ#TÑ#VÔ#VÐ Ø”[Ô.Ô9ˆ
Ø.×5Ò5ÀÈ
ÐYcÐ5ÑdÔdˆØ)×0Ò0¸1À:ÐT^Ð0Ñ_Ô_ˆØ /× 3Ò 3¸Ð 3Ñ AÔ AÀAÒ E×KÒKÑMÔMÐð *˜Ô)ð 
Ø%Ð<PÐ^bð
ð 
Øflð
ð 
ˆð ,Ô=Ðð ŸšÐ(;Ñ<Ô<ˆØ&4ˆÔ#àÐr'   aØ  
        Inputs fed to the model can have an arbitrary number of images. To account for this, pixel_values fed to
        the model have image padding -> (batch_size, max_num_images, 3, max_heights, max_widths) where
        max_num_images is the maximum number of images among the batch_size samples in the batch.
        Padding images are not needed beyond padding the pixel_values at the entrance of the model.
        For efficiency, we only pass through the vision_model's forward the real images by
        discarding the padding images i.e. pixel_values of size (image_batch_size, 3, height, width) where
        image_batch_size would be 7 when num_images_per_sample=[1, 3, 1, 2] and max_num_images would be 3.
        Úattention_maskÚposition_idsÚpast_key_valuesÚ	use_cachec
           	      óŠ  — |�|j         \  }}n|�|j         \  }}}nt          d¦  «        ‚|	r|€t          | j        ¬¦  «        }|€: | j                             ¦   «         |¦  «                             |j        ¦  «        }|�|�t          d¦  «        ‚|�8|                      ||d¬¦  «        j	        }|                     |j        ¦  «        }n#|�!|                     | j
        |j        ¬¦  «        }|�|                      |||¬¦  «        } | j        d
|||||	dœ|
¤Ž}t          |j        |j        |j        |j        |¬	¦  «        S )Nz5You have to specify either input_ids or inputs_embeds)rJ   zMYou cannot specify both pixel_values and image_hidden_states at the same timeT)rt   r>   )r:   r;   r<   )r;   r�   rŽ   r�   r�   )r   r�   Úhidden_statesÚ
attentionsr<   r&   )rF   Ú
ValueErrorr   rJ   Ú
text_modelrG   ru   r@   rŒ   r�   r?   rc   r7   r   r�   r’   r“   )rU   r:   r�   rŽ   r�   r;   re   rf   r<   r�   rg   r‚   Ú
seq_lengthrV   Úoutputss                  r(   ÚforwardzSmolVLMModel.forwardÊ   s±  € ð4 Ð Ø%.¤_Ñ"ˆJ˜
˜
ØÐ&Ø(5Ô(;Ñ%ˆJ˜
 A AåÐTÑUÔUÐUàð 	?˜Ð0Ý*°$´+Ð>Ñ>Ô>ˆOàÐ ØB˜DœO×@Ò@ÑBÔBÀ9ÑMÔM×PÒPÐQZÔQaÑbÔbˆMàÐ#Ð(;Ð(GÝÐlÑmÔmÐmàÐ#Ø"&×"9Ò"9ØÐ2Àð #:ñ #ô #äð  ð #6×"8Ò"8¸Ô9MÑ"NÔ"NÐÐØ Ð,Ø"5×"8Ò"8¸t¼zÐR_ÔRfÐ"8Ñ"gÔ"gÐàÐ*Ø ×.Ò.Ø#Ø+Ø$7ð /ñ ô ˆMð "�$”/ð 
Ø'Ø)Ø%Ø+Øð
ð 
ð ð
ð 
ˆõ .Ø%Ô7Ø#Ô3Ø!Ô/ØÔ)Ø 3ð
ñ 
ô 
ð 	
r'   )N)	NNNNNNNNN)r!   r"   r#   r$   rH   Ú
LongTensorÚTensorrc   r   r   ÚFloatTensorr   r   Útupler
   rŒ   r   r   Ú
BoolTensorr{   r	   r7   r˜   r&   r'   r(   r9   r9   k   s  € € € € € ðð ð
ØÔ)ðØ:?¼,ðØ]bÔ]iðð ð ð ðB Ø€^Ømðñ ô ð 9=ð2ð 2àÔ'ð2ð $Ô.°Ñ5ð2ð Ð+Ô,ð	2ð
 
Ð+Ñ	+ð2ð 2ð 2ñô ñ Ôð2ðh  ØØ€^ðð
ñ 
ô 
ð .2Ø.2Ø04Ø(,Ø26Ø15Ø8<Ø8<Ø!%ð;
ð ;
àÔ# dÑ*ð;
ð œ tÑ+ð;
ð Ô&¨Ñ-ð	;
ð
  ™ð;
ð Ô(¨4Ñ/ð;
ð Ô'¨$Ñ.ð;
ð $Ô.°Ñ5ð;
ð #Ô.°Ñ5ð;
ð ˜$‘;ð;
ð Ð-Ô.ð;
ð 
Ð/Ñ	/ð;
ð ;
ð ;
ñ
ô 
ñ Ôñ  Ôð;
ð ;
ð ;
r'   r9   c                   ó0   ‡ — e Zd ZddiZˆ fd„Zˆ fd„Zˆ xZS )ÚSmolVLMForConditionalGenerationzlm_head.weightz$model.text_model.embed_tokens.weightc                 ó@  •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          j        |¦  «        | j        j        _        t          j	        |j
        j        |j
        j        d¬¦  «        | _        |                      ¦   «          d S )NF)Úbias)ÚsuperÚ__init__r9   Úmodelr   Úfrom_model_configr•   Úgeneration_configr   ÚLinearÚtext_configÚhidden_sizeÚ
vocab_sizeÚlm_headÚ	post_init)rU   rJ   Ú	__class__s     €r(   r£   z(SmolVLMForConditionalGeneration.__init__  s~   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý! &Ñ)Ô)ˆŒ
Ý2BÔ2TÐU[Ñ2\Ô2\ˆŒ
ÔÔ/Ý”y Ô!3Ô!?ÀÔASÔA^ÐejÐkÑkÔkˆŒØ�ŠÑÔÐÐÐr'   c                 ó:   •—  t          ¦   «         j        di |¤Ž dS )aÔ	  
        pixel_attention_mask (`torch.Tensor` of shape `(batch_size, image_size, image_size)`, *optional*):
            Mask to avoid performing attention on padding pixel indices.
        image_hidden_states (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The hidden states of the image encoder after modality projection.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            config.vocab_size]` or `model.image_token_id`. Tokens with indices set to `model.image_token_id` are
            ignored (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.

        Example:

        ```python
        >>> import httpx
        >>> from io import BytesIO
        >>> import torch
        >>> from PIL import Image
        >>> from io import BytesIO

        >>> from transformers import AutoProcessor, AutoModelForImageTextToText
        >>> from transformers.image_utils import load_image

        >>> # Note that passing the image urls (instead of the actual pil images) to the processor is also possible
        >>> image1 = load_image("https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg")
        >>> image2 = load_image("https://cdn.britannica.com/59/94459-050-DBA42467/Skyline-Chicago.jpg")
        >>> image3 = load_image("https://cdn.britannica.com/68/170868-050-8DDE8263/Golden-Gate-Bridge-San-Francisco.jpg")

        >>> processor = AutoProcessor.from_pretrained("HuggingFaceTB/SmolVLM2-2.2B-Instruct")
        >>> model = AutoModelForImageTextToText.from_pretrained("HuggingFaceTB/SmolVLM2-2.2B-Instruct", dtype=torch.bfloat16, device_map="auto")

        >>> # Create inputs
        >>> messages = [
        ...     {
        ...         "role": "user",
        ...         "content": [
        ...             {"type": "video", "path": path/to/video},
        ...             {"type": "text", "text": "What is happening in this video?"},
        ...         ]
        ...     }
        ... ]

        >>> inputs = processor.apply_chat_template([messages], add_generation_prompt=True)

        >>> # Generate
        >>> generated_ids = model.generate(**inputs, max_new_tokens=256)
        >>> generated_texts = processor.batch_decode(generated_ids, skip_special_tokens=True)

        >>> print(generated_texts)
        ```Nr&   )r¢   r˜   )rU   Úsuper_kwargsr­   s     €r(   r˜   z'SmolVLMForConditionalGeneration.forward  s(   ø€ ðd 	�‰ŒŒÐ'Ð'˜,Ð'Ð'Ð'Ð'Ð'r'   )r!   r"   r#   Ú_tied_weights_keysr£   r˜   Ú__classcell__)r­   s   @r(   rŸ   rŸ     s]   ø€ € € € € Ø*Ð,RÐSÐðð ð ð ð ð2(ð 2(ð 2(ð 2(ð 2(ð 2(ð 2(ð 2(ð 2(r'   rŸ   )r   r0   r3   r5   rŸ   r*   r9   r.   )1rH   Úhuggingface_hub.dataclassesr   r   Úcache_utilsr   r   Ú
generationr   Úmodeling_flash_attention_utilsr	   Úmodeling_outputsr
   Úprocessing_utilsr   Úutilsr   r   r   r   r   Úutils.genericr   Úidefics3.configuration_idefics3r   r   Ú"idefics3.image_processing_idefics3r   Ú&idefics3.image_processing_pil_idefics3r   Úidefics3.modeling_idefics3r   r   r   r   r   Ú
get_loggerr!   Úloggerr   r*   r.   r0   r3   r5   r7   r9   rŸ   Ú__all__r&   r'   r(   ú<module>rÁ      sR  ðð" €€€Ø .Ð .Ð .Ð .Ð .Ð .Ø Ð Ð Ð Ð Ð à .Ð .Ð .Ð .Ð .Ð .Ð .Ð .Ø *Ð *Ð *Ð *Ð *Ð *Ø BÐ BÐ BÐ BÐ BÐ BØ :Ð :Ð :Ð :Ð :Ð :Ø &Ð &Ð &Ð &Ð &Ð &Ø jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jØ 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø RÐ RÐ RÐ RÐ RÐ RÐ RÐ RØ GÐ GÐ GÐ GÐ GÐ GØ NÐ NÐ NÐ NÐ NÐ Nðð ð ð ð ð ð ð ð ð ð ð ð ð ð 
ˆÔ	˜HÑ	%Ô	%€ð €ÐAÐBÑBÔBØð"ð "ð "ð "ð "Ð.ñ "ô "ñ „ñ CÔBð"ð*	ð 	ð 	ð 	ð 	Ð4ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð8ñ 	ô 	ð 	ð €ÐAÐBÑBÔBØðð ð ð ð �Nñ ô ñ „ñ CÔBðð&	ð 	ð 	ð 	ð 	Ð2ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð8ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð%Dñ 	ô 	ð 	ðg
ð g
ð g
ð g
ð g
�=ñ g
ô g
ð g
ðT<(ð <(ð <(ð <(ð <(Ð&Fñ <(ô <(ð <(ð~	ð 	ð 	€€€r'   