§
    ‚Štj:;  ã                   óh  — d Z ddlZddlmZ ddlmZ ddlmZ ddlmZm	Z	 ddl
mZ dd	lmZ dd
lmZ ddlmZmZmZmZmZ ddlmZ  ej        e¦  «        Ze G d„ de¦  «        ¦   «         Z ed¬¦  «         G d„ de¦  «        ¦   «         Z ed¬¦  «         G d„ dee¦  «        ¦   «         Zg d¢ZdS )zPyTorch Fuyu model.é    N)Únné   )ÚCache)ÚGenerationMixin)ÚBaseModelOutputWithPoolingÚCausalLMOutputWithPast)ÚPreTrainedModel)Ú	AutoModel)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚcan_return_tupleÚloggingÚtorch_compilable_checké   )Ú
FuyuConfigc                   ó@   — e Zd ZU eed<   dZdZdZdZdZ	dZ
dZg ZdgZdS )ÚFuyuPreTrainedModelÚconfigÚmodel)ÚimageÚtextTÚpast_key_valuesN)Ú__name__Ú
__module__Ú__qualname__r   Ú__annotations__Úbase_model_prefixÚinput_modalitiesÚsupports_gradient_checkpointingÚ_supports_attention_backendÚ_supports_flash_attnÚ_supports_sdpaÚ_supports_flex_attnÚ_no_split_modulesÚ_skip_keys_device_placement© ó    úd/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/fuyu/modeling_fuyu.pyr   r       sV   € € € € € € àÐÐÑØÐØ(ÐØ&*Ð#Ø"&ÐØÐØ€NØÐØÐØ#4Ð"5ÐÐÐr(   r   zt
    The Fuyu model which consists of a vision backbone and a language model, without a language modeling head.
    )Úcustom_introc                   óÒ  ‡ — e Zd Zdefˆ fd„Zdej        deej                 dej        dej        fd„Ze	e
dej        d	ee         deez  fd
„¦   «         ¦   «         Zdej        dej        dej        fd„Ze	e
	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dedz  dej        dz  dedz  d	ee         deez  fd„¦   «         ¦   «         Zˆ xZS )Ú	FuyuModelr   c                 ó^  •— t          ¦   «                              |¦  «         |j        | _        |j        j        | _        t          j        |j        ¦  «        | _        t          j
        |j        |j        z  |j        z  |j        ¦  «        | _        d| _        |                      ¦   «          d S )NF)ÚsuperÚ__init__Úpad_token_idÚpadding_idxÚtext_configÚ
vocab_sizer
   Úfrom_configÚlanguage_modelr   ÚLinearÚ
patch_sizeÚnum_channelsÚhidden_sizeÚvision_embed_tokensÚgradient_checkpointingÚ	post_init©Úselfr   Ú	__class__s     €r)   r/   zFuyuModel.__init__4   s˜   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø!Ô.ˆÔØ Ô,Ô7ˆŒÝ'Ô3°FÔ4FÑGÔGˆÔÝ#%¤9ØÔ Ô 1Ñ1°FÔ4GÑGÈÔI[ñ$
ô $
ˆÔ ð ',ˆÔ#à�ŠÑÔÐÐÐr(   Úword_embeddingsÚcontinuous_embeddingsÚimage_patch_input_indicesÚreturnc           
      óR  — |j         d         t          |¦  «        k    s-t          dt          |¦  «        ›d|j         d         ›�¦  «        ‚|                     ¦   «         }t	          |j         d         ¦  «        D ]¬}t          j        ||         dk    d¬¦  «        d         }||         |         }|j         d         ||         j         d         k    r)t          d||         j         ›d|j         ›d|› d	�¦  «        ‚||         |                              |j        ¦  «        |||f<   Œ­|S )
aÙ  This function places the continuous_embeddings into the word_embeddings at the locations
        indicated by image_patch_input_indices. Different batch elements can have different numbers of continuous
        embeddings.

        Args:
            word_embeddings (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
                Tensor of word embeddings.
            continuous_embeddings (`torch.FloatTensor` of shape `(batch_size, num_patches, hidden_size)`):
                Tensor of continuous embeddings. The length of the list is the batch size. Each entry is shape
                [num_image_embeddings, hidden], and num_image_embeddings needs to match the number of non-negative
                indices in image_patch_input_indices for that batch element.
            image_patch_input_indices (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                Tensor of indices of the image patches in the input_ids tensor.
        r   z7Batch sizes must match! Got len(continuous_embeddings)=z and word_embeddings.shape[0]=T)Úas_tuplezGNumber of continuous embeddings continuous_embeddings[batch_idx].shape=zA does not match number of continuous token ids src_indices.shape=z in batch element ú.)	ÚshapeÚlenÚ
ValueErrorÚcloneÚrangeÚtorchÚnonzeroÚtoÚdevice)r>   r@   rA   rB   Úoutput_embeddingsÚ	batch_idxÚdst_indicesÚsrc_indicess           r)   Úgather_continuous_embeddingsz&FuyuModel.gather_continuous_embeddingsA   s{  € ð(  Ô% aÔ(­CÐ0EÑ,FÔ,FÒFÐFÝØl­sÐ3HÑ/IÔ/IÐlÐlÐQ`ÔQfÐghÔQiÐlÐlñô ð ð ,×1Ò1Ñ3Ô3ÐÝ˜Ô4°QÔ7Ñ8Ô8ð 	ð 	ˆIõ  œ-Ð(AÀ)Ô(LÐPQÒ(QÐ\`ÐaÑaÔaÐbcÔdˆKð 4°IÔ>¸{ÔKˆKàÔ  Ô#Ð&;¸IÔ&FÔ&LÈQÔ&OÒOÐOÝ ðiÐ7LÈYÔ7WÔ7]ð ið iØ6AÔ6Gðið iØ\eðið ið iñô ð ð 9NÈiÔ8XÐYdÔ8e×8hÒ8hØ!Ô(ñ9ô 9Ð˜i¨Ð4Ñ5Ð5ð !Ð r(   Úpixel_valuesÚkwargsc                 óL   — |                       |¦  «        }t          |¬¦  «        S )z®
        pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The tensors corresponding to the input images.
        )Úlast_hidden_state)r:   r   )r>   rU   rV   Úpatch_embeddingss       r)   Úget_image_featureszFuyuModel.get_image_featuresm   s*   € ð  ×3Ò3°LÑAÔAÐÝ)Ð<LÐMÑMÔMÐMr(   Ú	input_idsÚinputs_embedsÚimage_featuresc                 ó   — |€e| |                       ¦   «         t          j        | j        j        t          j        |j        ¬¦  «        ¦  «        k    }|                     d¦  «        }n|| j        j        k    }|                     ¦   «         }|j	        d         |j	        d         z  }| 
                    d¦  «                             |j        ¦  «        }t          ||j	        d         z  |                     ¦   «         k    d|› d|› �¦  «         |S )zï
        Obtains multimodal placeholder mask from `input_ids` or `inputs_embeds`, and checks that the placeholder token count is
        equal to the length of multimodal features. If the lengths are different, an error is raised.
        N©ÚdtyperO   éÿÿÿÿr   r   z6Image features and image tokens do not match, tokens: z, features: )Úget_input_embeddingsrL   Útensorr   Úimage_token_idÚlongrO   ÚallÚsumrG   Ú	unsqueezerN   r   Únumel)r>   r[   r\   r]   Úspecial_image_maskÚn_image_tokensÚn_image_featuress          r)   Úget_placeholder_maskzFuyuModel.get_placeholder_masky   s  € ð ÐØ!.Ð2M°$×2KÒ2KÑ2MÔ2MÝ”˜Tœ[Ô7½u¼zÐR_ÔRfÐgÑgÔgñ3ô 3ò "Ðð "4×!7Ò!7¸Ñ!;Ô!;ÐÐà!*¨d¬kÔ.HÒ!HÐà+×/Ò/Ñ1Ô1ˆØ)Ô/°Ô2°^Ô5IÈ!Ô5LÑLÐØ/×9Ò9¸"Ñ=Ô=×@Ò@ÀÔAUÑVÔVÐÝØ˜]Ô0°Ô4Ñ4¸×8LÒ8LÑ8NÔ8NÒNØsÀ^ÐsÐsÐaqÐsÐsñ	
ô 	
ð 	
ð "Ð!r(   NÚimage_patchesÚimage_patches_indicesÚattention_maskÚposition_idsr   Ú	use_cachec	           	      ó^  — |du |duz  rt          d¦  «        ‚|€" | j                             ¦   «         |¦  «        }|j        d         }
|€b|�|j        n|j        }|�|                     ¦   «         nd}t          j        ||
|z   t          j        |¬¦  «        }| 	                    d¦  «        }|�j|  
                    |d¬¦  «        j        }|                     |j        |j        ¦  «        }|                      |||¬¦  «        }|                     ||¦  «        } | j        d
|||||d	œ|	¤Ž}|S )aä  
        image_patches (`torch.FloatTensor` of shape `(batch_size, num_total_patches, patch_size_ x patch_size x num_channels)`, *optional*):
            Image patches to be used as continuous embeddings. The patches are flattened and then projected to the
            hidden size of the model.
        image_patches_indices (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Tensor of indices of the image patches in the input_ids tensor.
        Nz:You must specify exactly one of input_ids or inputs_embedsr   r   r_   T)Úreturn_dict)r\   r]   )r\   rp   rq   r   rr   r'   )rI   r5   rb   rG   rO   Úget_seq_lengthrL   Úarangere   rh   rZ   rX   rN   r`   rm   Úmasked_scatter)r>   r[   rn   ro   rp   rq   r   r\   rr   rV   Úseq_lenrO   Úpast_key_values_lengthrY   rj   Úoutputss                   r)   ÚforwardzFuyuModel.forward‘   sŒ  € ð, ˜Ð -°tÐ";Ñ<ð 	[ÝÐYÑZÔZÐZàÐ ØF˜DÔ/×DÒDÑFÔFÀyÑQÔQˆMàÔ% aÔ(ˆàÐØ)2Ð)>�YÔ%Ð%ÀMÔDXˆFØIXÐId _×%CÒ%CÑ%EÔ%EÐ%EÐjkÐ"Ý œ<Ø&¨Ð2HÑ(HÕPUÔPZÐciðñ ô ˆLð (×1Ò1°!Ñ4Ô4ˆLàÐ$Ø#×6Ò6°}ÐRVÐ6ÑWÔWÔiÐØ/×2Ò2°=Ô3GÈÔI\Ñ]Ô]ÐØ!%×!:Ò!:Ø¨ÐGWð ";ñ "ô "Ðð *×8Ò8Ð9KÐM]Ñ^Ô^ˆMà%�$Ô%ð 
Ø'Ø)Ø%Ø+Øð
ð 
ð ð
ð 
ˆð ˆr(   )NNNNNNNN)r   r   r   r   r/   rL   ÚTensorÚlistrT   r   r   ÚFloatTensorr   r   Útupler   rZ   Ú
LongTensorrm   r   Úboolr   r{   Ú__classcell__©r?   s   @r)   r,   r,   .   s  ø€ € € € € ð˜zð ð ð ð ð ð ð*!àœð*!ð  $ E¤LÔ1ð*!ð $)¤<ð	*!ð
 
Œð*!ð *!ð *!ð *!ðX ØðNØ!Ô-ðNØ9?Ð@RÔ9SðNà	Ð+Ñ	+ðNð Nð Nñ „^ñ ÔðNð"ØÔ)ð"Ø:?Ô:Kð"Ø]bÔ]nð"ð "ð "ð "ð0 Øð .2à-1Ø59Ø.2Ø04Ø(,Ø26Ø!%ð5ð 5àÔ# dÑ*ð5ð ”| dÑ*ð	5ð
  %œ|¨dÑ2ð5ð œ tÑ+ð5ð Ô&¨Ñ-ð5ð  ™ð5ð Ô(¨4Ñ/ð5ð ˜$‘;ð5ð Ð+Ô,ð5ð 
Ð'Ñ	'ð5ð 5ð 5ñ „^ñ Ôð5ð 5ð 5ð 5ð 5r(   r,   zz
    Fuyu Model with a language modeling head on top for causal language model conditioned on image patches and text.
    c                   óF  ‡ — e Zd ZddiZdefˆ fd„Zee	 	 	 	 	 	 	 	 	 	 ddej	        dz  dej
        dz  d	ej
        dz  d
ej
        dz  dej	        dz  dedz  dej        dz  dedz  dej
        dz  dedz  dee         deez  fd„¦   «         ¦   «         Z	 	 	 	 	 	 dˆ fd„	Zˆ xZS )ÚFuyuForCausalLMzlm_head.weightz(model.language_model.embed_tokens.weightr   c                 óú   •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          j        |j        j        |j        j        d¬¦  «        | _	        |  
                    ¦   «          d S )NF)Úbias)r.   r/   r,   r   r   r6   r2   r9   r3   Úlm_headr<   r=   s     €r)   r/   zFuyuForCausalLM.__init__Ó   se   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý˜vÑ&Ô&ˆŒ
Ý”y Ô!3Ô!?ÀÔASÔA^ÐejÐkÑkÔkˆŒØ�ŠÑÔÐÐÐr(   Nr   r[   rn   ro   rp   rq   r   r\   rr   ÚlabelsÚlogits_to_keeprV   rC   c                 ó`  —  | j         d||||||||dœ|¤Ž}|d         }t          |
t          ¦  «        rt          |
 d¦  «        n|
}|                      |dd…|dd…f         ¦  «        }d}|	�  | j        d||	| j        j        j        dœ|¤Ž}t          |||j
        |j        |j        ¬¦  «        S )a‘  
        image_patches (`torch.FloatTensor` of shape `(batch_size, num_total_patches, patch_size_ x patch_size x num_channels)`, *optional*):
            Image patches to be used as continuous embeddings. The patches are flattened and then projected to the
            hidden size of the model.
        image_patches_indices (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Tensor of indices of the image patches in the input_ids tensor.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            config.text_config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.text_config.vocab_size]`.

        Examples:

        ```python
        >>> from transformers import FuyuProcessor, FuyuForCausalLM
        >>> from PIL import Image
        >>> import httpx
        >>> from io import BytesIO

        >>> processor = FuyuProcessor.from_pretrained("adept/fuyu-8b")
        >>> model = FuyuForCausalLM.from_pretrained("adept/fuyu-8b")

        >>> url = "https://huggingface.co/datasets/hf-internal-testing/fixtures-captioning/resolve/main/bus.png"
        >>> with httpx.stream("GET", url) as response:
        ...     image = Image.open(BytesIO(response.read()))
        >>> prompt = "Generate a coco-style caption.\n"

        >>> inputs = processor(images=image, text=prompt, return_tensors="pt")
        >>> outputs = model(**inputs)

        >>> generated_ids = model.generate(**inputs, max_new_tokens=7)
        >>> generation_text = processor.batch_decode(generated_ids[:, -7:], skip_special_tokens=True)
        >>> print(generation_text[0])
        A blue bus parked on the side of a road.
        ```)r[   rn   ro   r\   rp   rq   r   rr   r   N)Úlogitsr‰   r3   )ÚlossrŒ   r   Úhidden_statesÚ
attentionsr'   )r   Ú
isinstanceÚintÚslicerˆ   Úloss_functionr   r2   r3   r   r   rŽ   r�   )r>   r[   rn   ro   rp   rq   r   r\   rr   r‰   rŠ   rV   rz   rŽ   Úslice_indicesrŒ   r�   s                    r)   r{   zFuyuForCausalLM.forwardÙ   s  € ðj �$”*ð 

ØØ'Ø"7Ø'Ø)Ø%Ø+Øð

ð 

ð ð

ð 

ˆð   œ
ˆå8BÀ>ÕSVÑ8WÔ8WÐk�˜~˜o¨tÑ4Ô4Ð4Ð]kˆØ—’˜m¨A¨A¨A¨}¸a¸a¸aÐ,?Ô@ÑAÔAˆàˆØÐØ%�4Ô%ð Ø f¸¼Ô9PÔ9[ðð Ø_eðð ˆDõ &ØØØ#Ô3Ø!Ô/ØÔ)ð
ñ 
ô 
ð 	
r(   Fc           
      óŽ   •—  t          ¦   «         j        |f||||||dœ|¤Ž}	|s |                     dd¦  «        r
d |	d<   d |	d<   |	S )N)r   rp   r\   rn   ro   Úis_first_iterationrr   Tro   rn   )r.   Úprepare_inputs_for_generationÚget)r>   r[   r   rp   r\   rn   ro   r–   rV   Úmodel_inputsr?   s             €r)   r—   z-FuyuForCausalLM.prepare_inputs_for_generation-  s€   ø€ ð =•u‘w”wÔ<Øð	
à+Ø)Ø'Ø'Ø"7Ø1ð	
ð 	
ð ð	
ð 	
ˆð "ð 	1 f§j¢j°¸dÑ&CÔ&Cð 	1à48ˆLÐ0Ñ1Ø,0ˆL˜Ñ)àÐr(   )
NNNNNNNNNr   )NNNNNF)r   r   r   Ú_tied_weights_keysr   r/   r   r   rL   r€   r|   r   r~   r�   r‘   r   r   r   r   r{   r—   r‚   rƒ   s   @r)   r…   r…   Ë   s¨  ø€ € € € € ð +Ð,VÐWÐð˜zð ð ð ð ð ð ð Øð .2à-1Ø59Ø.2Ø04Ø(,Ø26Ø!%Ø&*Ø%&ðP
ð P
àÔ# dÑ*ðP
ð ”| dÑ*ð	P
ð
  %œ|¨dÑ2ðP
ð œ tÑ+ðP
ð Ô&¨Ñ-ðP
ð  ™ðP
ð Ô(¨4Ñ/ðP
ð ˜$‘;ðP
ð ”˜tÑ#ðP
ð ˜d™
ðP
ð Ð+Ô,ðP
ð 
Ð'Ñ	'ðP
ð P
ð P
ñ „^ñ ÔðP
ðj ØØØØ"Ø ðð ð ð ð ð ð ð ð ð r(   r…   )r…   r   r,   )Ú__doc__rL   r   Úcache_utilsr   Ú
generationr   Úmodeling_outputsr   r   Úmodeling_utilsr	   Úmodels.auto.modeling_autor
   Úprocessing_utilsr   Úutilsr   r   r   r   r   Úconfiguration_fuyur   Ú
get_loggerr   Úloggerr   r,   r…   Ú__all__r'   r(   r)   ú<module>r§      sæ  ðð Ð à €€€Ø Ð Ð Ð Ð Ð à  Ð  Ð  Ð  Ð  Ð  Ø )Ð )Ð )Ð )Ð )Ð )Ø RÐ RÐ RÐ RÐ RÐ RÐ RÐ RØ -Ð -Ð -Ð -Ð -Ð -Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2Ø &Ð &Ð &Ð &Ð &Ð &Ø jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jÐ jØ *Ð *Ð *Ð *Ð *Ð *ð 
ˆÔ	˜HÑ	%Ô	%€ð ð
6ð 
6ð 
6ð 
6ð 
6˜/ñ 
6ô 
6ñ „ð
6ð €ððñ ô ð
Uð Uð Uð Uð UÐ#ñ Uô Uñô ð
Uðp €ððñ ô ð
zð zð zð zð zÐ)¨?ñ zô zñô ð
zðz BÐ
AÐ
A€€€r(   