§
    ‚Štjë_  ã                   ó  — d dl Z d dlmZ d dlmZmZmZ d dlmZm	Z	m
Z
mZmZmZmZmZ ddlmZ ddlmZ ddlmZmZ  ed	¬
¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z ed	¬
¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z ed	¬
¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de
¦  «        Z G d„ de¦  «        Z G d„ de	¦  «        Zg d¢Z dS )é    N)Ústrict)ÚInstructBlipConfigÚInstructBlipQFormerConfigÚInstructBlipVisionConfig)Ú'BaseModelOutputWithVisionQformerOutputsÚ$InstructBlipForConditionalGenerationÚ/InstructBlipForConditionalGenerationModelOutputÚInstructBlipModelÚInstructBlipPreTrainedModelÚInstructBlipQFormerModelÚInstructBlipVisionModelÚTransformersKwargsé   )ÚBaseModelOutputWithPooling)ÚUnpack)Úauto_docstringÚcan_return_tuplez"Salesforce/instructblip-flan-t5-xl)Ú
checkpointc                   ó   — e Zd ZdZdS )ÚInstructBlipVideoVisionConfigaH  
    Example:

    ```python
    >>> from transformers import InstructBlipVideoVisionConfig, InstructBlipVideoVisionModel

    >>> # Initializing a InstructBlipVideoVisionConfig with Salesforce/instructblip-flan-t5-xl style configuration
    >>> configuration = InstructBlipVideoVisionConfig()

    >>> # Initializing a InstructBlipVideoVisionModel (with random weights) from the Salesforce/instructblip-flan-t5-xl style configuration
    >>> model = InstructBlipVideoVisionModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```N©Ú__name__Ú
__module__Ú__qualname__Ú__doc__© ó    ú}/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/instructblipvideo/modular_instructblipvideo.pyr   r   (   s   € € € € € ðð ð ð r   r   c                   ó   — e Zd ZdZdS )ÚInstructBlipVideoQFormerConfiga3  
    cross_attention_frequency (`int`, *optional*, defaults to 2):
        The frequency of adding cross-attention to the Transformer layers.
    encoder_hidden_size (`int`, *optional*, defaults to 1408):
        The hidden size of the hidden states for cross-attention.

    Examples:

    ```python
    >>> from transformers import InstructBlipVideoQFormerConfig, InstructBlipVideoQFormerModel

    >>> # Initializing a InstructBlipVideo Salesforce/instructblip-flan-t5-xl style configuration
    >>> configuration = InstructBlipVideoQFormerConfig()

    >>> # Initializing a model (with random weights) from the Salesforce/instructblip-flan-t5-xl style configuration
    >>> model = InstructBlipVideoQFormerModel(configuration)
    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Nr   r   r   r   r    r    <   s   € € € € € ðð ð ð r   r    c                   óD   — e Zd ZU dZddiZdZedz  ed<    e¦   «         Z	dS )ÚInstructBlipVideoConfiga  
    qformer_config (`dict`, *optional*):
        Dictionary of configuration options used to initialize [`InstructBlipVideoQFormerConfig`].
    num_query_tokens (`int`, *optional*, defaults to 32):
        The number of query tokens passed through the Transformer.

    Example:

    ```python
    >>> from transformers import (
    ...     InstructBlipVideoVisionConfig,
    ...     InstructBlipVideoQFormerConfig,
    ...     OPTConfig,
    ...     InstructBlipVideoConfig,
    ...     InstructBlipVideoForConditionalGeneration,
    ... )

    >>> # Initializing a InstructBlipVideoConfig with Salesforce/instructblip-flan-t5-xl style configuration
    >>> configuration = InstructBlipVideoConfig()

    >>> # Initializing a InstructBlipVideoForConditionalGeneration (with random weights) from the Salesforce/instructblip-flan-t5-xl style configuration
    >>> model = InstructBlipVideoForConditionalGeneration(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config

    >>> # We can also initialize a InstructBlipVideoConfig from a InstructBlipVideoVisionConfig, InstructBlipVideoQFormerConfig and any PreTrainedConfig

    >>> # Initializing Instructblipvideo vision, Instructblipvideo Q-Former and language model configurations
    >>> vision_config = InstructBlipVideoVisionConfig()
    >>> qformer_config = InstructBlipVideoQFormerConfig()
    >>> text_config = OPTConfig()

    >>> config = InstructBlipVideoConfig(vision_config=vision_config, qformer_config=qformer_config, text_config=text_config)
    ```Úvideo_token_idÚvideo_token_indexN)
r   r   r   r   Úattribute_mapr$   ÚintÚ__annotations__ÚAttributeErrorÚimage_token_indexr   r   r   r"   r"   T   sM   € € € € € € ð"ð "ðH &Ð':Ð;€MØ$(Ð�s˜T‘zÐ(Ð(Ñ(Ø&˜Ñ(Ô(ÐÐÐr   r"   c                   ó   — e Zd ZdZdS )Ú InstructBlipVideoPreTrainedModel)ÚvideoÚtextN©r   r   r   Úinput_modalitiesr   r   r   r+   r+   €   s   € € € € € Ø(ÐÐÐr   r+   c                   ó   — e Zd ZdZdS )ÚInstructBlipVideoVisionModelr,   Nr.   r   r   r   r1   r1   „   s   € € € € € ØÐÐÐr   r1   c                   ó   — e Zd ZdS )ÚInstructBlipVideoQFormerModelN©r   r   r   r   r   r   r3   r3   ˆ   ó   € € € € € Ø€Dr   r3   c                   ó   — e Zd ZdS )Ú4InstructBlipVideoForConditionalGenerationModelOutputNr4   r   r   r   r7   r7   Œ   r5   r   r7   c                   ó  — e Zd Zee	 	 	 	 	 	 	 	 ddej        dej        dej        dz  dej        dz  dej        dz  dej        dz  d	ej        dz  d
ej        dz  de	de	dz  de
e         deez  fd„¦   «         ¦   «         ZdS )ÚInstructBlipVideoModelNFÚpixel_valuesÚqformer_input_idsÚqformer_attention_maskÚ	input_idsÚattention_maskÚdecoder_input_idsÚdecoder_attention_maskÚinputs_embedsÚinterpolate_pos_encodingÚ	use_cacheÚkwargsÚreturnc           	      óR  — |j         \  }}}}}|                     ||z  |||¦  «        } | j        d||	dœ|¤Ž}|d         }t          j        |                     ¦   «         d d…         t          j        |j        ¬¦  «        }| j         	                    |j         d         dd¦  «        }t          j        |                     ¦   «         d d…         t          j        |j        ¬¦  «        }|€t          j
        |¦  «        }|                     |d¬¦  «        }|                     |d¬¦  «        }|                     |j        ¦  «        }t          j        ||gd¬¦  «        } | j        d|||||dœ|¤Ž}|d         d d …d |                     d¦  «        …d d …f         }|                      |¦  «        }|                     || j        j        |z  d¦  «        }|€I | j                             ¦   «         |¦  «        }|| j        j        k    }|€t          j
        |¦  «        }nd| |                      ¦   «         t          j        | j        j        t          j        |j        ¬¦  «        ¦  «        k    }|                     d¦  «        }|                     d¦  «                             |j        ¦  «        }|                     |j        |j        ¦  «        }|                     ||¦  «        }| j        j        r | j        d|||
dœ|¤Ž}n | j        d|||||
d	œ|¤Ž}t7          |||¬
¦  «        S )N©r:   rB   r   éÿÿÿÿ©ÚdtypeÚdevice©Údimé   ©r=   r>   Úquery_embedsÚencoder_hidden_statesÚencoder_attention_mask©rA   r>   rC   )rA   r>   r?   r@   rC   )Úvision_outputsÚqformer_outputsÚlanguage_model_outputsr   )ÚshapeÚreshapeÚvision_modelÚtorchÚonesÚsizeÚlongrK   Úquery_tokensÚexpandÚ	ones_likeÚrepeat_interleaveÚtoÚcatÚqformerÚlanguage_projectionÚconfigÚnum_query_tokensÚlanguage_modelÚget_input_embeddingsr#   ÚtensorÚallÚ	unsqueezerJ   Úmasked_scatterÚuse_decoder_only_language_modelr7   )Úselfr:   r;   r<   r=   r>   r?   r@   rA   rB   rC   rD   Ú
batch_sizeÚframesÚchannelÚheightÚwidthrT   Úimage_embedsÚimage_attention_maskr^   Úquery_attention_maskÚquery_outputsÚquery_outputÚlanguage_model_inputsÚspecial_image_maskÚoutputss                              r   ÚforwardzInstructBlipVideoModel.forward‘   s¹  € ð$ 6BÔ5GÑ2ˆ
�F˜G V¨UØ#×+Ò+¨J¸Ñ,?ÀÈ&ÐRWÑXÔXˆà*˜Ô*ð 
Ø%Ø%=ð
ð 
ð ð
ð 
ˆð
 & aÔ(ˆõ  %œz¨,×*;Ò*;Ñ*=Ô*=¸c¸r¸cÔ*BÍ%Ì*Ð]iÔ]pÐqÑqÔqÐð Ô(×/Ò/°Ô0BÀ1Ô0EÀrÈ2ÑNÔNˆÝ$œz¨,×*;Ò*;Ñ*=Ô*=¸c¸r¸cÔ*BÍ%Ì*Ð]iÔ]pÐqÑqÔqÐà!Ð)Ý%*¤_Ð5FÑ%GÔ%GÐ"à-×?Ò?ÀÈAÐ?ÑNÔNÐØ!7×!IÒ!IÈ&ÐVWÐ!IÑ!XÔ!XÐØ!7×!:Ò!:Ð;OÔ;VÑ!WÔ!WÐÝ!&¤Ð,@ÐBXÐ+YÐ_`Ð!aÑ!aÔ!aÐØ$˜œð 
Ø'Ø1Ø%Ø".Ø#7ð
ð 
ð ð
ð 
ˆð % QÔ'¨¨¨Ð+A¨\×->Ò->¸qÑ-AÔ-AÐ+AÀ1À1À1Ð(DÔEˆð !%× 8Ò 8¸Ñ FÔ FÐð !6× =Ò =¸jÈ$Ì+ÔJfÐioÑJoÐqsÑ tÔ tÐØÐ ØF˜DÔ/×DÒDÑFÔFÀyÑQÔQˆMØ!*¨d¬kÔ.HÒ!HÐØÐ%Ý!&¤°Ñ!;Ô!;�øà!.Ð2M°$×2KÒ2KÑ2MÔ2MÝ”˜Tœ[Ô7½u¼zÐR_ÔRfÐgÑgÔgñ3ô 3ò "Ðð "4×!7Ò!7¸Ñ!;Ô!;Ðà/×9Ò9¸"Ñ=Ô=×@Ò@ÀÔAUÑVÔVÐØ 5× 8Ò 8¸Ô9MÈ}ÔObÑ cÔ cÐØ%×4Ò4Ð5GÐI^Ñ_Ô_ˆàŒ;Ô6ð 	Ø)�dÔ)ð Ø+Ø-Ø#ðð ð ð	ð ˆGˆGð *�dÔ)ð Ø+Ø-Ø"3Ø'=Ø#ðð ð ðð ˆGõ DØ)Ø)Ø#*ð
ñ 
ô 
ð 	
r   )NNNNNNFN)r   r   r   r   r   rZ   ÚFloatTensorÚ
LongTensorÚTensorÚboolr   r   Útupler7   r}   r   r   r   r9   r9   �   s*  € € € € € ØØð
 ;?Ø.2Ø26Ø59Ø:>Ø-1Ø).Ø!%ð[
ð [
àÔ'ð[
ð !Ô,ð[
ð !&Ô 0°4Ñ 7ð	[
ð
 Ô$ tÑ+ð[
ð Ô(¨4Ñ/ð[
ð !Ô+¨dÑ2ð[
ð !&Ô 0°4Ñ 7ð[
ð ”| dÑ*ð[
ð #'ð[
ð ˜$‘;ð[
ð Ð+Ô,ð[
ð 
ÐEÑ	Eð[
ð [
ð [
ñ „^ñ Ôð[
ð [
ð [
r   r9   c                   óŠ  — e Zd Zee	 	 ddej        dej        dej        dz  dedz  de	e
         deez  fd	„¦   «         ¦   «         Zd
„ Zdej        dej        fd„Zee	 	 	 	 	 	 	 	 	 ddej        dej        dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dededz  de	e
         deez  fd„¦   «         ¦   «         Z ej        ¦   «         	 	 	 	 	 	 ddej        dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dedej        fd„¦   «         ZdS )Ú)InstructBlipVideoForConditionalGenerationNFr:   r;   r<   rB   rD   rE   c           	      ó  — |j         \  }}}}	}
|                     ||z  ||	|
¦  «        } | j        d
||dœ|¤Ž}t          |j        |j        |j        |j        |d¬¦  «        }|d         }t          j	        | 
                    ¦   «         dd…         t          j        |j        ¬¦  «        }| j                             |j         d         dd¦  «        }t          j	        | 
                    ¦   «         dd…         t          j        |j        ¬¦  «        }|€t          j        |¦  «        }|                     |d¬¦  «        }|                     |d¬¦  «        }|                     |j        ¦  «        }t          j        ||gd¬¦  «        } | j        d
|||||d	œ|¤Ž}||_        |d         dd…d| 
                    d¦  «        …dd…f         }|                      |¦  «        }|                     || j        j        |z  d¦  «        }||_        |S )a  
        pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
            The tensors corresponding to the input images.
        qformer_input_ids (`torch.LongTensor` of shape (batch_size, sequence_length)):
            The sequence used as a prompt to be fed to the Q-Former module.
        qformer_attention_mask (`torch.LongTensor` of shape (batch_size, sequence_length), *optional*):
            Mask to avoid performing attention on padding token indices.
        rG   N)Úlast_hidden_stateÚpooler_outputÚhidden_statesÚ
attentionsrT   rU   r   rH   rI   rL   rN   rO   r   )rW   rX   rY   r   r†   r‡   rˆ   r‰   rZ   r[   r\   r]   rK   r^   r_   r`   ra   rb   rc   rd   rU   re   rf   rg   )ro   r:   r;   r<   rB   rD   rp   rq   rr   rs   rt   rT   ru   rv   r^   rw   rU   ry   Úvideo_featuress                      r   Úget_video_featuresz<InstructBlipVideoForConditionalGeneration.get_video_featuresò   si  € ð( 6BÔ5GÑ2ˆ
�F˜G V¨UØ#×+Ò+¨J¸Ñ,?ÀÈ&ÐRWÑXÔXˆà5F°TÔ5Fð 6
Ø%Ø%=ð6
ð 6
ð ð6
ð 6
ˆõ
 AØ,Ô>Ø(Ô6Ø(Ô6Ø%Ô0Ø)Ø ð
ñ 
ô 
ˆð & aÔ(ˆõ  %œz¨,×*;Ò*;Ñ*=Ô*=¸c¸r¸cÔ*BÍ%Ì*Ð]iÔ]pÐqÑqÔqÐð Ô(×/Ò/°Ô0BÀ1Ô0EÀrÈ2ÑNÔNˆÝ$œz¨,×*;Ò*;Ñ*=Ô*=¸c¸r¸cÔ*BÍ%Ì*Ð]iÔ]pÐqÑqÔqÐà!Ð)Ý%*¤_Ð5FÑ%GÔ%GÐ"à-×?Ò?ÀÈAÐ?ÑNÔNÐØ!7×!IÒ!IÈ&ÐVWÐ!IÑ!XÔ!XÐØ!7×!:Ò!:Ð;OÔ;VÑ!WÔ!WÐÝ!&¤Ð,@ÐBXÐ+YÐ_`Ð!aÑ!aÔ!aÐØ&˜$œ,ð 
Ø'Ø1Ø%Ø".Ø#7ð
ð 
ð ð
ð 
ˆð *9ˆÔ&Ø& qÔ)¨!¨!¨!Ð-C¨|×/@Ò/@ÀÑ/CÔ/CÐ-CÀQÀQÀQÐ*FÔGˆð ×1Ò1°,Ñ?Ô?ˆð (×/Ò/°
¸D¼KÔ<XÐ[aÑ<aÐceÑfÔfˆØ'5ˆÔ$àÐr   c                  ó    — t          d¦  «        ‚)Nz=No need to inherit as this architecture only supports videos.)r(   )Úsuper_kwargss    r   Úget_image_featuresz<InstructBlipVideoForConditionalGeneration.get_image_features:  s   € ÝÐ\Ñ]Ô]Ð]r   r=   rA   c                 óN  — |€e| |                       ¦   «         t          j        | j        j        t          j        |j        ¬¦  «        ¦  «        k    }|                     d¦  «        }n|| j        j        k    }|                     d¦  «         	                    |j        ¦  «        }|S )zZ
        Obtains multimodal placeholder mask from `input_ids` or `inputs_embeds`.
        NrI   rH   )
ri   rZ   rj   rf   r#   r]   rK   rk   rl   rb   )ro   r=   rA   r{   s       r   Úget_placeholder_maskz>InstructBlipVideoForConditionalGeneration.get_placeholder_mask=  s£   € ð ÐØ!.Ð2M°$×2KÒ2KÑ2MÔ2MÝ”˜Tœ[Ô7½u¼zÐR_ÔRfÐgÑgÔgñ3ô 3ò "Ðð "4×!7Ò!7¸Ñ!;Ô!;ÐÐà!*¨d¬kÔ.HÒ!HÐà/×9Ò9¸"Ñ=Ô=×@Ò@ÀÔAUÑVÔVÐØ!Ð!r   r>   r?   r@   ÚlabelsrC   c           
      óT  —  | j         |f|||
dœ|¤Ž}|j        }|j        }|j        }|€ |                      ¦   «         |¦  «        }|€t          j        |¦  «        }|                     |j        |j	        ¦  «        }|  
                    ||¬¦  «        }|                     ||¦  «        }| j        j        r> | j        d	|||dœ|¤Ž}|d         }d}|	�  | j        d	||	| j        j        j        dœ|¤Ž}n" | j        d	|||||	|dœ|¤Ž}|j        }|j        }t)          |||||¬¦  «        S )
a˜  
        qformer_input_ids (`torch.LongTensor` of shape (batch_size, sequence_length)):
            The sequence used as a prompt to be fed to the Q-Former module.
        qformer_attention_mask (`torch.LongTensor` of shape (batch_size, sequence_length), *optional*):
            Mask to avoid performing attention on padding token indices.

        Examples:

        ```python
        >>> from transformers import InstructBlipVideoProcessor, InstructBlipVideoForConditionalGeneration
        >>> import torch
        >>> from huggingface_hub import hf_hub_download
        >>> import av
        >>> import numpy as np

        >>> def read_video_pyav(container, indices):
        ...     '''
        ...     Decode the video with PyAV decoder.
        ...     Args:
        ...         container (`av.container.input.InputContainer`): PyAV container.
        ...         indices (`list[int]`): List of frame indices to decode.
        ...     Returns:
        ...         result (np.ndarray): np array of decoded frames of shape (num_frames, height, width, 3).
        ...     '''
        ...     frames = []
        ...     container.seek(0)
        ...     start_index = indices[0]
        ...     end_index = indices[-1]
        ...     for i, frame in enumerate(container.decode(video=0)):
        ...         if i > end_index:
        ...             break
        ...         if i >= start_index and i in indices:
        ...             frames.append(frame)
        ...     return np.stack([x.to_ndarray(format="rgb24") for x in frames])

        >>> model = InstructBlipVideoForConditionalGeneration.from_pretrained("Salesforce/instructblip-vicuna-7b", device_map="auto")
        >>> processor = InstructBlipVideoProcessor.from_pretrained("Salesforce/instructblip-vicuna-7b")

        >>> file_path = hf_hub_download(
        ...       repo_id="nielsr/video-demo", filename="eating_spaghetti.mp4", repo_type="dataset"
        ... )
        >>> container = av.open(file_path)

        >>> # sample uniformly 4 frames from the videWhy is this video funny?o
        >>> total_frames = container.streams.video[0].frames
        >>> indices = np.arange(0, total_frames, total_frames / 4).astype(int)
        >>> clip = read_video_pyav(container, indices)

        >>> prompt = "What is happening in the video?"
        >>> inputs = processor(text=prompt, images=clip, return_tensors="pt").to(model.device)

        >>> outputs = model.generate(
        ...     **inputs,
        ...     do_sample=False,
        ...     num_beams=5,
        ...     max_length=256,
        ...     repetition_penalty=1.5,
        ...     length_penalty=1.0,
        ... )
        >>> generated_text = processor.batch_decode(outputs, skip_special_tokens=True)[0].strip()
        >>> print(generated_text)
        "A person is eating a bowl of pasta, and they are using a fork to eat it. The person is sitting at a table, and the plate of pasta is on the table in front"
        ```©r;   r<   rB   N©rA   rS   r   )Úlogitsr‘   Ú
vocab_size)rA   r>   r?   r@   r‘   rC   )Úlossr•   rT   rU   rV   r   )r‹   r‡   rU   rT   ri   rZ   r`   rb   rK   rJ   r�   rm   rf   rn   rh   Úloss_functionÚtext_configr–   r—   r•   r7   )ro   r:   r;   r<   r=   r>   r?   r@   rA   r‘   rB   rC   rD   rŠ   rz   rU   rT   r{   r|   r•   r—   s                        r   r}   z1InstructBlipVideoForConditionalGeneration.forwardL  sÚ  € ð` CZÀ$ÔBYØðC
à/Ø#9Ø%=ð	C
ð C
ð
 ðC
ð C
ˆð !/Ô <ÐØ(Ô8ˆØ'Ô6ˆàÐ Ø7˜D×5Ò5Ñ7Ô7¸	ÑBÔBˆMàÐ!Ý"œ_¨YÑ7Ô7ˆNà 5× 8Ò 8¸Ô9MÈ}ÔObÑ cÔ cÐØ!×6Ò6°yÐP]Ð6Ñ^Ô^ÐØ%×4Ò4Ð5GÐI^Ñ_Ô_ˆàŒ;Ô6ð 	$Ø)�dÔ)ð Ø+Ø-Ø#ðð ð ð	ð ˆGð ˜Q”ZˆFØˆDØÐ!Ø)�tÔ)ð Ø!¨&¸T¼[Ô=TÔ=_ðð Øciðð �øð
 *�dÔ)ð Ø+Ø-Ø"3Ø'=ØØ#ðð ð ðð ˆGð ”<ˆDØ”^ˆFåCØØØ)Ø+Ø#*ð
ñ 
ô 
ð 	
r   c                 óì  — t          | d¦  «        r|                      ¦   «          |j        d         }	|                      ||||¬¦  «        }
|
j        }|€Ž|€o| j        j        g| j        j        z  dz  }|| j        j        j	        gz   }t          j        |gt          j        |j        ¬¦  «        }|                     |	d¦  «        } |                      ¦   «         |¦  «        }|€t          j        |¦  «        }|                     |j        |j        ¦  «        }|                      ||¬¦  «        }|                     ||¦  «        }||d	œ}| j        j        j        s||d
<    | j        j        di |¤|¤Ž}|S )aÙ  
        Overrides `generate` function to be able to use the model as a conditional generator.

        Args:
            pixel_values (`torch.FloatTensor` of shape (batch_size, num_channels, height, width) or
                (batch_size, num_frames, num_channels, height, width)): Input images or videos to be processed.
            qformer_input_ids (`torch.LongTensor` of shape (batch_size, sequence_length), *optional*):
                The sequence used as a prompt to be fed to the Q-Former module.
            qformer_attention_mask (`torch.LongTensor` of shape (batch_size, sequence_length), *optional*):
                Mask to avoid performing attention on padding token indices.
            input_ids (`torch.LongTensor` of shape (batch_size, sequence_length), *optional*):
                The sequence used as a prompt for the generation.
            attention_mask (`torch.LongTensor` of shape (batch_size, sequence_length), *optional*):
                Mask to avoid performing attention on padding token indices.
            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
                Embedded representation of the inputs. Should be float, not int tokens.
            interpolate_pos_encoding (`bool`, *optional*, defaults to `False`):
                Whether to interpolate the positional encoding of the image embeddings.

        Returns:
            captions (list): A list of strings of length batch_size * num_captions.
        Úhf_device_mapr   r“   Né   rI   rN   r”   )rA   r>   r=   r   )ÚhasattrÚ_preprocess_acceleraterW   r‹   r‡   rf   r$   rg   r™   Úbos_token_idrZ   rj   r]   rK   Úrepeatri   r`   rb   rJ   r�   rm   rh   Úis_encoder_decoderÚgenerate)ro   r:   r;   r<   r=   r>   rA   rB   Úgenerate_kwargsrp   rŠ   rz   Úvideo_tokensÚstart_tokensr{   Úinputsr|   s                    r   r¢   z2InstructBlipVideoForConditionalGeneration.generateÔ  s¦  € õD �4˜Ñ)Ô)ð 	*à×'Ò'Ñ)Ô)Ð)à!Ô'¨Ô*ˆ
ØBF×BYÒBYØØ/Ø#9Ø%=ð	 CZñ C
ô C
ˆð !/Ô <ÐàÐ ØÐ Ø $¤Ô =Ð>ÀÄÔA]Ñ]Ð`aÑa�Ø+¨t¬{Ô/FÔ/SÐ.TÑT�Ý!œL¨,¨½u¼zÐR^ÔReÐfÑfÔf�	Ø%×,Ò,¨Z¸Ñ;Ô;�	Ø7˜D×5Ò5Ñ7Ô7¸	ÑBÔBˆMàÐ!Ý"œ_¨YÑ7Ô7ˆNà 5× 8Ò 8¸Ô9MÈ}ÔObÑ cÔ cÐØ!×6Ò6°yÐP]Ð6Ñ^Ô^ÐØ%×4Ò4Ð5GÐI^Ñ_Ô_ˆà#0ÀNÐSÐSˆØÔ"Ô)Ô<ð 	,Ø"+ˆF�;Ñà.�$Ô%Ô.ÐKÐK°ÐK¸?ÐKÐKˆàˆr   )NF)	NNNNNNNFN)NNNNNF)r   r   r   r   r   rZ   r~   r   r�   r   r   r‚   r   r‹   rŽ   r�   r7   r}   Úno_gradr¢   r   r   r   r„   r„   ñ   sâ  € € € € € ØØð
 ;?Ø05ðDð DàÔ'ðDð !Ô+ðDð !&Ô 0°4Ñ 7ð	Dð
 #'¨¡+ðDð Ð+Ô,ðDð 
Ð8Ñ	8ðDð Dð Dñ „^ñ ÔðDðL^ð ^ð ^ð"¨eÔ.>ð "ÈuÔO`ð "ð "ð "ð "ð Øð
 ;?Ø.2Ø26Ø59Ø:>Ø26Ø*.Ø).Ø!%ðD
ð D
àÔ'ðD
ð !Ô,ðD
ð !&Ô 0°4Ñ 7ð	D
ð
 Ô$ tÑ+ðD
ð Ô(¨4Ñ/ðD
ð !Ô+¨dÑ2ðD
ð !&Ô 0°4Ñ 7ðD
ð Ô(¨4Ñ/ðD
ð Ô  4Ñ'ðD
ð #'ðD
ð ˜$‘;ðD
ð Ð+Ô,ðD
ð 
ÐEÑ	EðD
ð D
ð D
ñ „^ñ ÔðD
ðL €U„]�_„_ð 6:Ø:>Ø-1Ø26Ø26Ø).ðCð CàÔ'ðCð !Ô+¨dÑ2ðCð !&Ô 0°4Ñ 7ð	Cð
 Ô# dÑ*ðCð Ô(¨4Ñ/ðCð Ô(¨4Ñ/ðCð #'ðCð 
Ô	ðCð Cð Cñ „_ðCð Cð Cr   r„   )r"   r    r   r1   r+   r3   r9   r„   )!rZ   Úhuggingface_hub.dataclassesr   Ú;transformers.models.instructblip.configuration_instructblipr   r   r   Ú6transformers.models.instructblip.modeling_instructblipr   r   r	   r
   r   r   r   r   Úmodeling_outputsr   Úprocessing_utilsr   Úutilsr   r   r   r    r"   r+   r1   r3   r7   r9   r„   Ú__all__r   r   r   ú<module>r¯      s  ðð  €€€Ø .Ð .Ð .Ð .Ð .Ð .ðð ð ð ð ð ð ð ð ð ð
	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	ð ;Ð :Ð :Ð :Ð :Ð :Ø &Ð &Ð &Ð &Ð &Ð &Ø 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5ð €Ð?Ð@Ñ@Ô@Øðð ð ð ð Ð$<ñ ô ñ „ñ AÔ@ðð$ €Ð?Ð@Ñ@Ô@Øðð ð ð ð Ð%>ñ ô ñ „ñ AÔ@ðð, €Ð?Ð@Ñ@Ô@Øð')ð ')ð ')ð ')ð ')Ð0ñ ')ô ')ñ „ñ AÔ@ð')ðT)ð )ð )ð )ð )Ð'Bñ )ô )ð )ðð ð ð ð Ð#:ñ ô ð ð	ð 	ð 	ð 	ð 	Ð$<ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð;jñ 	ô 	ð 	ð^
ð ^
ð ^
ð ^
ð ^
Ð.ñ ^
ô ^
ð ^
ðBgð gð gð gð gÐ0Tñ gô gð gðT		ð 	ð 	€€€r   