§
    ‚ŠtjA  ã                   óØ   — d dl mZmZ ddlmZ ddlmZmZ ddlm	Z	m
Z
mZmZ ddlmZmZ ddlmZmZ  e¦   «         rd dlZ G d	„ d
e
d¬¦  «        Ze G d„ de¦  «        ¦   «         ZdgZdS )é    )ÚOptionalÚUnioné   )ÚBatchFeature)Ú
ImageInputÚis_valid_image)ÚMultiModalDataÚProcessingKwargsÚProcessorMixinÚUnpack)ÚPreTokenizedInputÚ	TextInput)Úauto_docstringÚis_torch_availableNc                   ó(   — e Zd ZddidddœddidœZd	S )
ÚColQwen2ProcessorKwargsÚpaddingÚlongestÚchannels_firstT)Údata_formatÚdo_convert_rgbÚreturn_tensorsÚpt)Útext_kwargsÚimages_kwargsÚcommon_kwargsN)Ú__name__Ú
__module__Ú__qualname__Ú	_defaults© ó    ún/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/colqwen2/processing_colqwen2.pyr   r   "   sB   € € € € € ð �yð
ð ,Ø"ð
ð 
ð +¨DÐ1ð	ð 	€I€I€Ir"   r   F)Útotalc                   ó¼  ‡ — e Zd Z	 	 	 	 	 ddedz  dedz  fˆ fd„Ze	 	 ddedz  deez  e	e         z  e	e         z  de
e         defd	„¦   «         Zdd
„Zed„ ¦   «         Zedefd„¦   «         Z	 ddedz  de
e         defd„Zdee	e         z  de
e         defd„Z	 	 	 ddede	d         f         dede	d         f         deded         dedef         ddfd„Zˆ xZS )ÚColQwen2ProcessorNÚvisual_prompt_prefixÚquery_prefixc                 óì   •— t          ¦   «                              |||¬¦  «         t          |d¦  «        sdn|j        | _        t          |d¦  «        sdn|j        | _        |pd| _        |pd| _        dS )	ar  
        visual_prompt_prefix (`str`, *optional*, defaults to `"<|im_start|>user\n<|vision_start|><|image_pad|><|vision_end|>Describe the image.<|im_end|><|endoftext|>"`):
            A string that gets tokenized and prepended to the image tokens.
        query_prefix (`str`, *optional*, defaults to `"Query: "`):
            A prefix to be used for the query.
        )Úchat_templateÚimage_tokenz<|image_pad|>Úvideo_tokenz<|video_pad|>zf<|im_start|>user
<|vision_start|><|image_pad|><|vision_end|>Describe the image.<|im_end|><|endoftext|>zQuery: N)ÚsuperÚ__init__Úhasattrr+   r,   r'   r(   )ÚselfÚimage_processorÚ	tokenizerr*   r'   r(   ÚkwargsÚ	__class__s          €r#   r.   zColQwen2Processor.__init__1   sŠ   ø€ õ 	‰Œ×Ò˜¨)À=ÐÑQÔQÐQÝ29¸)À]Ñ2SÔ2SÐn˜?˜?ÐYbÔYnˆÔÝ29¸)À]Ñ2SÔ2SÐn˜?˜?ÐYbÔYnˆÔà$8ð %
Øuð 	Ô!ð )Ð5¨IˆÔÐÐr"   ÚimagesÚtextr3   Úreturnc                 ó&  —  | j         t          fd| j        j        i|¤Ž}|d                              dd¦  «        }|du}|€|€t          d¦  «        ‚|�|�t          d¦  «        ‚|���t          |¦  «        r|g}n…t          |t          ¦  «        rt          |d         ¦  «        rnZt          |t          ¦  «        r6t          |d         t          ¦  «        rt          |d         d         ¦  «        st          d¦  «        ‚| j	        gt          |¦  «        z  } | j        dd	|i|d
         ¤Ž}|d         }	|	�º| j        j        dz  }
d}t          t          |¦  «        ¦  «        D ]Œ}| j        ||         v rW||                              | j        d|	|                              ¦   «         |
z  z  d¦  «        ||<   |dz  }| j        ||         v °W||                              d| j        ¦  «        ||<   Œ� | j        |fddi|d         ¤Ž}t#          i |¥|¥¬¦  «        }|d         dd…df         |d         dd…df         z  }t          t%          j        |d         |                     ¦   «         ¦  «        ¦  «        }t$          j        j        j                             |d¬¦  «        |d<   |r=|d                              |d         dk    d¦  «        }|                     d|i¦  «         |S |�¥t          |t6          ¦  «        r|g}n?t          |t          ¦  «        rt          |d         t6          ¦  «        st          d¦  «        ‚|€
| j        dz  }g }|D ]$}| j        |z   |z   }|                     |¦  «         Œ% | j        |fddi|d         ¤Ž}|S dS )a  
        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
        Útokenizer_init_kwargsr   ÚsuffixNz&Either text or images must be providedz5Only one of text or images can be processed at a timer   zAimages must be an image, list of images or list of list of imagesr5   r   Úimage_grid_thwé   z<|placeholder|>é   Úreturn_token_type_idsF)ÚdataÚpixel_valuesT)Úbatch_firstÚ	input_idsÚtoken_type_idsiœÿÿÿÚlabelsz*Text must be a string or a list of stringsé
   r!   )Ú_merge_kwargsr   r2   Úinit_kwargsÚpopÚ
ValueErrorr   Ú
isinstanceÚlistr'   Úlenr1   Ú
merge_sizeÚranger+   ÚreplaceÚprodr   ÚtorchÚsplitÚtolistÚnnÚutilsÚrnnÚpad_sequenceÚmasked_fillÚupdateÚstrÚquery_augmentation_tokenr(   Úappend)r0   r5   r6   r3   Úoutput_kwargsr:   r>   Ú	texts_docÚimage_inputsr;   Úmerge_lengthÚindexÚiÚtext_inputsÚreturn_dataÚoffsetsr@   rD   Útexts_queryÚqueryÚaugmented_queryÚbatch_querys                         r#   Ú__call__zColQwen2Processor.__call__I   sg  € ð" +˜Ô*Ý#ð
ð 
à"&¤.Ô"<ð
ð ð
ð 
ˆð
 ˜}Ô-×1Ò1°(¸DÑAÔAˆà &¨dÐ 2Ðàˆ<˜F˜NÝÐEÑFÔFÐFØÐ Ð 2ÝÐTÑUÔUÐUàÑÝ˜fÑ%Ô%ð fØ ˜��Ý˜F¥DÑ)Ô)ð f­n¸VÀA¼YÑ.GÔ.Gð fØÝ  ­Ñ.Ô.ð fµ:¸fÀQ¼iÍÑ3NÔ3Nð fÕSaÐbhÐijÔbkÐlmÔbnÑSoÔSoð fÝ Ð!dÑeÔeÐeàÔ2Ð3µc¸&±k´kÑAˆIà/˜4Ô/Ð`Ð`°vÐ`ÀÈÔA_Ð`Ð`ˆLØ)Ð*:Ô;ˆNàÐ)Ø#Ô3Ô>ÀÑA�Ø�Ý�s 9™~œ~Ñ.Ô.ð ]ð ]�AØÔ*¨i¸¬lÐ:Ð:Ø'0°¤|×';Ò';Ø Ô,Ð.?À>ÐRWÔCX×C]ÒC]ÑC_ÔC_ÐcoÑCoÑ.pÐrsñ(ô (˜	 !™ð  ™
˜ð	 Ô*¨i¸¬lÐ:Ð:ð
 $-¨Q¤<×#7Ò#7Ð8IÈ4ÔK[Ñ#\Ô#\�I˜a‘L�Là(˜$œ.Øðð à&+ðð   Ô.ðð ˆKõ 'Ð,K¨{Ð,K¸lÐ,KÐLÑLÔLˆKð "Ð"2Ô3°A°A°A°q°DÔ9¸KÐHXÔ<YÐZ[ÐZ[ÐZ[Ð]^ÐZ^Ô<_Ñ_ˆGõ  Ý”˜K¨Ô7¸¿ºÑ9IÔ9IÑJÔJñô ˆLõ
 +0¬(¬.Ô*<×*IÒ*IØ¨$ð +Jñ +ô +ˆK˜Ñ'ð %ð 7Ø$ [Ô1×=Ò=¸kÐJZÔ>[Ð_`Ò>`ÐbfÑgÔg�Ø×"Ò" H¨fÐ#5Ñ6Ô6Ð6àÐàÐÝ˜$¥Ñ$Ô$ð OØ�v��Ý  ¥tÑ,Ô,ð Oµ¸DÀ¼GÅSÑ1IÔ1Ið OÝ Ð!MÑNÔNÐNàˆ~ØÔ6¸Ñ;�à%'ˆKàð 4ð 4�Ø"&Ô"3°eÑ";¸fÑ"D�Ø×"Ò" ?Ñ3Ô3Ð3Ð3à(˜$œ.Øðð à&+ðð   Ô.ðð ˆKð Ðð+ Ðr"   c                 ó@  ‡ ‡‡— i }|�Œt           j                             di ¦  «        Š‰                     |¦  «         ‰                     dd¦  «        p‰ j        j        Šˆˆ fd„|D ¦   «         }ˆfd„|D ¦   «         }|                     ||dœ¦  «         t          di |¤ŽS )a¹  
        Computes the number of placeholder tokens needed for multimodal inputs with the given sizes.
        Args:
            image_sizes (`list[list[int]]`, *optional*):
                The input sizes formatted as (height, width) per each image.
        Returns:
            `MultiModalData`: A `MultiModalData` object holding number of tokens per each of the provided
            input modalities, along with other useful data.
        Nr   rM   c                 ó8   •— g | ]} ‰j         j        g |¢‰‘R Ž ‘ŒS r!   )r1   Úget_number_of_image_patches)Ú.0Ú
image_sizer   r0   s     €€r#   ú
<listcomp>z@ColQwen2Processor._get_num_multimodal_tokens.<locals>.<listcomp>Ã   sE   ø€ ð !ð !ð !àð A�Ô$Ô@Ð\À*Ð\ÈmÐ\Ð\Ð\ð!ð !ð !r"   c                 ó    •— g | ]
}|‰d z  z  ‘ŒS )r<   r!   )rn   Únum_patchesrM   s     €r#   rp   z@ColQwen2Processor._get_num_multimodal_tokens.<locals>.<listcomp>Ç   s"   ø€ ÐdÐdÐdÀ; °
¸A±Ñ!=ÐdÐdÐdr"   )Únum_image_tokensÚnum_image_patchesr!   )r   r    ÚgetrY   r1   rM   r	   )r0   Úimage_sizesr3   Úvision_datart   rs   r   rM   s   `     @@r#   Ú_get_num_multimodal_tokensz,ColQwen2Processor._get_num_multimodal_tokens²   sÝ   øøø€ ð ˆØÐ"Ý3Ô=×AÒAÀ/ÐSUÑVÔVˆMØ× Ò  Ñ(Ô(Ð(Ø&×*Ò*¨<¸Ñ>Ô>ÐaÀ$ÔBVÔBaˆJð!ð !ð !ð !ð !à"-ð!ñ !ô !Ðð  eÐdÐdÐdÐRcÐdÑdÔdÐØ×ÒÐ4DÐ[lÐmÐmÑnÔnÐnåÐ,Ð, Ð,Ð,Ð,r"   c                 óT   — | j         j        }| j        j        }d„ |D ¦   «         }||z   S )Nc                 ó   — g | ]}|d v¯|‘Œ	S ))Úpixel_values_videosÚvideo_grid_thwr!   )rn   Únames     r#   rp   z7ColQwen2Processor.model_input_names.<locals>.<listcomp>Ó   s*   € ð '
ð '
ð '
Ø¸DÐHqÐ<qÐ<qˆDÐ<qÐ<qÐ<qr"   )r2   Úmodel_input_namesr1   )r0   Útokenizer_input_namesÚimage_processor_input_namess      r#   r~   z#ColQwen2Processor.model_input_namesÌ   sF   € à $¤Ô @ÐØ&*Ô&:Ô&LÐ#ð'
ð '
Ø8ð'
ñ '
ô '
Ð#ð %Ð'BÑBÐBr"   c                 ó   — | j         j        S )zŠ
        Return the query augmentation token.

        Query augmentation buffers are used as reasoning buffers during inference.
        )r2   Ú	pad_token)r0   s    r#   r[   z*ColQwen2Processor.query_augmentation_tokenØ   s   € ð Œ~Ô'Ð'r"   c                 ó    —  | j         dd|i|¤ŽS )a  
        Prepare for the model one or several image(s). This method is a wrapper around the `__call__` method of the ColQwen2Processor's
        [`ColQwen2Processor.__call__`].

        This method forwards the `images` and `kwargs` arguments to the image processor.

        Args:
            images (`PIL.Image.Image`, `np.ndarray`, `torch.Tensor`, `list[PIL.Image.Image]`, `list[np.ndarray]`, `list[torch.Tensor]`):
                The image or batch of images to be prepared. Each image can be a PIL image, NumPy array or PyTorch
                tensor. In case of a NumPy array/PyTorch tensor, each image should be of shape (C, H, W), where C is a
                number of channels, H and W are image height and width.
            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors of a particular framework. Acceptable values are:

                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return NumPy `np.ndarray` objects.

        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
        r5   r!   ©rj   )r0   r5   r3   s      r#   Úprocess_imagesz ColQwen2Processor.process_imagesá   s!   € ð> ˆtŒ}Ð5Ð5 FÐ5¨fÐ5Ð5Ð5r"   c                 ó    —  | j         dd|i|¤ŽS )ai  
        Prepare for the model one or several texts. This method is a wrapper around the `__call__` method of the ColQwen2Processor's
        [`ColQwen2Processor.__call__`].

        This method forwards the `text` and `kwargs` arguments to the tokenizer.

        Args:
            text (`str`, `list[str]`, `list[list[str]]`):
                The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings
                (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set
                `is_split_into_words=True` (to lift the ambiguity with a batch of sequences).
            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors of a particular framework. Acceptable values are:

                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return NumPy `np.ndarray` objects.

        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
        r6   r!   r„   )r0   r6   r3   s      r#   Úprocess_queriesz!ColQwen2Processor.process_queries  s!   € ð< ˆtŒ}Ð1Ð1 $Ð1¨&Ð1Ð1Ð1r"   é€   ÚcpuÚquery_embeddingsztorch.TensorÚpassage_embeddingsÚ
batch_sizeÚoutput_dtypeztorch.dtypeÚoutput_deviceztorch.devicec           	      ó8  — t          |¦  «        dk    rt          d¦  «        ‚t          |¦  «        dk    rt          d¦  «        ‚|d         j        |d         j        k    rt          d¦  «        ‚|d         j        |d         j        k    rt          d¦  «        ‚|€|d         j        }g }t	          dt          |¦  «        |¦  «        D �]:}g }t
          j        j        j         	                    ||||z   …         dd¬¦  «        }	t	          dt          |¦  «        |¦  «        D ]�}
t
          j        j        j         	                    ||
|
|z   …         dd¬¦  «        }| 
                    t          j        d	|	|¦  «                             d
¬¦  «        d                              d¬¦  «        ¦  «         Œ‘| 
                    t          j        |d¬¦  «                             |¦  «                             |¦  «        ¦  «         �Œ<t          j        |d¬¦  «        S )a[  
        Compute the late-interaction/MaxSim score (ColBERT-like) for the given multi-vector
        query embeddings (`qs`) and passage embeddings (`ps`). For ColQwen2, a passage is the
        image of a document page.

        Because the embedding tensors are multi-vector and can thus have different shapes, they
        should be fed as:
        (1) a list of tensors, where the i-th tensor is of shape (sequence_length_i, embedding_dim)
        (2) a single tensor of shape (n_passages, max_sequence_length, embedding_dim) -> usually
            obtained by padding the list of tensors.

        Args:
            query_embeddings (`Union[torch.Tensor, list[torch.Tensor]`): Query embeddings.
            passage_embeddings (`Union[torch.Tensor, list[torch.Tensor]`): Passage embeddings.
            batch_size (`int`, *optional*, defaults to 128): Batch size for computing scores.
            output_dtype (`torch.dtype`, *optional*, defaults to `torch.float32`): The dtype of the output tensor.
                If `None`, the dtype of the input embeddings is used.
            output_device (`torch.device` or `str`, *optional*, defaults to "cpu"): The device of the output tensor.

        Returns:
            `torch.Tensor`: A tensor of shape `(n_queries, n_passages)` containing the scores. The score
            tensor is saved on the "cpu" device.
        r   zNo queries providedzNo passages providedz/Queries and passages must be on the same devicez-Queries and passages must have the same dtypeNT)rA   Úpadding_valuezbnd,csd->bcnsr   )Údimr<   r=   )rL   rI   ÚdeviceÚdtyperN   rQ   rT   rU   rV   rW   r\   ÚeinsumÚmaxÚsumÚcatÚto)r0   rŠ   r‹   rŒ   r�   rŽ   Úscoresrb   Úbatch_scoresÚbatch_queriesÚjÚbatch_passagess               r#   Úscore_retrievalz!ColQwen2Processor.score_retrieval"  s*  € õ@ ÐÑ Ô  AÒ%Ð%ÝÐ2Ñ3Ô3Ð3ÝÐ!Ñ"Ô" aÒ'Ð'ÝÐ3Ñ4Ô4Ð4à˜AÔÔ%Ð);¸AÔ)>Ô)EÒEÐEÝÐNÑOÔOÐOà˜AÔÔ$Ð(:¸1Ô(=Ô(CÒCÐCÝÐLÑMÔMÐMàÐØ+¨AÔ.Ô4ˆLà%'ˆå�q�#Ð.Ñ/Ô/°Ñ<Ô<ð 	]ñ 	]ˆAØ/1ˆLÝ!œHœNÔ.×;Ò;Ø   Q¨¡^Ð!3Ô4À$ÐVWð <ñ ô ˆMõ ˜1�cÐ"4Ñ5Ô5°zÑBÔBð ð �Ý!&¤¤Ô!3×!@Ò!@Ø& q¨1¨z©>Ð'9Ô:ÈÐ\]ð "Añ "ô "�ð ×#Ò#Ý”L °-ÀÑPÔP×TÒTÐYZÐTÑ[Ô[Ð\]Ô^×bÒbÐghÐbÑiÔiñô ð ð ð �MŠM�%œ) L°aÐ8Ñ8Ô8×;Ò;¸LÑIÔI×LÒLÈ]Ñ[Ô[Ñ\Ô\Ð\Ñ\åŒy˜ QÐ'Ñ'Ô'Ð'r"   )NNNNN)NN)N)rˆ   Nr‰   )r   r   r   rZ   r.   r   r   r   r   rK   r   r   r   rj   rx   Úpropertyr~   r[   r…   r‡   r   Úintr   rž   Ú__classcell__)r4   s   @r#   r&   r&   /   s[  ø€ € € € € ð ØØØ+/Ø#'ð6ð 6ð
 " D™jð6ð ˜D‘jð6ð 6ð 6ð 6ð 6ð 6ð0 ð %)ØZ^ðfð fà˜TÑ!ðfð Ð+Ñ+¨d°9¬oÑ=ÀÐEVÔ@WÑWðfð Ð0Ô1ð	fð
 
ðfð fð fñ „^ðfðP-ð -ð -ð -ð4 ð	Cð 	Cñ „Xð	Cð ð(¨#ð (ð (ð (ñ „Xð(ð %)ð6ð 6à˜TÑ!ð6ð Ð0Ô1ð6ð 
ð	6ð 6ð 6ð 6ðB2à˜$˜yœ/Ñ)ð2ð Ð0Ô1ð2ð 
ð	2ð 2ð 2ð 2ðH Ø04Ø49ð>(ð >(à °°^Ô0DÐ DÔEð>(ð " .°$°~Ô2FÐ"FÔGð>(ð ð	>(ð
 ˜}Ô-ð>(ð ˜^¨SÐ0Ô1ð>(ð 
ð>(ð >(ð >(ð >(ð >(ð >(ð >(ð >(r"   r&   )Útypingr   r   Úfeature_extraction_utilsr   Úimage_utilsr   r   Úprocessing_utilsr	   r
   r   r   Útokenization_utils_baser   r   rU   r   r   rQ   r   r&   Ú__all__r!   r"   r#   ú<module>r¨      s8  ðð* #Ð "Ð "Ð "Ð "Ð "Ð "Ð "à 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5Ø XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XØ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CØ 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7ð ÐÑÔð Ø€L€L€Lð
ð 
ð 
ð 
ð 
Ð.°eð 
ñ 
ô 
ð 
ð ðp(ð p(ð p(ð p(ð p(˜ñ p(ô p(ñ „ðp(ðf	 Ð
€€€r"   