§
    ‚Štjx<  ã                   óÒ  — d dl mZ ddlmZ ddlmZ ddlmZmZ ddl	m
Z
mZmZmZ ddlmZmZ ddlmZmZmZmZmZ d	d
lmZmZ d	dlmZ ddlmZ  e¦   «         rd dlZ ej        e ¦  «        Z! G d„ ded¬¦  «        Z" G d„ de¦  «        Z# G d„ de¦  «        Z$ ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z% ed¬¦  «         G d„ de¦  «        ¦   «         Z&g d¢Z'dS )é    )Ú	dataclassé   )ÚCache)ÚBatchFeature)Ú
ImageInputÚis_valid_image)ÚMultiModalDataÚProcessingKwargsÚProcessorMixinÚUnpack)ÚPreTokenizedInputÚ	TextInput)ÚModelOutputÚauto_docstringÚcan_return_tupleÚis_torch_availableÚloggingé   )ÚColPaliForRetrievalÚColPaliPreTrainedModel)ÚColPaliProcessoré   )ÚColQwen2ConfigNc                   ó(   — e Zd ZddidddœddidœZd	S )
ÚColQwen2ProcessorKwargsÚpaddingÚlongestÚchannels_firstT)Údata_formatÚdo_convert_rgbÚreturn_tensorsÚpt)Útext_kwargsÚimages_kwargsÚcommon_kwargsN)Ú__name__Ú
__module__Ú__qualname__Ú	_defaults© ó    úk/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/colqwen2/modular_colqwen2.pyr   r   "   sB   € € € € € ð �yð
ð ,Ø"ð
ð 
ð +¨DÐ1ð	ð 	€I€I€Ir+   r   F)Útotalc            	       ó®   — e Zd Z	 	 	 	 	 ddedz  dedz  fd„Z	 	 ddedz  deez  ee         z  ee         z  de	e
         defd	„Zdd
„Zed„ ¦   «         ZdS )ÚColQwen2ProcessorNÚvisual_prompt_prefixÚquery_prefixc                 óÒ   — t          j        | |||¬¦  «         t          |d¦  «        sdn|j        | _        t          |d¦  «        sdn|j        | _        |pd| _        |pd| _        dS )	ar  
        visual_prompt_prefix (`str`, *optional*, defaults to `"<|im_start|>user\n<|vision_start|><|image_pad|><|vision_end|>Describe the image.<|im_end|><|endoftext|>"`):
            A string that gets tokenized and prepended to the image tokens.
        query_prefix (`str`, *optional*, defaults to `"Query: "`):
            A prefix to be used for the query.
        )Úchat_templateÚimage_tokenz<|image_pad|>Úvideo_tokenz<|video_pad|>zf<|im_start|>user
<|vision_start|><|image_pad|><|vision_end|>Describe the image.<|im_end|><|endoftext|>zQuery: N)r   Ú__init__Úhasattrr4   r5   r0   r1   )ÚselfÚimage_processorÚ	tokenizerr3   r0   r1   Úkwargss          r,   r6   zColQwen2Processor.__init__0   s…   € õ 	Ô  o°yÐP]Ð^Ñ^Ô^Ð^Ý29¸)À]Ñ2SÔ2SÐn˜?˜?ÐYbÔYnˆÔÝ29¸)À]Ñ2SÔ2SÐn˜?˜?ÐYbÔYnˆÔà$8ð %
Øuð 	Ô!ð )Ð5¨IˆÔÐÐr+   ÚimagesÚtextr;   Úreturnc                 ó&  —  | j         t          fd| j        j        i|¤Ž}|d                              dd¦  «        }|du}|€|€t          d¦  «        ‚|�|�t          d¦  «        ‚|���t          |¦  «        r|g}n…t          |t          ¦  «        rt          |d         ¦  «        rnZt          |t          ¦  «        r6t          |d         t          ¦  «        rt          |d         d         ¦  «        st          d¦  «        ‚| j	        gt          |¦  «        z  } | j        dd	|i|d
         ¤Ž}|d         }	|	�º| j        j        dz  }
d}t          t          |¦  «        ¦  «        D ]Œ}| j        ||         v rW||                              | j        d|	|                              ¦   «         |
z  z  d¦  «        ||<   |dz  }| j        ||         v °W||                              d| j        ¦  «        ||<   Œ� | j        |fddi|d         ¤Ž}t#          i |¥|¥¬¦  «        }|d         dd…df         |d         dd…df         z  }t          t%          j        |d         |                     ¦   «         ¦  «        ¦  «        }t$          j        j        j                             |d¬¦  «        |d<   |r=|d                              |d         dk    d¦  «        }|                     d|i¦  «         |S |�¥t          |t6          ¦  «        r|g}n?t          |t          ¦  «        rt          |d         t6          ¦  «        st          d¦  «        ‚|€
| j        dz  }g }|D ]$}| j        |z   |z   }|                     |¦  «         Œ% | j        |fddi|d         ¤Ž}|S dS )a  
        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
        Útokenizer_init_kwargsr#   ÚsuffixNz&Either text or images must be providedz5Only one of text or images can be processed at a timer   zAimages must be an image, list of images or list of list of imagesr<   r$   Úimage_grid_thwr   z<|placeholder|>r   Úreturn_token_type_idsF)ÚdataÚpixel_valuesT)Úbatch_firstÚ	input_idsÚtoken_type_idsiœÿÿÿÚlabelsz*Text must be a string or a list of stringsé
   r*   )Ú_merge_kwargsr   r:   Úinit_kwargsÚpopÚ
ValueErrorr   Ú
isinstanceÚlistr0   Úlenr9   Ú
merge_sizeÚranger4   ÚreplaceÚprodr   ÚtorchÚsplitÚtolistÚnnÚutilsÚrnnÚpad_sequenceÚmasked_fillÚupdateÚstrÚquery_augmentation_tokenr1   Úappend)r8   r<   r=   r;   Úoutput_kwargsrA   rC   Ú	texts_docÚimage_inputsrB   Úmerge_lengthÚindexÚiÚtext_inputsÚreturn_dataÚoffsetsrE   rI   Útexts_queryÚqueryÚaugmented_queryÚbatch_querys                         r,   Ú__call__zColQwen2Processor.__call__H   sg  € ð  +˜Ô*Ý#ð
ð 
à"&¤.Ô"<ð
ð ð
ð 
ˆð
 ˜}Ô-×1Ò1°(¸DÑAÔAˆà &¨dÐ 2Ðàˆ<˜F˜NÝÐEÑFÔFÐFØÐ Ð 2ÝÐTÑUÔUÐUàÑÝ˜fÑ%Ô%ð fØ ˜��Ý˜F¥DÑ)Ô)ð f­n¸VÀA¼YÑ.GÔ.Gð fØÝ  ­Ñ.Ô.ð fµ:¸fÀQ¼iÍÑ3NÔ3Nð fÕSaÐbhÐijÔbkÐlmÔbnÑSoÔSoð fÝ Ð!dÑeÔeÐeàÔ2Ð3µc¸&±k´kÑAˆIà/˜4Ô/Ð`Ð`°vÐ`ÀÈÔA_Ð`Ð`ˆLØ)Ð*:Ô;ˆNàÐ)Ø#Ô3Ô>ÀÑA�Ø�Ý�s 9™~œ~Ñ.Ô.ð ]ð ]�AØÔ*¨i¸¬lÐ:Ð:Ø'0°¤|×';Ò';Ø Ô,Ð.?À>ÐRWÔCX×C]ÒC]ÑC_ÔC_ÐcoÑCoÑ.pÐrsñ(ô (˜	 !™ð  ™
˜ð	 Ô*¨i¸¬lÐ:Ð:ð
 $-¨Q¤<×#7Ò#7Ð8IÈ4ÔK[Ñ#\Ô#\�I˜a‘L�Là(˜$œ.Øðð à&+ðð   Ô.ðð ˆKõ 'Ð,K¨{Ð,K¸lÐ,KÐLÑLÔLˆKð "Ð"2Ô3°A°A°A°q°DÔ9¸KÐHXÔ<YÐZ[ÐZ[ÐZ[Ð]^ÐZ^Ô<_Ñ_ˆGõ  Ý”˜K¨Ô7¸¿ºÑ9IÔ9IÑJÔJñô ˆLõ
 +0¬(¬.Ô*<×*IÒ*IØ¨$ð +Jñ +ô +ˆK˜Ñ'ð %ð 7Ø$ [Ô1×=Ò=¸kÐJZÔ>[Ð_`Ò>`ÐbfÑgÔg�Ø×"Ò" H¨fÐ#5Ñ6Ô6Ð6àÐàÐÝ˜$¥Ñ$Ô$ð OØ�v��Ý  ¥tÑ,Ô,ð Oµ¸DÀ¼GÅSÑ1IÔ1Ið OÝ Ð!MÑNÔNÐNàˆ~ØÔ6¸Ñ;�à%'ˆKàð 4ð 4�Ø"&Ô"3°eÑ";¸fÑ"D�Ø×"Ò" ?Ñ3Ô3Ð3Ð3à(˜$œ.Øðð à&+ðð   Ô.ðð ˆKð Ðð+ Ðr+   c                 ó@  ‡ ‡‡— i }|�Œt           j                             di ¦  «        Š‰                     |¦  «         ‰                     dd¦  «        p‰ j        j        Šˆˆ fd„|D ¦   «         }ˆfd„|D ¦   «         }|                     ||dœ¦  «         t          di |¤ŽS )a¹  
        Computes the number of placeholder tokens needed for multimodal inputs with the given sizes.
        Args:
            image_sizes (`list[list[int]]`, *optional*):
                The input sizes formatted as (height, width) per each image.
        Returns:
            `MultiModalData`: A `MultiModalData` object holding number of tokens per each of the provided
            input modalities, along with other useful data.
        Nr$   rR   c                 ó8   •— g | ]} ‰j         j        g |¢‰‘R Ž ‘ŒS r*   )r9   Úget_number_of_image_patches)Ú.0Ú
image_sizer$   r8   s     €€r,   ú
<listcomp>z@ColQwen2Processor._get_num_multimodal_tokens.<locals>.<listcomp>Á   sE   ø€ ð !ð !ð !àð A�Ô$Ô@Ð\À*Ð\ÈmÐ\Ð\Ð\ð!ð !ð !r+   c                 ó    •— g | ]
}|‰d z  z  ‘ŒS )r   r*   )rs   Únum_patchesrR   s     €r,   ru   z@ColQwen2Processor._get_num_multimodal_tokens.<locals>.<listcomp>Å   s"   ø€ ÐdÐdÐdÀ; °
¸A±Ñ!=ÐdÐdÐdr+   )Únum_image_tokensÚnum_image_patchesr*   )r   r)   Úgetr^   r9   rR   r	   )r8   Úimage_sizesr;   Úvision_datary   rx   r$   rR   s   `     @@r,   Ú_get_num_multimodal_tokensz,ColQwen2Processor._get_num_multimodal_tokens°   sÝ   øøø€ ð ˆØÐ"Ý3Ô=×AÒAÀ/ÐSUÑVÔVˆMØ× Ò  Ñ(Ô(Ð(Ø&×*Ò*¨<¸Ñ>Ô>ÐaÀ$ÔBVÔBaˆJð!ð !ð !ð !ð !à"-ð!ñ !ô !Ðð  eÐdÐdÐdÐRcÐdÑdÔdÐØ×ÒÐ4DÐ[lÐmÐmÑnÔnÐnåÐ,Ð, Ð,Ð,Ð,r+   c                 óT   — | j         j        }| j        j        }d„ |D ¦   «         }||z   S )Nc                 ó   — g | ]}|d v¯|‘Œ	S ))Úpixel_values_videosÚvideo_grid_thwr*   )rs   Únames     r,   ru   z7ColQwen2Processor.model_input_names.<locals>.<listcomp>Ñ   s*   € ð '
ð '
ð '
Ø¸DÐHqÐ<qÐ<qˆDÐ<qÐ<qÐ<qr+   )r:   Úmodel_input_namesr9   )r8   Útokenizer_input_namesÚimage_processor_input_namess      r,   rƒ   z#ColQwen2Processor.model_input_namesÊ   sF   € à $¤Ô @ÐØ&*Ô&:Ô&LÐ#ð'
ð '
Ø8ð'
ñ '
ô '
Ð#ð %Ð'BÑBÐBr+   )NNNNN)NN©N)r&   r'   r(   r_   r6   r   r   r   rP   r   r   r   ro   r}   Úpropertyrƒ   r*   r+   r,   r/   r/   /   s  € € € € € ð ØØØ+/Ø#'ð6ð 6ð
 " D™jð6ð ˜D‘jð6ð 6ð 6ð 6ð4 %)ØZ^ðfð fà˜TÑ!ðfð Ð+Ñ+¨d°9¬oÑ=ÀÐEVÔ@WÑWðfð Ð0Ô1ð	fð
 
ðfð fð fð fðP-ð -ð -ð -ð4 ð	Cð 	Cñ „Xð	Cð 	Cð 	Cr+   r/   c                   ó   — e Zd ZdS )ÚColQwen2PreTrainedModelN)r&   r'   r(   r*   r+   r,   r‰   r‰   ×   s   € € € € € Ø€Dr+   r‰   z4
    Base class for ColQwen2 embeddings output.
    )Úcustom_introc                   ó¸   — e Zd ZU dZdZej        dz  ed<   dZej	        dz  ed<   dZ
edz  ed<   dZeej                 dz  ed<   dZeej                 dz  ed<   dS )ÚColQwen2ForRetrievalOutputaÝ  
    loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
        Language modeling loss (for next-token prediction).
    embeddings (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
        The embeddings of the model.
    past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
        It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

        Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see
        `past_key_values` input) to speed up sequential decoding.
    NÚlossÚ
embeddingsÚpast_key_valuesÚhidden_statesÚ
attentions)r&   r'   r(   Ú__doc__r�   rV   ÚFloatTensorÚ__annotations__rŽ   ÚTensorr�   r   r�   Útupler‘   r*   r+   r,   rŒ   rŒ   Û   s›   € € € € € € ð
ð 
ð &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø&*€J�”˜tÑ#Ð*Ð*Ñ*Ø$(€O�U˜T‘\Ð(Ð(Ñ(Ø59€M�5˜Ô*Ô+¨dÑ2Ð9Ð9Ñ9Ø26€J��eÔ'Ô(¨4Ñ/Ð6Ð6Ñ6Ð6Ð6r+   rŒ   uG  
    Following the ColPali approach, ColQwen2 leverages VLMs to construct efficient multi-vector embeddings directly
    from document images (â€œscreenshotsâ€�) for document retrieval. The model is trained to maximize the similarity
    between these document embeddings and the corresponding query embeddings, using the late interaction method
    introduced in ColBERT.

    Using ColQwen2 removes the need for potentially complex and brittle layout recognition and OCR pipelines with
    a single model that can take into account both the textual and visual content (layout, charts, ...) of a document.

    ColQwen2 is part of the ColVision model family, which was introduced with ColPali in the following paper:
    [*ColPali: Efficient Document Retrieval with Vision Language Models*](https://huggingface.co/papers/2407.01449).
    c                   ó(  ‡ — e Zd Zdefˆ fd„Zee	 	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej	        dz  dej        dz  de
dz  dej        dz  d	ej        dz  d
edz  dedz  dedz  dedz  dej	        dz  dej        dz  defd„¦   «         ¦   «         Zˆ xZS )ÚColQwen2ForRetrievalÚconfigc                 óN   •— t          ¦   «                              |¦  «         | `d S r†   )Úsuperr6   Ú_tied_weights_keys)r8   r™   Ú	__class__s     €r,   r6   zColQwen2ForRetrieval.__init__  s'   ø€ Ý‰Œ×Ò˜Ñ Ô Ð ØÐ#Ð#Ð#r+   NrG   Úattention_maskÚposition_idsr�   rI   Úinputs_embedsÚ	use_cacheÚoutput_attentionsÚoutput_hidden_statesÚreturn_dictrE   rB   r>   c                 ó  — |�u|�s|dd…df         |dd…df         z  }t          j        |j        d         |j        ¬¦  «        }|                     d¦  «        |                     d¦  «        k     }||         }|�|n| j        j        }|	�|	n| j        j        }	|
�|
n| j        j        }
|€¤ | j	         
                    ¦   «         |¦  «        }|�€| j	                             ||d¬¦  «        j        }|| j        j        j        k                         d¦  «        }|                     |j        |j        ¦  «        }|                     ||¦  «        }|  	                    d|||||||	|
¬	¦	  «	        }|	r|j        nd}|d         }| j        j        j        }|                      |                     |¦  «        ¦  «        }||                     dd¬
¦  «        z  }|�||                     d¦  «        z  }t-          ||j        ||j        ¬¦  «        S )z¯
        image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
            The temporal, height and width of feature shape of each image in LLM.
        Nr   r   )Údevicer   T)Úgrid_thwr¤   éÿÿÿÿ)	rG   rŸ   rž   r�   r    r¡   r¢   r£   r¤   )ÚdimÚkeepdim)rŽ   r�   r�   r‘   )rV   ÚarangeÚshaper¦   Ú	unsqueezer™   r¢   r£   r¤   ÚvlmÚget_input_embeddingsÚvisualÚpooler_outputÚ
vlm_configÚimage_token_idÚtoÚdtypeÚmasked_scatterr�   Úembedding_proj_layerÚweightÚnormrŒ   r�   r‘   )r8   rG   rž   rŸ   r�   rI   r    r¡   r¢   r£   r¤   rE   rB   r;   rj   r«   ÚmaskÚimage_embedsÚ
image_maskÚ
vlm_outputÚvlm_hidden_statesÚlast_hidden_statesÚ
proj_dtyperŽ   s                           r,   ÚforwardzColQwen2ForRetrieval.forward  sF  € ð. Ð#¨Ð(Bà$ Q Q Q¨ TÔ*¨^¸A¸A¸A¸q¸DÔ-AÑAˆGÝ”\ ,Ô"4°QÔ"7ÀÄÐOÑOÔOˆFØ×#Ò# AÑ&Ô&¨×):Ò):¸1Ñ)=Ô)=Ò=ˆDØ'¨Ô-ˆLà1BÐ1NÐ-Ð-ÐTXÔT_ÔTqÐð %9Ð$DÐ Ð È$Ì+ÔJjð 	ð &1Ð%<�k�kÀ$Ä+ÔBYˆð Ð Ø;˜DœH×9Ò9Ñ;Ô;¸IÑFÔFˆMàÐ'Ø#œxŸš¨|ÀnÐbf˜ÑgÔgÔu�Ø'¨4¬;Ô+AÔ+PÒP×[Ò[Ð\^Ñ_Ô_�
Ø+Ÿš¨}Ô/CÀ]ÔEXÑYÔY�Ø -× <Ò <¸ZÈÑ VÔ V�à—X’XØØ%Ø)Ø+Ø'ØØ/Ø!5Ø#ð ñ 

ô 

ˆ
ð 9MÐV˜JÔ4Ð4ÐRVÐà'¨œ]ÐØÔ.Ô5Ô;ˆ
Ø×.Ò.Ð/A×/DÒ/DÀZÑ/PÔ/PÑQÔQˆ
ð   *§/¢/°bÀ$ /Ñ"GÔ"GÑGˆ
ØÐ%Ø# n×&>Ò&>¸rÑ&BÔ&BÑBˆJå)Ø!Ø&Ô6Ø+Ø!Ô,ð	
ñ 
ô 
ð 	
r+   )NNNNNNNNNNNN)r&   r'   r(   r   r6   r   r   rV   Ú
LongTensorr•   r   r“   ÚboolrŒ   rÁ   Ú__classcell__)r�   s   @r,   r˜   r˜   õ   sz  ø€ € € € € ð$˜~ð $ð $ð $ð $ð $ð $ð Øð .2Ø.2Ø04Ø(,Ø*.Ø26Ø!%Ø)-Ø,0Ø#'Ø,0Ø26ðI
ð I
àÔ# dÑ*ðI
ð œ tÑ+ðI
ð Ô&¨Ñ-ð	I
ð
  ™ðI
ð Ô  4Ñ'ðI
ð Ô(¨4Ñ/ðI
ð ˜$‘;ðI
ð   $™;ðI
ð # T™kðI
ð ˜D‘[ðI
ð ”l TÑ)ðI
ð Ô(¨4Ñ/ðI
ð 
$ðI
ð I
ð I
ñ „^ñ ÔðI
ð I
ð I
ð I
ð I
r+   r˜   )r˜   r‰   r/   )(Údataclassesr   Úcache_utilsr   Úfeature_extraction_utilsr   Úimage_utilsr   r   Úprocessing_utilsr	   r
   r   r   Útokenization_utils_baser   r   rZ   r   r   r   r   r   Úcolpali.modeling_colpalir   r   Úcolpali.processing_colpalir   Úconfiguration_colqwen2r   rV   Ú
get_loggerr&   Úloggerr   r/   r‰   rŒ   r˜   Ú__all__r*   r+   r,   ú<module>rÑ      s‡  ðð "Ð !Ð !Ð !Ð !Ð !à  Ð  Ð  Ð  Ð  Ð  Ø 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5Ø XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XÐ XØ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CØ _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ø RÐ RÐ RÐ RÐ RÐ RÐ RÐ RØ 9Ð 9Ð 9Ð 9Ð 9Ð 9Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2ð ÐÑÔð Ø€L€L€Là	ˆÔ	˜HÑ	%Ô	%€ð
ð 
ð 
ð 
ð 
Ð.°eð 
ñ 
ô 
ð 
ðeCð eCð eCð eCð eCÐ(ñ eCô eCð eCðP	ð 	ð 	ð 	ð 	Ð4ñ 	ô 	ð 	ð €ððñ ô ð
 ð7ð 7ð 7ð 7ð 7 ñ 7ô 7ñ „ñô ð7ð( €ððñ ô ðP
ð P
ð P
ð P
ð P
Ð.ñ P
ô P
ñô ðP
ðfð ð €€€r+   