§
    ‚Štjl8  ã            
       ób  — d Z ddlZddlmZ ddlmZmZ ddlm	Z	m
Z
mZ ddlmZmZ ddlmZ  G d	„ d
e	d¬¦  «        Zdee         dedeee                  fd„Zdeeee                           deee                  dededej        f
d„Zdedededefd„Ze G d„ de
¦  «        ¦   «         ZdgZdS )zProcessor class for Mllama.é    Né   )ÚBatchFeature)Ú
ImageInputÚmake_nested_list_of_images)ÚProcessingKwargsÚProcessorMixinÚUnpack)ÚPreTokenizedInputÚ	TextInput)Úauto_docstringc                   ó   — e Zd ZdddiiZdS )ÚMllamaProcessorKwargsÚimage_kwargsÚmax_image_tilesé   N)Ú__name__Ú
__module__Ú__qualname__Ú	_defaults© ó    új/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/mllama/processing_mllama.pyr   r      s"   € € € € € àØ˜qð
ð€I€I€Ir   r   F)ÚtotalÚ	input_idsÚimage_token_idÚreturnc                 óÈ  ‡— ˆfd„t          | ¦  «        D ¦   «         }t          |¦  «        dk    rg S t          |¦  «        dk    r|d         dggS d„ t          |dd…         |dd…         ¦  «        D ¦   «         }|                     |d         t          | ¦  «        g¦  «         |d         d         }|ddd…         D ]$}|d         |d         dz
  k    r||d<   |d         }Œ%|S )aó  
    Generate a cross-attention token mask for image tokens in the input sequence.

    This function identifies the positions of image tokens in the input sequence and creates
    a mask that defines which subsequent tokens each image token should attend to.

    Args:
        input_ids (list[int]): A list of token ids representing the input sequence.
        image_token_id (int): The id of the token used to represent images in the sequence.

    Returns:
        list[list[int]]: A list of [start, end] pairs, where each pair represents the range
        of tokens an image token should attend to.

    Notes:
        - If no image tokens are present, an empty list is returned.
        - For a single image token, it attends to all subsequent tokens until the end of the sequence.
        - For multiple image tokens, each attends to tokens up to the next image token or the end of the sequence.
        - Consecutive image tokens are treated as a group and attend to all subsequent tokens together.
    c                 ó&   •— g | ]\  }}|‰k    ¯|‘ŒS r   r   )Ú.0ÚiÚtokenr   s      €r   ú
<listcomp>z2get_cross_attention_token_mask.<locals>.<listcomp>8   s(   ø€ Ð_Ð_Ð_¡8 1 eÀuÐP^ÒG^ÐG^˜QÐG^ÐG^ÐG^r   r   é   éÿÿÿÿc                 ó   — g | ]	\  }}||g‘Œ
S r   r   )r   Úloc1Úloc2s      r   r"   z2get_cross_attention_token_mask.<locals>.<listcomp>A   s    € ÐnÐnÐn¡Z T¨4�T˜4�LÐnÐnÐnr   N)Ú	enumerateÚlenÚzipÚappend)r   r   Úimage_token_locationsÚvision_masksÚlast_mask_endÚvision_masks    `    r   Úget_cross_attention_token_maskr0   "   s"  ø€ ð, `Ð_Ð_Ð_­y¸Ñ/CÔ/CÐ_Ñ_Ô_Ðå
Ð Ñ!Ô! QÒ&Ð&Øˆ	õ Ð Ñ!Ô! QÒ&Ð&Ø& qÔ)¨2Ð.Ð/Ð/ànÐnµ3Ð7LÈSÈbÈSÔ7QÐShÐijÐikÐikÔSlÑ3mÔ3mÐnÑnÔn€Lð ×ÒÐ.¨rÔ2µC¸	±N´NÐCÑDÔDÐDð
 ! Ô$ QÔ'€MØ# D D b DÔ)ð 'ð 'ˆØ�qŒ>˜[¨œ^¨aÑ/Ò/Ð/Ø*ˆK˜‰NØ# AœˆˆàÐr   Úcross_attention_token_maskÚ	num_tilesÚmax_num_tilesÚlengthc           	      ó°  — t          | ¦  «        }t          d„ | D ¦   «         ¦  «        }t          j        ||||ft          j        ¬¦  «        }t          t          | |¦  «        ¦  «        D ]k\  }\  }}	t          t          ||	¦  «        ¦  «        D ]E\  }
\  }}t          |¦  «        dk    r*|\  }}t          ||¦  «        }|dk    r|}d||||…|
d|…f<   ŒFŒl|S )a  
    Convert the cross attention mask indices to a cross attention mask 4D array.

    This function takes a sparse representation of cross attention masks and converts it to a dense 4D numpy array.
    The sparse representation is a nested list structure that defines attention ranges for each image in each batch item.

    Args:
        cross_attention_token_mask (list[list[list[int]]]): A nested list structure where:
            - The outer list represents the batch dimension.
            - The middle list represents different images within each batch item.
            - The inner list contains pairs of integers [start, end] representing token ranges for each image.
        num_tiles (list[list[int]]): A nested list structure specifying the number of tiles for each image in each batch item.
        max_num_tiles (int): The maximum possible number of tiles.
        length (int): The total sequence length of the input.

    Returns:
        np.ndarray: A 4D numpy array of shape (batch_size, length, max_num_images, max_num_tiles)
            The array contains `1` where attention is allowed and `0` where it is not.

    Note:
        - Special handling is done for cases where the end token is -1, which is interpreted as attending to the end of the sequence.
    c              3   ó4   K  — | ]}t          |¦  «        V — Œd S ©N©r)   )r   Úmaskss     r   ú	<genexpr>z?convert_sparse_cross_attention_mask_to_dense.<locals>.<genexpr>p   s(   è è € ÐLÐL¨�˜U™œÐLÐLÐLÐLÐLÐLr   )ÚshapeÚdtypeé   r$   r#   N)r)   ÚmaxÚnpÚzerosÚint64r(   r*   Úmin)r1   r2   r3   r4   Ú
batch_sizeÚmax_num_imagesÚcross_attention_maskÚ
sample_idxÚsample_masksÚsample_num_tilesÚmask_idxÚ	locationsÚmask_num_tilesÚstartÚends                  r   Ú,convert_sparse_cross_attention_mask_to_denserN   R   s  € õ: Ð/Ñ0Ô0€JÝÐLÐLÐ1KÐLÑLÔLÑLÔL€Nåœ8Ø˜6 >°=ÐAÝŒhðñ ô Ðõ
 9BÅ#ÐF`ÐbkÑBlÔBlÑ8mÔ8mð [ð [Ñ4ˆ
Ñ4�\Ð#3Ý5>½sÀ<ÐQaÑ?bÔ?bÑ5cÔ5cð 	[ð 	[Ñ1ˆHÑ1�y .Ý�9‰~Œ~ Ò"Ð"Ø&‘
��sÝ˜#˜vÑ&Ô&�Ø˜"’9�9Ø �CØYZÐ$ Z°°s°¸HÀoÀ~ÀoÐ%UÑVøð	[ð  Ðr   ÚpromptÚ	bos_tokenÚimage_tokenc                 ó´   — || v r| S d}|                       |¦  «        r1| t          |¦  «        d…         } |dz  }|                       |¦  «        °1||z  › |› | › �S )a\  
    Builds a string from the input prompt by adding `bos_token` if not already present.

    Args:
        prompt (`str`):
            The input prompt string.
        bos_token (`str`):
            The beginning of sentence token to be added.
        image_token (`str`):
            The image token used to identify the start of an image sequence.

    Returns:
        str: The modified prompt string with the `bos_token` added if necessary.

    Examples:
        >>> build_string_from_input("Hello world", "<begin_of_text>", "<|image|>")
        '<begin_of_text>Hello world'

        >>> build_string_from_input("<|image|>Hello world", "<begin_of_text>", "<|image|>")
        '<|image|><begin_of_text>Hello world'

        >>> build_string_from_input("<begin_of_text>Hello world", "<begin_of_text>", "<|image|>")
        '<begin_of_text>Hello world'
    r   Nr#   )Ú
startswithr)   )rO   rP   rQ   Únum_image_tokens_on_starts       r   Úbuild_string_from_inputrU   ‚   s‰   € ð4 �FÐÐØˆà !ÐØ
×
Ò
˜KÑ
(Ô
(ð 'Ø�˜KÑ(Ô(Ð*Ð*Ô+ˆØ! QÑ&Ð!ð ×
Ò
˜KÑ
(Ô
(ð 'ð Ð5Ñ5ÐJ°yÐJÀ&ÐJÐJÐJr   c            
       óx  ‡ — e Zd ZeZdˆ fd„	Ze	 	 ddedz  dee	z  e
e         z  e
e	         z  dz  dee         defd„¦   «         Z	 	 ddedz  dee	z  e
e         z  e
e	         z  dz  fˆ fd„Z	 	 ddedz  dee	z  e
e         z  e
e	         z  dz  dee         fˆ fd	„Zd
ededefd„Z	 dd„Zed„ ¦   «         Zˆ xZS )ÚMllamaProcessorNc                 óR  •— t          |d¦  «        s'd| _        |                     | j        ¦  «        | _        n|j        | _        |j        | _        d| _        |                     | j        ¦  «        | _        |j        | _        t          ¦   «                              |||¬¦  «         d S )NrQ   z	<|image|>z<|python_tag|>)Úchat_template)	ÚhasattrrQ   Úconvert_tokens_to_idsr   Úpython_tokenÚpython_token_idrP   ÚsuperÚ__init__)ÚselfÚimage_processorÚ	tokenizerrY   Ú	__class__s       €r   r_   zMllamaProcessor.__init__«   sŸ   ø€ Ý�y -Ñ0Ô0ð 	;Ø*ˆDÔØ"+×"AÒ"AÀ$ÔBRÑ"SÔ"SˆDÔÐà(Ô4ˆDÔØ"+Ô":ˆDÔà,ˆÔØ(×>Ò>¸tÔ?PÑQÔQˆÔØ"Ô,ˆŒÝ‰Œ×Ò˜¨)À=ÐÑQÔQÐQÐQÐQr   ÚimagesÚtextÚkwargsr   c           
      óh  ‡ — ‰                       ||¬¦  «        \  }} ‰ j        d||dœ|¤Ž  ‰ j        t          fd‰ j        j        i|¤Ž}|d                              dd¦  «        }i }|�- ‰ j        |fi |d         ¤Ž}‰                      ||dg¬¦  «         i }|�, ‰ j        |fi |d         ¤Ž\  }}|                     d	¦  «        }	|�U|�Sˆ fd
„|d         D ¦   «         }
t          |
|	‰ j
        j        t          d„ |d         D ¦   «         ¦  «        ¬¦  «        }||d<   t          i |¥|¥|¬¦  «        S )a—  
        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model. Returned when `text` is not `None`.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
            TODO: add aspect_ratio_ids and aspect_ratio_mask and cross_attention_mask
        ©rd   re   Útokenizer_init_kwargsÚtext_kwargsÚreturn_tensorsNÚimage)Ú
modalitiesÚimages_kwargsr2   c                 ó:   •— g | ]}t          |‰j        ¦  «        ‘ŒS r   )r0   r   )r   Ú	token_idsr`   s     €r   r"   z,MllamaProcessor.__call__.<locals>.<listcomp>à   s6   ø€ ð *ð *ð *àõ /¨y¸$Ô:MÑNÔNð*ð *ð *r   r   c              3   ó4   K  — | ]}t          |¦  «        V — Œd S r7   r8   )r   r   s     r   r:   z+MllamaProcessor.__call__.<locals>.<genexpr>è   s(   è è € ÐTÐT¨i�3˜y™>œ>ÐTÐTÐTÐTÐTÐTr   )r2   r3   r4   rE   )ÚdataÚtensor_typer   )Úprepare_inputs_layoutÚvalidate_inputsÚ_merge_kwargsr   rb   Úinit_kwargsÚpopÚ_check_special_mm_tokensÚ_process_imagesrN   ra   r   r>   r   )r`   rd   re   rf   Úoutput_kwargsrk   Útext_inputsÚimage_inputsÚ_r2   r1   rE   s   `           r   Ú__call__zMllamaProcessor.__call__¸   sÅ  ø€ ð$ ×1Ò1¸ÀdÐ1ÑKÔK‰ˆ�ØˆÔÐ@ F°Ð@Ð@¸Ð@Ð@Ð@à*˜Ô*Ý!ð
ð 
à"&¤.Ô"<ð
ð ð
ð 
ˆð
 ' }Ô5×9Ò9Ð:JÈDÑQÔQˆàˆØÐØ(˜$œ.¨ÐNÐN°¸}Ô1MÐNÐNˆKØ×)Ò)¨$°ÈÈ	Ð)ÑRÔRÐRàˆØÐØ2˜dÔ2°6Ð\Ð\¸]È?Ô=[Ð\Ð\‰OˆL˜!Ø$×(Ò(¨Ñ5Ô5ˆIð Ð $Ð"2ð*ð *ð *ð *à!,¨[Ô!9ð*ñ *ô *Ð&õ $PØ*Ø#Ø"Ô2ÔBÝÐTÐT¸;À{Ô;SÐTÑTÔTÑTÔTð	$ñ $ô $Ð ð 3GˆKÐ.Ñ/åÐ!@ KÐ!@°<Ð!@ÈnÐ]Ñ]Ô]Ð]r   c                 óŽ   •‡ —  t          ¦   «         j        d||dœ|¤Ž^}}}|�t          |¦  «        }|�ˆ fd„|D ¦   «         }||fS )Nrh   c                 óF   •— g | ]}t          |‰j        ‰j        ¦  «        ‘ŒS r   )rU   rP   rQ   )r   Ú	text_itemr`   s     €r   r"   z9MllamaProcessor.prepare_inputs_layout.<locals>.<listcomp>û   s,   ø€ ÐoÐoÐoÐ]fÕ+¨I°t´~ÀtÔGWÑXÔXÐoÐoÐor   r   )r^   rt   r   )r`   rd   re   rf   r~   rc   s   `    €r   rt   z%MllamaProcessor.prepare_inputs_layoutî   sp   øø€ ð 9�5™7œ7Ô8Ð\ÀÈTÐ\Ð\ÐU[Ð\Ð\Ðˆ��qð ÐÝ/°Ñ7Ô7ˆFàÐØoÐoÐoÐoÐjnÐoÑoÔoˆDà�tˆ|Ðr   c                 óü  •‡ —  t          ¦   «         j        ||fi |¤Ž |�Øˆ fd„|D ¦   «         }t          |¦  «        dk    r|€t          d¦  «        ‚|�¦t	          |¦  «        }d„ |D ¦   «         }t          d„ |D ¦   «         ¦  «        r(t          d„ |D ¦   «         ¦  «        st          d¦  «        ‚||k    rFd}t          |¦  «        t          |¦  «        k    r||k    rd	}t          d
|› d|› d|› �¦  «        ‚d S d S d S )Nc                 óD   •— g | ]}|                      ‰j        ¦  «        ‘ŒS r   )ÚcountrQ   )r   Útr`   s     €r   r"   z3MllamaProcessor.validate_inputs.<locals>.<listcomp>  s(   ø€ ÐHÐHÐH¸a §¢¨Ô(8Ñ 9Ô 9ÐHÐHÐHr   r   z@No image were provided, but there are image tokens in the promptc                 ó,   — g | ]}t          |¦  «        ‘ŒS r   r8   )r   Úsamples     r   r"   z3MllamaProcessor.validate_inputs.<locals>.<listcomp>  s   € Ð%GÐ%GÐ%G°f¥c¨&¡k¤kÐ%GÐ%GÐ%Gr   c              3   ó"   K  — | ]
}|d k    V — ŒdS ©r   Nr   ©r   Ú	batch_imgs     r   r:   z2MllamaProcessor.validate_inputs.<locals>.<genexpr>  s&   è è € ÐHÐH¨)�y A’~ÐHÐHÐHÐHÐHÐHr   c              3   ó"   K  — | ]
}|d k    V — ŒdS rŠ   r   r‹   s     r   r:   z2MllamaProcessor.validate_inputs.<locals>.<genexpr>  s?   è è € ð Uð UØ'0�I ’NðUð Uð Uð Uð Uð Ur   zaIf a batch of text is provided, there should be either no images or at least one image per sampleÚ zZMake sure to pass your images as a nested list, where each sub-list holds images per batchz)The number of image tokens in each text (zA) should be the same as the number of provided images per batch (z). )r^   ru   ÚsumÚ
ValueErrorr   ÚanyÚall)r`   rd   re   rf   Ún_images_in_textÚn_images_in_imagesÚadd_messagerc   s   `      €r   ru   zMllamaProcessor.validate_inputsÿ   s¢  øø€ ð 	 �‰ŒÔ ¨Ð7Ð7°Ð7Ð7Ð7àÐØHÐHÐHÐHÀ4ÐHÑHÔHÐåÐ#Ñ$Ô$ qÒ(Ð(¨V¨^Ý Ð!cÑdÔdÐdØÐ#Ý3°FÑ;Ô;�Ø%GÐ%GÀÐ%GÑ%GÔ%GÐ"åÐHÐHÐ7GÐHÑHÔHÑHÔHð ÕQTð Uð UØ4DðUñ Uô Uñ Rô Rð õ %Ø{ñô ð ð &Ð)9Ò9Ð9Ø"$�KÝÐ-Ñ.Ô.µ#Ð6FÑ2GÔ2GÒGÐGÐL^ÐbrÒLrÐLrð 'C˜å$ðeÐDTð eð eØ@Rðeð eØWbðeð eñô ð ð+ Ðð
 $Ð#ð :Ð9r   r}   Ú	image_idxc                 ó   — | j         S r7   )rQ   )r`   r}   r–   s      r   Úreplace_image_tokenz#MllamaProcessor.replace_image_token!  s   € ð ÔÐr   TFc                 ó.   —  | j         j        |f||dœ|¤ŽS )aš  
        Post-process the output of the model to decode the text.

        Args:
            generated_outputs (`torch.Tensor` or `np.ndarray`):
                The output of the model `generate` function. The output is expected to be a tensor of shape `(batch_size, sequence_length)`
                or `(sequence_length,)`.
            skip_special_tokens (`bool`, *optional*, defaults to `True`):
                Whether or not to remove special tokens in the output. Argument passed to the tokenizer's `batch_decode` method.
            clean_up_tokenization_spaces (`bool`, *optional*, defaults to `False`):
                Whether or not to clean up the tokenization spaces. Argument passed to the tokenizer's `batch_decode` method.
            **kwargs:
                Additional arguments to be passed to the tokenizer's `batch_decode method`.

        Returns:
            `list[str]`: The decoded text.
        )Úskip_special_tokensÚclean_up_tokenization_spaces)rb   Úbatch_decode)r`   Úgenerated_outputsrš   r›   rf   s        r   Úpost_process_image_text_to_textz/MllamaProcessor.post_process_image_text_to_text&  s:   € ð( +ˆtŒ~Ô*Øð
à 3Ø)Eð
ð 
ð ð	
ð 
ð 	
r   c                 óv   — | j         j        }| j        j        }d„ |D ¦   «         }t          ||z   dgz   ¦  «        S )Nc                 ó   — g | ]
}|d k    ¯|‘ŒS )r2   r   )r   Únames     r   r"   z5MllamaProcessor.model_input_names.<locals>.<listcomp>H  s$   € Ð&kÐ&kÐ&k°ÐW[Ð_jÒWjÐWj tÐWjÐWjÐWjr   rE   )rb   Úmodel_input_namesra   Úlist)r`   Útokenizer_input_namesÚimage_processor_input_namess      r   r¢   z!MllamaProcessor.model_input_namesA  sO   € à $¤Ô @ÐØ&*Ô&:Ô&LÐ#ð 'lÐ&kÐ8SÐ&kÑ&kÔ&kÐ#ÝÐ)Ð,GÑGÐKaÐJbÑbÑcÔcÐcr   r7   )NN)TF)r   r   r   r   Úvalid_processor_kwargsr_   r   r   r   r
   r£   r	   r   r   rt   r   ru   ÚdictÚintÚstrr˜   rž   Úpropertyr¢   Ú__classcell__)rc   s   @r   rW   rW   §   s  ø€ € € € € à2ÐðRð Rð Rð Rð Rð Rð ð %)Øaeð3^ð 3^à˜TÑ!ð3^ð Ð+Ñ+¨d°9¬oÑ=ÀÐEVÔ@WÑWÐZ^Ñ^ð3^ð Ð.Ô/ð	3^ð
 
ð3^ð 3^ð 3^ñ „^ð3^ðn %)Øaeðð à˜TÑ!ðð Ð+Ñ+¨d°9¬oÑ=ÀÐEVÔ@WÑWÐZ^Ñ^ðð ð ð ð ð ð& %)Øaeð ð  à˜TÑ!ð ð Ð+Ñ+¨d°9¬oÑ=ÀÐEVÔ@WÑWÐZ^Ñ^ð ð Ð)Ô*ð	 ð  ð  ð  ð  ð  ðD °ð  Àð  Èð  ð  ð  ð  ð Y^ð
ð 
ð 
ð 
ð6 ðdð dñ „Xðdð dð dð dð dr   rW   )Ú__doc__Únumpyr?   Úfeature_extraction_utilsr   Úimage_utilsr   r   Úprocessing_utilsr   r   r	   Útokenization_utils_baser
   r   Úutilsr   r   r£   r¨   r0   ÚndarrayrN   r©   rU   rW   Ú__all__r   r   r   ú<module>rµ      sà  ðð "Ð !à Ð Ð Ð à 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø AÐ AÐ AÐ AÐ AÐ AÐ AÐ AØ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HØ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CØ #Ð #Ð #Ð #Ð #Ð #ðð ð ð ð Ð,°Eð ñ ô ð ð-¨d°3¬ið -Èð -ÐQUÐVZÐ[^ÔV_ÔQ`ð -ð -ð -ð -ð`- Ø $ T¨$¨s¬)¤_Ô 5ð- à�D˜”IŒð- ð ð- ð ð	- ð
 „Zð- ð - ð - ð - ð`"K Cð "K°Cð "KÀcð "KÈcð "Kð "Kð "Kð "KðJ ðadð adð adð adð ad�nñ adô adñ „ðadðH Ð
€€€r   