§
    ‚Štj0-  ã                   óÎ  — d Z ddlmZ ddlZddlmZ ddlmZ ddlm	Z	 ddl
mZ dd	lmZ dd
lmZ ddlmZmZmZmZ ddlmZ ddlmZ  ej        e¦  «        Z ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Ze G d„ de¦  «        ¦   «         Zdd„Z G d„ dej        ¦  «        Z  G d„ dej        ¦  «        Z! ed¬¦  «         G d„ de¦  «        ¦   «         Z"ddgZ#dS )zPyTorch VitPose model.é    )Ú	dataclassN)Únné   )Úinitialization)Úload_backbone)ÚBackboneOutput)ÚPreTrainedModel)ÚUnpack)ÚModelOutputÚTransformersKwargsÚauto_docstringÚlogging)Úcan_return_tupleé   )ÚVitPoseConfigz6
    Class for outputs of pose estimation models.
    )Úcustom_introc                   ó¬   — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dZ	e
ej        df         dz  ed<   dZe
ej        df         dz  ed<   dS )ÚVitPoseEstimatorOutputaH  
    loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
        Loss is not supported at this moment. See https://github.com/ViTAE-Transformer/ViTPose/tree/main/mmpose/models/losses for further detail.
    heatmaps (`torch.FloatTensor` of shape `(batch_size, num_keypoints, height, width)`):
        Heatmaps as predicted by the model.
    hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
        Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
        one for the output of each stage) of shape `(batch_size, sequence_length, hidden_size)`. Hidden-states
        (also called feature maps) of the model at the output of each stage.
    NÚlossÚheatmaps.Úhidden_statesÚ
attentions)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   ÚtorchÚFloatTensorÚ__annotations__r   r   Útupler   © ó    új/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/vitpose/modeling_vitpose.pyr   r   $   s’   € € € € € € ð	ð 	ð &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø)-€HˆeÔ $Ñ&Ð-Ð-Ñ-Ø:>€M�5˜Ô*¨CÐ/Ô0°4Ñ7Ð>Ð>Ñ>Ø7;€J��eÔ'¨Ð,Ô-°Ñ4Ð;Ð;Ñ;Ð;Ð;r"   r   c                   ó”   ‡ — e Zd ZU eed<   dZdZdZdZ e	j
        ¦   «         dej        ej        z  ej        z  fˆ fd„¦   «         Zˆ xZS )ÚVitPosePreTrainedModelÚconfigÚvitÚpixel_values)ÚimageTÚmodulec                 ó*  •— t          ¦   «                              |¦  «         t          |t          j        t          j        f¦  «        rHt          j        |j        d| j	        j
        ¬¦  «         |j        �t          j        |j        ¦  «         dS dS dS )zInitialize the weightsg        )ÚmeanÚstdN)ÚsuperÚ_init_weightsÚ
isinstancer   ÚLinearÚConv2dÚinitÚtrunc_normal_Úweightr&   Úinitializer_rangeÚbiasÚzeros_)Úselfr*   Ú	__class__s     €r#   r/   z$VitPosePreTrainedModel._init_weightsD   s‡   ø€ õ 	‰Œ×Ò˜fÑ%Ô%Ð%Ý�f�rœy­"¬)Ð4Ñ5Ô5ð 	)ÝÔ˜vœ}°3¸D¼KÔ<YÐZÑZÔZÐZØŒ{Ð&Ý”˜FœKÑ(Ô(Ð(Ð(Ð(ð	)ð 	)à&Ð&r"   )r   r   r   r   r   Úbase_model_prefixÚmain_input_nameÚinput_modalitiesÚsupports_gradient_checkpointingr   Úno_gradr   r1   r2   Ú	LayerNormr/   Ú__classcell__©r:   s   @r#   r%   r%   <   s‹   ø€ € € € € € àÐÐÑØÐØ$€OØ!ÐØ&*Ð#à€U„]�_„_ð) B¤I°´	Ñ$9¸B¼LÑ$Hð )ð )ð )ð )ð )ñ „_ð)ð )ð )ð )ð )r"   r%   úgaussian-heatmapc                 ó&  — |dvrt          d¦  «        ‚| j        dk    rt          d¦  «        ‚| j        \  }}}}d}|dk    r2d}|                      ¦   «         } | dd…ddd…d	f          | dd…ddd…d	f<   |                      |d
|||¦  «        } |                      ¦   «         }|                     d
¦  «        \  }	}
| dd…|
d	f         |dd…|	d	f<   | dd…|	d	f         |dd…|
d	f<   |                     ||||f¦  «        }|                     d
¦  «        }|S )aÃ  Flip the flipped heatmaps back to the original form.

    Args:
        output_flipped (`torch.tensor` of shape `(batch_size, num_keypoints, height, width)`):
            The output heatmaps obtained from the flipped images.
        flip_pairs (`torch.Tensor` of shape `(num_keypoints, 2)`):
            Pairs of keypoints which are mirrored (for example, left ear -- right ear).
        target_type (`str`, *optional*, defaults to `"gaussian-heatmap"`):
            Target type to use. Can be gaussian-heatmap or combined-target.
            gaussian-heatmap: Classification target with gaussian distribution.
            combined-target: The combination of classification target (response map) and regression target (offset map).
            Paper ref: Huang et al. The Devil is in the Details: Delving into Unbiased Data Processing for Human Pose Estimation (CVPR 2020).

    Returns:
        torch.Tensor: heatmaps that flipped back to the original image
    )rC   úcombined-targetz9target_type should be gaussian-heatmap or combined-targeté   zCoutput_flipped should be [batch_size, num_keypoints, height, width]r   rE   r   N.éÿÿÿÿ)Ú
ValueErrorÚndimÚshapeÚcloneÚreshapeÚunbindÚflip)Úoutput_flippedÚ
flip_pairsÚtarget_typeÚ
batch_sizeÚnum_keypointsÚheightÚwidthÚchannelsÚoutput_flipped_backÚleft_indicesÚright_indicess              r#   Ú	flip_backrZ   N   st  € ð" ÐAÐAÐAÝÐTÑUÔUÐUàÔ˜aÒÐÝÐ^Ñ_Ô_Ð_à/=Ô/CÑ,€J�˜v uØ€HØÐ'Ò'Ð'ØˆØ'×-Ò-Ñ/Ô/ˆØ(6°q°q°q¸!¸$¸Q¸$À°|Ô(DÐ'Dˆ�q�q�q˜!˜$˜Q˜$ �|Ñ$Ø#×+Ò+¨J¸¸HÀfÈeÑTÔT€NØ(×.Ò.Ñ0Ô0Ðð #-×"3Ò"3°BÑ"7Ô"7Ñ€L�-Ø0>¸q¸q¸qÀ-ÐQTÐ?TÔ0UÐ˜˜˜˜<¨Ð,Ñ-Ø1?ÀÀÀÀ<ÐQTÐ@TÔ1UÐ˜˜˜˜=¨#Ð-Ñ.Ø-×5Ò5°zÀ=ÐRXÐZ_Ð6`ÑaÔaÐà-×2Ò2°2Ñ6Ô6ÐØÐr"   c                   ób   ‡ — e Zd ZdZdefˆ fd„Zd	dej        dej        dz  dej        fd„Zˆ xZ	S )
ÚVitPoseSimpleDecoderz�
    Simple decoding head consisting of a ReLU activation, 4x upsampling and a 3x3 convolution, turning the
    feature maps into heatmaps.
    r&   c                 ó  •— t          ¦   «                              ¦   «          t          j        ¦   «         | _        t          j        |j        dd¬¦  «        | _        t          j        |j	        j
        |j        ddd¬¦  «        | _        d S )NÚbilinearF)Úscale_factorÚmodeÚalign_cornersr   r   ©Úkernel_sizeÚstrideÚpadding)r.   Ú__init__r   ÚReLUÚ
activationÚUpsampler_   Ú
upsamplingr2   Úbackbone_configÚhidden_sizeÚ
num_labelsÚconv©r9   r&   r:   s     €r#   rf   zVitPoseSimpleDecoder.__init__~   st   ø€ Ý‰Œ×ÒÑÔÐåœ'™)œ)ˆŒÝœ+°6Ô3FÈZÐglÐmÑmÔmˆŒÝ”IØÔ"Ô.°Ô0AÈqÐYZÐdeð
ñ 
ô 
ˆŒ	ˆ	ˆ	r"   NÚhidden_staterP   Úreturnc                 ó¨   — |                       |¦  «        }|                      |¦  «        }|                      |¦  «        }|�t          ||¦  «        }|S ©N)rh   rj   rn   rZ   ©r9   rp   rP   r   s       r#   ÚforwardzVitPoseSimpleDecoder.forward‡   sO   € à—’ |Ñ4Ô4ˆØ—’ |Ñ4Ô4ˆØ—9’9˜\Ñ*Ô*ˆàÐ!Ý  ¨:Ñ6Ô6ˆHàˆr"   rs   ©
r   r   r   r   r   rf   r   ÚTensorru   rA   rB   s   @r#   r\   r\   x   s‰   ø€ € € € € ðð ð

˜}ð 
ð 
ð 
ð 
ð 
ð 
ð	ð 	 E¤Lð 	¸e¼lÈTÑ>Qð 	Ð]bÔ]ið 	ð 	ð 	ð 	ð 	ð 	ð 	ð 	r"   r\   c                   óT   ‡ — e Zd ZdZdefˆ fd„Zddej        dej        dz  fd„Zˆ xZ	S )	ÚVitPoseClassicDecoderzš
    Classic decoding head consisting of a 2 deconvolutional blocks, followed by a 1x1 convolution layer,
    turning the feature maps into heatmaps.
    r&   c                 óâ  •— t          ¦   «                              ¦   «          t          j        |j        j        ddddd¬¦  «        | _        t          j        d¦  «        | _        t          j	        ¦   «         | _
        t          j        dddddd¬¦  «        | _        t          j        d¦  «        | _        t          j	        ¦   «         | _        t          j        d|j        ddd¬¦  «        | _        d S )	Né   rF   é   r   F)rc   rd   re   r7   r   rb   )r.   rf   r   ÚConvTranspose2drk   rl   Údeconv1ÚBatchNorm2dÚ
batchnorm1rg   Úrelu1Údeconv2Ú
batchnorm2Úrelu2r2   rm   rn   ro   s     €r#   rf   zVitPoseClassicDecoder.__init__™   sÊ   ø€ Ý‰Œ×ÒÑÔÐåÔ)ØÔ"Ô.°ÀÈ1ÐVWÐ^cð
ñ 
ô 
ˆŒõ œ.¨Ñ-Ô-ˆŒÝ”W‘Y”YˆŒ
åÔ)¨#¨sÀÈ!ÐUVÐ]bÐcÑcÔcˆŒÝœ.¨Ñ-Ô-ˆŒÝ”W‘Y”YˆŒ
å”I˜c 6Ô#4À!ÈAÐWXÐYÑYÔYˆŒ	ˆ	ˆ	r"   Nrp   rP   c                 óP  — |                       |¦  «        }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }|�t          ||¦  «        }|S rs   )r~   r€   r�   r‚   rƒ   r„   rn   rZ   rt   s       r#   ru   zVitPoseClassicDecoder.forward¨   s“   € Ø—|’| LÑ1Ô1ˆØ—’ |Ñ4Ô4ˆØ—z’z ,Ñ/Ô/ˆà—|’| LÑ1Ô1ˆØ—’ |Ñ4Ô4ˆØ—z’z ,Ñ/Ô/ˆà—9’9˜\Ñ*Ô*ˆàÐ!Ý  ¨:Ñ6Ô6ˆHàˆr"   rs   rv   rB   s   @r#   ry   ry   “   s…   ø€ € € € € ðð ð
Z˜}ð Zð Zð Zð Zð Zð Zðð  E¤Lð ¸e¼lÈTÑ>Qð ð ð ð ð ð ð ð r"   ry   z?
    The VitPose model with a pose estimation head on top.
    c                   ó²   ‡ — e Zd Zdefˆ fd„Zee	 	 	 ddej        dej        dz  dej        dz  dej        dz  de	e
         d	efd
„¦   «         ¦   «         Zˆ xZS )ÚVitPoseForPoseEstimationr&   c                 óä  •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          | j        j        d¦  «        st          d¦  «        ‚t	          | j        j        d¦  «        st          d¦  «        ‚t	          | j        j        d¦  «        st          d¦  «        ‚|j        rt          |¦  «        nt          |¦  «        | _
        |                      ¦   «          d S )Nrl   z0The backbone should have a hidden_size attributeÚ
image_sizez0The backbone should have an image_size attributeÚ
patch_sizez/The backbone should have a patch_size attribute)r.   rf   r   ÚbackboneÚhasattrr&   rH   Úuse_simple_decoderr\   ry   ÚheadÚ	post_initro   s     €r#   rf   z!VitPoseForPoseEstimation.__init__¿   sà   ø€ Ý‰Œ×Ò˜Ñ Ô Ð å% fÑ-Ô-ˆŒõ �t”}Ô+¨]Ñ;Ô;ð 	QÝÐOÑPÔPÐPÝ�t”}Ô+¨\Ñ:Ô:ð 	QÝÐOÑPÔPÐPÝ�t”}Ô+¨\Ñ:Ô:ð 	PÝÐNÑOÔOÐOà4:Ô4MÐpÕ(¨Ñ0Ô0Ð0ÕShÐioÑSpÔSpˆŒ	ð 	�ŠÑÔÐÐÐr"   Nr(   Údataset_indexrP   ÚlabelsÚkwargsrq   c                 ó,  — d}|�t          d¦  «        ‚ | j        j        |fd|i|¤Ž}|j        d         }|j        d         }	| j        j        j        d         | j        j        j        d         z  }
| j        j        j        d         | j        j        j        d         z  }| 	                    ddd¦  «        }| 
                    |	d|
|¦  «                             ¦   «         }|                      ||¬¦  «        }t          |||j        |j        ¬	¦  «        S )
a´  
        dataset_index (`torch.Tensor` of shape `(batch_size,)`):
            Index to use in the Mixture-of-Experts (MoE) blocks of the backbone.

            This corresponds to the dataset index used during training, e.g. For the single dataset index 0 refers to the corresponding dataset. For the multiple datasets index 0 refers to dataset A (e.g. MPII) and index 1 refers to dataset B (e.g. CrowdPose).
        flip_pairs (`torch.tensor`, *optional*):
            Whether to mirror pairs of keypoints (for example, left ear -- right ear).

        Examples:

        ```python
        >>> from transformers import AutoImageProcessor, VitPoseForPoseEstimation
        >>> import torch
        >>> from PIL import Image
        >>> import httpx
        >>> from io import BytesIO

        >>> processor = AutoImageProcessor.from_pretrained("usyd-community/vitpose-base-simple")
        >>> model = VitPoseForPoseEstimation.from_pretrained("usyd-community/vitpose-base-simple")

        >>> url = "http://images.cocodataset.org/val2017/000000039769.jpg"
        >>> with httpx.stream("GET", url) as response:
        ...     image = Image.open(BytesIO(response.read()))
        >>> boxes = [[[412.8, 157.61, 53.05, 138.01], [384.43, 172.21, 15.12, 35.74]]]
        >>> inputs = processor(image, boxes=boxes, return_tensors="pt")

        >>> with torch.no_grad():
        ...     outputs = model(**inputs)
        >>> heatmaps = outputs.heatmaps
        ```NzTraining is not yet supportedr�   rG   r   r   r|   )rP   )r   r   r   r   )ÚNotImplementedErrorr‹   Úforward_with_filtered_kwargsÚfeature_mapsrJ   r&   rk   r‰   rŠ   ÚpermuterL   Ú
contiguousrŽ   r   r   r   )r9   r(   r�   rP   r‘   r’   r   ÚoutputsÚsequence_outputrR   Úpatch_heightÚpatch_widthr   s                r#   ru   z VitPoseForPoseEstimation.forwardÑ   s0  € ðR ˆØÐÝ%Ð&EÑFÔFÐFà"L $¤-Ô"LØð#
ð #
à'ð#
ð ð#
ð #
ˆð "Ô.¨rÔ2ˆØ$Ô*¨1Ô-ˆ
Ø”{Ô2Ô=¸aÔ@ÀDÄKÔD_ÔDjÐklÔDmÑmˆØ”kÔ1Ô<¸QÔ?À4Ä;ÔC^ÔCiÐjkÔClÑlˆØ)×1Ò1°!°Q¸Ñ:Ô:ˆØ)×1Ò1°*¸bÀ,ÐP[Ñ\Ô\×gÒgÑiÔiˆà—9’9˜_¸�9ÑDÔDˆå%ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r"   )NNN)r   r   r   r   rf   r   r   r   rw   r
   r   r   ru   rA   rB   s   @r#   r‡   r‡   ¹   så   ø€ € € € € ð˜}ð ð ð ð ð ð ð$ Øð .2Ø*.Ø&*ð@
ð @
à”lð@
ð ”| dÑ*ð@
ð ”L 4Ñ'ð	@
ð
 ”˜tÑ#ð@
ð Ð+Ô,ð@
ð 
 ð@
ð @
ð @
ñ „^ñ Ôð@
ð @
ð @
ð @
ð @
r"   r‡   )rC   )$r   Údataclassesr   r   r   Ú r   r3   Úbackbone_utilsr   Úmodeling_outputsr   Úmodeling_utilsr	   Úprocessing_utilsr
   Úutilsr   r   r   r   Úutils.genericr   Úconfiguration_vitposer   Ú
get_loggerr   Úloggerr   r%   rZ   ÚModuler\   ry   r‡   Ú__all__r!   r"   r#   ú<module>rª      sY  ðð Ð à !Ð !Ð !Ð !Ð !Ð !à €€€Ø Ð Ð Ð Ð Ð à &Ð &Ð &Ð &Ð &Ð &Ø +Ð +Ð +Ð +Ð +Ð +Ø .Ð .Ð .Ð .Ð .Ð .Ø -Ð -Ð -Ð -Ð -Ð -Ø &Ð &Ð &Ð &Ð &Ð &Ø MÐ MÐ MÐ MÐ MÐ MÐ MÐ MÐ MÐ MÐ MÐ MØ -Ð -Ð -Ð -Ð -Ð -Ø 0Ð 0Ð 0Ð 0Ð 0Ð 0ð 
ˆÔ	˜HÑ	%Ô	%€ð
 €ððñ ô ð
 ð<ð <ð <ð <ð <˜[ñ <ô <ñ „ñô ð<ð$ ð)ð )ð )ð )ð )˜_ñ )ô )ñ „ð)ð"'ð 'ð 'ð 'ðTð ð ð ð ˜2œ9ñ ô ð ð6#ð #ð #ð #ð #˜BœIñ #ô #ð #ðL €ððñ ô ð
U
ð U
ð U
ð U
ð U
Ð5ñ U
ô U
ñô ð
U
ðp $Ð%?Ð
@€€€r"   