§
    ‚Štj†4  ã                   óà   — d dl mZ ddlmZ ddlmZ ddlmZ  ed¬¦  «        e G d„ d	e¦  «        ¦   «         ¦   «         Z ed
¬¦  «        e G d„ dee¦  «        ¦   «         ¦   «         Z	dd	gZ
dS )é    )Ústricté   )ÚBackboneConfigMixin)ÚPreTrainedConfig)Úauto_docstringzfacebook/sapiens2-seg-0.4b)Ú
checkpointc                   óÂ  ‡ — e Zd ZU dZdZdZdZee         dz  e	d<   dZ
ee         dz  e	d<   dZee	d<   dZedz  e	d	<   dZee         dz  e	d
<   dZee         dz  e	d<   dZee	d<   dZee         dz  e	d<   dZee         dz  e	d<   dZee	d<   dZedz  e	d<   dZee         dz  e	d<   ˆ fd„Zdeee         z  eeef         z  deee         z  eeef         z  ddfd„Zˆ xZS )ÚSapiens2HeadConfigal	  
    upsample_out_channels (`list[int]`, *optional*):
        Output channel counts for each upsample block.
        The first block takes `hidden_size` channels as input; subsequent blocks use the previous output.
    upsample_kernel_sizes (`list[int]`, *optional*):
        Kernel size for each upsample block. Auto-filled with `[4, ...]` when
        `upsample_out_channels` is set but this is `None`.
        Must have the same length as `upsample_out_channels`.
    upsample_kernel_size (`int`, defaults to 4):
        Default kernel size for upsample blocks when `upsample_kernel_sizes` is not set.
    use_pixel_shuffle (`bool`, *optional*):
        Whether the upsample head uses pixel-shuffle upsampling instead of transposed convolutions.
        When `None` (default), the head uses transposed convolutions.
    conv_out_channels (`list[int]`, *optional*):
        Output channel counts for the refinement conv layers that follow the upsample blocks.
    conv_kernel_sizes (`list[int]`, *optional*):
        Kernel size for each refinement conv layer. Auto-filled with `[1, ...]` when
        `conv_out_channels` is set but this is `None`.
        Must have the same length as `conv_out_channels`.
    conv_kernel_size (`int`, defaults to 1):
        Default kernel size for conv layers when `conv_kernel_sizes` is not set.
    scale_conv_out_channels (`list[int]`, *optional*):
        Output channel counts for the stride-2 conv layers used to predict the focal-length scale.
        When `None` (default), no scale branch is built.
    scale_conv_kernel_sizes (`list[int]`, *optional*):
        Kernel size for each scale conv layer. Auto-filled with `[1, ...]` when
        `scale_conv_out_channels` is set but this is `None`.
        Must have the same length as `scale_conv_out_channels`.
    scale_conv_kernel_size (`int`, defaults to 1):
        Default kernel size for scale conv layers when `scale_conv_kernel_sizes` is not set.
    scale_final_input_size (`int`, *optional*):
        Flattened feature size passed into the scale MLP.
        When `None` (default), it is automatically inferred from `image_size` and `patch_size`
        in the parent [`Sapiens2Config`].
    scale_final_hidden_sizes (`list[int]`, *optional*):
        Hidden-layer sizes for the MLP that maps flattened scale features to the scalar scale output.
        When `None` (default), no scale branch is built.
    Úsapiens2_headÚhead_configNÚupsample_out_channelsÚupsample_kernel_sizesé   Úupsample_kernel_sizeÚuse_pixel_shuffleÚconv_out_channelsÚconv_kernel_sizesé   Úconv_kernel_sizeÚscale_conv_out_channelsÚscale_conv_kernel_sizesÚscale_conv_kernel_sizeÚscale_final_input_sizeÚscale_final_hidden_sizesc                 óZ  •— | j         �)| j        €"| j        gt          | j         ¦  «        z  | _        | j        �)| j        €"| j        gt          | j        ¦  «        z  | _        | j        �)| j        €"| j	        gt          | j        ¦  «        z  | _         t          ¦   «         j        di |¤Ž d S )N© )r   r   r   Úlenr   r   r   r   r   r   ÚsuperÚ__post_init__©ÚselfÚkwargsÚ	__class__s     €úq/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/sapiens2/configuration_sapiens2.pyr   z Sapiens2HeadConfig.__post_init__T   s³   ø€ ØÔ%Ð1°dÔ6PÐ6XØ*.Ô*CÐ)DÅsÈ4ÔKeÑGfÔGfÑ)fˆDÔ&ØÔ!Ð-°$Ô2HÐ2PØ&*Ô&;Ð%<½sÀ4ÔCYÑ?ZÔ?ZÑ%ZˆDÔ"ØÔ'Ð3¸Ô8TÐ8\Ø,0Ô,GÐ+HÍ3ÈtÔOkÑKlÔKlÑ+lˆDÔ(Ø�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'ó    Ú
image_sizeÚ
patch_sizeÚreturnc                 ó¦  — | j         €| j        �| j        €d S t          |t          t
          f¦  «        r|n||f\  }}t          |t          ¦  «        r|n|d         }t          |t          ¦  «        r|n|d         }||z  }||z  }| j        D ],}	|	dz
  dz  }
|d|
z  z   |	z
  dz  dz   }|d|
z  z   |	z
  dz  dz   }Œ-||z  | j        d         z  | _         d S )Nr   r   é   éÿÿÿÿ)r   r   r   Ú
isinstanceÚlistÚtupleÚint)r!   r&   r'   Úimage_heightÚimage_widthÚpatch_heightÚpatch_widthÚfeatures_heightÚfeatures_widthÚkernel_sizeÚpaddings              r$   Ú_init_scale_final_input_sizez/Sapiens2HeadConfig._init_scale_final_input_size]   s  € ð Ô'Ð3ØÔ+Ð3ØÔ+Ð3àˆFÝ2<¸ZÍ$ÕPUÈÑ2WÔ2WÐ$u J JÐ^hÐjtÐ]uÑ!ˆ�kÝ%/°
½CÑ%@Ô%@ÐS�z�zÀjÐQRÄmˆÝ$.¨z½3Ñ$?Ô$?ÐR�j�jÀZÐPQÄ]ˆØ&¨,Ñ6ˆØ$¨Ñ3ˆØÔ7ð 	Sð 	SˆKØ" Q‘¨1Ñ,ˆGØ.°°W±Ñ<¸{ÑJÈqÑPÐSTÑTˆOØ,¨q°7©{Ñ:¸[ÑHÈQÑNÐQRÑRˆNˆNØ&5¸Ñ&FÈÔIeÐfhÔIiÑ&iˆÔ#Ð#Ð#r%   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚbase_config_keyr   r-   r/   Ú__annotations__r   r   r   Úboolr   r   r   r   r   r   r   r   r   r.   r8   Ú__classcell__©r#   s   @r$   r
   r
      sË  ø€ € € € € € ð%ð %ðN !€JØ#€Oà.2Ð˜4 œ9 tÑ+Ð2Ð2Ñ2Ø.2Ð˜4 œ9 tÑ+Ð2Ð2Ñ2Ø !Ð˜#Ð!Ð!Ñ!Ø%)Ð�t˜d‘{Ð)Ð)Ñ)Ø*.Ð�t˜C”y 4Ñ'Ð.Ð.Ñ.Ø*.Ð�t˜C”y 4Ñ'Ð.Ð.Ñ.ØÐ�cÐÐÑØ04Ð˜T #œY¨Ñ-Ð4Ð4Ñ4Ø04Ð˜T #œY¨Ñ-Ð4Ð4Ñ4Ø"#Ð˜CÐ#Ð#Ñ#Ø)-Ð˜C $™JÐ-Ð-Ñ-Ø15Ð˜d 3œi¨$Ñ.Ð5Ð5Ñ5ð(ð (ð (ð (ð (ðjØ  S¤	™/¨E°#°s°(¬OÑ;ðjØILÈtÐTWÌyÉÐ[`ÐadÐfiÐaiÔ[jÑIjðjà	ðjð jð jð jð jð jð jð jr%   r
   zfacebook/sapiens2-pretrain-0.4bc                   ó  ‡ — e Zd ZU dZdZdZeee         z  eeef         z  e	d<   dZ
ee	d<   dZee	d<   d	Zee	d
<   dZee	d<   dZee	d<   dZeez  e	d<   dZee	d<   dZee	d<   dZeee         z  eeef         z  e	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZeez  e	d <   dZee	d!<   d"Zee	d#<   d$Z ed$z  e	d%<   d$Z!ed$z  e	d&<   d'Z"ed$z  e	d(<   d$Z#ee         d$z  e	d)<   d$Z$ee         d$z  e	d*<   dZ%ee	d+<   d,e&iZ'd-Z(ee	d.<   d/Z)ee	d0<   dZ*ee	d1<   dZ+ee	d2<   d$Z,ee         d$z  e	d3<   d"Z-ee	d4<   d"Z.ee	d5<   d"Z/ee	d6<   d7Z0ee	d8<   d$Z1eee                  d$z  e	d9<   d$Z2e&e3z  d$z  e	d,<   ˆ fd:„Z4ˆ xZ5S );ÚSapiens2ConfiguF  
    rope_theta (`float`, *optional*, defaults to 100.0):
        The base period of the RoPE embeddings.
    query_bias (`bool`, *optional*, defaults to `True`):
        Whether to add a bias to the query projection.
    key_bias (`bool`, *optional*, defaults to `False`):
        Whether to add a bias to the key projection.
    value_bias (`bool`, *optional*, defaults to `True`):
        Whether to add a bias to the value projection.
    proj_bias (`bool`, *optional*, defaults to `True`):
        Whether to add a bias to the output projection.
    layerscale_value (`float`, *optional*, defaults to 1.0):
        Initial value to use for layer scale.
    use_gated_mlp (`bool`, *optional*, defaults to `False`):
        Whether to use the SwiGLU feedforward neural network.
    num_register_tokens (`int`, *optional*, defaults to 0):
        The number of register tokens.
    pos_embed_shift (`float`, *optional*):
        Amount to randomly shift position embedding coordinates in [-shift, shift],
        applied only in training mode if not `None`.
    pos_embed_jitter (`float`, *optional*):
        Amount to randomly jitter position embedding coordinates in log-uniform value in [1/jitter, jitter],
        applied only in training mode if not `None`.
    pos_embed_rescale (`float`, *optional*, defaults to 2.0):
        Amount to randomly rescale position embedding coordinates in log-uniform value in [1/rescale, rescale],
        applied only in training mode if not `None`.
    reshape_hidden_states (`bool`, *optional*, defaults to `True`):
        Whether to reshape the hidden states to spatial dimensions when used as backbone.
    use_mask_token (`bool`, *optional*, defaults to `False`):
        Whether to use a mask token in the embeddings (needed for masked image modeling pretraining).
    rms_norm_eps (`float`, *optional*, defaults to 1e-6):
        Epsilon for the RMS normalization layers.
    normalize_backbone_outputs (`bool`, *optional*, defaults to `True`):
        Whether to apply RMSNorm to the backbone `feature_maps` and `cls_tokens` outputs before
        returning them from the forward pass. Only applies when the model is used as a backbone.
    use_qk_norm (`bool`, *optional*, defaults to `True`):
        Whether to apply RMSNorm to queries and keys before RoPE in attention layers.
    num_key_value_heads_per_layer (`list[int]`, *optional*):
        Number of key/value heads for each transformer layer. Setting a layer's value equal to
        `num_attention_heads` gives full multi-head attention; a smaller value gives grouped-query
        attention. Defaults to `num_attention_heads` for the first `num_first_full_attention_layers`
        and last `num_last_full_attention_layers` layers and `num_key_valueattention_heads` for all other
        layers.
    num_key_value_attention_heads (`int`):
        Number of key/value heads for layers that use grouped-query attention when `num_key_value_heads_per_layer`
        is not set. Ignored when `num_key_value_heads_per_layer` is set.
    num_first_full_attention_layers (`int`, *optional*, defaults to 8):
        Number of leading transformer layers that use full multi-head attention.
        Only used when `num_key_value_heads_per_layer` is `None`.
    num_last_full_attention_layers (`int`, *optional*, defaults to 8):
        Number of trailing transformer layers that use full multi-head attention.
        Only used when `num_key_value_heads_per_layer` is `None`.
    semantic_loss_ignore_index (`int`, *optional*, defaults to 255):
        Label index ignored when computing the segmentation loss.
    flip_pairs (`list[list[int]]`, *optional*):
        Pairs of keypoint indices that are mirrored horizontally (e.g., left ear â†” right ear).
        Each pair is a two-element list `[left_index, right_index]`. Used for test-time
        horizontal flip augmentation in pose estimation: pass these pairs to the second
        forward call so the model flips heatmaps back before returning them.
    head_config (`Sapiens2HeadConfig`, *optional*):
        Configuration for the decode head. See [`Sapiens2HeadConfig`] for the available options.
    Úsapiens2é   r'   i   Úhidden_sizei   Úintermediate_sizeé   Únum_hidden_layersÚnum_attention_headsÚsiluÚ
hidden_actg        Úattention_dropoutg{®Gáz”?Úinitializer_rangeg      Y@Ú
rope_thetaéà   r&   r   Únum_channelsTÚ
query_biasÚkey_biasÚ
value_biasÚ	proj_biasÚmlp_biasg      ð?Úlayerscale_valueÚdrop_path_rateÚuse_gated_mlpé   Únum_register_tokensNÚpos_embed_shiftÚpos_embed_jitterg       @Úpos_embed_rescaleÚ_out_featuresÚ_out_indicesÚreshape_hidden_statesr   FÚuse_mask_tokeng�íµ ÷Æ°>Úrms_norm_epsÚnormalize_backbone_outputsÚuse_qk_normÚnum_key_value_heads_per_layerÚnum_key_value_attention_headsÚnum_first_full_attention_layersÚnum_last_full_attention_layerséÿ   Úsemantic_loss_ignore_indexÚ
flip_pairsc                 ó"  •‡ — ‰ j         €%ˆ fd„t          ‰ j        ¦  «        D ¦   «         ‰ _         t          ‰ j        t
          ¦  «        rt          d	i ‰ j        ¤Ž‰ _        ‰ j        �&‰ j                             ‰ j        ‰ j	        ¬¦  «         dgd„ t          d‰ j        dz   ¦  «        D ¦   «         z   ‰ _
        ‰                      |                     dd ¦  «        |                     dd ¦  «        ¬¦  «          t          ¦   «         j        d	i |¤Ž d S )
Nc                 óh   •— g | ].}|‰j         k     s|‰j        ‰j        z
  k    r‰j        n‰j        ‘Œ/S r   )ri   rJ   rj   rK   rh   )Ú.0Úlayer_indexr!   s     €r$   ú
<listcomp>z0Sapiens2Config.__post_init__.<locals>.<listcomp>à   s_   ø€ ð 2ð 2ð 2ð  ð	   $Ô"FÒFÐFØ" dÔ&<¸tÔ?bÑ&bÒbÐbð Ô(Ð(ð
 Ô7ð2ð 2ð 2r%   )r&   r'   Ústemc                 ó   — g | ]}d |› �‘ŒS )Ústager   )rp   Úis     r$   rr   z0Sapiens2Config.__post_init__.<locals>.<listcomp>í   s   € Ð&aÐ&aÐ&a°q {¨q { {Ð&aÐ&aÐ&ar%   r   Úout_indicesÚout_features)rw   rx   r   )rg   ÚrangerJ   r,   r   Údictr
   r8   r&   r'   Ústage_namesÚ"set_output_features_output_indicesÚpopr   r   r    s   ` €r$   r   zSapiens2Config.__post_init__Þ   s7  øø€ ØÔ-Ð5ð2ð 2ð 2ð 2õ $)¨Ô)?Ñ#@Ô#@ð2ñ 2ô 2ˆDÔ.õ �dÔ&­Ñ-Ô-ð 	FÝ1ÐEÐE°DÔ4DÐEÐEˆDÔØÔÐ'ØÔ×9Ò9ÀTÄ_ÐaeÔapÐ9ÑqÔqÐqØ"˜8Ð&aÐ&a½EÀ!ÀTÔE[Ð^_ÑE_Ñ<`Ô<`Ð&aÑ&aÔ&aÑaˆÔØ×/Ò/ØŸ
š
 =°$Ñ7Ô7ÀfÇjÂjÐQ_ÐaeÑFfÔFfð 	0ñ 	
ô 	
ð 	
ð 	�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'r%   )6r9   r:   r;   r<   r=   r'   r/   r-   r.   r?   rG   rH   rJ   rK   rM   ÚstrrN   ÚfloatrO   rP   r&   rR   rS   r@   rT   rU   rV   rW   rX   rY   rZ   r\   r]   r^   r_   r`   ra   rb   r
   Úsub_configsrc   rd   re   rf   rg   rh   ri   rj   rl   rm   r   rz   r   rA   rB   s   @r$   rD   rD   r   s(  ø€ € € € € € ð=ð =ð~ €Jà46€J��d˜3”i‘ %¨¨S¨¤/Ñ1Ð6Ð6Ñ6à€K�ÐÐÑØ!Ð�sÐ!Ð!Ñ!ØÐ�sÐÐÑØ!Ð˜Ð!Ð!Ñ!Ø€J�ÐÐÑØ%(Ð�u˜s‘{Ð(Ð(Ñ(Ø#Ð�uÐ#Ð#Ñ#Ø€J�ÐÐÑØ47€J��d˜3”i‘ %¨¨S¨¤/Ñ1Ð7Ð7Ñ7Ø€L�#ÐÐÑØ€J�ÐÐÑØ€HˆdÐÐÑØ€J�ÐÐÑØ€IˆtÐÐÑØ€HˆdÐÐÑØ!Ð�eÐ!Ð!Ñ!Ø"%€N�E˜C‘KÐ%Ð%Ñ%Ø€M�4ÐÐÑØ Ð˜Ð Ð Ñ Ø$(€O�U˜T‘\Ð(Ð(Ñ(Ø%)Ð�e˜d‘lÐ)Ð)Ñ)Ø&)Ð�u˜t‘|Ð)Ð)Ñ)Ø&*€M�4˜”9˜tÑ#Ð*Ð*Ñ*Ø%)€L�$�s”)˜dÑ"Ð)Ð)Ñ)Ø"&Ð˜4Ð&Ð&Ñ&Ø Ð"4Ð5€KØ €N�DÐ Ð Ñ Ø€L�%ÐÐÑØ'+Ð Ð+Ð+Ñ+Ø€K�ÐÐÑØ6:Ð! 4¨¤9¨tÑ#3Ð:Ð:Ñ:Ø)*Ð! 3Ð*Ð*Ñ*Ø+,Ð# SÐ,Ð,Ñ,Ø*+Ð" CÐ+Ð+Ñ+Ø&)Ð Ð)Ð)Ñ)Ø)-€J��T˜#”Y” $Ñ&Ð-Ð-Ñ-Ø48€KÐ# dÑ*¨TÑ1Ð8Ð8Ñ8ð(ð (ð (ð (ð (ð (ð (ð (ð (r%   rD   N)Úhuggingface_hub.dataclassesr   Úbackbone_utilsr   Úconfiguration_utilsr   Úutilsr   r
   rD   Ú__all__r   r%   r$   ú<module>r†      s  ðð& /Ð .Ð .Ð .Ð .Ð .à 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø #Ð #Ð #Ð #Ð #Ð #ð €Ð7Ð8Ñ8Ô8ØðSjð Sjð Sjð Sjð SjÐ)ñ Sjô Sjñ „ñ 9Ô8ðSjðl €Ð<Ð=Ñ=Ô=Øð}(ð }(ð }(ð }(ð }(Ð(Ð*:ñ }(ô }(ñ „ñ >Ô=ð}(ð@ Ð1Ð
2€€€r%   