§
    ‚Štj~1  ã                   óÖ  — d Z ddlmZ ddlmZ ddlmZ ddlmZm	Z	  ed¬	¦  «        e G d
„ de¦  «        ¦   «         ¦   «         Z
 ed¬	¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z ed¬	¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z ed¬	¦  «        e G d„ de¦  «        ¦   «         ¦   «         Z ed¬	¦  «        e G d„ de¦  «        ¦   «         ¦   «         Zg d¢ZdS )zSAM2 model configurationé    )Ústricté   )ÚPreTrainedConfig)Úauto_docstringé   )ÚCONFIG_MAPPINGÚ
AutoConfigzfacebook/sam2.1-hiera-tiny)Ú
checkpointc                   ó  ‡ — e Zd ZU dZdZdZdZeed<   dZ	eed<   dZ
eed	<   d
Zeee         z  d
z  ed<   d
Zeee         z  d
z  ed<   d
Zeee         z  d
z  ed<   d
Zeee         z  d
z  ed<   d
Zeee         z  d
z  ed<   d
Zee         d
z  ed<   dZeed<   d
Zee         d
z  ed<   d
Zee         d
z  ed<   d
Zee         d
z  ed<   d
Zee         d
z  ed<   d
Zee         d
z  ed<   dZeed<   dZeed<   dZeed<   dZeed<   ˆ fd„Zˆ xZS ) ÚSam2HieraDetConfiga,  
    patch_kernel_size (`list[int]`, *optional*, defaults to `[7, 7]`):
        The kernel size of the patch.
    patch_stride (`list[int]`, *optional*, defaults to `[4, 4]`):
        The stride of the patch.
    patch_padding (`list[int]`, *optional*, defaults to `[3, 3]`):
        The padding of the patch.
    query_stride (`list[int]`, *optional*, defaults to `[2, 2]`):
        The downsample stride between stages.
    window_positional_embedding_background_size (`list[int]`, *optional*, defaults to `[7, 7]`):
        The window size per stage when not using global attention.
    num_query_pool_stages (`int`, *optional*, defaults to 3):
        The number of query pool stages.
    blocks_per_stage (`list[int]`, *optional*, defaults to `[1, 2, 7, 2]`):
        The number of blocks per stage.
    embed_dim_per_stage (`list[int]`, *optional*, defaults to `[96, 192, 384, 768]`):
        The embedding dimension per stage.
    num_attention_heads_per_stage (`list[int]`, *optional*, defaults to `[1, 2, 4, 8]`):
        The number of attention heads per stage.
    window_size_per_stage (`list[int]`, *optional*, defaults to `[8, 4, 14, 7]`):
        The window size per stage.
    global_attention_blocks (`list[int]`, *optional*, defaults to `[5, 7, 9]`):
        The blocks where global attention is used.
    Úbackbone_configÚsam2_hiera_det_modelé`   Úhidden_sizeé   Únum_attention_headsr   Únum_channelsNÚ
image_sizeÚpatch_kernel_sizeÚpatch_strideÚpatch_paddingÚquery_strideÚ+window_positional_embedding_background_sizeÚnum_query_pool_stagesÚblocks_per_stageÚembed_dim_per_stageÚnum_attention_heads_per_stageÚwindow_size_per_stageÚglobal_attention_blocksg      @Ú	mlp_ratioÚgeluÚ
hidden_actç�íµ ÷Æ°>Úlayer_norm_epsç{®Gáz”?Úinitializer_rangec                 ó4  •— | j         �| j         nddg| _         | j        �| j        nddg| _        | j        �| j        nddg| _        | j        �| j        nddg| _        | j        �| j        nddg| _        | j        �| j        nddg| _        | j        �| j        ng d¢| _        | j        �| j        ng d¢| _        | j        �| j        ng d¢| _        | j	        �| j	        ng d	¢| _	        | j
        �| j
        ng d
¢| _
         t          ¦   «         j        di |¤Ž d S )Né   é   é   r   r   )r   r   r)   r   )r   éÀ   é€  é   )r   r   r*   é   )r.   r*   é   r)   )é   r)   é	   © )r   r   r   r   r   r   r   r   r   r   r   ÚsuperÚ__post_init__©ÚselfÚkwargsÚ	__class__s     €úi/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/sam2/configuration_sam2.pyr4   z Sam2HieraDetConfig.__post_init__J   s‘  ø€ Ø-1¬_Ð-H˜$œ/˜/ÈtÐUYÈlˆŒØ;?Ô;QÐ;] Ô!7Ð!7ÐdeÐghÐciˆÔØ15Ô1BÐ1N˜DÔ-Ð-ÐUVÐXYÐTZˆÔØ37Ô3EÐ3Q˜TÔ/Ð/ÐXYÐ[\ÐW]ˆÔØ15Ô1BÐ1N˜DÔ-Ð-ÐUVÐXYÐTZˆÔð Ô?ÐKð Ô<Ð<à�Q�ð 	Ô8ð
 :>Ô9NÐ9Z Ô 5Ð 5Ð`lÐ`lÐ`lˆÔà(,Ô(@Ð(LˆDÔ$Ð$ÐReÐReÐReð 	Ô ð 37Ô2TÐ2`ˆDÔ.Ð.ÐfrÐfrÐfrð 	Ô*ð +/Ô*DÐ*PˆDÔ&Ð&ÐVcÐVcÐVcð 	Ô"ð -1Ô,HÐ,TˆDÔ(Ð(ÐZcÐZcÐZcð 	Ô$ð 	�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'ó    ) Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úbase_config_keyÚ
model_typer   ÚintÚ__annotations__r   r   r   Úlistr   r   r   r   r   r   r   r   r   r   r   r    Úfloatr"   Ústrr$   r&   r4   Ú__classcell__©r8   s   @r9   r   r      s  ø€ € € € € € ðð ð2 (€OØ'€Jà€K�ÐÐÑØ Ð˜Ð Ð Ñ Ø€L�#ÐÐÑØ)-€J��d˜3”i‘ $Ñ&Ð-Ð-Ñ-Ø04Ð�s˜T #œY‘¨Ñ-Ð4Ð4Ñ4Ø+/€L�#˜˜Sœ	‘/ DÑ(Ð/Ð/Ñ/Ø,0€M�3˜˜cœ‘? TÑ)Ð0Ð0Ñ0Ø+/€L�#˜˜Sœ	‘/ DÑ(Ð/Ð/Ñ/ØDHÐ/°°c´¸TÑ1AÐHÐHÑHØ!"Ð˜3Ð"Ð"Ñ"Ø)-Ð�d˜3”i $Ñ&Ð-Ð-Ñ-Ø,0Ð˜˜cœ TÑ)Ð0Ð0Ñ0Ø6:Ð! 4¨¤9¨tÑ#3Ð:Ð:Ñ:Ø.2Ð˜4 œ9 tÑ+Ð2Ð2Ñ2Ø04Ð˜T #œY¨Ñ-Ð4Ð4Ñ4Ø€IˆuÐÐÑØ€J�ÐÐÑØ €N�EÐ Ð Ñ Ø#Ð�uÐ#Ð#Ñ#ð(ð (ð (ð (ð (ð (ð (ð (ð (r:   r   c                   ó  ‡ — e Zd ZU dZdZdZdeiZdZe	e
z  dz  ed<   dZee         dz  ed<   dZedz  ed<   dZeed	<   d
Zeed<   d
Zeed<   dZeed<   dZee         dz  ed<   dZeed<   dZeed<   dZeed<   dZeed<   ˆ fd„Zˆ xZS )ÚSam2VisionConfigaß  
    backbone_channel_list (`List[int]`, *optional*, defaults to `[768, 384, 192, 96]`):
        The list of channel dimensions for the backbone.
    backbone_feature_sizes (`List[List[int]]`, *optional*, defaults to `[[256, 256], [128, 128], [64, 64]]`):
        The spatial sizes of the feature maps from the backbone.
    fpn_hidden_size (`int`, *optional*, defaults to 256):
        The hidden dimension of the FPN.
    fpn_kernel_size (`int`, *optional*, defaults to 1):
        The kernel size for the convolutions in the neck.
    fpn_stride (`int`, *optional*, defaults to 1):
        The stride for the convolutions in the neck.
    fpn_padding (`int`, *optional*, defaults to 0):
        The padding for the convolutions in the neck.
    fpn_top_down_levels (`List[int]`, *optional*, defaults to `[2, 3]`):
        The levels for the top-down FPN connections.
    num_feature_levels (`int`, *optional*, defaults to 3):
        The number of feature levels from the FPN to use.
    Úvision_configÚsam2_vision_modelr   NÚbackbone_channel_listÚbackbone_feature_sizesé   Úfpn_hidden_sizer   Úfpn_kernel_sizeÚ
fpn_strider   Úfpn_paddingÚfpn_top_down_levelsr   Únum_feature_levelsr!   r"   r#   r$   r%   r&   c                 óÐ  •— | j         €g d¢n| j         | _         | j        €ddgddgddggn| j        | _        | j        €ddgn| j        | _        t          | j        t
          ¦  «        rK| j                             dd¦  «        | j        d<   t          | j        d                  d	i | j        ¤Ž| _        n| j        €t          ¦   «         | _         t          ¦   «         j
        d	i |¤Ž d S )
N)r-   r,   r+   r   rN   é€   é@   r   r   r@   r   r2   )rL   rM   rS   Ú
isinstancer   ÚdictÚgetr   r   r3   r4   r5   s     €r9   r4   zSam2VisionConfig.__post_init__Ž   s  ø€ à#'Ô#=Ð#EÐÐÐÐÈ4ÔKeð 	Ô"ð 37Ô2MÐ2Uˆc�3ˆZ˜#˜s˜ b¨" XÐ.Ð.Ð[_Ô[vð 	Ô#ð .2Ô-EÐ-M A q 6 6ÐSWÔSkˆÔ å�dÔ*­DÑ1Ô1ð 	8Ø15Ô1E×1IÒ1IÈ,ÐXnÑ1oÔ1oˆDÔ  Ñ.Ý#1°$Ô2FÀ|Ô2TÔ#UÐ#mÐ#mÐX\ÔXlÐ#mÐ#mˆDÔ Ð ØÔ!Ð)Ý#5Ñ#7Ô#7ˆDÔ à�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'r:   )r;   r<   r=   r>   r?   r@   r	   Úsub_configsr   rY   r   rB   rL   rC   rA   rM   rO   rP   rQ   rR   rS   rT   r"   rE   r$   rD   r&   r4   rF   rG   s   @r9   rI   rI   e   sD  ø€ € € € € € ðð ð& &€OØ$€Jà˜:ð€Kð 7;€O�TÐ,Ñ,¨tÑ3Ð:Ð:Ñ:Ø.2Ð˜4 œ9 tÑ+Ð2Ð2Ñ2Ø*.Ð˜D 4™KÐ.Ð.Ñ.Ø€O�SÐÐÑØ€O�SÐÐÑØ€J�ÐÐÑØ€K�ÐÐÑØ,0Ð˜˜cœ TÑ)Ð0Ð0Ñ0ØÐ˜ÐÐÑØ€J�ÐÐÑØ €N�EÐ Ð Ñ Ø#Ð�uÐ#Ð#Ñ#ð(ð (ð (ð (ð (ð (ð (ð (ð (r:   rI   c                   óØ   — e Zd ZU dZdZdZeed<   dZee	e         z  e
eef         z  ed<   dZee	e         z  e
eef         z  ed<   dZeed	<   d
Zeed<   dZeed<   dZeed<   dZeed<   dS )ÚSam2PromptEncoderConfigaY  
    mask_input_channels (`int`, *optional*, defaults to 16):
        The number of channels to be fed to the `MaskDecoder` module.
    num_point_embeddings (`int`, *optional*, defaults to 4):
        The number of point embeddings to be used.
    scale (`float`, *optional*, defaults to 1):
        The scale factor for the prompt encoder.
    Úprompt_encoder_configrN   r   r(   r   é   Ú
patch_sizeÚmask_input_channelsr*   Únum_point_embeddingsr!   r"   r#   r$   r   ÚscaleN)r;   r<   r=   r>   r?   r   rA   rB   r   rC   Útupler`   ra   rb   r"   rE   r$   rD   rc   r2   r:   r9   r]   r]       sÊ   € € € € € € ðð ð .€Oà€K�ÐÐÑØ48€J��d˜3”i‘ %¨¨S¨¤/Ñ1Ð8Ð8Ñ8Ø46€J��d˜3”i‘ %¨¨S¨¤/Ñ1Ð6Ð6Ñ6Ø!Ð˜Ð!Ð!Ñ!Ø !Ð˜#Ð!Ð!Ñ!Ø€J�ÐÐÑØ €N�EÐ Ð Ñ Ø€Eˆ3€N€N�N€N€Nr:   r]   c                   óÀ   — e Zd ZU dZdZdZeed<   dZe	ed<   dZ
eed<   d	Zeed
<   dZeed<   d	Zeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dS )ÚSam2MaskDecoderConfiga±  
    mlp_dim (`int`, *optional*, defaults to 2048):
        The dimension of the MLP in the two-way transformer.
    attention_downsample_rate (`int`, *optional*, defaults to 2):
        The downsample rate for the attention layers.
    num_multimask_outputs (`int`, *optional*, defaults to 3):
        The number of multimask outputs.
    iou_head_depth (`int`, *optional*, defaults to 3):
        The depth of the IoU head.
    iou_head_hidden_dim (`int`, *optional*, defaults to 256):
        The hidden dimension of the IoU head.
    dynamic_multimask_via_stability (`bool`, *optional*, defaults to `True`):
        Whether to use dynamic multimask via stability.
    dynamic_multimask_stability_delta (`float`, *optional*, defaults to 0.05):
        The stability delta for the dynamic multimask.
    dynamic_multimask_stability_thresh (`float`, *optional*, defaults to 0.98):
        The stability threshold for the dynamic multimask.
    Úmask_decoder_configrN   r   r!   r"   i   Úmlp_dimr   Únum_hidden_layersr.   r   Úattention_downsample_rater   Únum_multimask_outputsÚiou_head_depthÚiou_head_hidden_dimTÚdynamic_multimask_via_stabilitygš™™™™™©?Ú!dynamic_multimask_stability_deltag\�Âõ(\ï?Ú"dynamic_multimask_stability_threshN)r;   r<   r=   r>   r?   r   rA   rB   r"   rE   rh   ri   r   rj   rk   rl   rm   rn   Úboolro   rD   rp   r2   r:   r9   rf   rf   ¸   së   € € € € € € ðð ð& ,€Oà€K�ÐÐÑØ€J�ÐÐÑØ€GˆSÐÐÑØÐ�sÐÐÑØ Ð˜Ð Ð Ñ Ø%&Ð˜sÐ&Ð&Ñ&Ø!"Ð˜3Ð"Ð"Ñ"Ø€N�CÐÐÑØ"Ð˜Ð"Ð"Ñ"Ø,0Ð# TÐ0Ð0Ñ0Ø/3Ð% uÐ3Ð3Ñ3Ø04Ð&¨Ð4Ð4Ñ4Ð4Ð4r:   rf   c                   ó�   ‡ — e Zd ZU dZdZeeedœZdZ	e
ez  dz  ed<   dZe
ez  dz  ed<   dZe
ez  dz  ed<   dZeed	<   ˆ fd
„Zˆ xZS )Ú
Sam2Configag  
    prompt_encoder_config (Union[`dict`, `Sam2PromptEncoderConfig`], *optional*):
        Dictionary of configuration options used to initialize [`Sam2PromptEncoderConfig`].
    mask_decoder_config (Union[`dict`, `Sam2MaskDecoderConfig`], *optional*):
        Dictionary of configuration options used to initialize [`Sam2MaskDecoderConfig`].

    Example:

    ```python
    >>> from transformers import (
    ...     Sam2VisionConfig,
    ...     Sam2PromptEncoderConfig,
    ...     Sam2MaskDecoderConfig,
    ...     Sam2Model,
    ... )

    >>> # Initializing a Sam2Config with `"facebook/sam2.1_hiera_tiny"` style configuration
    >>> configuration = Sam2Config()

    >>> # Initializing a Sam2Model (with random weights) from the `"facebook/sam2.1_hiera_tiny"` style configuration
    >>> model = Sam2Model(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config

    >>> # We can also initialize a Sam2Config from a Sam2VisionConfig, Sam2PromptEncoderConfig, and Sam2MaskDecoderConfig

    >>> # Initializing SAM2 vision encoder, memory attention, and memory encoder configurations
    >>> vision_config = Sam2VisionConfig()
    >>> prompt_encoder_config = Sam2PromptEncoderConfig()
    >>> mask_decoder_config = Sam2MaskDecoderConfig()

    >>> config = Sam2Config(vision_config, prompt_encoder_config, mask_decoder_config)
    ```Úsam2)rJ   r^   rg   NrJ   r^   rg   r%   r&   c                 óp  •— t          | j        t          ¦  «        rK| j                             dd¦  «        | j        d<   t	          | j        d                  di | j        ¤Ž| _        n | j        €t	          d         ¦   «         | _        t          | j        t          ¦  «        rt          di | j        ¤Ž| _        n| j        €t          ¦   «         | _        t          | j        t          ¦  «        rt          di | j        ¤Ž| _        n| j        €t          ¦   «         | _         t          ¦   «         j
        di |¤Ž d S )Nr@   rK   r2   )rX   rJ   rY   rZ   r   r^   r]   rg   rf   r3   r4   r5   s     €r9   r4   zSam2Config.__post_init__  s5  ø€ Ý�dÔ(­$Ñ/Ô/ð 	GØ/3Ô/A×/EÒ/EÀlÐTgÑ/hÔ/hˆDÔ˜|Ñ,Ý!/°Ô0BÀ<Ô0PÔ!QÐ!gÐ!gÐTXÔTfÐ!gÐ!gˆDÔÐØÔÐ'Ý!/Ð0CÔ!DÑ!FÔ!FˆDÔå�dÔ0µ$Ñ7Ô7ð 	CÝ)@Ð)^Ð)^À4ÔC]Ð)^Ð)^ˆDÔ&Ð&ØÔ'Ð/Ý)@Ñ)BÔ)BˆDÔ&å�dÔ.µÑ5Ô5ð 	?Ý'<Ð'XÐ'X¸tÔ?WÐ'XÐ'XˆDÔ$Ð$ØÔ%Ð-Ý'<Ñ'>Ô'>ˆDÔ$à�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'r:   )r;   r<   r=   r>   r@   r	   r]   rf   r[   rJ   rY   r   rB   r^   rg   r&   rD   r4   rF   rG   s   @r9   rs   rs   Þ   sÇ   ø€ € € € € € ð!ð !ðF €Jà#Ø!8Ø4ðð €Kð 59€M�4Ð*Ñ*¨TÑ1Ð8Ð8Ñ8Ø<@Ð˜4Ð"2Ñ2°TÑ9Ð@Ð@Ñ@Ø:>Ð˜Ð 0Ñ0°4Ñ7Ð>Ð>Ñ>Ø#Ð�uÐ#Ð#Ñ#ð(ð (ð (ð (ð (ð (ð (ð (ð (r:   rs   )rs   r   rI   r]   rf   N)r>   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   Úautor   r	   r   rI   r]   rf   rs   Ú__all__r2   r:   r9   ú<module>r{      s  ðð Ð à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø #Ð #Ð #Ð #Ð #Ð #Ø -Ð -Ð -Ð -Ð -Ð -Ð -Ð -ð €Ð7Ð8Ñ8Ô8ØðI(ð I(ð I(ð I(ð I(Ð)ñ I(ô I(ñ „ñ 9Ô8ðI(ðX €Ð7Ð8Ñ8Ô8Øð6(ð 6(ð 6(ð 6(ð 6(Ð'ñ 6(ô 6(ñ „ñ 9Ô8ð6(ðr €Ð7Ð8Ñ8Ô8Øðð ð ð ð Ð.ñ ô ñ „ñ 9Ô8ðð, €Ð7Ð8Ñ8Ô8Øð!5ð !5ð !5ð !5ð !5Ð,ñ !5ô !5ñ „ñ 9Ô8ð!5ðH €Ð7Ð8Ñ8Ô8ØðA(ð A(ð A(ð A(ð A(Ð!ñ A(ô A(ñ „ñ 9Ô8ðA(ðHð ð €€€r:   