§
    ‚Štj£   ã                   óJ  — d Z ddlmZ ddlmZ ddlmZmZ  ej        e	¦  «        Z
 ed¬¦  «        e G d„ d	e¦  «        ¦   «         ¦   «         Z ed¬¦  «        e G d
„ de¦  «        ¦   «         ¦   «         Z ed¬¦  «        e G d„ de¦  «        ¦   «         ¦   «         Zg d¢ZdS )zPix2Struct model configurationé    )Ústricté   )ÚPreTrainedConfig)Úauto_docstringÚloggingzgoogle/pix2struct-base)Ú
checkpointc                   ót  — e Zd ZU dZdZdgZddddddddœZdZee	d	<   d
Z
ee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZeez  e	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZed z  e	d!<   d"Zeee         z  d z  e	d#<   d Zed z  e	d$<   dZee	d%<   d&Zee	d'<   dZ ee	d(<   d S ))ÚPix2StructTextConfiga®  
    relative_attention_num_buckets (`int`, *optional*, defaults to 32):
        The number of buckets to use for each attention layer.
    relative_attention_max_distance (`int`, *optional*, defaults to 128):
        The maximum distance of the longer sequences for the bucket separation.
    dense_act_fn (`Union[Callable, str]`, *optional*, defaults to `"gelu_new"`):
        The non-linear activation function (function or string).

    Example:

    ```python
    >>> from transformers import Pix2StructTextConfig, Pix2StructTextModel

    >>> # Initializing a Pix2StructTextConfig with google/pix2struct-base style configuration
    >>> configuration = Pix2StructTextConfig()

    >>> # Initializing a Pix2StructTextModel (with random weights) from the google/pix2struct-base style configuration
    >>> model = Pix2StructTextModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úpix2struct_text_modelÚpast_key_valuesÚhidden_sizeÚ	num_headsÚ
num_layers)r   Únum_attention_headsÚnum_hidden_layersÚdecoder_attention_headsÚencoder_attention_headsÚencoder_layersÚdecoder_layersiDÄ  Ú
vocab_sizeé   é@   Úd_kvé   Úd_ffé   é    Úrelative_attention_num_bucketsé€   Úrelative_attention_max_distancegš™™™™™¹?Údropout_rateç�íµ ÷Æ°>Úlayer_norm_epsilonç      ð?Úinitializer_factorÚgelu_newÚdense_act_fnr   Údecoder_start_token_idFÚ	use_cacheNÚpad_token_idé   Úeos_token_idÚbos_token_idÚtie_word_embeddingsTÚ
is_decoderÚadd_cross_attention)!Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚattribute_mapr   ÚintÚ__annotations__r   r   r   r   r   r   r    r!   Úfloatr#   r%   r'   Ústrr(   r)   Úboolr*   r,   Úlistr-   r.   r/   r0   © ó    úu/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/pix2struct/configuration_pix2struct.pyr
   r
      s¬  € € € € € € ðð ð. )€JØ#4Ð"5Ðà$Ø*Ø)Ø#.Ø#.Ø&Ø&ðð €Mð €J�ÐÐÑØ€K�ÐÐÑØ€Dˆ#€N€N�NØ€Dˆ#ÐÐÑØ€J�ÐÐÑØ€IˆsÐÐÑØ*,Ð" CÐ,Ð,Ñ,Ø+.Ð# SÐ.Ð.Ñ.Ø #€L�%˜#‘+Ð#Ð#Ñ#Ø $Ð˜Ð$Ð$Ñ$Ø #Ð˜Ð#Ð#Ñ#Ø"€L�#Ð"Ð"Ñ"Ø"#Ð˜CÐ#Ð#Ñ#Ø€IˆtÐÐÑØ €L�#˜‘*Ð Ð Ñ Ø+,€L�#˜˜Sœ	‘/ DÑ(Ð,Ð,Ñ,Ø#€L�#˜‘*Ð#Ð#Ñ#Ø %Ð˜Ð%Ð%Ñ%Ø€J�ÐÐÑØ %Ð˜Ð%Ð%Ñ%Ð%Ð%r?   r
   c                   óö   — e Zd ZU dZdZdZeed<   dZeed<   dZ	eed<   dZ
eed	<   d
Zeed<   d
Zeed<   dZeed<   dZeed<   dZeez  ed<   dZeez  ed<   dZeed<   dZeed<   dZeed<   dZeed<   dZeed<   dS )ÚPix2StructVisionConfiga’  
    patch_embed_hidden_size (`int`, *optional*, defaults to 768):
        Dimensionality of the input patch_embedding layer in the Transformer encoder.
    d_ff (`int`, *optional*, defaults to 2048):
        Dimensionality of the "intermediate" (i.e., feed-forward) layer in the Transformer encoder.
    d_kv (`int`, *optional*, defaults to 64):
        Dimensionality of the key, query, value projections per attention head.
    The non-linear activation function (function or string) in the encoder and pooler. If string, `"gelu"`,
        `"relu"`, `"selu"` and `"gelu_new"` `"gelu"` are supported.
    dense_act_fn (`Union[Callable, str]`, *optional*, defaults to `"gelu_new"`):
        The non-linear activation function (function or string).
    seq_len (`int`, *optional*, defaults to 4096):
        Maximum sequence length (here number of patches) supported by the model.
    relative_attention_num_buckets (`int`, *optional*, defaults to 32):
        The number of buckets to use for each attention layer.
    relative_attention_max_distance (`int`, *optional*, defaults to 128):
        The maximum distance (in tokens) to use for each attention layer.

    Example:

    ```python
    >>> from transformers import Pix2StructVisionConfig, Pix2StructVisionModel

    >>> # Initializing a Pix2StructVisionConfig with google/pix2struct-base style configuration
    >>> configuration = Pix2StructVisionConfig()

    >>> # Initializing a Pix2StructVisionModel (with random weights) from the google/pix2struct-base style configuration
    >>> model = Pix2StructVisionModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úpix2struct_vision_modelr   r   Úpatch_embed_hidden_sizer   r   r   r   r   r   r   r&   r'   r"   Úlayer_norm_epsg        r!   Úattention_dropoutg»½×Ùß|Û=Úinitializer_ranger$   r%   i   Úseq_lenr   r   r   r    N)r1   r2   r3   r4   r5   r   r8   r9   rD   r   r   r   r   r'   r;   rE   r:   r!   rF   rG   r%   rH   r   r    r>   r?   r@   rB   rB   U   s!  € € € € € € ðð ðB +€Jà€K�ÐÐÑØ#&Ð˜SÐ&Ð&Ñ&Ø€Dˆ#ÐÐÑØ€Dˆ#€N€N�NØÐ�sÐÐÑØ!Ð˜Ð!Ð!Ñ!Ø"€L�#Ð"Ð"Ñ"Ø €N�EÐ Ð Ñ Ø #€L�%˜#‘+Ð#Ð#Ñ#Ø%(Ð�u˜s‘{Ð(Ð(Ñ(Ø$Ð�uÐ$Ð$Ñ$Ø #Ð˜Ð#Ð#Ñ#Ø€GˆSÐÐÑØ*,Ð" CÐ,Ð,Ñ,Ø+.Ð# SÐ.Ð.Ñ.Ð.Ð.r?   rB   c                   ó¬   ‡ — e Zd ZU dZdZeedœZdZe	e
z  dz  ed<   dZe	e
z  dz  ed<   dZeed<   d	Zeed
<   dZeed<   dZeed<   dZeed<   ˆ fd„Zˆ xZS )ÚPix2StructConfiga  
    is_vqa (`bool`, *optional*, defaults to `False`):
        Whether the model has been fine-tuned for VQA or not.

    Example:

    ```python
    >>> from transformers import Pix2StructConfig, Pix2StructForConditionalGeneration

    >>> # Initializing a Pix2StructConfig with google/pix2struct-base style configuration
    >>> configuration = Pix2StructConfig()

    >>> # Initializing a Pix2StructForConditionalGeneration (with random weights) from the google/pix2struct-base style configuration
    >>> model = Pix2StructForConditionalGeneration(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config

    >>> # We can also initialize a Pix2StructConfig from a Pix2StructTextConfig and a Pix2StructVisionConfig

    >>> # Initializing a Pix2Struct text and Pix2Struct vision configuration
    >>> config_text = Pix2StructTextConfig()
    >>> config_vision = Pix2StructVisionConfig()

    >>> config = Pix2StructConfig(text_config=config_text, vision_config=config_vision)
    ```Ú
pix2struct)Útext_configÚvision_configNrL   rM   r$   r%   g{®Gáz”?rG   FÚis_vqar.   TÚis_encoder_decoderc                 óÎ  •— | j         €;t          | j        | j        ¬¦  «        | _         t                               d¦  «         nNt          | j         t          ¦  «        r4| j        | j         d<   | j        | j         d<   t          di | j         ¤Ž| _         | j        €.t          ¦   «         | _        t                               d¦  «         n0t          | j        t          ¦  «        rt          di | j        ¤Ž| _        | j         j
        | _
        | j         j        | _        | j         j        | _        | j        | j         _        | j        | j        _         t          ¦   «         j        di |¤Ž d S )N)rO   r.   zU`text_config` is `None`. initializing the `Pix2StructTextConfig` with default values.rO   r.   zY`vision_config` is `None`. initializing the `Pix2StructVisionConfig` with default values.r>   )rL   r
   rO   r.   ÚloggerÚinfoÚ
isinstanceÚdictrM   rB   r(   r*   r,   rG   ÚsuperÚ__post_init__)ÚselfÚkwargsÚ	__class__s     €r@   rV   zPix2StructConfig.__post_init__µ   s]  ø€ ØÔÐ#Ý3Ø#'Ô#:Ø$(Ô$<ð ñ  ô  ˆDÔõ �KŠKÐoÑpÔpÐpÐpÝ˜Ô(­$Ñ/Ô/ð 	HØ59Ô5LˆDÔÐ1Ñ2Ø6:Ô6NˆDÔÐ2Ñ3Ý3ÐGÐG°dÔ6FÐGÐGˆDÔàÔÐ%Ý!7Ñ!9Ô!9ˆDÔÝ�KŠKÐsÑtÔtÐtÐtÝ˜Ô*­DÑ1Ô1ð 	NÝ!7Ð!MÐ!M¸$Ô:LÐ!MÐ!MˆDÔà&*Ô&6Ô&MˆÔ#Ø Ô,Ô9ˆÔØ Ô,Ô9ˆÔà-1Ô-CˆÔÔ*Ø/3Ô/EˆÔÔ,à�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'r?   )r1   r2   r3   r4   r5   r
   rB   Úsub_configsrL   rT   r   r9   rM   r%   r:   rG   rN   r<   r.   rO   rV   Ú__classcell__)rY   s   @r@   rJ   rJ   Œ   sã   ø€ € € € € € ðð ð6 €JØ"6ÐI_Ð`Ð`€Kà26€K�Ð(Ñ(¨4Ñ/Ð6Ð6Ñ6Ø48€M�4Ð*Ñ*¨TÑ1Ð8Ð8Ñ8Ø #Ð˜Ð#Ð#Ñ#Ø#Ð�uÐ#Ð#Ñ#Ø€FˆDÐÐÑØ %Ð˜Ð%Ð%Ñ%Ø#Ð˜Ð#Ð#Ñ#ð(ð (ð (ð (ð (ð (ð (ð (ð (r?   rJ   )rJ   r
   rB   N)r4   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   r   Ú
get_loggerr1   rQ   r
   rB   rJ   Ú__all__r>   r?   r@   ú<module>ra      sj  ðð %Ð $à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø ,Ð ,Ð ,Ð ,Ð ,Ð ,Ð ,Ð ,ð 
ˆÔ	˜HÑ	%Ô	%€ð €Ð3Ð4Ñ4Ô4Øð7&ð 7&ð 7&ð 7&ð 7&Ð+ñ 7&ô 7&ñ „ñ 5Ô4ð7&ðt €Ð3Ð4Ñ4Ô4Øð2/ð 2/ð 2/ð 2/ð 2/Ð-ñ 2/ô 2/ñ „ñ 5Ô4ð2/ðj €Ð3Ð4Ñ4Ô4Øð@(ð @(ð @(ð @(ð @(Ð'ñ @(ô @(ñ „ñ 5Ô4ð@(ðF QÐ
PÐ
P€€€r?   