§
    ‚Štjã  ã                   ó¨   — d Z ddlmZ ddlmZ ddlmZmZ  ej        e	¦  «        Z
 ed¬¦  «        e G d„ d	e¦  «        ¦   «         ¦   «         Zd	gZd
S )zXLNet configurationé    )Ústricté   )ÚPreTrainedConfig)Úauto_docstringÚloggingzxlnet/xlnet-large-cased)Ú
checkpointc                   óB  ‡ — e Zd ZU dZdZdgZdddddœZd	Zee	d<   d
Z
ee	d<   dZee	d<   dZee	d<   dZee	d<   dZedz  e	d<   dZee	d<   dZee	d<   dZee	d<   dZee	d<   dZeez  e	d<   dZedz  e	d<   dZedz  e	d<   dZee	d<   d Zee	d!<   d Zee	d"<   d#Zee	d$<   d Zee	d%<   d&Zee	d'<   dZee	d(<   d)Z ee	d*<   dZ!eez  e	d+<   d,Z"ee	d-<   d,Z#ee	d.<   d,Z$edz  e	d/<   d0Z%edz  e	d1<   d2Z&ee'e         z  dz  e	d3<   dZ(ee	d4<   ˆ fd5„Z)d6„ Z*e+d7„ ¦   «         Z,e,j-        d8„ ¦   «         Z,ˆ xZ.S )9ÚXLNetConfigaï  
    ff_activation (`str` or `Callable`, *optional*, defaults to `"gelu"`):
        The non-linear activation function (function or string) in the If string, `"gelu"`, `"relu"`, `"silu"` and
        `"gelu_new"` are supported.
    attn_type (`str`, *optional*, defaults to `"bi"`):
        The attention type used by the model. Set `"bi"` for XLNet, `"uni"` for Transformer-XL.
    mem_len (`int` or `None`, *optional*):
        The number of tokens to cache. The key/value pairs that have already been pre-computed in a previous
        forward pass won't be re-computed. See the
        [quickstart](https://huggingface.co/transformers/quickstart.html#using-the-past) for more information.
    reuse_len (`int`, *optional*):
        The number of tokens in the current batch to be cached and reused in the future.
    use_mems_eval (`bool`, *optional*, defaults to `True`):
        Whether or not the model should make use of the recurrent memory mechanism in evaluation mode.
    use_mems_train (`bool`, *optional*, defaults to `False`):
        Whether or not the model should make use of the recurrent memory mechanism in train mode.
        <Tip>
        For pretraining, it is recommended to set `use_mems_train` to `True`. For fine-tuning, it is recommended to
        set `use_mems_train` to `False` as discussed
        [here](https://github.com/zihangdai/xlnet/issues/41#issuecomment-505102587). If `use_mems_train` is set to
        `True`, one has to make sure that the train batches are correctly pre-processed, *e.g.* `batch_1 = [[This
        line is], [This is the]]` and `batch_2 = [[ the first line], [ second line]]` and that all batches are of
        equal size.
        </Tip>
    bi_data (`bool`, *optional*, defaults to `False`):
        Whether or not to use bidirectional input pipeline. Usually set to `True` during pretraining and `False`
        during finetuning.
    clamp_len (`int`, *optional*, defaults to -1):
        Clamp all relative distances larger than clamp_len. Setting this attribute to -1 means no clamping.
    same_length (`bool`, *optional*, defaults to `False`):
        Whether or not to use the same attention length for each token.
    summary_type (`str`, *optional*, defaults to "last"):
        Argument used when doing sequence summary. Used in the sequence classification and multiple choice models.
        Has to be one of the following options:
            - `"last"`: Take the last token hidden state (like XLNet).
            - `"first"`: Take the first token hidden state (like BERT).
            - `"mean"`: Take the mean of all tokens hidden states.
            - `"cls_index"`: Supply a Tensor of classification token position (like GPT/GPT-2).
            - `"attn"`: Not implemented now, use multi-head attention.
    summary_use_proj (`bool`, *optional*, defaults to `True`):
        Argument used when doing sequence summary. Used in the sequence classification and multiple choice models.
        Whether or not to add a projection after the vector extraction.
    summary_activation (`str`, *optional*):
        Argument used when doing sequence summary. Used in the sequence classification and multiple choice models.
        Pass `"tanh"` for a tanh activation to the output, any other value will result in no activation.
    summary_last_dropout (`float`, *optional*, defaults to 0.1):
        Used in the sequence classification and multiple choice models.
        The dropout ratio to be used after the projection and activation.
    start_n_top (`int`, *optional*, defaults to 5):
        Used in the SQuAD evaluation script.
    end_n_top (`int`, *optional*, defaults to 5):
        Used in the SQuAD evaluation script.

    Examples:

    ```python
    >>> from transformers import XLNetConfig, XLNetModel

    >>> # Initializing a XLNet configuration
    >>> configuration = XLNetConfig()

    >>> # Initializing a model (with random weights) from the configuration
    >>> model = XLNetModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```ÚxlnetÚmemsÚ
vocab_sizeÚd_modelÚn_headÚn_layer)Ún_tokenÚhidden_sizeÚnum_attention_headsÚnum_hidden_layersi }  i   é   é   i   Úd_innerNÚd_headÚgeluÚff_activationÚbiÚ	attn_typeg{®Gáz”?Úinitializer_rangegê-�™—q=Úlayer_norm_epsgš™™™™™¹?Údropouti   Úmem_lenÚ	reuse_lenTÚuse_mems_evalFÚuse_mems_trainÚbi_dataéÿÿÿÿÚ	clamp_lenÚsame_lengthÚlastÚsummary_typeÚsummary_use_projÚtanhÚsummary_activationÚsummary_last_dropouté   Ústart_n_topÚ	end_n_topÚpad_token_idé   Úbos_token_idé   Úeos_token_idÚtie_word_embeddingsc                 óp   •— | j         p| j        | j        z  | _          t          ¦   «         j        di |¤Ž d S )N© )r   r   r   ÚsuperÚ__post_init__)ÚselfÚkwargsÚ	__class__s     €úk/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/xlnet/configuration_xlnet.pyr:   zXLNetConfig.__post_init__‡   s=   ø€ Ø”kÐ@ T¤\°T´[Ñ%@ˆŒØ�‰ŒÔÐ'Ð' Ð'Ð'Ð'Ð'Ð'ó    c                 óì   — | j         | j        z  dk    r t          d| j         | j        z  › d�¦  «        ‚| j        | j         | j        z  k    r(t          d| j        › d| j         | j        z  › d�¦  «        ‚dS )zOPart of `@strict`-powered validation. Validates the architecture of the config.r   z'd_model % n_head' (z) should be equal to 0z
`d_head` (z*) should be equal to `d_model // n_head` (ú)N)r   r   Ú
ValueErrorr   ©r;   s    r>   Úvalidate_architecturez!XLNetConfig.validate_architecture‹   sŽ   € àŒ<˜$œ+Ñ%¨Ò*Ð*ÝÐf°D´LÀ4Ä;Ñ4NÐfÐfÐfÑgÔgÐgØŒ;˜$œ,¨$¬+Ñ5Ò5Ð5ÝØr˜Tœ[ÐrÐrÐTXÔT`ÐdhÔdoÑToÐrÐrÐrñô ð ð 6Ð5r?   c                 óL   — t                                d| j        › d�¦  «         dS )Nú
The model ú< is one of the few models that has no sequence length limit.r%   )ÚloggerÚinfoÚ
model_typerC   s    r>   Úmax_position_embeddingsz#XLNetConfig.max_position_embeddings”   s'   € å�ŠÐn ¤ÐnÐnÐnÑoÔoÐoØˆrr?   c                 ó2   — t          d| j        › d�¦  «        ‚)NrF   rG   )ÚNotImplementedErrorrJ   )r;   Úvalues     r>   rK   z#XLNetConfig.max_position_embeddings™   s&   € õ "Øf˜œÐfÐfÐfñ
ô 
ð 	
r?   )/Ú__name__Ú
__module__Ú__qualname__Ú__doc__rJ   Úkeys_to_ignore_at_inferenceÚattribute_mapr   ÚintÚ__annotations__r   r   r   r   r   r   Ústrr   r   Úfloatr   r   r    r!   r"   Úboolr#   r$   r&   r'   r)   r*   r,   r-   r/   r0   r1   r3   r5   Úlistr6   r:   rD   ÚpropertyrK   ÚsetterÚ__classcell__)r=   s   @r>   r
   r
      s§  ø€ € € € € € ðBð BðH €JØ#) (ÐàØ Ø'Ø&ð	ð €Mð €J�ÐÐÑØ€GˆSÐÐÑØ€GˆSÐÐÑØ€FˆCÐÐÑØ€GˆSÐÐÑØ€FˆC�$‰JÐÐÑØ€M�3ÐÐÑØ€IˆsÐÐÑØ#Ð�uÐ#Ð#Ñ#Ø!€N�EÐ!Ð!Ñ!Ø€GˆU�S‰[ÐÐÑØ€GˆS�4‰ZÐÐÑØ €Iˆs�T‰zÐ Ð Ñ Ø€M�4ÐÐÑØ €N�DÐ Ð Ñ Ø€GˆTÐÐÑØ€IˆsÐÐÑØ€K�ÐÐÑØ€L�#ÐÐÑØ!Ð�dÐ!Ð!Ñ!Ø$Ð˜Ð$Ð$Ñ$Ø(+Ð˜% #™+Ð+Ð+Ñ+Ø€K�ÐÐÑØ€IˆsÐÐÑØ €L�#˜‘*Ð Ð Ñ Ø €L�#˜‘*Ð Ð Ñ Ø+,€L�#˜˜Sœ	‘/ DÑ(Ð,Ð,Ñ,Ø $Ð˜Ð$Ð$Ñ$ð(ð (ð (ð (ð (ðð ð ð ðð ñ „Xðð Ô#ð
ð 
ñ $Ô#ð
ð 
ð 
ð 
ð 
r?   r
   N)rR   Úhuggingface_hub.dataclassesr   Úconfiguration_utilsr   Úutilsr   r   Ú
get_loggerrO   rH   r
   Ú__all__r8   r?   r>   ú<module>rc      sÃ   ðð Ð à .Ð .Ð .Ð .Ð .Ð .à 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø ,Ð ,Ð ,Ð ,Ð ,Ð ,Ð ,Ð ,ð 
ˆÔ	˜HÑ	%Ô	%€ð €Ð4Ð5Ñ5Ô5ØðB
ð B
ð B
ð B
ð B
Ð"ñ B
ô B
ñ „ñ 6Ô5ðB
ðJ ˆ/€€€r?   