§
    ‚Štj- ã                   ó²  — d Z ddlZddlmZ ddlmZ ddlZddlZddlm	Z	 ddl
mZmZmZ ddlmZ dd	lmZmZ dd
lmZmZ ddlmZ ddlmZmZmZmZmZmZ ddl m!Z! ddl"m#Z# ddl$m%Z%m&Z&m'Z' ddl(m)Z)  e'j*        e+¦  «        Z,d„ Z-d@d„Z. G d„ de	j/        ¦  «        Z0 G d„ de	j/        ¦  «        Z1 e&d¬¦  «         G d„ de	j/        ¦  «        ¦   «         Z2 e&d¬¦  «        e G d„ de%¦  «        ¦   «         ¦   «         Z3 G d„ d e	j/        ¦  «        Z4 G d!„ d"e	j/        ¦  «        Z5 G d#„ d$e	j/        ¦  «        Z6 G d%„ d&e	j/        ¦  «        Z7 G d'„ d(e	j/        ¦  «        Z8e& G d)„ d*e!¦  «        ¦   «         Z9e& G d+„ d,e9¦  «        ¦   «         Z: e&d-¬¦  «         G d.„ d/e9e¦  «        ¦   «         Z; e&d0¬¦  «         G d1„ d2e9¦  «        ¦   «         Z<e& G d3„ d4e9¦  «        ¦   «         Z= e&d5¬¦  «         G d6„ d7e9¦  «        ¦   «         Z> e&d8¬¦  «        e G d9„ d:e%¦  «        ¦   «         ¦   «         Z?e& G d;„ d<e9¦  «        ¦   «         Z@e& G d=„ d>e9¦  «        ¦   «         ZAg d?¢ZBdS )Az%PyTorch Flaubert model, based on XLM.é    N)ÚCallable)Ú	dataclass)Únn)ÚBCEWithLogitsLossÚCrossEntropyLossÚMSELossé   )Úinitialization)ÚgeluÚget_activation)ÚDynamicCacheÚEncoderDecoderCache)ÚGenerationMixin)ÚBaseModelOutputÚMaskedLMOutputÚMultipleChoiceModelOutputÚQuestionAnsweringModelOutputÚSequenceClassifierOutputÚTokenClassifierOutput)ÚPreTrainedModel)Úapply_chunking_to_forward)ÚModelOutputÚauto_docstringÚloggingé   )ÚFlaubertConfigc           	      óŒ  ‡— t          j        ˆfd„t          | ¦  «        D ¦   «         ¦  «        }d|_        t	          j        t          j        |d d …dd d…f         ¦  «        ¦  «        |d d …dd d…f<   t	          j        t          j        |d d …dd d…f         ¦  «        ¦  «        |d d …dd d…f<   |                     ¦   «          |S )Nc                 óJ   •‡— g | ]Šˆˆfd „t          ‰¦  «        D ¦   «         ‘ŒS )c           	      óR   •— g | ]#}‰t          j        d d|dz  z  ‰z  ¦  «        z  ‘Œ$S )i'  é   )ÚnpÚpower)Ú.0ÚjÚdimÚposs     €€úl/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/flaubert/modeling_flaubert.pyú
<listcomp>z;create_sinusoidal_embeddings.<locals>.<listcomp>.<listcomp>0   s7   ø€ Ð\Ð\Ð\ÈA˜c¥B¤H¨U°A¸¸a¹±LÀ3Ñ4FÑ$GÔ$GÑGÐ\Ð\Ð\ó    )Úrange)r#   r&   r%   s    @€r'   r(   z0create_sinusoidal_embeddings.<locals>.<listcomp>0   s=   øø€ ÐuÐuÐuÐadÐ\Ð\Ð\Ð\Ð\ÕQVÐWZÑQ[ÔQ[Ð\Ñ\Ô\ÐuÐuÐur)   Fr   r    r   )	r!   Úarrayr*   Úrequires_gradÚtorchÚFloatTensorÚsinÚcosÚdetach_)Ún_posr%   ÚoutÚposition_encs    `  r'   Úcreate_sinusoidal_embeddingsr5   /   sÉ   ø€ Ý”8ÐuÐuÐuÐuÕhmÐnsÑhtÔhtÐuÑuÔuÑvÔv€LØ€CÔÝÔ$¥R¤V¨L¸¸¸¸A¸D¸q¸D¸Ô,AÑ%BÔ%BÑCÔC€Cˆˆˆˆ1ˆ4ˆaˆ4ˆ�LÝÔ$¥R¤V¨L¸¸¸¸A¸D¸q¸D¸Ô,AÑ%BÔ%BÑCÔC€Cˆˆˆˆ1ˆ4ˆaˆ4ˆ�LØ‡K‚K�M„M€MØ€Jr)   c                 óè  — t          j        | t           j        |j        ¬¦  «        }|�|}n<|                     ¦   «                              ¦   «         | k    sJ ‚||dd…df         k     }|                     d¦  «        }|r2|dddd…f                              || d¦  «        |ddd…df         k    }n|}|                     ¦   «         || fk    sJ ‚|du s|                     ¦   «         || | fk    sJ ‚||fS )zH
    Generate hidden states mask, and optionally an attention mask.
    ©ÚdtypeÚdeviceNr   r   F)r-   ÚarangeÚlongr9   ÚmaxÚitemÚsizeÚrepeat)ÚslenÚlengthsÚcausalÚpadding_maskÚalenÚmaskÚbsÚ	attn_masks           r'   Ú	get_masksrH   9   s  € õ Œ<˜¥E¤J°w´~ÐFÑFÔF€DØÐØˆˆà�{Š{‰}Œ}×!Ò!Ñ#Ô# tÒ+Ð+Ð+Ð+Ø�g˜a˜a˜a ˜gÔ&Ò&ˆð 
�Š�a‰Œ€BØð Ø˜˜t Q Q Q˜Ô'×.Ò.¨r°4¸Ñ;Ô;¸tÀDÈ!È!È!ÈTÀMÔ?RÒRˆ	ˆ	àˆ	ð �9Š9‰;Œ;˜2˜t˜*Ò$Ð$Ð$Ð$Ø�Uˆ?ˆ?˜iŸnšnÑ.Ô.°2°t¸TÐ2BÒBÐBÐBÐBà�ˆ?Ðr)   c                   ó4   ‡ — e Zd Zddefˆ fd„Z	 	 	 dd„Zˆ xZS )	ÚMultiHeadAttentionr   Ú	layer_idxc                 ó˜  •— t          ¦   «                              ¦   «          || _        || _        || _        ||z  | _        |j        | _        | j        | j        z  dk    sJ ‚t          j	        ||¦  «        | _
        t          j	        ||¦  «        | _        t          j	        ||¦  «        | _        t          j	        ||¦  «        | _        d S )Nr   )ÚsuperÚ__init__Úlayer_idr%   Ún_headsÚhead_dimÚattention_dropoutÚdropoutr   ÚLinearÚq_linÚk_linÚv_linÚout_lin)ÚselfrP   r%   ÚconfigrK   Ú	__class__s        €r'   rN   zMultiHeadAttention.__init__T   s­   ø€ Ý‰Œ×ÒÑÔÐØ!ˆŒØˆŒØˆŒØ˜w™ˆŒØÔ/ˆŒØŒx˜$œ,Ñ&¨!Ò+Ð+Ð+Ð+å”Y˜s CÑ(Ô(ˆŒ
Ý”Y˜s CÑ(Ô(ˆŒ
Ý”Y˜s CÑ(Ô(ˆŒ
Ý”y  cÑ*Ô*ˆŒˆˆr)   NFc                 óÂ  — |                      ¦   «         \  }}}	|du}
|                     ¦   «         dk    r|d|dfn|dddf}|                      |¦  «                             |d| j        | j        ¦  «                             dd¦  «        }|�Ht          |t          ¦  «        r1|j	         
                    | j        ¦  «        }|
r|j        }n
|j        }n|}|
r|n|}|
r)|�'|r%|j        | j                 }|j        | j                 }nÈ|                      |¦  «        }|                      |¦  «        }|                     |d| j        | j        ¦  «                             dd¦  «        }|                     |d| j        | j        ¦  «                             dd¦  «        }|�0|                     ||| j        ¦  «        \  }}|
rd|j	        | j        <   |t'          j        | j        ¦  «        z  }t+          j        ||                     dd¦  «        ¦  «        }|dk                         |¦  «                             |¦  «        }|                     |t+          j        |j        ¦  «        j        ¦  «         t8          j                             |                     ¦   «         d¬¦  «                              |¦  «        }t8          j         !                    || j!        | j"        ¬	¦  «        }t+          j        ||¦  «        }|                     dd¦  «         #                    ¦   «                              |d| j        | j        z  ¦  «        }|  $                    |¦  «        f}|r||fz   }|S )
zd
        Self-attention (if kv is None) or attention over source sentence (provided by kv).
        Nr	   r   éÿÿÿÿr    Tr   ©r%   ©ÚpÚtraining)%r>   r%   rU   ÚviewrP   rQ   Ú	transposeÚ
isinstancer   Ú
is_updatedÚgetrO   Úcross_attention_cacheÚself_attention_cacheÚ	key_cacheÚvalue_cacherV   rW   ÚupdateÚmathÚsqrtr-   ÚmatmulÚ	expand_asÚmasked_fill_Úfinfor8   Úminr   Ú
functionalÚsoftmaxÚfloatÚtype_asrS   ra   Ú
contiguousrX   )rY   ÚinputrE   ÚkvÚcacheÚoutput_attentionsÚkwargsrF   Úqlenr%   Úis_cross_attentionÚmask_reshapeÚqre   Úcurr_past_key_valuesÚcurrent_statesÚkÚvÚscoresÚweightsÚcontextÚoutputss                         r'   ÚforwardzMultiHeadAttention.forwardb   s  € ð Ÿ
š
™œ‰ˆˆD�#Ø t˜^ÐØ,0¯HªH©J¬J¸!ªO¨O˜˜A˜t RÐ(Ð(À"ÀaÈÈBÀˆà�JŠJ�uÑÔ×"Ò" 2 r¨4¬<¸¼ÑGÔG×QÒQÐRSÐUVÑWÔWˆØÐÝ˜%Õ!4Ñ5Ô5ð -Ø"Ô-×1Ò1°$´-Ñ@Ô@�
Ø%ð Fà+0Ô+FÐ(Ð(à+0Ô+EÐ(Ð(à',Ð$à1Ð<˜˜°uˆØð 	; %Ð"3¸
Ð"3à$Ô.¨t¬}Ô=ˆAØ$Ô0°´Ô?ˆAˆAà—
’
˜>Ñ*Ô*ˆAØ—
’
˜>Ñ*Ô*ˆAØ—’�r˜2˜tœ|¨T¬]Ñ;Ô;×EÒEÀaÈÑKÔKˆAØ—’�r˜2˜tœ|¨T¬]Ñ;Ô;×EÒEÀaÈÑKÔKˆAàÐ à+×2Ò2°1°a¸¼ÑGÔG‘��1à%ð ;Ø6:�EÔ$ T¤]Ñ3à•”	˜$œ-Ñ(Ô(Ñ(ˆÝ”˜a §¢¨Q°Ñ!2Ô!2Ñ3Ô3ˆØ˜’	×Ò Ñ-Ô-×7Ò7¸Ñ?Ô?ˆØ×Ò˜D¥%¤+¨f¬lÑ";Ô";Ô"?Ñ@Ô@Ð@å”-×'Ò'¨¯ª©¬¸BÐ'Ñ?Ô?×GÒGÈÑOÔOˆÝ”-×'Ò'¨°4´<È$Ì-Ð'ÑXÔXˆå”,˜w¨Ñ*Ô*ˆØ×#Ò# A qÑ)Ô)×4Ò4Ñ6Ô6×;Ò;¸BÀÀDÄLÐSWÔS`ÑD`ÑaÔaˆà—<’< Ñ(Ô(Ð*ˆØð 	+Ø  
Ñ*ˆGØˆr)   )r   )NNF)Ú__name__Ú
__module__Ú__qualname__ÚintrN   r‰   Ú__classcell__©r[   s   @r'   rJ   rJ   S   sh   ø€ € € € € ð+ð +¸ð +ð +ð +ð +ð +ð +ð$ ØØð>ð >ð >ð >ð >ð >ð >ð >r)   rJ   c                   ó*   ‡ — e Zd Zˆ fd„Zd„ Zd„ Zˆ xZS )ÚTransformerFFNc                 ó6  •— t          ¦   «                              ¦   «          |j        | _        t          j        ||¦  «        | _        t          j        ||¦  «        | _        |j        rt          nt          j	        j
        | _        |j        | _        d| _        d S ©Nr   )rM   rN   rS   r   rT   Úlin1Úlin2Úgelu_activationr   rs   ÚreluÚactÚchunk_size_feed_forwardÚseq_len_dim)rY   Úin_dimÚ
dim_hiddenÚout_dimrZ   r[   s        €r'   rN   zTransformerFFN.__init__¥   sy   ø€ Ý‰Œ×ÒÑÔÐØ”~ˆŒÝ”I˜f jÑ1Ô1ˆŒ	Ý”I˜j¨'Ñ2Ô2ˆŒ	Ø!Ô1ÐI•4�4µr´}Ô7IˆŒØ'-Ô'EˆÔ$ØˆÔÐÐr)   c                 óD   — t          | j        | j        | j        |¦  «        S ©N)r   Úff_chunkr™   rš   )rY   rx   s     r'   r‰   zTransformerFFN.forward®   s    € Ý(¨¬¸Ô8TÐVZÔVfÐhmÑnÔnÐnr)   c                 óÜ   — |                       |¦  «        }|                      |¦  «        }|                      |¦  «        }t          j                             || j        | j        ¬¦  «        }|S )Nr_   )r”   r˜   r•   r   rs   rS   ra   )rY   rx   Úxs      r'   r    zTransformerFFN.ff_chunk±   sV   € Ø�IŠI�eÑÔˆØ�HŠH�Q‰KŒKˆØ�IŠI�a‰LŒLˆÝŒM×!Ò! ! t¤|¸d¼mÐ!ÑLÔLˆØˆr)   )rŠ   r‹   rŒ   rN   r‰   r    rŽ   r�   s   @r'   r‘   r‘   ¤   sY   ø€ € € € € ðð ð ð ð ðoð oð oðð ð ð ð ð ð r)   r‘   zl
    The bare Flaubert Model transformer outputting raw hidden-states without any specific head on top.
    )Úcustom_introc                   ó*   ‡ — e Zd ZdZˆ fd„Zdd„Zˆ xZS )ÚFlaubertPredLayerz?
    Prediction layer (cross_entropy or adaptive_softmax).
    c                 óP  •— t          ¦   «                              ¦   «          |j        | _        |j        | _        |j        | _        |j        }|j        du r#t          j        ||j        d¬¦  «        | _        d S t          j	        ||j        |j
        |j        d¬¦  «        | _        d S )NFT©Úbias)Úin_featuresÚ	n_classesÚcutoffsÚ	div_valueÚ	head_bias)rM   rN   ÚasmÚn_wordsÚ	pad_indexÚemb_dimr   rT   ÚprojÚAdaptiveLogSoftmaxWithLossÚasm_cutoffsÚasm_div_value)rY   rZ   r%   r[   s      €r'   rN   zFlaubertPredLayer.__init__Ä   s›   ø€ Ý‰Œ×ÒÑÔÐØ”:ˆŒØ”~ˆŒØÔ)ˆŒØŒnˆàŒ:˜ÐÐÝœ	 # v¤~¸DÐAÑAÔAˆDŒIˆIˆIåÔ5ØØ œ.ØÔ*Ø Ô.Øðñ ô ˆDŒIˆIˆIr)   Nc                 ó‚  — d}| j         du rr|                      |¦  «        }|f|z   }|�Tt          j                             |                     d| j        ¦  «        |                     d¦  «        d¬¦  «        }|f|z   }nA| j                             |¦  «        }|f|z   }|�|                      ||¦  «        \  }}|f|z   }|S )z,Compute the loss, and optionally the scores.© FNr]   Úmean)Ú	reduction)r®   r²   r   rs   Úcross_entropyrb   r¯   Úlog_prob)rY   r¢   Úyrˆ   r…   ÚlossÚ_s          r'   r‰   zFlaubertPredLayer.forwardÖ   sÊ   € àˆØŒ8�uÐÐØ—Y’Y˜q‘\”\ˆFØ�i 'Ñ)ˆGØˆ}Ý”}×2Ò2°6·;²;¸rÀ4Ä<Ñ3PÔ3PÐRS×RXÒRXÐY[ÑR\ÔR\ÐhnÐ2ÑoÔo�Ø˜' GÑ+�øà”Y×'Ò'¨Ñ*Ô*ˆFØ�i 'Ñ)ˆGØˆ}ØŸ)š) A q™/œ/‘��4Ø˜' GÑ+�àˆr)   rŸ   )rŠ   r‹   rŒ   Ú__doc__rN   r‰   rŽ   r�   s   @r'   r¥   r¥   ¹   sV   ø€ € € € € ðð ðð ð ð ð ð$ð ð ð ð ð ð ð r)   r¥   zl
    Base class for outputs of question answering models using a [`~modeling_utils.FlaubertSQuADHead`].
    c                   óÈ   — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dZ	ej
        dz  ed<   dZej        dz  ed<   dZej
        dz  ed<   dZej        dz  ed<   dS )	ÚFlaubertSquadHeadOutputá9  
    loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned if both `start_positions` and `end_positions` are provided):
        Classification loss as the sum of start token, end token (and is_impossible if provided) classification
        losses.
    start_top_log_probs (`torch.FloatTensor` of shape `(batch_size, config.start_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
        Log probabilities for the top config.start_n_top start token possibilities (beam-search).
    start_top_index (`torch.LongTensor` of shape `(batch_size, config.start_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
        Indices for the top config.start_n_top start token possibilities (beam-search).
    end_top_log_probs (`torch.FloatTensor` of shape `(batch_size, config.start_n_top * config.end_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
        Log probabilities for the top `config.start_n_top * config.end_n_top` end token possibilities
        (beam-search).
    end_top_index (`torch.LongTensor` of shape `(batch_size, config.start_n_top * config.end_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
        Indices for the top `config.start_n_top * config.end_n_top` end token possibilities (beam-search).
    cls_logits (`torch.FloatTensor` of shape `(batch_size,)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
        Log probabilities for the `is_impossible` label of the answers.
    Nr½   Ústart_top_log_probsÚstart_top_indexÚend_top_log_probsÚend_top_indexÚ
cls_logits)rŠ   r‹   rŒ   r¿   r½   r-   r.   Ú__annotations__rÃ   rÄ   Ú
LongTensorrÅ   rÆ   rÇ   r·   r)   r'   rÁ   rÁ   é   s°   € € € € € € ðð ð" &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø48Ð˜Ô*¨TÑ1Ð8Ð8Ñ8Ø/3€O�UÔ%¨Ñ,Ð3Ð3Ñ3Ø26Ð�uÔ(¨4Ñ/Ð6Ð6Ñ6Ø-1€M�5Ô# dÑ*Ð1Ð1Ñ1Ø+/€J�Ô! DÑ(Ð/Ð/Ñ/Ð/Ð/r)   rÁ   c                   ób   ‡ — e Zd ZdZdefˆ fd„Zd	dej        dej        dz  dej        fd„Zˆ xZ	S )
ÚFlaubertPoolerStartLogitszÐ
    Compute SQuAD start logits from sequence hidden states.

    Args:
        config ([`FlaubertConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model.
    rZ   c                 ó†   •— t          ¦   «                              ¦   «          t          j        |j        d¦  «        | _        d S r“   )rM   rN   r   rT   Úhidden_sizeÚdense©rY   rZ   r[   s     €r'   rN   z"FlaubertPoolerStartLogits.__init__  s3   ø€ Ý‰Œ×ÒÑÔÐÝ”Y˜vÔ1°1Ñ5Ô5ˆŒ
ˆ
ˆ
r)   NÚhidden_statesÚp_maskÚreturnc                 ó¾   — |                       |¦  «                             d¦  «        }|�2|j        t          j        k    r|d|z
  z  d|z  z
  }n|d|z
  z  d|z  z
  }|S )aì  
        Args:
            hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
                The final hidden states of the model.
            p_mask (`torch.FloatTensor` of shape `(batch_size, seq_len)`, *optional*):
                Mask for tokens at invalid position, such as query and special symbols (PAD, SEP, CLS). 1.0 means token
                should be masked.

        Returns:
            `torch.FloatTensor`: The start logits for SQuAD.
        r]   Nr   éÜÿ  çêŒ 9Y>)F)rÎ   Úsqueezer8   r-   Úfloat16)rY   rÐ   rÑ   r¢   s       r'   r‰   z!FlaubertPoolerStartLogits.forward  sm   € ð �JŠJ�}Ñ%Ô%×-Ò-¨bÑ1Ô1ˆàÐØŒ|�uœ}Ò,Ð,Ø˜˜V™Ñ$ u¨v¡~Ñ5��à˜˜V™Ñ$ t¨f¡}Ñ4�àˆr)   rŸ   )
rŠ   r‹   rŒ   r¿   r   rN   r-   r.   r‰   rŽ   r�   s   @r'   rË   rË     sŒ   ø€ € € € € ðð ð6˜~ð 6ð 6ð 6ð 6ð 6ð 6ðð  UÔ%6ð ÀÔ@QÐTXÑ@Xð ÐdiÔduð ð ð ð ð ð ð ð r)   rË   c                   ó�   ‡ — e Zd ZdZdefˆ fd„Z	 	 	 ddej        dej        dz  dej        dz  dej        dz  d	ej        f
d
„Z	ˆ xZ
S )ÚFlaubertPoolerEndLogitszú
    Compute SQuAD end logits from sequence hidden states.

    Args:
        config ([`FlaubertConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model and the `layer_norm_eps`
            to use.
    rZ   c                 óN  •— t          ¦   «                              ¦   «          t          j        |j        dz  |j        ¦  «        | _        t          j        ¦   «         | _        t          j        |j        |j	        ¬¦  «        | _        t          j        |j        d¦  «        | _
        d S )Nr    ©Úepsr   )rM   rN   r   rT   rÍ   Údense_0ÚTanhÚ
activationÚ	LayerNormÚlayer_norm_epsÚdense_1rÏ   s     €r'   rN   z FlaubertPoolerEndLogits.__init__:  sz   ø€ Ý‰Œ×ÒÑÔÐÝ”y Ô!3°aÑ!7¸Ô9KÑLÔLˆŒÝœ'™)œ)ˆŒÝœ fÔ&8¸fÔ>SÐTÑTÔTˆŒÝ”y Ô!3°QÑ7Ô7ˆŒˆˆr)   NrÐ   Ústart_statesÚstart_positionsrÑ   rÒ   c                 óJ  — |€|€
J d¦   «         ‚|�a|j         dd…         \  }}|dd…ddf                              dd|¦  «        }|                     d|¦  «        }|                     d|d¦  «        }|                      t	          j        ||gd¬¦  «        ¦  «        }|                      |¦  «        }|                      |¦  «        }|                      |¦  «         	                    d¦  «        }|�2|j
        t          j        k    r|d|z
  z  d|z  z
  }n|d|z
  z  d|z  z
  }|S )	aë  
        Args:
            hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
                The final hidden states of the model.
            start_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`, *optional*):
                The hidden states of the first tokens for the labeled span.
            start_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                The position of the first token for the labeled span.
            p_mask (`torch.FloatTensor` of shape `(batch_size, seq_len)`, *optional*):
                Mask for tokens at invalid position, such as query and special symbols (PAD, SEP, CLS). 1.0 means token
                should be masked.

        <Tip>

        One of `start_states` or `start_positions` should be not `None`. If both are set, `start_positions` overrides
        `start_states`.

        </Tip>

        Returns:
            `torch.FloatTensor`: The end logits for SQuAD.
        Nú7One of start_states, start_positions should be not Noneéþÿÿÿr]   r^   r   rÔ   rÕ   )ÚshapeÚexpandÚgatherrÝ   r-   Úcatrß   rà   râ   rÖ   r8   r×   )rY   rÐ   rã   rä   rÑ   r@   Úhszr¢   s           r'   r‰   zFlaubertPoolerEndLogits.forwardA  s>  € ð: Ð'¨?Ð+FÐ+FØEñ ,GÔ+FÐFð Ð&Ø%Ô+¨B¨C¨CÔ0‰IˆD�#Ø-¨a¨a¨a°°t¨mÔ<×CÒCÀBÈÈCÑPÔPˆOØ(×/Ò/°°OÑDÔDˆLØ'×.Ò.¨r°4¸Ñ<Ô<ˆLà�LŠL�œ M°<Ð#@ÀbÐIÑIÔIÑJÔJˆØ�OŠO˜AÑÔˆØ�NŠN˜1ÑÔˆØ�LŠL˜‰OŒO×#Ò# BÑ'Ô'ˆàÐØŒ|�uœ}Ò,Ð,Ø˜˜V™Ñ$ u¨v¡~Ñ5��à˜˜V™Ñ$ t¨f¡}Ñ4�àˆr)   ©NNN©rŠ   r‹   rŒ   r¿   r   rN   r-   r.   rÉ   r‰   rŽ   r�   s   @r'   rÙ   rÙ   0  sÀ   ø€ € € € € ðð ð8˜~ð 8ð 8ð 8ð 8ð 8ð 8ð 26Ø37Ø+/ð1ð 1àÔ(ð1ð Ô'¨$Ñ.ð1ð Ô)¨DÑ0ð	1ð
 Ô! DÑ(ð1ð 
Ô	ð1ð 1ð 1ð 1ð 1ð 1ð 1ð 1r)   rÙ   c                   ó�   ‡ — e Zd ZdZdefˆ fd„Z	 	 	 ddej        dej        dz  dej        dz  dej        dz  d	ej        f
d
„Z	ˆ xZ
S )ÚFlaubertPoolerAnswerClasszë
    Compute SQuAD 2.0 answer class from classification and start tokens hidden states.

    Args:
        config ([`FlaubertConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model.
    rZ   c                 ó  •— t          ¦   «                              ¦   «          t          j        |j        dz  |j        ¦  «        | _        t          j        ¦   «         | _        t          j        |j        dd¬¦  «        | _        d S )Nr    r   Fr§   )	rM   rN   r   rT   rÍ   rÝ   rÞ   rß   râ   rÏ   s     €r'   rN   z"FlaubertPoolerAnswerClass.__init__  sc   ø€ Ý‰Œ×ÒÑÔÐÝ”y Ô!3°aÑ!7¸Ô9KÑLÔLˆŒÝœ'™)œ)ˆŒÝ”y Ô!3°Q¸UÐCÑCÔCˆŒˆˆr)   NrÐ   rã   rä   Ú	cls_indexrÒ   c                 ó`  — |j         d         }|€|€
J d¦   «         ‚|�K|dd…ddf                              dd|¦  «        }|                     d|¦  «                             d¦  «        }|�L|dd…ddf                              dd|¦  «        }|                     d|¦  «                             d¦  «        }n|dd…ddd…f         }|                      t          j        ||gd¬¦  «        ¦  «        }|                      |¦  «        }|                      |¦  «                             d¦  «        }|S )a¸  
        Args:
            hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
                The final hidden states of the model.
            start_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`, *optional*):
                The hidden states of the first tokens for the labeled span.
            start_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                The position of the first token for the labeled span.
            cls_index (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                Position of the CLS token for each sentence in the batch. If `None`, takes the last token.

        <Tip>

        One of `start_states` or `start_positions` should be not `None`. If both are set, `start_positions` overrides
        `start_states`.

        </Tip>

        Returns:
            `torch.FloatTensor`: The SQuAD 2.0 answer class.
        r]   Nræ   rç   r^   )	rè   ré   rê   rÖ   rÝ   r-   rë   rß   râ   )rY   rÐ   rã   rä   rò   rì   Úcls_token_stater¢   s           r'   r‰   z!FlaubertPoolerAnswerClass.forward…  s@  € ð: Ô! "Ô%ˆØÐ'¨?Ð+FÐ+FØEñ ,GÔ+FÐFð Ð&Ø-¨a¨a¨a°°t¨mÔ<×CÒCÀBÈÈCÑPÔPˆOØ(×/Ò/°°OÑDÔD×LÒLÈRÑPÔPˆLàÐ Ø! ! ! ! T¨4 -Ô0×7Ò7¸¸BÀÑDÔDˆIØ+×2Ò2°2°yÑAÔA×IÒIÈ"ÑMÔMˆOˆOà+¨A¨A¨A¨r°1°1°1¨HÔ5ˆOà�LŠL�œ L°/Ð#BÈÐKÑKÔKÑLÔLˆØ�OŠO˜AÑÔˆØ�LŠL˜‰OŒO×#Ò# BÑ'Ô'ˆàˆr)   rí   rî   r�   s   @r'   rð   rð   v  sÇ   ø€ € € € € ðð ðD˜~ð Dð Dð Dð Dð Dð Dð 26Ø37Ø-1ð/ð /àÔ(ð/ð Ô'¨$Ñ.ð/ð Ô)¨DÑ0ð	/ð
 Ô# dÑ*ð/ð 
Ô	ð/ð /ð /ð /ð /ð /ð /ð /r)   rð   c                   óä   ‡ — e Zd ZdZdefˆ fd„Ze	 	 	 	 	 	 ddej        dej	        dz  dej	        dz  d	ej	        dz  d
ej	        dz  dej        dz  de
deeej                 z  fd„¦   «         Zˆ xZS )ÚFlaubertSQuADHeadzä
    A SQuAD head inspired by XLNet.

    Args:
        config ([`FlaubertConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model and the `layer_norm_eps`
            to use.
    rZ   c                 óð   •— t          ¦   «                              ¦   «          |j        | _        |j        | _        t	          |¦  «        | _        t          |¦  «        | _        t          |¦  «        | _	        d S rŸ   )
rM   rN   Ústart_n_topÚ	end_n_toprË   Ústart_logitsrÙ   Ú
end_logitsrð   Úanswer_classrÏ   s     €r'   rN   zFlaubertSQuADHead.__init__Â  sc   ø€ Ý‰Œ×ÒÑÔÐØ!Ô-ˆÔØÔ)ˆŒå5°fÑ=Ô=ˆÔÝ1°&Ñ9Ô9ˆŒÝ5°fÑ=Ô=ˆÔÐÐr)   NFrÐ   rä   Úend_positionsrò   Úis_impossiblerÑ   Úreturn_dictrÒ   c                 ó¾  — |                       ||¬¦  «        }|�Ø|�Ö||||fD ]1}	|	�-|	                     ¦   «         dk    r|	                     d¦  «         Œ2|                      |||¬¦  «        }
t	          ¦   «         } |||¦  «        } ||
|¦  «        }||z   dz  }|�A|�?|                      |||¬¦  «        }t          j        ¦   «         } |||¦  «        }||dz  z  }|rt          |¬	¦  «        n|fS | 	                    ¦   «         \  }}}t          j
                             |d¬
¦  «        }t          j        || j        d¬
¦  «        \  }}|                     d¦  «                             dd|¦  «        }t          j        |d|¦  «        }|                     d¦  «                             d|dd¦  «        }|                     d¦  «                             |¦  «        }|�|                     d¦  «        nd}|                      |||¬¦  «        }
t          j
                             |
d¬
¦  «        }t          j        || j        d¬
¦  «        \  }}|                     d| j        | j        z  ¦  «        }|                     d| j        | j        z  ¦  «        }t          j        d||¦  «        }|                      |||¬¦  «        }|s|||||fS t          |||||¬¦  «        S )a  
        hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
            Final hidden states of the model on the sequence tokens.
        start_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Positions of the first token for the labeled span.
        end_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Positions of the last token for the labeled span.
        cls_index (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Position of the CLS token for each sentence in the batch. If `None`, takes the last token.
        is_impossible (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Whether the question has a possible answer in the paragraph or not.
        p_mask (`torch.FloatTensor` of shape `(batch_size, seq_len)`, *optional*):
            Mask for tokens at invalid position, such as query and special symbols (PAD, SEP, CLS). 1.0 means token
            should be masked.
        )rÑ   Nr   r]   )rä   rÑ   r    )rä   rò   g      à?)r½   r^   rç   )rã   rÑ   z
blh,bl->bh)rã   rò   )rÃ   rÄ   rÅ   rÆ   rÇ   )rú   r%   Úsqueeze_rû   r   rü   r   r   rÁ   r>   rs   rt   r-   Útopkrø   Ú	unsqueezeré   rê   ro   rù   rb   Úeinsum)rY   rÐ   rä   rý   rò   rþ   rÑ   rÿ   rú   r¢   rû   Úloss_fctÚ
start_lossÚend_lossÚ
total_lossrÇ   Úloss_fct_clsÚcls_lossÚbszr@   rì   Ústart_log_probsrÃ   rÄ   Ústart_top_index_exprã   Úhidden_states_expandedÚend_log_probsrÅ   rÆ   s                                 r'   r‰   zFlaubertSQuADHead.forwardË  s:  € ð4 ×(Ò(¨¸vÐ(ÑFÔFˆàÐ&¨=Ð+Dà% }°iÀÐOð #ð #�Ø�= Q§U¢U¡W¤W¨q¢[ [Ø—J’J˜r‘N”N�Nøð Ÿš¨ÈÐ`f˜ÑgÔgˆJå'Ñ)Ô)ˆHØ!˜ ,°Ñ@Ô@ˆJØ�x 
¨MÑ:Ô:ˆHØ$ xÑ/°1Ñ4ˆJàÐ$¨Ð)Bà!×.Ò.¨}ÈoÐirÐ.ÑsÔs�
Ý!Ô3Ñ5Ô5�Ø'˜<¨
°MÑBÔB�ð ˜h¨™nÑ,�
à?JÐ]Õ*°
Ð;Ñ;Ô;Ð;ÐQ[ÐP]Ð]ð +×/Ò/Ñ1Ô1‰NˆC��sÝ œm×3Ò3°LÀbÐ3ÑIÔIˆOå38´:Ø Ô!1°rð4ñ 4ô 4Ñ0Ð ð #2×";Ò";¸BÑ"?Ô"?×"FÒ"FÀrÈ2ÈsÑ"SÔ"SÐÝ œ<¨°rÐ;NÑOÔOˆLØ'×1Ò1°!Ñ4Ô4×;Ò;¸BÀÀbÈ"ÑMÔMˆLà%2×%<Ò%<¸QÑ%?Ô%?×%IÒ%IØñ&ô &Ð"ð .4Ð-?�V×%Ò% bÑ)Ô)Ð)ÀTˆFØŸšÐ)?ÈlÐci˜ÑjÔjˆJÝœM×1Ò1°*À!Ð1ÑDÔDˆMå/4¬zØ˜tœ~°1ð0ñ 0ô 0Ñ,Ð˜}ð !2× 6Ò 6°r¸4Ô;KÈdÌnÑ;\Ñ ]Ô ]ÐØ)×.Ò.¨r°4Ô3CÀdÄnÑ3TÑUÔUˆMå œ<¨°mÀ_ÑUÔUˆLØ×*Ò*¨=À|Ð_hÐ*ÑiÔiˆJàð 	Ø+¨_Ð>OÐQ^Ð`jÐkÐkå.Ø(;Ø$3Ø&7Ø"/Ø)ðñ ô ð r)   )NNNNNF)rŠ   r‹   rŒ   r¿   r   rN   r   r-   r.   rÉ   ÚboolrÁ   Útupler‰   rŽ   r�   s   @r'   rö   rö   ¸  s  ø€ € € € € ðð ð>˜~ð >ð >ð >ð >ð >ð >ð ð 48Ø15Ø-1Ø15Ø+/Ø!ðYð YàÔ(ðYð Ô)¨DÑ0ðYð Ô'¨$Ñ.ð	Yð
 Ô# dÑ*ðYð Ô'¨$Ñ.ðYð Ô! DÑ(ðYð ðYð 
! 5¨Ô):Ô#;Ñ	;ðYð Yð Yñ „^ðYð Yð Yð Yð Yr)   rö   c                   ód   ‡ — e Zd ZdZdefˆ fd„Z	 d	dej        dej        dz  dej        fd„Z	ˆ xZ
S )
ÚFlaubertSequenceSummaryaÐ  
    Compute a single vector summary of a sequence hidden states.

    Args:
        config ([`FlaubertConfig`]):
            The config used by the model. Relevant arguments in the config class of the model are (refer to the actual
            config class of your model for the default values it uses):

            - **summary_type** (`str`) -- The method to use to make this summary. Accepted values are:

                - `"last"` -- Take the last token hidden state (like XLNet)
                - `"first"` -- Take the first token hidden state (like Bert)
                - `"mean"` -- Take the mean of all tokens hidden states
                - `"cls_index"` -- Supply a Tensor of classification token position (GPT/GPT-2)
                - `"attn"` -- Not implemented now, use multi-head attention

            - **summary_use_proj** (`bool`) -- Add a projection after the vector extraction.
            - **summary_proj_to_labels** (`bool`) -- If `True`, the projection outputs to `config.num_labels` classes
              (otherwise to `config.hidden_size`).
            - **summary_activation** (`Optional[str]`) -- Set to `"tanh"` to add a tanh activation to the output,
              another string or `None` will add no activation.
            - **summary_first_dropout** (`float`) -- Optional dropout probability before the projection and activation.
            - **summary_last_dropout** (`float`)-- Optional dropout probability after the projection and activation.
    rZ   c                 óV  •— t          ¦   «                              ¦   «          t          |dd¦  «        | _        | j        dk    rt          ‚t          j        ¦   «         | _        t          |d¦  «        rW|j	        rPt          |d¦  «        r|j
        r|j        dk    r|j        }n|j        }t          j        |j        |¦  «        | _        t          |dd ¦  «        }|rt          |¦  «        nt          j        ¦   «         | _        t          j        ¦   «         | _        t          |d¦  «        r)|j        dk    rt          j        |j        ¦  «        | _        t          j        ¦   «         | _        t          |d	¦  «        r+|j        dk    r"t          j        |j        ¦  «        | _        d S d S d S )
NÚsummary_typeÚlastÚattnÚsummary_use_projÚsummary_proj_to_labelsr   Úsummary_activationÚsummary_first_dropoutÚsummary_last_dropout)rM   rN   Úgetattrr  ÚNotImplementedErrorr   ÚIdentityÚsummaryÚhasattrr  r  Ú
num_labelsrÍ   rT   r   rß   Úfirst_dropoutr  ÚDropoutÚlast_dropoutr  )rY   rZ   Únum_classesÚactivation_stringr[   s       €r'   rN   z FlaubertSequenceSummary.__init__C  sœ  ø€ Ý‰Œ×ÒÑÔÐå# F¨N¸FÑCÔCˆÔØÔ Ò&Ð&õ &Ð%å”{‘}”}ˆŒÝ�6Ð-Ñ.Ô.ð 	F°6Ô3Jð 	FÝ�vÐ7Ñ8Ô8ð 1¸VÔ=Zð 1Ð_eÔ_pÐstÒ_tÐ_tØ$Ô/��à$Ô0�Ýœ9 VÔ%7¸ÑEÔEˆDŒLå# FÐ,@À$ÑGÔGÐØIZÐ$m¥NÐ3DÑ$EÔ$EÐ$EÕ`bÔ`kÑ`mÔ`mˆŒåœ[™]œ]ˆÔÝ�6Ð2Ñ3Ô3ð 	J¸Ô8TÐWXÒ8XÐ8XÝ!#¤¨FÔ,HÑ!IÔ!IˆDÔåœK™MœMˆÔÝ�6Ð1Ñ2Ô2ð 	H°vÔ7RÐUVÒ7VÐ7VÝ "¤
¨6Ô+FÑ GÔ GˆDÔÐÐð	Hð 	HÐ7VÐ7Vr)   NrÐ   rò   rÒ   c                 ó:  — | j         dk    r|dd…df         }�n-| j         dk    r|dd…df         }�n| j         dk    r|                     d¬¦  «        }nò| j         d	k    rÕ|€=t          j        |d
dd…dd…f         |j        d         dz
  t          j        ¬¦  «        }nl|                     d¦  «                             d¦  «        }|                     d|                     ¦   «         dz
  z  | 	                    d¦  «        fz   ¦  «        }| 
                    d|¦  «                             d¦  «        }n| j         dk    rt          ‚|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }|S )ak  
        Compute a single vector summary of a sequence hidden states.

        Args:
            hidden_states (`torch.FloatTensor` of shape `[batch_size, seq_len, hidden_size]`):
                The hidden states of the last layer.
            cls_index (`torch.LongTensor` of shape `[batch_size]` or `[batch_size, ...]` where ... are optional leading dimensions of `hidden_states`, *optional*):
                Used if `summary_type == "cls_index"` and takes the last token of the sequence as classification token.

        Returns:
            `torch.FloatTensor`: The summary of the sequence hidden states.
        r  Nr]   Úfirstr   r¸   r   r^   rò   .rç   )r8   )r]   r  )r  r¸   r-   Ú	full_likerè   r;   r  ré   r%   r>   rê   rÖ   r  r#  r   rß   r%  )rY   rÐ   rò   Úoutputs       r'   r‰   zFlaubertSequenceSummary.forward`  sª  € ð Ô Ò&Ð&Ø" 1 1 1 b 5Ô)ˆF‰FØÔ 'Ò)Ð)Ø" 1 1 1 a 4Ô(ˆF‰FØÔ &Ò(Ð(Ø"×'Ò'¨AÐ'Ñ.Ô.ˆFˆFØÔ +Ò-Ð-ØÐ Ý!œOØ! # r¨ r¨1¨1¨1 *Ô-Ø!Ô'¨Ô+¨aÑ/Ýœ*ðñ ô �	�	ð &×/Ò/°Ñ3Ô3×=Ò=¸bÑAÔA�	Ø%×,Ò,¨U°i·m²m±o´oÈÑ6IÑ-JÈm×N`ÒN`ÐacÑNdÔNdÐMfÑ-fÑgÔg�	à"×)Ò)¨"¨iÑ8Ô8×@Ò@ÀÑDÔDˆFˆFØÔ &Ò(Ð(Ý%Ð%à×#Ò# FÑ+Ô+ˆØ—’˜fÑ%Ô%ˆØ—’ Ñ(Ô(ˆØ×"Ò" 6Ñ*Ô*ˆàˆr)   rŸ   rî   r�   s   @r'   r  r  )  s›   ø€ € € € € ðð ð2H˜~ð Hð Hð Hð Hð Hð Hð< VZð)ð )Ø"Ô.ð)Ø;@Ô;KÈdÑ;Rð)à	Ô	ð)ð )ð )ð )ð )ð )ð )ð )r)   r  c                   ón   ‡ — e Zd ZU eed<   dZed„ ¦   «         Z ej	        ¦   «         ˆ fd„¦   «         Z
ˆ xZS )ÚFlaubertPreTrainedModelrZ   Útransformerc                 óú   — t          j        g d¢g d¢g d¢g¦  «        }t          j        g d¢g d¢g d¢g¦  «        }| j        j        r.| j        j        dk    rt          j        g d¢g d¢g d¢g¦  «        }nd }|||dœS )	N)é   é   r   r   r   )r   r    r	   r   r   )r   r   r   é   é   )r   r   r   r   r   )r   r   r   r   r   )r   r   r   r   r   r   )Ú	input_idsÚattention_maskÚlangs)r-   ÚtensorrZ   Úuse_lang_embÚn_langs)rY   Úinputs_listÚ
attns_listÚ
langs_lists       r'   Údummy_inputsz$FlaubertPreTrainedModel.dummy_inputs’  sœ   € å”l O O O°_°_°_ÀoÀoÀoÐ#VÑWÔWˆÝ”\ ? ? ?°O°O°OÀ_À_À_Ð"UÑVÔVˆ
ØŒ;Ô#ð 	¨¬Ô(;¸aÒ(?Ð(?Ýœ   ¸¸¸ÈÈÈÐ&YÑZÔZˆJˆJàˆJØ(¸JÐQ[Ð\Ð\Ð\r)   c           
      ó  •— t          ¦   «                              |¦  «         t          |t          j        ¦  «        rz| j        �2| j        j        �&t          j        |j	        d| j        j        ¬¦  «         |j
        �:t          |j	        dd¦  «        s$t          j        |j	        |j
                 ¦  «         t          |t          ¦  «        r¼| j        j        r_t          j        |j        j	        t#          | j        j        | j        j        t)          j        |j        j	        ¦  «        ¬¦  «        ¦  «         t          j        |j        t)          j        |j        j        d         ¦  «                             d¦  «        ¦  «         dS dS )	zInitialize the weights.Nr   )r¸   ÚstdÚ_is_hf_initializedF)r3   r]   ©r   r]   )rM   Ú_init_weightsrd   r   Ú	EmbeddingrZ   Úembed_init_stdÚinitÚnormal_ÚweightÚpadding_idxr  Úzeros_ÚFlaubertModelÚsinusoidal_embeddingsÚcopy_Úposition_embeddingsr5   Úmax_position_embeddingsr±   r-   Ú
empty_likeÚposition_idsr:   rè   ré   )rY   Úmoduler[   s     €r'   rB  z%FlaubertPreTrainedModel._init_weightsœ  sZ  ø€ õ 	‰Œ×Ò˜fÑ%Ô%Ð%Ý�f�bœlÑ+Ô+ð 	?ØŒ{Ð&¨4¬;Ô+EÐ+QÝ”˜Vœ]°¸¼Ô8RÐSÑSÔSÐSàÔ!Ð-µg¸f¼mÐMaÐchÑ6iÔ6iÐ-Ý”˜FœM¨&Ô*<Ô=Ñ>Ô>Ð>Ý�f�mÑ,Ô,ð 
	iØŒ{Ô0ð Ý”
ØÔ.Ô5Ý0ØœÔ;ØœÔ+Ý!Ô,¨VÔ-GÔ-NÑOÔOðñ ô ñô ð õ ŒJ�vÔ*­E¬L¸Ô9LÔ9RÐSUÔ9VÑ,WÔ,W×,^Ò,^Ð_fÑ,gÔ,gÑhÔhÐhÐhÐhð
	ið 
	ir)   )rŠ   r‹   rŒ   r   rÈ   Úbase_model_prefixÚpropertyr=  r-   Úno_gradrB  rŽ   r�   s   @r'   r-  r-  Œ  s‡   ø€ € € € € € ð ÐÐÑØ%Ðàð]ð ]ñ „Xð]ð €U„]�_„_ðið ið ið iñ „_ðið ið ið ið ir)   r-  c                   ó2  ‡ — e Zd Zˆ fd„Zd„ Zd„ Ze	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej	        dz  dej
        dz  dej        dz  d	ej        dz  d
ej        dz  deeej	        f         dz  dej	        dz  dedz  dedz  dedz  deez  fd„¦   «         Zˆ xZS )rJ  c           	      ó  •— t          ¦   «                              |¦  «         |j        | _        |j         | _        | j        rt	          d¦  «        ‚|j        | _        |j        | _        |j        | _        |j        | _        |j	        | _	        |j
        | _
        |j        | _        | j        dz  | _        |j        | _        |j        | _        |j        | _        |j        | _        | j        | j        z  dk    s
J d¦   «         ‚t%          j        |j        | j        ¦  «        | _        |j        dk    r+|j        r$t%          j        | j        | j        ¦  «        | _        t%          j        | j        | j        | j
        ¬¦  «        | _        t%          j        | j        |j        ¬¦  «        | _        t%          j        ¦   «         | _        t%          j        ¦   «         | _        t%          j        ¦   «         | _        t%          j        ¦   «         | _        tA          | j        ¦  «        D ]á}| j         !                    tE          | j        | j        ||¬¦  «        ¦  «         | j         !                    t%          j        | j        |j        ¬¦  «        ¦  «         | j         !                    tG          | j        | j        | j        |¬	¦  «        ¦  «         | j         !                    t%          j        | j        |j        ¬¦  «        ¦  «         ŒâtI          |d
d¦  «        | _%        tI          |dd¦  «        | _&        |  '                    dtQ          j)        |j        ¦  «         *                    d¦  «        d¬¦  «         |  +                    ¦   «          d S )Nz1Currently Flaubert can only be used as an encoderr2  r   z-transformer dim must be a multiple of n_headsr   )rH  rÛ   )rZ   rK   ©rZ   Ú	layerdropg        Úpre_normFrP  rA  )Ú
persistent),rM   rN   Ú
is_encoderÚ
is_decoderr  rB   r9  r8  r¯   Ú	eos_indexr°   r±   r%   Ú
hidden_dimrP   Ún_layersrS   rR   r   rC  rN  rM  Úlang_embeddingsÚ
embeddingsrà   rá   Úlayer_norm_embÚ
ModuleListÚ
attentionsÚlayer_norm1ÚffnsÚlayer_norm2r*   ÚappendrJ   r‘   r  rX  rY  Úregister_bufferr-   r:   ré   Ú	post_init)rY   rZ   Úir[   s      €r'   rN   zFlaubertModel.__init__µ  sú  ø€ Ý‰Œ×Ò˜Ñ Ô Ð ð !Ô+ˆŒØ$Ô/Ð/ˆŒØŒ?ð 	[Ý%Ð&YÑZÔZÐZà”mˆŒð ”~ˆŒØ"Ô/ˆÔØ”~ˆŒØÔ)ˆŒØÔ)ˆŒð ”>ˆŒØœ( Q™,ˆŒØ”~ˆŒØœˆŒØ”~ˆŒØ!'Ô!9ˆÔØŒx˜$œ,Ñ&¨!Ò+Ð+Ð+Ð-\Ñ+Ô+Ð+õ $&¤<°Ô0NÐPTÔPXÑ#YÔ#YˆÔ ØŒ>˜AÒÐ &Ô"5ÐÝ#%¤<°´¸d¼hÑ#GÔ#GˆDÔ Ýœ, t¤|°T´XÈ4Ì>ÐZÑZÔZˆŒÝ œl¨4¬8¸Ô9NÐOÑOÔOˆÔõ œ-™/œ/ˆŒÝœ=™?œ?ˆÔÝ”M‘O”OˆŒ	Ýœ=™?œ?ˆÔõ
 �t”}Ñ%Ô%ð 	Wð 	WˆAØŒO×"Ò"Õ#5°d´lÀDÄHÐU[ÐghÐ#iÑ#iÔ#iÑjÔjÐjØÔ×#Ò#¥B¤L°´¸vÔ?TÐ$UÑ$UÔ$UÑVÔVÐVð ŒI×Ò�^¨D¬H°d´oÀtÄxÐX^Ð_Ñ_Ô_Ñ`Ô`Ð`ØÔ×#Ò#¥B¤L°´¸vÔ?TÐ$UÑ$UÔ$UÑVÔVÐVÐVå  ¨°cÑ:Ô:ˆŒÝ ¨
°EÑ:Ô:ˆŒØ×ÒØ�EœL¨Ô)GÑHÔH×OÒOÐPWÑXÔXÐejð 	ñ 	
ô 	
ð 	
ð
 	�ŠÑÔÐÐÐr)   c                 ó   — | j         S rŸ   ©ra  ©rY   s    r'   Úget_input_embeddingsz"FlaubertModel.get_input_embeddingsø  s
   € ØŒÐr)   c                 ó   — || _         d S rŸ   rm  ©rY   Únew_embeddingss     r'   Úset_input_embeddingsz"FlaubertModel.set_input_embeddingsü  s   € Ø(ˆŒˆˆr)   Nr4  r5  r6  Útoken_type_idsrP  rA   rz   Úinputs_embedsr{   Úoutput_hidden_statesrÿ   rÒ   c                 óf  — |	�|	n| j         j        }	|
�|
n| j         j        }
|�|n| j         j        }|�|                     ¦   «         \  }}n|                     ¦   «         dd…         \  }}|�|j        n|j        }|€6t          t          | j         ¬¦  «        t          | j         ¬¦  «        ¦  «        }|€W|�2|| j        k     	                    d¬¦  «         
                    ¦   «         }n#t          j        |f||t          j
        ¬¦  «        }|                     d¦  «        |k    sJ ‚|                     ¦   «                              ¦   «         |k    sJ ‚t          ||| j        |¬¦  «        \  }}|€‡t#          | d	¦  «        r+| j        dd…d|…f         }|                     ||f¦  «        }nht          j        |t          j
        |¬
¦  «        }|                     d¦  «                             ||f¦  «        }n|                     ¦   «         ||fk    sJ ‚|�|                     ¦   «         ||fk    sJ ‚|�f|�d||                     ¦   «         z
  }|dd…| d…f         }|dd…| d…f         }|�|dd…| d…f         }|dd…| d…f         }|dd…| d…f         }|€|                      |¦  «        }||                      |¦  «                             |¦  «        z   }|�/| j        r(| j         j        dk    r||                      |¦  «        z   }|�||                      |¦  «        z   }|                      |¦  «        }t<          j                              || j         | j!        ¬¦  «        }||                     d¦  «         "                    |j#        ¦  «        z  }|
rdnd}|	rdnd}tI          | j%        ¦  «        D �]Ã}| j!        r t          j&        g ¦  «        }|| j'        k     rŒ*|
r||fz   }| j(        sx | j)        |         ||||	¬¦  «        }|d         }|	r||d         fz   }t<          j                              || j         | j!        ¬¦  «        }||z   } | j*        |         |¦  «        }n| | j*        |         |¦  «        } | j)        |         ||||         ¬¦  «        }|d         }|	r||d         fz   }t<          j                              || j         | j!        ¬¦  «        }||z   }| j(        s0| | j+        |         |¦  «        z   } | j,        |         |¦  «        }n/ | j,        |         |¦  «        }| | j+        |         |¦  «        z   }||                     d¦  «         "                    |j#        ¦  «        z  }�ŒÅ|
r||fz   }|st[          d„ |||fD ¦   «         ¦  «        S t]          |||¬¦  «        S )a  
        langs (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            A parallel sequence of tokens to be used to indicate the language of each token in the input. Indices are
            languages ids which can be obtained from the language names by using two conversion mappings provided in
            the configuration of the model (only provided for multilingual models). More precisely, the *language name
            to language id* mapping is in `model.config.lang2id` (which is a dictionary string to int) and the
            *language id to language name* mapping is in `model.config.id2lang` (dictionary int to string).

            See usage examples detailed in the [multilingual documentation](../multilingual).
        lengths (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Length of each sentence that can be used to avoid performing attention on padding token indices. You can
            also use `attention_mask` for the same result (see above), kept here for compatibility. Indices selected in
            `[0, ..., input_ids.size(-1)]`:
        cache (`dict[str, torch.FloatTensor]`, *optional*):
            Dictionary strings to `torch.FloatTensor` that contains precomputed hidden-states (key and values in the
            attention blocks) as computed by the model (see `cache` output below). Can be used to speed up sequential
            decoding. The dictionary object will be modified in-place during the forward pass to add newly computed
            hidden-states.
        Nr]   rW  r   r^   )r9   r8   r   )rC   rP  r7   r_   r·   )rz   r{   )rz   c              3   ó   K  — | ]}|®|V — Œ	d S rŸ   r·   )r#   r„   s     r'   ú	<genexpr>z(FlaubertModel.forward.<locals>.<genexpr>¥  s"   è è € ÐYÐY˜qÈ1È=˜È=È=È=È=ÐYÐYr)   )Úlast_hidden_staterÐ   rd  )/rZ   r{   rv  rÿ   r>   r9   r   r   r°   Úsumr;   r-   Úfullr<   r=   rH   rB   r!  rP  ré   r:   r  Úget_seq_lengthra  rM  ro   r8  r9  r`  rb  r   rs   rS   ra   Útor8   r*   r_  ÚrandrX  rY  rd  re  rf  rg  r  r   )rY   r4  r5  r6  rt  rP  rA   rz   ru  r{   rv  rÿ   r|   rF   r@   r9   rE   rG   Ú_slenr7  rÐ   rd  rk  Údropout_probabilityÚattn_outputsr  Útensor_normalizeds                              r'   r‰   zFlaubertModel.forwardÿ  sf  € ðF 2CÐ1NÐ-Ð-ÐTXÔT_ÔTqÐà$8Ð$DÐ Ð È$Ì+ÔJjð 	ð &1Ð%<�k�kÀ$Ä+ÔBYˆð Ð Ø —~’~Ñ'Ô'‰HˆB��à$×)Ò)Ñ+Ô+¨C¨R¨CÔ0‰HˆB�à%.Ð%:�Ô!Ð!ÀÔ@Tˆàˆ=Ý'­¸D¼KÐ(HÑ(HÔ(HÍ,Ð^bÔ^iÐJjÑJjÔJjÑkÔkˆEàˆ?ØÐ$Ø$¨¬Ò6×;Ò;ÀÐ;ÑBÔB×GÒGÑIÔI��åœ* b U¨D¸ÅuÄzÐRÑRÔR�ð �|Š|˜A‰Œ "Ò$Ð$Ð$Ð$Ø�{Š{‰}Œ}×!Ò!Ñ#Ô# tÒ+Ð+Ð+Ð+õ $ D¨'°4´;È^Ð\Ñ\Ô\‰ˆˆið ÐÝ�t˜^Ñ,Ô,ð LØ#Ô0°°°°E°T°E°Ô:�Ø+×2Ò2°B¸°:Ñ>Ô>��å$œ|¨D½¼
È6ÐRÑRÔR�Ø+×5Ò5°aÑ8Ô8×?Ò?ÀÀTÀ
ÑKÔK��à×$Ò$Ñ&Ô&¨2¨t¨*Ò4Ð4Ð4Ð4ð ÐØ—:’:‘<”< B¨ :Ò-Ð-Ð-Ð-ð Ð Ð!6Ø˜5×/Ò/Ñ1Ô1Ñ1ˆEØ! ! ! ! e V W W *Ô-ˆIØ'¨¨¨¨E¨6¨7¨7¨
Ô3ˆLØÐ Ø˜a˜a˜a %   ˜jÔ)�Ø˜˜˜˜E˜6˜7˜7˜
Ô#ˆDØ! ! ! ! e V W W *Ô-ˆIð Ð Ø ŸOšO¨IÑ6Ô6ˆMà ×!9Ò!9¸,Ñ!GÔ!G×!QÒ!QÐR_Ñ!`Ô!`Ñ`ˆØÐ Ô!2Ð°t´{Ô7JÈQÒ7NÐ7NØ˜d×2Ò2°5Ñ9Ô9Ñ9ˆFØÐ%Ø˜dŸošo¨nÑ=Ô=Ñ=ˆFØ×$Ò$ VÑ,Ô,ˆÝ”×&Ò& v°´ÈÌÐ&ÑVÔVˆØ�$—.’. Ñ$Ô$×'Ò'¨¬Ñ5Ô5Ñ5ˆð 3Ð<˜˜¸ˆØ,Ð6�R�R°$ˆ
Ý�t”}Ñ%Ô%ð )	:ñ )	:ˆAàŒ}ð Ý&+¤j°¡n¤nÐ#Ø&¨¬Ò7Ð7Øà#ð :Ø -°°	Ñ 9�ð ”=ð 'Ø1˜tœ¨qÔ1ØØØØ&7ð	 ñ  ô  �ð $ A”�Ø$ð AØ!+¨|¸A¬Ð.@Ñ!@�JÝ”}×,Ò,¨T°T´\ÈDÌMÐ,ÑZÔZ�Ø $™�Ø,˜Ô)¨!Ô,¨VÑ4Ô4��à$7 DÔ$4°QÔ$7¸Ñ$?Ô$?Ð!Ø1˜tœ¨qÔ1Ð2CÀYÐV[Ð\]ÔV^Ð_Ñ_Ô_�Ø# A”�Ø$ð AØ!+¨|¸A¬Ð.@Ñ!@�JÝ”}×,Ò,¨T°T´\ÈDÌMÐ,ÑZÔZ�Ø $™�ð ”=ð BØ , $¤)¨A¤,¨vÑ"6Ô"6Ñ6�Ø,˜Ô)¨!Ô,¨VÑ4Ô4��à$7 DÔ$4°QÔ$7¸Ñ$?Ô$?Ð!Ø , $¤)¨A¤,Ð/@Ñ"AÔ"AÑA�à�d—n’n RÑ(Ô(×+Ò+¨F¬LÑ9Ô9Ñ9ˆF‰Fð  ð 	6Ø)¨V¨IÑ5ˆMàð 	ZÝÐYÐY V¨]¸JÐ$GÐYÑYÔYÑYÔYÐYå°À}ÐakÐlÑlÔlÐlr)   )NNNNNNNNNNN)rŠ   r‹   rŒ   rN   ro  rs  r   r-   rÉ   r.   ÚTensorÚdictÚstrr  r  r   r‰   rŽ   r�   s   @r'   rJ  rJ  ³  sž  ø€ € € € € ð@ð @ð @ð @ð @ðFð ð ð)ð )ð )ð ð .2Ø37Ø%)Ø26Ø04Ø+/Ø59Ø26Ø)-Ø,0Ø#'ðgmð gmàÔ# dÑ*ðgmð Ô)¨DÑ0ðgmð Œ|˜dÑ"ð	gmð
 Ô(¨4Ñ/ðgmð Ô&¨Ñ-ðgmð Ô! DÑ(ðgmð �C˜Ô*Ð*Ô+¨dÑ2ðgmð Ô(¨4Ñ/ðgmð   $™;ðgmð # T™kðgmð ˜D‘[ðgmð 
�Ñ	 ðgmð gmð gmñ „^ðgmð gmð gmð gmð gmr)   rJ  z‹
    The Flaubert Model transformer with a language modeling head on top (linear layer with weights tied to the input
    embeddings).
    c                   óV  ‡ — e Zd ZddiZˆ fd„Zd„ Zd„ Zd„ Ze	 	 	 	 	 	 	 	 	 	 	 	 dde	j
        dz  d	e	j
        dz  d
e	j
        dz  de	j
        dz  de	j
        dz  de	j
        dz  deee	j
        f         dz  de	j
        dz  de	j
        dz  dedz  dedz  dedz  deez  fd„¦   «         Zˆ xZS )ÚFlaubertWithLMHeadModelzpred_layer.proj.weightztransformer.embeddings.weightc                 óÂ   •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          |¦  «        | _        |                      ¦   «          d S rŸ   )rM   rN   rJ  r.  r¥   Ú
pred_layerrj  rÏ   s     €r'   rN   z FlaubertWithLMHeadModel.__init__³  sR   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý(¨Ñ0Ô0ˆÔÝ+¨FÑ3Ô3ˆŒð 	�ŠÑÔÐÐÐr)   c                 ó   — | j         j        S rŸ   ©rŠ  r²   rn  s    r'   Úget_output_embeddingsz-FlaubertWithLMHeadModel.get_output_embeddings»  s   € ØŒÔ#Ð#r)   c                 ó   — || j         _        d S rŸ   rŒ  rq  s     r'   Úset_output_embeddingsz-FlaubertWithLMHeadModel.set_output_embeddings¾  s   € Ø-ˆŒÔÐÐr)   c                 ó  — | j         j        }| j         j        }|j        d         }t	          j        |df|t          j        |j        ¬¦  «        }t	          j        ||gd¬¦  «        }|�t	          j	        ||¦  «        }nd }||dœS )Nr   r   r7   r^   )r4  r6  )
rZ   Úmask_token_idÚlang_idrè   r-   r|  r;   r9   rë   r*  )rY   r4  r|   r‘  r’  Úeffective_batch_sizeÚ
mask_tokenr6  s           r'   Úprepare_inputs_for_generationz5FlaubertWithLMHeadModel.prepare_inputs_for_generationÁ  s�   € ð œÔ1ˆØ”+Ô%ˆà(œ¨qÔ1ÐÝ”ZÐ!5°qÐ 9¸=ÕPUÔPZÐclÔcsÐtÑtÔtˆ
Ý”I˜y¨*Ð5¸1Ð=Ñ=Ô=ˆ	ØÐÝ”O I¨wÑ7Ô7ˆEˆEàˆEØ&°Ð7Ð7Ð7r)   Nr4  r5  r6  rt  rP  rA   rz   ru  Úlabelsr{   rv  rÿ   rÒ   c                 ó*  — |�|n| j         j        }|                      |||||||||
||¬¦  «        }|d         }|                      ||	¦  «        }|s||dd…         z   S t	          |	�|d         nd|	€|d         n|d         |j        |j        ¬¦  «        S )aÂ  
        langs (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            A parallel sequence of tokens to be used to indicate the language of each token in the input. Indices are
            languages ids which can be obtained from the language names by using two conversion mappings provided in
            the configuration of the model (only provided for multilingual models). More precisely, the *language name
            to language id* mapping is in `model.config.lang2id` (which is a dictionary string to int) and the
            *language id to language name* mapping is in `model.config.id2lang` (dictionary int to string).

            See usage examples detailed in the [multilingual documentation](../multilingual).
        lengths (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Length of each sentence that can be used to avoid performing attention on padding token indices. You can
            also use `attention_mask` for the same result (see above), kept here for compatibility. Indices selected in
            `[0, ..., input_ids.size(-1)]`:
        cache (`dict[str, torch.FloatTensor]`, *optional*):
            Dictionary strings to `torch.FloatTensor` that contains precomputed hidden-states (key and values in the
            attention blocks) as computed by the model (see `cache` output below). Can be used to speed up sequential
            decoding. The dictionary object will be modified in-place during the forward pass to add newly computed
            hidden-states.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for language modeling. Note that the labels **are shifted** inside the model, i.e. you can set
            `labels = input_ids` Indices are selected in `[-100, 0, ..., config.vocab_size]` All labels set to `-100`
            are ignored (masked), the loss is only computed for labels in `[0, ..., config.vocab_size]`
        N©
r5  r6  rt  rP  rA   rz   ru  r{   rv  rÿ   r   r   ©r½   ÚlogitsrÐ   rd  )rZ   rÿ   r.  rŠ  r   rÐ   rd  )rY   r4  r5  r6  rt  rP  rA   rz   ru  r–  r{   rv  rÿ   r|   Útransformer_outputsr+  rˆ   s                    r'   r‰   zFlaubertWithLMHeadModel.forwardÐ  sÖ   € ðP &1Ð%<�k�kÀ$Ä+ÔBYˆà"×.Ò.ØØ)ØØ)Ø%ØØØ'Ø/Ø!5Ø#ð /ñ 
ô 
Ðð % QÔ'ˆØ—/’/ &¨&Ñ1Ô1ˆàð 	5ØÐ0°°°Ô4Ñ4Ð4åØ%Ð1�˜”�°tØ!' �7˜1”:�:°W¸Q´ZØ-Ô;Ø*Ô5ð	
ñ 
ô 
ð 	
r)   ©NNNNNNNNNNNN)rŠ   r‹   rŒ   Ú_tied_weights_keysrN   r�  r�  r•  r   r-   r„  r…  r†  r  r  r   r‰   rŽ   r�   s   @r'   rˆ  rˆ  ª  s®  ø€ € € € € ð 3Ð4SÐTÐðð ð ð ð ð$ð $ð $ð.ð .ð .ð8ð 8ð 8ð ð *.Ø.2Ø%)Ø.2Ø,0Ø'+Ø04Ø-1Ø&*Ø)-Ø,0Ø#'ðB
ð B
à”< $Ñ&ðB
ð œ tÑ+ðB
ð Œ|˜dÑ"ð	B
ð
 œ tÑ+ðB
ð ”l TÑ)ðB
ð ” Ñ$ðB
ð �C˜œÐ%Ô&¨Ñ-ðB
ð ”| dÑ*ðB
ð ”˜tÑ#ðB
ð   $™;ðB
ð # T™kðB
ð ˜D‘[ðB
ð 
�Ñ	ðB
ð B
ð B
ñ „^ðB
ð B
ð B
ð B
ð B
r)   rˆ  z”
    Flaubert Model with a sequence classification/regression head on top (a linear layer on top of the pooled output)
    e.g. for GLUE tasks.
    c                   ó<  ‡ — e Zd Zˆ fd„Ze	 	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	eeej        f         dz  d
ej        dz  dej        dz  de	dz  de	dz  de	dz  de
ez  fd„¦   «         Zˆ xZS )Ú!FlaubertForSequenceClassificationc                 óè   •— t          ¦   «                              |¦  «         |j        | _        || _        t	          |¦  «        | _        t          |¦  «        | _        |                      ¦   «          d S rŸ   )	rM   rN   r"  rZ   rJ  r.  r  Úsequence_summaryrj  rÏ   s     €r'   rN   z*FlaubertForSequenceClassification.__init__  sd   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒØˆŒå(¨Ñ0Ô0ˆÔÝ 7¸Ñ ?Ô ?ˆÔð 	�ŠÑÔÐÐÐr)   Nr4  r5  r6  rt  rP  rA   rz   ru  r–  r{   rv  rÿ   rÒ   c                 óÈ  — |�|n| j         j        }|                      |||||||||
||¬¦  «        }|d         }|                      |¦  «        }d}|	��Z| j         j        €f| j        dk    rd| j         _        nN| j        dk    r7|	j        t          j        k    s|	j        t          j	        k    rd| j         _        nd| j         _        | j         j        dk    rWt          ¦   «         }| j        dk    r1 ||                     ¦   «         |	                     ¦   «         ¦  «        }nŽ |||	¦  «        }n�| j         j        dk    rGt          ¦   «         } ||                     d| j        ¦  «        |	                     d¦  «        ¦  «        }n*| j         j        dk    rt          ¦   «         } |||	¦  «        }|s|f|dd…         z   }|�|f|z   n|S t          |||j        |j        ¬	¦  «        S )
a®  
        langs (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            A parallel sequence of tokens to be used to indicate the language of each token in the input. Indices are
            languages ids which can be obtained from the language names by using two conversion mappings provided in
            the configuration of the model (only provided for multilingual models). More precisely, the *language name
            to language id* mapping is in `model.config.lang2id` (which is a dictionary string to int) and the
            *language id to language name* mapping is in `model.config.id2lang` (dictionary int to string).

            See usage examples detailed in the [multilingual documentation](../multilingual).
        lengths (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Length of each sentence that can be used to avoid performing attention on padding token indices. You can
            also use *attention_mask* for the same result (see above), kept here for compatibility. Indices selected in
            `[0, ..., input_ids.size(-1)]`.
        cache (`dict[str, torch.FloatTensor]`, *optional*):
            Instance of `EncoderDecoderCache` that contains precomputed KV states. Can be used to speed up sequential
            decoding.
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the sequence classification/regression loss. Indices should be in `[0, ...,
            config.num_labels - 1]`. If `config.num_labels == 1` a regression loss is computed (Mean-Square loss), If
            `config.num_labels > 1` a classification loss is computed (Cross-Entropy).
        Nr˜  r   r   Ú
regressionÚsingle_label_classificationÚmulti_label_classificationr]   r™  )rZ   rÿ   r.  r¡  Úproblem_typer"  r8   r-   r;   r�   r   rÖ   r   rb   r   r   rÐ   rd  )rY   r4  r5  r6  rt  rP  rA   rz   ru  r–  r{   rv  rÿ   r|   r›  r+  rš  r½   r  s                      r'   r‰   z)FlaubertForSequenceClassification.forward)  s  € ðL &1Ð%<�k�kÀ$Ä+ÔBYˆà"×.Ò.ØØ)ØØ)Ø%ØØØ'Ø/Ø!5Ø#ð /ñ 
ô 
Ðð % QÔ'ˆØ×&Ò& vÑ.Ô.ˆàˆØÑØŒ{Ô'Ð/Ø”? aÒ'Ð'Ø/;�D”KÔ,Ð,Ø”_ qÒ(Ð(¨f¬l½e¼jÒ.HÐ.HÈFÌLÕ\aÔ\eÒLeÐLeØ/L�D”KÔ,Ð,à/K�D”KÔ,àŒ{Ô'¨<Ò7Ð7Ý"™9œ9�Ø”? aÒ'Ð'Ø#˜8 F§N¢NÑ$4Ô$4°f·n²nÑ6FÔ6FÑGÔG�D�Dà#˜8 F¨FÑ3Ô3�D�DØ”Ô)Ð-JÒJÐJÝ+Ñ-Ô-�Ø�x §¢¨B°´Ñ @Ô @À&Ç+Â+ÈbÁ/Ä/ÑRÔR��Ø”Ô)Ð-IÒIÐIÝ,Ñ.Ô.�Ø�x ¨Ñ/Ô/�àð 	FØ�YÐ!4°Q°R°RÔ!8Ñ8ˆFØ)-Ð)9�T�G˜fÑ$Ð$¸vÐEå'ØØØ-Ô;Ø*Ô5ð	
ñ 
ô 
ð 	
r)   rœ  )rŠ   r‹   rŒ   rN   r   r-   r„  r…  r†  r  r  r   r‰   rŽ   r�   s   @r'   rŸ  rŸ    st  ø€ € € € € ð	ð 	ð 	ð 	ð 	ð ð *.Ø.2Ø%)Ø.2Ø,0Ø'+Ø04Ø-1Ø&*Ø)-Ø,0Ø#'ðX
ð X
à”< $Ñ&ðX
ð œ tÑ+ðX
ð Œ|˜dÑ"ð	X
ð
 œ tÑ+ðX
ð ”l TÑ)ðX
ð ” Ñ$ðX
ð �C˜œÐ%Ô&¨Ñ-ðX
ð ”| dÑ*ðX
ð ”˜tÑ#ðX
ð   $™;ðX
ð # T™kðX
ð ˜D‘[ðX
ð 
Ð)Ñ	)ðX
ð X
ð X
ñ „^ðX
ð X
ð X
ð X
ð X
r)   rŸ  c                   ó<  ‡ — e Zd Zˆ fd„Ze	 	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	eeej        f         dz  d
ej        dz  dej        dz  de	dz  de	dz  de	dz  de
ez  fd„¦   «         Zˆ xZS )ÚFlaubertForTokenClassificationc                 ó6  •— t          ¦   «                              |¦  «         |j        | _        t          |¦  «        | _        t          j        |j        ¦  «        | _        t          j        |j	        |j        ¦  «        | _
        |                      ¦   «          d S rŸ   )rM   rN   r"  rJ  r.  r   r$  rS   rT   rÍ   Ú
classifierrj  rÏ   s     €r'   rN   z'FlaubertForTokenClassification.__init__ˆ  sy   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒå(¨Ñ0Ô0ˆÔÝ”z &¤.Ñ1Ô1ˆŒÝœ) FÔ$6¸Ô8IÑJÔJˆŒð 	�ŠÑÔÐÐÐr)   Nr4  r5  r6  rt  rP  rA   rz   ru  r–  r{   rv  rÿ   rÒ   c                 óÈ  — |�|n| j         j        }|                      |||||||||
||¬¦  «        }|d         }|                      |¦  «        }|                      |¦  «        }d}|	�Ft          ¦   «         } ||                     d| j        ¦  «        |	                     d¦  «        ¦  «        }|s|f|dd…         z   }|�|f|z   n|S t          |||j	        |j
        ¬¦  «        S )aü  
        langs (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            A parallel sequence of tokens to be used to indicate the language of each token in the input. Indices are
            languages ids which can be obtained from the language names by using two conversion mappings provided in
            the configuration of the model (only provided for multilingual models). More precisely, the *language name
            to language id* mapping is in `model.config.lang2id` (which is a dictionary string to int) and the
            *language id to language name* mapping is in `model.config.id2lang` (dictionary int to string).

            See usage examples detailed in the [multilingual documentation](../multilingual).
        lengths (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Length of each sentence that can be used to avoid performing attention on padding token indices. You can
            also use *attention_mask* for the same result (see above), kept here for compatibility. Indices selected in
            `[0, ..., input_ids.size(-1)]`.
        cache (`dict[str, torch.FloatTensor]`, *optional*):
            Instance of `EncoderDecoderCache` that contains precomputed KV states. Can be used to speed up sequential
            decoding.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the token classification loss. Indices should be in `[0, ..., config.num_labels - 1]`.
        Nr˜  r   r]   r   r™  )rZ   rÿ   r.  rS   rª  r   rb   r"  r   rÐ   rd  )rY   r4  r5  r6  rt  rP  rA   rz   ru  r–  r{   rv  rÿ   r|   rˆ   Úsequence_outputrš  r½   r  r+  s                       r'   r‰   z&FlaubertForTokenClassification.forward“  s  € ðH &1Ð%<�k�kÀ$Ä+ÔBYˆà×"Ò"ØØ)ØØ)Ø%ØØØ'Ø/Ø!5Ø#ð #ñ 
ô 
ˆð " !œ*ˆàŸ,š, Ñ7Ô7ˆØ—’ Ñ1Ô1ˆàˆØÐÝ'Ñ)Ô)ˆHØ�8˜FŸKšK¨¨D¬OÑ<Ô<¸f¿kºkÈ"¹o¼oÑNÔNˆDàð 	FØ�Y ¨¨¨¤Ñ,ˆFØ)-Ð)9�T�G˜fÑ$Ð$¸vÐEå$ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r)   rœ  )rŠ   r‹   rŒ   rN   r   r-   r„  r…  r†  r  r  r   r‰   rŽ   r�   s   @r'   r¨  r¨  …  st  ø€ € € € € ð	ð 	ð 	ð 	ð 	ð ð *.Ø.2Ø%)Ø.2Ø,0Ø'+Ø04Ø-1Ø&*Ø)-Ø,0Ø#'ðF
ð F
à”< $Ñ&ðF
ð œ tÑ+ðF
ð Œ|˜dÑ"ð	F
ð
 œ tÑ+ðF
ð ”l TÑ)ðF
ð ” Ñ$ðF
ð �C˜œÐ%Ô&¨Ñ-ðF
ð ”| dÑ*ðF
ð ”˜tÑ#ðF
ð   $™;ðF
ð # T™kðF
ð ˜D‘[ðF
ð 
Ð&Ñ	&ðF
ð F
ð F
ñ „^ðF
ð F
ð F
ð F
ð F
r)   r¨  zá
    Flaubert Model with a span classification head on top for extractive question-answering tasks like SQuAD (a linear
    layers on top of the hidden-states output to compute `span start logits` and `span end logits`).
    c                   óR  ‡ — e Zd Zˆ fd„Ze	 	 	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	eeej        f         dz  d
ej        dz  dej        dz  dej        dz  de	dz  de	dz  de	dz  de
ez  fd„¦   «         Zˆ xZS )Ú"FlaubertForQuestionAnsweringSimplec                 óâ   •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          j        |j        |j        ¦  «        | _        |  	                    ¦   «          d S rŸ   )
rM   rN   rJ  r.  r   rT   rÍ   r"  Ú
qa_outputsrj  rÏ   s     €r'   rN   z+FlaubertForQuestionAnsweringSimple.__init__å  s\   ø€ Ý‰Œ×Ò˜Ñ Ô Ð å(¨Ñ0Ô0ˆÔÝœ) FÔ$6¸Ô8IÑJÔJˆŒð 	�ŠÑÔÐÐÐr)   Nr4  r5  r6  rt  rP  rA   rz   ru  rä   rý   r{   rv  rÿ   rÒ   c                 ó´  — |�|n| j         j        }|                      |||||||||||¬¦  «        }|d         }|                      |¦  «        }|                     dd¬¦  «        \  }}|                     d¦  «                             ¦   «         }|                     d¦  «                             ¦   «         }d}|	�ç|
�åt          |	                     ¦   «         ¦  «        dk    r|	                     d¦  «        }	t          |
                     ¦   «         ¦  «        dk    r|
                     d¦  «        }
|                     d¦  «        }|	 	                    d|¦  «        }	|
 	                    d|¦  «        }
t          |¬¦  «        } |||	¦  «        } |||
¦  «        }||z   dz  }|s||f|dd…         z   }|�|f|z   n|S t          ||||j        |j        ¬	¦  «        S )
a*  
        langs (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            A parallel sequence of tokens to be used to indicate the language of each token in the input. Indices are
            languages ids which can be obtained from the language names by using two conversion mappings provided in
            the configuration of the model (only provided for multilingual models). More precisely, the *language name
            to language id* mapping is in `model.config.lang2id` (which is a dictionary string to int) and the
            *language id to language name* mapping is in `model.config.id2lang` (dictionary int to string).

            See usage examples detailed in the [multilingual documentation](../multilingual).
        lengths (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Length of each sentence that can be used to avoid performing attention on padding token indices. You can
            also use *attention_mask* for the same result (see above), kept here for compatibility. Indices selected in
            `[0, ..., input_ids.size(-1)]`.
        cache (`dict[str, torch.FloatTensor]`, *optional*):
            Instance of `EncoderDecoderCache` that contains precomputed KV states. Can be used to speed up sequential
            decoding.
        Nr˜  r   r   r]   r^   )Úignore_indexr    )r½   rú   rû   rÐ   rd  )rZ   rÿ   r.  r°  ÚsplitrÖ   rw   Úlenr>   Úclampr   r   rÐ   rd  )rY   r4  r5  r6  rt  rP  rA   rz   ru  rä   rý   r{   rv  rÿ   r|   r›  r¬  rš  rú   rû   r  Úignored_indexr  r  r  r+  s                             r'   r‰   z*FlaubertForQuestionAnsweringSimple.forwardî  s   € ðF &1Ð%<�k�kÀ$Ä+ÔBYˆà"×.Ò.ØØ)ØØ)Ø%ØØØ'Ø/Ø!5Ø#ð /ñ 
ô 
Ðð .¨aÔ0ˆà—’ Ñ1Ô1ˆØ#)§<¢<°°r <Ñ#:Ô#:Ñ ˆ�jØ#×+Ò+¨BÑ/Ô/×:Ò:Ñ<Ô<ˆØ×'Ò'¨Ñ+Ô+×6Ò6Ñ8Ô8ˆ
àˆ
ØÐ&¨=Ð+Då�?×'Ò'Ñ)Ô)Ñ*Ô*¨QÒ.Ð.Ø"1×"9Ò"9¸"Ñ"=Ô"=�Ý�=×%Ò%Ñ'Ô'Ñ(Ô(¨1Ò,Ð,Ø -× 5Ò 5°bÑ 9Ô 9�à(×-Ò-¨aÑ0Ô0ˆMØ-×3Ò3°A°}ÑEÔEˆOØ)×/Ò/°°=ÑAÔAˆMå'°]ÐCÑCÔCˆHØ!˜ ,°Ñ@Ô@ˆJØ�x 
¨MÑ:Ô:ˆHØ$ xÑ/°1Ñ4ˆJàð 	RØ" JÐ/Ð2EÀaÀbÀbÔ2IÑIˆFØ/9Ð/E�Z�M FÑ*Ð*È6ÐQå+ØØ%Ø!Ø-Ô;Ø*Ô5ð
ñ 
ô 
ð 	
r)   )NNNNNNNNNNNNN)rŠ   r‹   rŒ   rN   r   r-   r„  r…  r†  r  r  r   r‰   rŽ   r�   s   @r'   r®  r®  Ý  s‰  ø€ € € € € ðð ð ð ð ð ð *.Ø.2Ø%)Ø.2Ø,0Ø'+Ø04Ø-1Ø/3Ø-1Ø)-Ø,0Ø#'ðT
ð T
à”< $Ñ&ðT
ð œ tÑ+ðT
ð Œ|˜dÑ"ð	T
ð
 œ tÑ+ðT
ð ”l TÑ)ðT
ð ” Ñ$ðT
ð �C˜œÐ%Ô&¨Ñ-ðT
ð ”| dÑ*ðT
ð œ¨Ñ,ðT
ð ”| dÑ*ðT
ð   $™;ðT
ð # T™kðT
ð ˜D‘[ðT
ð  
Ð-Ñ	-ð!T
ð T
ð T
ñ „^ðT
ð T
ð T
ð T
ð T
r)   r®  zR
    Base class for outputs of question answering models using a `SquadHead`.
    c                   ó  — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dZ	ej
        dz  ed<   dZej        dz  ed<   dZej
        dz  ed<   dZej        dz  ed<   dZeej                 dz  ed	<   dZeej                 dz  ed
<   dS )Ú"FlaubertForQuestionAnsweringOutputrÂ   Nr½   rÃ   rÄ   rÅ   rÆ   rÇ   rÐ   rd  )rŠ   r‹   rŒ   r¿   r½   r-   r.   rÈ   rÃ   rÄ   rÉ   rÅ   rÆ   rÇ   rÐ   r  rd  r·   r)   r'   r¸  r¸  F  sê   € € € € € € ðð ð" &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø48Ð˜Ô*¨TÑ1Ð8Ð8Ñ8Ø/3€O�UÔ%¨Ñ,Ð3Ð3Ñ3Ø26Ð�uÔ(¨4Ñ/Ð6Ð6Ñ6Ø-1€M�5Ô# dÑ*Ð1Ð1Ñ1Ø+/€J�Ô! DÑ(Ð/Ð/Ñ/Ø59€M�5˜Ô*Ô+¨dÑ2Ð9Ð9Ñ9Ø26€J��eÔ'Ô(¨4Ñ/Ð6Ð6Ñ6Ð6Ð6r)   r¸  c            %       ó”  ‡ — e Zd Zˆ fd„Ze	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	eeej        f         dz  d
ej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  de	dz  de	dz  de	dz  de
ez  f"d„¦   «         Zˆ xZS )ÚFlaubertForQuestionAnsweringc                 óÂ   •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          |¦  «        | _        |                      ¦   «          d S rŸ   )rM   rN   rJ  r.  rö   r°  rj  rÏ   s     €r'   rN   z%FlaubertForQuestionAnswering.__init__l  sR   ø€ Ý‰Œ×Ò˜Ñ Ô Ð å(¨Ñ0Ô0ˆÔÝ+¨FÑ3Ô3ˆŒð 	�ŠÑÔÐÐÐr)   Nr4  r5  r6  rt  rP  rA   rz   ru  rä   rý   rþ   rò   rÑ   r{   rv  rÿ   rÒ   c                 óF  — |�|n| j         j        }|                      |||||||||||¬¦  «        }|d         }|                      ||	|
||||¬¦  «        }|s||dd…         z   S t	          |j        |j        |j        |j        |j	        |j
        |j        |j        ¬¦  «        S )am
  
        langs (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            A parallel sequence of tokens to be used to indicate the language of each token in the input. Indices are
            languages ids which can be obtained from the language names by using two conversion mappings provided in
            the configuration of the model (only provided for multilingual models). More precisely, the *language name
            to language id* mapping is in `model.config.lang2id` (which is a dictionary string to int) and the
            *language id to language name* mapping is in `model.config.id2lang` (dictionary int to string).

            See usage examples detailed in the [multilingual documentation](../multilingual).
        lengths (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Length of each sentence that can be used to avoid performing attention on padding token indices. You can
            also use *attention_mask* for the same result (see above), kept here for compatibility. Indices selected in
            `[0, ..., input_ids.size(-1)]`.
        cache (`dict[str, torch.FloatTensor]`, *optional*):
            Instance of `EncoderDecoderCache` that contains precomputed KV states. Can be used to speed up sequential
            decoding.
        is_impossible (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels whether a question has an answer or no answer (SQuAD 2.0)
        cls_index (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for position (index) of the classification token to use as input for computing plausibility of the
            answer.
        p_mask (`torch.FloatTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Optional mask of tokens which can't be in answers (e.g. [CLS], [PAD], ...). 1.0 means token should be
            masked. 0.0 mean token is not masked.

        Example:

        ```python
        >>> from transformers import AutoTokenizer, FlaubertForQuestionAnswering
        >>> import torch

        >>> tokenizer = AutoTokenizer.from_pretrained("FacebookAI/xlm-mlm-en-2048")
        >>> model = FlaubertForQuestionAnswering.from_pretrained("FacebookAI/xlm-mlm-en-2048")

        >>> input_ids = torch.tensor(tokenizer.encode("Hello, my dog is cute", add_special_tokens=True)).unsqueeze(
        ...     0
        ... )  # Batch size 1
        >>> start_positions = torch.tensor([1])
        >>> end_positions = torch.tensor([3])

        >>> outputs = model(input_ids, start_positions=start_positions, end_positions=end_positions)
        >>> loss = outputs.loss
        ```Nr˜  r   )rä   rý   rò   rþ   rÑ   rÿ   r   )r½   rÃ   rÄ   rÅ   rÆ   rÇ   rÐ   rd  )rZ   rÿ   r.  r°  r¸  r½   rÃ   rÄ   rÅ   rÆ   rÇ   rÐ   rd  )rY   r4  r5  r6  rt  rP  rA   rz   ru  rä   rý   rþ   rò   rÑ   r{   rv  rÿ   r|   r›  r+  rˆ   s                        r'   r‰   z$FlaubertForQuestionAnswering.forwardu  sò   € ð@ &1Ð%<�k�kÀ$Ä+ÔBYˆà"×.Ò.ØØ)ØØ)Ø%ØØØ'Ø/Ø!5Ø#ð /ñ 
ô 
Ðð % QÔ'ˆà—/’/ØØ+Ø'ØØ'ØØ#ð "ñ 
ô 
ˆð ð 	5ØÐ0°°°Ô4Ñ4Ð4å1Ø”Ø 'Ô ;Ø#Ô3Ø%Ô7Ø!Ô/ØÔ)Ø-Ô;Ø*Ô5ð	
ñ 	
ô 	
ð 		
r)   )NNNNNNNNNNNNNNNN)rŠ   r‹   rŒ   rN   r   r-   r„  r…  r†  r  r  r¸  r‰   rŽ   r�   s   @r'   rº  rº  i  sÈ  ø€ € € € € ðð ð ð ð ð ð *.Ø.2Ø%)Ø.2Ø,0Ø'+Ø04Ø-1Ø/3Ø-1Ø-1Ø)-Ø&*Ø)-Ø,0Ø#'ð#g
ð g
à”< $Ñ&ðg
ð œ tÑ+ðg
ð Œ|˜dÑ"ð	g
ð
 œ tÑ+ðg
ð ”l TÑ)ðg
ð ” Ñ$ðg
ð �C˜œÐ%Ô&¨Ñ-ðg
ð ”| dÑ*ðg
ð œ¨Ñ,ðg
ð ”| dÑ*ðg
ð ”| dÑ*ðg
ð ”< $Ñ&ðg
ð ”˜tÑ#ðg
ð   $™;ðg
ð  # T™kð!g
ð" ˜D‘[ð#g
ð& 
Ð3Ñ	3ð'g
ð g
ð g
ñ „^ðg
ð g
ð g
ð g
ð g
r)   rº  c                   ó<  ‡ — e Zd Zˆ fd„Ze	 	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	eeej        f         dz  d
ej        dz  dej        dz  de	dz  de	dz  de	dz  de
ez  fd„¦   «         Zˆ xZS )ÚFlaubertForMultipleChoicec                 óø   •—  t          ¦   «         j        |g|¢R i |¤Ž t          |¦  «        | _        t	          |¦  «        | _        t          j        |j        d¦  «        | _	        |  
                    ¦   «          d S r“   )rM   rN   rJ  r.  r  r¡  r   rT   r"  Úlogits_projrj  )rY   rZ   Úinputsr|   r[   s       €r'   rN   z"FlaubertForMultipleChoice.__init__ã  sy   ø€ Ø�‰ŒÔ˜Ð3 &Ð3Ð3Ð3¨FÐ3Ð3Ð3å(¨Ñ0Ô0ˆÔÝ 7¸Ñ ?Ô ?ˆÔÝœ9 VÔ%6¸Ñ:Ô:ˆÔð 	�ŠÑÔÐÐÐr)   Nr4  r5  r6  rt  rP  rA   rz   ru  r–  r{   rv  rÿ   rÒ   c                 óT  — |�|n| j         j        }|�|j        d         n|j        d         }|�)|                     d|                     d¦  «        ¦  «        nd}|�)|                     d|                     d¦  «        ¦  «        nd}|�)|                     d|                     d¦  «        ¦  «        nd}|�)|                     d|                     d¦  «        ¦  «        nd}|�)|                     d|                     d¦  «        ¦  «        nd}|�=|                     d|                     d¦  «        |                     d¦  «        ¦  «        nd}|�t
                               d¦  «         d}|                      |||||||||
||¬¦  «        }|d         }|                      |¦  «        }|  	                    |¦  «        }|                     d|¦  «        }d}|	�t          ¦   «         } |||	¦  «        }|s|f|dd…         z   }|�|f|z   n|S t          |||j        |j        ¬¦  «        S )	a‰  
        input_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`):
            Indices of input sequence tokens in the vocabulary.

            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are input IDs?](../glossary#input-ids)
        langs (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`, *optional*):
            A parallel sequence of tokens to be used to indicate the language of each token in the input. Indices are
            languages ids which can be obtained from the language names by using two conversion mappings provided in
            the configuration of the model (only provided for multilingual models). More precisely, the *language name
            to language id* mapping is in `model.config.lang2id` (which is a dictionary string to int) and the
            *language id to language name* mapping is in `model.config.id2lang` (dictionary int to string).

            See usage examples detailed in the [multilingual documentation](../multilingual).
        token_type_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,
            1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.

            [What are token type IDs?](../glossary#token-type-ids)
        position_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`, *optional*):
            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
            config.max_position_embeddings - 1]`.

            [What are position IDs?](../glossary#position-ids)
        lengths (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Length of each sentence that can be used to avoid performing attention on padding token indices. You can
            also use *attention_mask* for the same result (see above), kept here for compatibility. Indices selected in
            `[0, ..., input_ids.size(-1)]`.
        cache (`dict[str, torch.FloatTensor]`, *optional*):
            Instance of `EncoderDecoderCache` that contains precomputed KV states. Can be used to speed up sequential
            decoding.
        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, num_choices, sequence_length, hidden_size)`, *optional*):
            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
            is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
            model's internal embedding lookup matrix.
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the multiple choice classification loss. Indices should be in `[0, ...,
            num_choices-1]` where `num_choices` is the size of the second dimension of the input tensors. (See
            `input_ids` above)
        Nr   r]   rç   zwThe `lengths` parameter cannot be used with the Flaubert multiple choice models. Please use the attention mask instead.)r4  r5  r6  rt  rP  rA   rz   ru  r{   rv  rÿ   r   r™  )rZ   rÿ   rè   rb   r>   ÚloggerÚwarningr.  r¡  rÀ  r   r   rÐ   rd  )rY   r4  r5  r6  rt  rP  rA   rz   ru  r–  r{   rv  rÿ   r|   Únum_choicesr›  r+  rš  Úreshaped_logitsr½   r  s                        r'   r‰   z!FlaubertForMultipleChoice.forwardí  sŠ  € ð| &1Ð%<�k�kÀ$Ä+ÔBYˆØ,5Ð,A�i”o aÔ(Ð(À}ÔGZÐ[\ÔG]ˆà>GÐ>S�I—N’N 2 y§~¢~°bÑ'9Ô'9Ñ:Ô:Ð:ÐY]ˆ	ØM[ÐMg˜×,Ò,¨R°×1DÒ1DÀRÑ1HÔ1HÑIÔIÐIÐmqˆØM[ÐMg˜×,Ò,¨R°×1DÒ1DÀRÑ1HÔ1HÑIÔIÐIÐmqˆØGSÐG_�|×(Ò(¨¨\×->Ò->¸rÑ-BÔ-BÑCÔCÐCÐeiˆØ27Ð2C�—
’
˜2˜uŸzšz¨"™~œ~Ñ.Ô.Ð.Èˆð Ð(ð ×Ò˜r =×#5Ò#5°bÑ#9Ô#9¸=×;MÒ;MÈbÑ;QÔ;QÑRÔRÐRàð 	ð ÐÝ�NŠNð*ñô ð ð ˆGà"×.Ò.ØØ)ØØ)Ø%ØØØ'Ø/Ø!5Ø#ð /ñ 
ô 
Ðð % QÔ'ˆØ×&Ò& vÑ.Ô.ˆØ×!Ò! &Ñ)Ô)ˆØ Ÿ+š+ b¨+Ñ6Ô6ˆàˆØÐÝ'Ñ)Ô)ˆHØ�8˜O¨VÑ4Ô4ˆDàð 	FØ%Ð'Ð*=¸a¸b¸bÔ*AÑAˆFØ)-Ð)9�T�G˜fÑ$Ð$¸vÐEå(ØØ"Ø-Ô;Ø*Ô5ð	
ñ 
ô 
ð 	
r)   rœ  )rŠ   r‹   rŒ   rN   r   r-   r„  r…  r†  r  r  r   r‰   rŽ   r�   s   @r'   r¾  r¾  à  st  ø€ € € € € ðð ð ð ð ð ð *.Ø.2Ø%)Ø.2Ø,0Ø'+Ø04Ø-1Ø&*Ø)-Ø,0Ø#'ðr
ð r
à”< $Ñ&ðr
ð œ tÑ+ðr
ð Œ|˜dÑ"ð	r
ð
 œ tÑ+ðr
ð ”l TÑ)ðr
ð ” Ñ$ðr
ð �C˜œÐ%Ô&¨Ñ-ðr
ð ”| dÑ*ðr
ð ”˜tÑ#ðr
ð   $™;ðr
ð # T™kðr
ð ˜D‘[ðr
ð 
Ð*Ñ	*ðr
ð r
ð r
ñ „^ðr
ð r
ð r
ð r
ð r
r)   r¾  )r¾  rº  r®  rŸ  r¨  rJ  rˆ  r-  rŸ   )Cr¿   rl   Úcollections.abcr   Údataclassesr   Únumpyr!   r-   r   Útorch.nnr   r   r   Ú r
   rE  Úactivationsr   r   Úcache_utilsr   r   Ú
generationr   Úmodeling_outputsr   r   r   r   r   r   Úmodeling_utilsr   Úpytorch_utilsr   Úutilsr   r   r   Úconfiguration_flaubertr   Ú
get_loggerrŠ   rÃ  r5   rH   ÚModulerJ   r‘   r¥   rÁ   rË   rÙ   rð   rö   r  r-  rJ  rˆ  rŸ  r¨  r®  r¸  rº  r¾  Ú__all__r·   r)   r'   ú<module>r×     s{  ðð ,Ð +à €€€Ø $Ð $Ð $Ð $Ð $Ð $Ø !Ð !Ð !Ð !Ð !Ð !à Ð Ð Ð Ø €€€Ø Ð Ð Ð Ð Ð Ø AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ Aà &Ð &Ð &Ð &Ð &Ð &Ø /Ð /Ð /Ð /Ð /Ð /Ð /Ð /Ø <Ð <Ð <Ð <Ð <Ð <Ð <Ð <Ø )Ð )Ð )Ð )Ð )Ð )ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð .Ð -Ð -Ð -Ð -Ð -Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6Ø 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2ð 
ˆÔ	˜HÑ	%Ô	%€ðð ð ðð ð ð ð4Mð Mð Mð Mð M˜œñ Mô Mð Mðbð ð ð ð �R”Yñ ô ð ð* €ððñ ô ð'ð 'ð 'ð 'ð '˜œ	ñ 'ô 'ñô ð'ðT €ððñ ô ð
 ð0ð 0ð 0ð 0ð 0˜kñ 0ô 0ñ „ñô ð0ð6!ð !ð !ð !ð ! ¤	ñ !ô !ð !ðJBð Bð Bð Bð B˜bœiñ Bô Bð BðL>ð >ð >ð >ð > ¤	ñ >ô >ð >ðDmð mð mð mð m˜œ	ñ mô mð mðb`ð `ð `ð `ð `˜bœiñ `ô `ð `ðF ð"ið "ið "ið "ið "i˜oñ "iô "iñ „ð"iðJ ðsmð smð smð smð smÐ+ñ smô smñ „ðsmðl €ððñ ô ðc
ð c
ð c
ð c
ð c
Ð5°ñ c
ô c
ñô ðc
ðL €ððñ ô ðe
ð e
ð e
ð e
ð e
Ð(?ñ e
ô e
ñô ðe
ðP ðS
ð S
ð S
ð S
ð S
Ð%<ñ S
ô S
ñ „ðS
ðl €ððñ ô ð_
ð _
ð _
ð _
ð _
Ð)@ñ _
ô _
ñô ð_
ðD €ððñ ô ð
 ð7ð 7ð 7ð 7ð 7¨ñ 7ô 7ñ „ñô ð7ð8 ðr
ð r
ð r
ð r
ð r
Ð#:ñ r
ô r
ñ „ðr
ðj ð~
ð ~
ð ~
ð ~
ð ~
Ð 7ñ ~
ô ~
ñ „ð~
ðB	ð 	ð 	€€€r)   