§
    ‚Štju{  ã                   ó8  — d Z ddlZddlmZ ddlmZmZmZ ddlmZ	 ddl
mZ ddlmZ ddlmZmZmZmZmZmZmZ dd	lmZ dd
lmZ ddlmZmZmZ ddlmZ ddl m!Z!m"Z"m#Z#m$Z$m%Z% ddl&m'Z'  ej(        e)¦  «        Z* G d„ de"¦  «        Z+ G d„ de%¦  «        Z, G d„ de!¦  «        Z- G d„ de#¦  «        Z.e G d„ de¦  «        ¦   «         Z/ G d„ de$¦  «        Z0 ed¬¦  «         G d„ d e/e¦  «        ¦   «         Z1e G d!„ d"e/¦  «        ¦   «         Z2 G d#„ d$ej3        ¦  «        Z4 ed%¬¦  «         G d&„ d'e/¦  «        ¦   «         Z5e G d(„ d)e/¦  «        ¦   «         Z6e G d*„ d+e/¦  «        ¦   «         Z7 G d,„ d-ej3        ¦  «        Z8e G d.„ d/e/¦  «        ¦   «         Z9g d0¢Z:dS )1zPyTorch RoBERTa model.é    N)ÚBCEWithLogitsLossÚCrossEntropyLossÚMSELossé   )Úinitialization)Úgelu)ÚGenerationMixin)Ú,BaseModelOutputWithPoolingAndCrossAttentionsÚ!CausalLMOutputWithCrossAttentionsÚMaskedLMOutputÚMultipleChoiceModelOutputÚQuestionAnsweringModelOutputÚSequenceClassifierOutputÚTokenClassifierOutput)ÚPreTrainedModel)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚlogging)Úcan_return_tupleé   )ÚBertCrossAttentionÚBertEmbeddingsÚ	BertLayerÚ	BertModelÚBertSelfAttentioné   )ÚRobertaConfigc                   ó´   ‡ — e Zd Zˆ fd„Z	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  def
d	„Ze	d
„ ¦   «         Z
e	dd„¦   «         Zˆ xZS )ÚRobertaEmbeddingsc                 óÀ   •— t          ¦   «                              |¦  «         | `| `|j        | _        t          j        |j        |j        | j        ¬¦  «        | _        d S )N)Úpadding_idx)	ÚsuperÚ__init__Úpad_token_idÚposition_embeddingsr"   ÚnnÚ	EmbeddingÚmax_position_embeddingsÚhidden_size©ÚselfÚconfigÚ	__class__s     €úi/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/roberta/modular_roberta.pyr$   zRobertaEmbeddings.__init__-   sa   ø€ Ý‰Œ×Ò˜Ñ Ô Ð àÐØÐ$à!Ô.ˆÔÝ#%¤<ØÔ*¨FÔ,>ÈDÔL\ð$
ñ $
ô $
ˆÔ Ð Ð ó    Nr   Ú	input_idsÚtoken_type_idsÚposition_idsÚinputs_embedsÚpast_key_values_lengthc                 ó*  — |€:|�|                       || j        |¦  «        }n|                      || j        ¦  «        }|�|                     ¦   «         }n|                     ¦   «         d d…         }|\  }}|€§t	          | d¦  «        rl| j                             |j        ¦  «                             |j	        d         d¦  «        }	t          j        |	d|¬¦  «        }	|	                     ||¦  «        }n+t          j        |t          j        | j        j        ¬¦  «        }|€|                      |¦  «        }|                      |¦  «        }
||
z   }|                      |¦  «        }||z   }|                      |¦  «        }|                      |¦  «        }|S )Néÿÿÿÿr2   r   r   )ÚdimÚindex©ÚdtypeÚdevice)Ú"create_position_ids_from_input_idsr"   Ú&create_position_ids_from_inputs_embedsÚsizeÚhasattrr2   Útor<   ÚexpandÚshapeÚtorchÚgatherÚzerosÚlongr3   Úword_embeddingsÚtoken_type_embeddingsr&   Ú	LayerNormÚdropout)r,   r1   r2   r3   r4   r5   Úinput_shapeÚ
batch_sizeÚ
seq_lengthÚbuffered_token_type_idsrI   Ú
embeddingsr&   s                r/   ÚforwardzRobertaEmbeddings.forward8   s¨  € ð ÐØÐ$à#×FÒFØ˜tÔ/Ð1Gñ ô  ��ð  $×JÒJÈ=ÐZ^ÔZjÑkÔk�àÐ Ø#Ÿ.š.Ñ*Ô*ˆKˆKà'×,Ò,Ñ.Ô.¨s°¨sÔ3ˆKà!,Ñˆ
�Jð
 Ð!Ý�tÐ-Ñ.Ô.ð mà*.Ô*=×*@Ò*@ÀÔATÑ*UÔ*U×*\Ò*\Ð]iÔ]oÐpqÔ]rÐtvÑ*wÔ*wÐ'Ý*/¬,Ð7NÐTUÐ]iÐ*jÑ*jÔ*jÐ'Ø!8×!?Ò!?À
ÈJÑ!WÔ!W��å!&¤¨[ÅÄ
ÐSWÔSdÔSkÐ!lÑ!lÔ!l�àÐ Ø ×0Ò0°Ñ;Ô;ˆMØ $× :Ò :¸>Ñ JÔ JÐØ"Ð%:Ñ:ˆ
à"×6Ò6°|ÑDÔDÐØÐ"5Ñ5ˆ
à—^’^ JÑ/Ô/ˆ
Ø—\’\ *Ñ-Ô-ˆ
ØÐr0   c                 óú   — |                       ¦   «         dd…         }|d         }t          j        |dz   ||z   dz   t          j        | j        ¬¦  «        }|                     d¦  «                             |¦  «        S )z×
        We are provided embeddings directly. We cannot infer which are padded so just generate sequential position ids.

        Args:
            inputs_embeds: torch.Tensor

        Returns: torch.Tensor
        Nr7   r   r:   r   )r?   rD   ÚarangerG   r<   Ú	unsqueezerB   )r4   r"   rL   Úsequence_lengthr3   s        r/   r>   z8RobertaEmbeddings.create_position_ids_from_inputs_embedsh   s~   € ð $×(Ò(Ñ*Ô*¨3¨B¨3Ô/ˆØ% aœ.ˆå”|Ø˜!‰O˜_¨{Ñ:¸QÑ>ÅeÄjÐYfÔYmð
ñ 
ô 
ˆð ×%Ò% aÑ(Ô(×/Ò/°Ñ<Ô<Ð<r0   c                 óÜ   — |                       |¦  «                             ¦   «         }t          j        |d¬¦  «                             |¦  «        |z   |z  }|                     ¦   «         |z   S )a  
        Replace non-padding symbols with their position numbers. Position numbers begin at padding_idx+1. Padding symbols
        are ignored. This is modified from fairseq's `utils.make_positions`.

        Args:
            x: torch.Tensor x:

        Returns: torch.Tensor
        r   ©r8   )ÚneÚintrD   ÚcumsumÚtype_asrG   )r1   r"   r5   ÚmaskÚincremental_indicess        r/   r=   z4RobertaEmbeddings.create_position_ids_from_input_idsz   sg   € ð �|Š|˜KÑ(Ô(×,Ò,Ñ.Ô.ˆÝ$œ|¨D°aÐ8Ñ8Ô8×@Ò@ÀÑFÔFÐI_Ñ_ÐcgÑgÐØ"×'Ò'Ñ)Ô)¨KÑ7Ð7r0   )NNNNr   )r   )Ú__name__Ú
__module__Ú__qualname__r$   rD   Ú
LongTensorÚFloatTensorrY   rQ   Ústaticmethodr>   r=   Ú__classcell__©r.   s   @r/   r    r    ,   sî   ø€ € € € € ð	
ð 	
ð 	
ð 	
ð 	
ð .2Ø26Ø04Ø26Ø&'ð.ð .àÔ# dÑ*ð.ð Ô(¨4Ñ/ð.ð Ô&¨Ñ-ð	.ð
 Ô(¨4Ñ/ð.ð !$ð.ð .ð .ð .ð` ð=ð =ñ „\ð=ð" ð8ð 8ð 8ñ „\ð8ð 8ð 8ð 8ð 8r0   r    c                   ó   — e Zd ZdS )ÚRobertaSelfAttentionN©r^   r_   r`   © r0   r/   rg   rg   ‹   ó   € € € € € Ø€Dr0   rg   c                   ó   — e Zd ZdS )ÚRobertaCrossAttentionNrh   ri   r0   r/   rl   rl   �   rj   r0   rl   c                   ó   — e Zd ZdS )ÚRobertaLayerNrh   ri   r0   r/   rn   rn   “   rj   r0   rn   c                   óp   ‡ — e Zd ZeZdZdZdZdZdZ	dZ
eeedœZ ej        ¦   «         ˆ fd„¦   «         Zˆ xZS )ÚRobertaPreTrainedModelÚrobertaT)Úhidden_statesÚ
attentionsÚcross_attentionsc                 ó¨  •— t          ¦   «                              |¦  «         t          |t          ¦  «        rt	          j        |j        ¦  «         dS t          |t          ¦  «        rjt	          j        |j	        t          j        |j	        j        d         ¦  «                             d¦  «        ¦  «         t	          j        |j        ¦  «         dS dS )zInitialize the weightsr7   )r   r7   N)r#   Ú_init_weightsÚ
isinstanceÚRobertaLMHeadÚinitÚzeros_Úbiasr    Úcopy_r3   rD   rS   rC   rB   r2   )r,   Úmoduler.   s     €r/   rv   z$RobertaPreTrainedModel._init_weights¦   s·   ø€ õ 	‰Œ×Ò˜fÑ%Ô%Ð%Ý�f�mÑ,Ô,ð 	/ÝŒK˜œÑ$Ô$Ð$Ð$Ð$Ý˜Õ 1Ñ2Ô2ð 	/ÝŒJ�vÔ*­E¬L¸Ô9LÔ9RÐSUÔ9VÑ,WÔ,W×,^Ò,^Ð_fÑ,gÔ,gÑhÔhÐhÝŒK˜Ô-Ñ.Ô.Ð.Ð.Ð.ð	/ð 	/r0   )r^   r_   r`   r   Úconfig_classÚbase_model_prefixÚsupports_gradient_checkpointingÚ_supports_flash_attnÚ_supports_sdpaÚ_supports_flex_attnÚ_supports_attention_backendrn   rg   rl   Ú_can_record_outputsrD   Úno_gradrv   rd   re   s   @r/   rp   rp   —   sŠ   ø€ € € € € à €LØ!ÐØ&*Ð#ØÐØ€NØÐØ"&Ðà%Ø*Ø1ðð Ðð €U„]�_„_ð/ð /ð /ð /ñ „_ð/ð /ð /ð /ð /r0   rp   c                   ó    ‡ — e Zd Zdˆ fd„	Zˆ xZS )ÚRobertaModelTc                 óL   •— t          ¦   «                              | |¦  «         d S ©N)r#   r$   )r,   r-   Úadd_pooling_layerr.   s      €r/   r$   zRobertaModel.__init__²   s#   ø€ Ý‰Œ×Ò˜˜vÑ&Ô&Ð&Ð&Ð&r0   )T)r^   r_   r`   r$   rd   re   s   @r/   rˆ   rˆ   ±   s=   ø€ € € € € ð'ð 'ð 'ð 'ð 'ð 'ð 'ð 'ð 'ð 'r0   rˆ   zS
    RoBERTa Model with a `language modeling` head on top for CLM fine-tuning.
    )Úcustom_introc                   óŽ  ‡ — e Zd ZdddœZˆ fd„Zd„ Zd„ Zee	 	 	 	 	 	 	 	 	 	 	 dd	e	j
        dz  d
e	j        dz  de	j
        dz  de	j
        dz  de	j        dz  de	j        dz  de	j        dz  de	j
        dz  deee	j                          dz  dedz  dee	j        z  dee         dee	j                 ez  fd„¦   «         ¦   «         Zˆ xZS )ÚRobertaForCausalLMú)roberta.embeddings.word_embeddings.weightúlm_head.bias©zlm_head.decoder.weightzlm_head.decoder.biasc                 ó  •— t          ¦   «                              |¦  «         |j        st                               d¦  «         t          |d¬¦  «        | _        t          |¦  «        | _        |  	                    ¦   «          d S )NzOIf you want to use `RobertaLMHeadModel` as a standalone, add `is_decoder=True.`F©r‹   ©
r#   r$   Ú
is_decoderÚloggerÚwarningrˆ   rq   rx   Úlm_headÚ	post_initr+   s     €r/   r$   zRobertaForCausalLM.__init__Á   su   ø€ Ý‰Œ×Ò˜Ñ Ô Ð àÔ ð 	nÝ�NŠNÐlÑmÔmÐmå# F¸eÐDÑDÔDˆŒÝ$ VÑ,Ô,ˆŒð 	�ŠÑÔÐÐÐr0   c                 ó   — | j         j        S rŠ   ©r˜   Údecoder©r,   s    r/   Úget_output_embeddingsz(RobertaForCausalLM.get_output_embeddingsÍ   ó   € ØŒ|Ô#Ð#r0   c                 ó   — || j         _        d S rŠ   r›   ©r,   Únew_embeddingss     r/   Úset_output_embeddingsz(RobertaForCausalLM.set_output_embeddingsÐ   ó   € Ø-ˆŒÔÐÐr0   Nr   r1   Úattention_maskr2   r3   r4   Úencoder_hidden_statesÚencoder_attention_maskÚlabelsÚpast_key_valuesÚ	use_cacheÚlogits_to_keepÚkwargsÚreturnc                 ól  — |�d}
 | j         |f|||||||	|
ddœ	|¤Ž}|j        }t          |t          ¦  «        rt	          | d¦  «        n|}|                      |dd…|dd…f         ¦  «        }d}|� | j        d||| j        j        dœ|¤Ž}t          |||j
        |j        |j        |j        ¬¦  «        S )am  
        token_type_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.
            This parameter can only be used when the model is initialized with `type_vocab_size` parameter with value
            >= 2. All the value in this tensor should be always < type_vocab_size.

            [What are token type IDs?](../glossary#token-type-ids)
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the left-to-right language modeling loss (next word prediction). Indices should be in
            `[-100, 0, ..., config.vocab_size]` (see `input_ids` docstring) Tokens with indices set to `-100` are
            ignored (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`

        Example:

        ```python
        >>> from transformers import AutoTokenizer, RobertaForCausalLM, AutoConfig
        >>> import torch

        >>> tokenizer = AutoTokenizer.from_pretrained("FacebookAI/roberta-base")
        >>> config = AutoConfig.from_pretrained("FacebookAI/roberta-base")
        >>> config.is_decoder = True
        >>> model = RobertaForCausalLM.from_pretrained("FacebookAI/roberta-base", config=config)

        >>> inputs = tokenizer("Hello, my dog is cute", return_tensors="pt")
        >>> outputs = model(**inputs)

        >>> prediction_logits = outputs.logits
        ```NFT)	r¥   r2   r3   r4   r¦   r§   r©   rª   Úreturn_dict)Úlogitsr¨   Ú
vocab_size)Úlossr°   r©   rr   rs   rt   ri   )rq   Úlast_hidden_staterw   rY   Úslicer˜   Úloss_functionr-   r±   r   r©   rr   rs   rt   )r,   r1   r¥   r2   r3   r4   r¦   r§   r¨   r©   rª   r«   r¬   Úoutputsrr   Úslice_indicesr°   r²   s                     r/   rQ   zRobertaForCausalLM.forwardÓ   s  € ð` ÐØˆIà@LÀÄØðA
à)Ø)Ø%Ø'Ø"7Ø#9Ø+ØØðA
ð A
ð ðA
ð A
ˆð  Ô1ˆå8BÀ>ÕSVÑ8WÔ8WÐk�˜~˜o¨tÑ4Ô4Ð4Ð]kˆØ—’˜m¨A¨A¨A¨}¸a¸a¸aÐ,?Ô@ÑAÔAˆàˆØÐØ%�4Ô%Ðp¨V¸FÈtÌ{ÔOeÐpÐpÐioÐpÐpˆDå0ØØØ#Ô3Ø!Ô/ØÔ)Ø$Ô5ð
ñ 
ô 
ð 	
r0   )NNNNNNNNNNr   )r^   r_   r`   Ú_tied_weights_keysr$   rž   r£   r   r   rD   ra   rb   ÚtupleÚboolrY   ÚTensorr   r   r   rQ   rd   re   s   @r/   rŽ   rŽ   ¶   sÅ  ø€ € € € € ð #NØ .ðð Ðð

ð 
ð 
ð 
ð 
ð$ð $ð $ð.ð .ð .ð Øð .2Ø37Ø26Ø04Ø26Ø:>Ø;?Ø*.ØBFØ!%Ø-.ðO
ð O
àÔ# dÑ*ðO
ð Ô)¨DÑ0ðO
ð Ô(¨4Ñ/ð	O
ð
 Ô&¨Ñ-ðO
ð Ô(¨4Ñ/ðO
ð  %Ô0°4Ñ7ðO
ð !&Ô 1°DÑ 8ðO
ð Ô  4Ñ'ðO
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðO
ð ˜$‘;ðO
ð ˜eœlÑ*ðO
ð Ð+Ô,ðO
ð 
ˆuŒ|Ô	Ð@Ñ	@ðO
ð O
ð O
ñ „^ñ ÔðO
ð O
ð O
ð O
ð O
r0   rŽ   c                   ó>  ‡ — e Zd ZdddœZˆ fd„Zd„ Zd„ Zee	 	 	 	 	 	 	 	 dde	j
        dz  d	e	j        dz  d
e	j
        dz  de	j
        dz  de	j        dz  de	j        dz  de	j        dz  de	j
        dz  dee         dee	j                 ez  fd„¦   «         ¦   «         Zˆ xZS )ÚRobertaForMaskedLMr�   r�   r‘   c                 ó  •— t          ¦   «                              |¦  «         |j        rt                               d¦  «         t          |d¬¦  «        | _        t          |¦  «        | _        |  	                    ¦   «          d S )NznIf you want to use `RobertaForMaskedLM` make sure `config.is_decoder=False` for bi-directional self-attention.Fr“   r”   r+   s     €r/   r$   zRobertaForMaskedLM.__init__.  s~   ø€ Ý‰Œ×Ò˜Ñ Ô Ð àÔð 	Ý�NŠNð1ñô ð õ
 $ F¸eÐDÑDÔDˆŒÝ$ VÑ,Ô,ˆŒð 	�ŠÑÔÐÐÐr0   c                 ó   — | j         j        S rŠ   r›   r�   s    r/   rž   z(RobertaForMaskedLM.get_output_embeddings=  rŸ   r0   c                 ó   — || j         _        d S rŠ   r›   r¡   s     r/   r£   z(RobertaForMaskedLM.set_output_embeddings@  r¤   r0   Nr1   r¥   r2   r3   r4   r¦   r§   r¨   r¬   r­   c	                 ót  —  | j         |f||||||ddœ|	¤Ž}
|
d         }|                      |¦  «        }d}|�e|                     |j        ¦  «        }t	          ¦   «         } ||                     d| j        j        ¦  «        |                     d¦  «        ¦  «        }t          |||
j	        |
j
        ¬¦  «        S )aô  
        token_type_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.
            This parameter can only be used when the model is initialized with `type_vocab_size` parameter with value
            >= 2. All the value in this tensor should be always < type_vocab_size.

            [What are token type IDs?](../glossary#token-type-ids)
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should be in `[-100, 0, ...,
            config.vocab_size]` (see `input_ids` docstring) Tokens with indices set to `-100` are ignored (masked), the
            loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`
        T)r¥   r2   r3   r4   r¦   r§   r¯   r   Nr7   ©r²   r°   rr   rs   )rq   r˜   rA   r<   r   Úviewr-   r±   r   rr   rs   )r,   r1   r¥   r2   r3   r4   r¦   r§   r¨   r¬   r¶   Úsequence_outputÚprediction_scoresÚmasked_lm_lossÚloss_fcts                  r/   rQ   zRobertaForMaskedLM.forwardC  së   € ð: �$”,Øð

à)Ø)Ø%Ø'Ø"7Ø#9Øð

ð 

ð ð

ð 

ˆð " !œ*ˆØ ŸLšL¨Ñ9Ô9ÐàˆØÐà—Y’YÐ0Ô7Ñ8Ô8ˆFÝ'Ñ)Ô)ˆHØ%˜XÐ&7×&<Ò&<¸RÀÄÔAWÑ&XÔ&XÐZ`×ZeÒZeÐfhÑZiÔZiÑjÔjˆNåØØ$Ø!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r0   )NNNNNNNN)r^   r_   r`   r¸   r$   rž   r£   r   r   rD   ra   rb   r   r   r¹   r»   r   rQ   rd   re   s   @r/   r½   r½   '  sj  ø€ € € € € ð #NØ .ðð Ðð
ð ð ð ð ð$ð $ð $ð.ð .ð .ð Øð .2Ø37Ø26Ø04Ø26Ø:>Ø;?Ø*.ð5
ð 5
àÔ# dÑ*ð5
ð Ô)¨DÑ0ð5
ð Ô(¨4Ñ/ð	5
ð
 Ô&¨Ñ-ð5
ð Ô(¨4Ñ/ð5
ð  %Ô0°4Ñ7ð5
ð !&Ô 1°DÑ 8ð5
ð Ô  4Ñ'ð5
ð Ð+Ô,ð5
ð 
ˆuŒ|Ô	˜~Ñ	-ð5
ð 5
ð 5
ñ „^ñ Ôð5
ð 5
ð 5
ð 5
ð 5
r0   r½   c                   ó(   ‡ — e Zd ZdZˆ fd„Zd„ Zˆ xZS )rx   z*Roberta Head for masked language modeling.c                 ó‚  •— t          ¦   «                              ¦   «          t          j        |j        |j        ¦  «        | _        t          j        |j        |j        ¬¦  «        | _        t          j        |j        |j	        ¦  «        | _
        t          j        t          j        |j	        ¦  «        ¦  «        | _        d S )N)Úeps)r#   r$   r'   ÚLinearr*   ÚdenserJ   Úlayer_norm_epsÚ
layer_normr±   rœ   Ú	ParameterrD   rF   r{   r+   s     €r/   r$   zRobertaLMHead.__init__€  s‰   ø€ Ý‰Œ×ÒÑÔÐÝ”Y˜vÔ1°6Ô3EÑFÔFˆŒ
Ýœ, vÔ'9¸vÔ?TÐUÑUÔUˆŒå”y Ô!3°VÔ5FÑGÔGˆŒÝ”L¥¤¨VÔ->Ñ!?Ô!?Ñ@Ô@ˆŒ	ˆ	ˆ	r0   c                 ó¢   — |                       |¦  «        }t          |¦  «        }|                      |¦  «        }|                      |¦  «        }|S rŠ   )rÌ   r   rÎ   rœ   ©r,   Úfeaturesr¬   Úxs       r/   rQ   zRobertaLMHead.forwardˆ  sE   € Ø�JŠJ�xÑ Ô ˆÝ�‰GŒGˆØ�OŠO˜AÑÔˆð �LŠL˜‰OŒOˆàˆr0   ©r^   r_   r`   Ú__doc__r$   rQ   rd   re   s   @r/   rx   rx   }  sR   ø€ € € € € Ø4Ð4ðAð Að Að Að Aðð ð ð ð ð ð r0   rx   zŸ
    RoBERTa Model transformer with a sequence classification/regression head on top (a linear layer on top of the
    pooled output) e.g. for GLUE tasks.
    c                   óü   ‡ — e Zd Zˆ fd„Zee	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	e	e
         d
eej                 ez  fd„¦   «         ¦   «         Zˆ xZS )Ú RobertaForSequenceClassificationc                 óì   •— t          ¦   «                              |¦  «         |j        | _        || _        t	          |d¬¦  «        | _        t          |¦  «        | _        |                      ¦   «          d S ©NFr“   )	r#   r$   Ú
num_labelsr-   rˆ   rq   ÚRobertaClassificationHeadÚ
classifierr™   r+   s     €r/   r$   z)RobertaForSequenceClassification.__init__š  sg   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒØˆŒå# F¸eÐDÑDÔDˆŒÝ3°FÑ;Ô;ˆŒð 	�ŠÑÔÐÐÐr0   Nr1   r¥   r2   r3   r4   r¨   r¬   r­   c           	      ó�  —  | j         |f||||ddœ|¤Ž}|d         }	|                      |	¦  «        }
d}|��t|                     |
j        ¦  «        }| j        j        €f| j        dk    rd| j        _        nN| j        dk    r7|j        t          j	        k    s|j        t          j
        k    rd| j        _        nd| j        _        | j        j        dk    rWt          ¦   «         }| j        dk    r1 ||
                     ¦   «         |                     ¦   «         ¦  «        }nŽ ||
|¦  «        }n�| j        j        dk    rGt          ¦   «         } ||
                     d	| j        ¦  «        |                     d	¦  «        ¦  «        }n*| j        j        dk    rt          ¦   «         } ||
|¦  «        }t!          ||
|j        |j        ¬
¦  «        S )aß  
        token_type_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.
            This parameter can only be used when the model is initialized with `type_vocab_size` parameter with value
            >= 2. All the value in this tensor should be always < type_vocab_size.

            [What are token type IDs?](../glossary#token-type-ids)
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the sequence classification/regression loss. Indices should be in `[0, ...,
            config.num_labels - 1]`. If `config.num_labels == 1` a regression loss is computed (Mean-Square loss), If
            `config.num_labels > 1` a classification loss is computed (Cross-Entropy).
        T©r¥   r2   r3   r4   r¯   r   Nr   Ú
regressionÚsingle_label_classificationÚmulti_label_classificationr7   rÂ   )rq   rÜ   rA   r<   r-   Úproblem_typerÚ   r;   rD   rG   rY   r   Úsqueezer   rÃ   r   r   rr   rs   ©r,   r1   r¥   r2   r3   r4   r¨   r¬   r¶   rÄ   r°   r²   rÇ   s                r/   rQ   z(RobertaForSequenceClassification.forward¥  sÝ  € ð6 �$”,Øð
à)Ø)Ø%Ø'Øð
ð 
ð ð
ð 
ˆð " !œ*ˆØ—’ Ñ1Ô1ˆàˆØÑà—Y’Y˜vœ}Ñ-Ô-ˆFØŒ{Ô'Ð/Ø”? aÒ'Ð'Ø/;�D”KÔ,Ð,Ø”_ qÒ(Ð(¨f¬l½e¼jÒ.HÐ.HÈFÌLÕ\aÔ\eÒLeÐLeØ/L�D”KÔ,Ð,à/K�D”KÔ,àŒ{Ô'¨<Ò7Ð7Ý"™9œ9�Ø”? aÒ'Ð'Ø#˜8 F§N¢NÑ$4Ô$4°f·n²nÑ6FÔ6FÑGÔG�D�Dà#˜8 F¨FÑ3Ô3�D�DØ”Ô)Ð-JÒJÐJÝ+Ñ-Ô-�Ø�x §¢¨B°´Ñ @Ô @À&Ç+Â+ÈbÁ/Ä/ÑRÔR��Ø”Ô)Ð-IÒIÐIÝ,Ñ.Ô.�Ø�x ¨Ñ/Ô/�å'ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r0   ©NNNNNN)r^   r_   r`   r$   r   r   rD   ra   rb   r   r   r¹   r»   r   rQ   rd   re   s   @r/   r×   r×   “  s  ø€ € € € € ð	ð 	ð 	ð 	ð 	ð Øð .2Ø37Ø26Ø04Ø26Ø*.ðC
ð C
àÔ# dÑ*ðC
ð Ô)¨DÑ0ðC
ð Ô(¨4Ñ/ð	C
ð
 Ô&¨Ñ-ðC
ð Ô(¨4Ñ/ðC
ð Ô  4Ñ'ðC
ð Ð+Ô,ðC
ð 
ˆuŒ|Ô	Ð7Ñ	7ðC
ð C
ð C
ñ „^ñ ÔðC
ð C
ð C
ð C
ð C
r0   r×   c                   óü   ‡ — e Zd Zˆ fd„Zee	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	e	e
         d
eej                 ez  fd„¦   «         ¦   «         Zˆ xZS )ÚRobertaForMultipleChoicec                 ó  •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          j        |j        ¦  «        | _        t	          j        |j	        d¦  «        | _
        |                      ¦   «          d S )Nr   )r#   r$   rˆ   rq   r'   ÚDropoutÚhidden_dropout_probrK   rË   r*   rÜ   r™   r+   s     €r/   r$   z!RobertaForMultipleChoice.__init__ï  sl   ø€ Ý‰Œ×Ò˜Ñ Ô Ð å# FÑ+Ô+ˆŒÝ”z &Ô"<Ñ=Ô=ˆŒÝœ) FÔ$6¸Ñ:Ô:ˆŒð 	�ŠÑÔÐÐÐr0   Nr1   r2   r¥   r¨   r3   r4   r¬   r­   c           	      ó†  — |�|j         d         n|j         d         }|�)|                     d|                     d¦  «        ¦  «        nd}	|�)|                     d|                     d¦  «        ¦  «        nd}
|�)|                     d|                     d¦  «        ¦  «        nd}|�)|                     d|                     d¦  «        ¦  «        nd}|�=|                     d|                     d¦  «        |                     d¦  «        ¦  «        nd} | j        |	f|
|||ddœ|¤Ž}|d         }|                      |¦  «        }|                      |¦  «        }|                     d|¦  «        }d}|�4|                     |j        ¦  «        }t          ¦   «         } |||¦  «        }t          |||j
        |j        ¬¦  «        S )a  
        input_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`):
            Indices of input sequence tokens in the vocabulary.

            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are input IDs?](../glossary#input-ids)
        token_type_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.
            This parameter can only be used when the model is initialized with `type_vocab_size` parameter with value
            >= 2. All the value in this tensor should be always < type_vocab_size.

            [What are token type IDs?](../glossary#token-type-ids)
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the multiple choice classification loss. Indices should be in `[0, ...,
            num_choices-1]` where `num_choices` is the size of the second dimension of the input tensors. (See
            `input_ids` above)
        position_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`, *optional*):
            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
            config.max_position_embeddings - 1]`.

            [What are position IDs?](../glossary#position-ids)
        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, num_choices, sequence_length, hidden_size)`, *optional*):
            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
            is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
            model's internal embedding lookup matrix.
        Nr   r7   éþÿÿÿT)r3   r2   r¥   r4   r¯   rÂ   )rC   rÃ   r?   rq   rK   rÜ   rA   r<   r   r   rr   rs   )r,   r1   r2   r¥   r¨   r3   r4   r¬   Únum_choicesÚflat_input_idsÚflat_position_idsÚflat_token_type_idsÚflat_attention_maskÚflat_inputs_embedsr¶   Úpooled_outputr°   Úreshaped_logitsr²   rÇ   s                       r/   rQ   z RobertaForMultipleChoice.forwardù  s   € ðV -6Ð,A�i”o aÔ(Ð(À}ÔGZÐ[\ÔG]ˆàCLÐCX˜Ÿš¨¨I¯NªN¸2Ñ,>Ô,>Ñ?Ô?Ð?Ð^bˆØLXÐLd˜L×-Ò-¨b°,×2CÒ2CÀBÑ2GÔ2GÑHÔHÐHÐjnÐØR`ÐRl˜n×1Ò1°"°n×6IÒ6IÈ"Ñ6MÔ6MÑNÔNÐNÐrvÐØR`ÐRl˜n×1Ò1°"°n×6IÒ6IÈ"Ñ6MÔ6MÑNÔNÐNÐrvÐð Ð(ð ×Ò˜r =×#5Ò#5°bÑ#9Ô#9¸=×;MÒ;MÈbÑ;QÔ;QÑRÔRÐRàð 	ð �$”,Øð
à*Ø.Ø.Ø,Øð
ð 
ð ð
ð 
ˆð   œ
ˆàŸš ]Ñ3Ô3ˆØ—’ Ñ/Ô/ˆØ Ÿ+š+ b¨+Ñ6Ô6ˆàˆØÐà—Y’Y˜Ô5Ñ6Ô6ˆFÝ'Ñ)Ô)ˆHØ�8˜O¨VÑ4Ô4ˆDå(ØØ"Ø!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r0   rå   )r^   r_   r`   r$   r   r   rD   ra   rb   r   r   r¹   r»   r   rQ   rd   re   s   @r/   rç   rç   í  s  ø€ € € € € ðð ð ð ð ð Øð .2Ø26Ø37Ø*.Ø04Ø26ðP
ð P
àÔ# dÑ*ðP
ð Ô(¨4Ñ/ðP
ð Ô)¨DÑ0ð	P
ð
 Ô  4Ñ'ðP
ð Ô&¨Ñ-ðP
ð Ô(¨4Ñ/ðP
ð Ð+Ô,ðP
ð 
ˆuŒ|Ô	Ð8Ñ	8ðP
ð P
ð P
ñ „^ñ ÔðP
ð P
ð P
ð P
ð P
r0   rç   c                   óü   ‡ — e Zd Zˆ fd„Zee	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	e	e
         d
eej                 ez  fd„¦   «         ¦   «         Zˆ xZS )ÚRobertaForTokenClassificationc                 óZ  •— t          ¦   «                              |¦  «         |j        | _        t          |d¬¦  «        | _        |j        �|j        n|j        }t          j        |¦  «        | _	        t          j
        |j        |j        ¦  «        | _        |                      ¦   «          d S rÙ   )r#   r$   rÚ   rˆ   rq   Úclassifier_dropoutrê   r'   ré   rK   rË   r*   rÜ   r™   ©r,   r-   rø   r.   s      €r/   r$   z&RobertaForTokenClassification.__init__P  sš   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒå# F¸eÐDÑDÔDˆŒà)/Ô)BÐ)NˆFÔ%Ð%ÐTZÔTnð 	õ ”zÐ"4Ñ5Ô5ˆŒÝœ) FÔ$6¸Ô8IÑJÔJˆŒð 	�ŠÑÔÐÐÐr0   Nr1   r¥   r2   r3   r4   r¨   r¬   r­   c           	      ó�  —  | j         |f||||ddœ|¤Ž}|d         }	|                      |	¦  «        }	|                      |	¦  «        }
d}|�`|                     |
j        ¦  «        }t          ¦   «         } ||
                     d| j        ¦  «        |                     d¦  «        ¦  «        }t          ||
|j	        |j
        ¬¦  «        S )a-  
        token_type_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.
            This parameter can only be used when the model is initialized with `type_vocab_size` parameter with value
            >= 2. All the value in this tensor should be always < type_vocab_size.

            [What are token type IDs?](../glossary#token-type-ids)
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the token classification loss. Indices should be in `[0, ..., config.num_labels - 1]`.
        TrÞ   r   Nr7   rÂ   )rq   rK   rÜ   rA   r<   r   rÃ   rÚ   r   rr   rs   rä   s                r/   rQ   z%RobertaForTokenClassification.forward^  sç   € ð2 �$”,Øð
à)Ø)Ø%Ø'Øð
ð 
ð ð
ð 
ˆð " !œ*ˆàŸ,š, Ñ7Ô7ˆØ—’ Ñ1Ô1ˆàˆØÐà—Y’Y˜vœ}Ñ-Ô-ˆFÝ'Ñ)Ô)ˆHØ�8˜FŸKšK¨¨D¬OÑ<Ô<¸f¿kºkÈ"¹o¼oÑNÔNˆDå$ØØØ!Ô/ØÔ)ð	
ñ 
ô 
ð 	
r0   rå   )r^   r_   r`   r$   r   r   rD   ra   rb   r   r   r¹   r»   r   rQ   rd   re   s   @r/   rö   rö   N  s  ø€ € € € € ðð ð ð ð ð Øð .2Ø37Ø26Ø04Ø26Ø*.ð2
ð 2
àÔ# dÑ*ð2
ð Ô)¨DÑ0ð2
ð Ô(¨4Ñ/ð	2
ð
 Ô&¨Ñ-ð2
ð Ô(¨4Ñ/ð2
ð Ô  4Ñ'ð2
ð Ð+Ô,ð2
ð 
ˆuŒ|Ô	Ð4Ñ	4ð2
ð 2
ð 2
ñ „^ñ Ôð2
ð 2
ð 2
ð 2
ð 2
r0   rö   c                   ó(   ‡ — e Zd ZdZˆ fd„Zd„ Zˆ xZS )rÛ   z-Head for sentence-level classification tasks.c                 ó4  •— t          ¦   «                              ¦   «          t          j        |j        |j        ¦  «        | _        |j        �|j        n|j        }t          j        |¦  «        | _	        t          j        |j        |j
        ¦  «        | _        d S rŠ   )r#   r$   r'   rË   r*   rÌ   rø   rê   ré   rK   rÚ   Úout_projrù   s      €r/   r$   z"RobertaClassificationHead.__init__˜  s   ø€ Ý‰Œ×ÒÑÔÐÝ”Y˜vÔ1°6Ô3EÑFÔFˆŒ
à)/Ô)BÐ)NˆFÔ%Ð%ÐTZÔTnð 	õ ”zÐ"4Ñ5Ô5ˆŒÝœ	 &Ô"4°fÔ6GÑHÔHˆŒˆˆr0   c                 óô   — |d d …dd d …f         }|                       |¦  «        }|                      |¦  «        }t          j        |¦  «        }|                       |¦  «        }|                      |¦  «        }|S )Nr   )rK   rÌ   rD   Útanhrý   rÑ   s       r/   rQ   z!RobertaClassificationHead.forward¡  sj   € Ø�Q�Q�Q˜˜1˜1˜1�WÔˆØ�LŠL˜‰OŒOˆØ�JŠJ�q‰MŒMˆÝŒJ�q‰MŒMˆØ�LŠL˜‰OŒOˆØ�MŠM˜!ÑÔˆØˆr0   rÔ   re   s   @r/   rÛ   rÛ   •  sR   ø€ € € € € Ø7Ð7ðIð Ið Ið Ið Iðð ð ð ð ð ð r0   rÛ   c                   ó  ‡ — e Zd Zˆ fd„Zee	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  dej        dz  dej        dz  d	ej        dz  d
e	e
         deej                 ez  fd„¦   «         ¦   «         Zˆ xZS )ÚRobertaForQuestionAnsweringc                 óþ   •— t          ¦   «                              |¦  «         |j        | _        t          |d¬¦  «        | _        t          j        |j        |j        ¦  «        | _        |  	                    ¦   «          d S rÙ   )
r#   r$   rÚ   rˆ   rq   r'   rË   r*   Ú
qa_outputsr™   r+   s     €r/   r$   z$RobertaForQuestionAnswering.__init__­  sj   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø Ô+ˆŒå# F¸eÐDÑDÔDˆŒÝœ) FÔ$6¸Ô8IÑJÔJˆŒð 	�ŠÑÔÐÐÐr0   Nr1   r¥   r2   r3   r4   Ústart_positionsÚend_positionsr¬   r­   c           	      óF  —  | j         |f||||ddœ|¤Ž}	|	d         }
|                      |
¦  «        }|                     dd¬¦  «        \  }}|                     d¦  «                             ¦   «         }|                     d¦  «                             ¦   «         }d}|�ç|�åt          |                     ¦   «         ¦  «        dk    r|                     d¦  «        }t          |                     ¦   «         ¦  «        dk    r|                     d¦  «        }|                     d¦  «        }|                     d|¦  «        }|                     d|¦  «        }t          |¬¦  «        } |||¦  «        } |||¦  «        }||z   d	z  }t          ||||	j
        |	j        ¬
¦  «        S )a[  
        token_type_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.
            This parameter can only be used when the model is initialized with `type_vocab_size` parameter with value
            >= 2. All the value in this tensor should be always < type_vocab_size.

            [What are token type IDs?](../glossary#token-type-ids)
        TrÞ   r   r   r7   rW   N)Úignore_indexr   )r²   Ústart_logitsÚ
end_logitsrr   rs   )rq   r  Úsplitrã   Ú
contiguousÚlenr?   Úclampr   r   rr   rs   )r,   r1   r¥   r2   r3   r4   r  r  r¬   r¶   rÄ   r°   r  r	  Ú
total_lossÚignored_indexrÇ   Ú
start_lossÚend_losss                      r/   rQ   z#RobertaForQuestionAnswering.forward·  sÏ  € ð0 �$”,Øð
à)Ø)Ø%Ø'Øð
ð 
ð ð
ð 
ˆð " !œ*ˆà—’ Ñ1Ô1ˆØ#)§<¢<°°r <Ñ#:Ô#:Ñ ˆ�jØ#×+Ò+¨BÑ/Ô/×:Ò:Ñ<Ô<ˆØ×'Ò'¨Ñ+Ô+×6Ò6Ñ8Ô8ˆ
àˆ
ØÐ&¨=Ð+Då�?×'Ò'Ñ)Ô)Ñ*Ô*¨QÒ.Ð.Ø"1×"9Ò"9¸"Ñ"=Ô"=�Ý�=×%Ò%Ñ'Ô'Ñ(Ô(¨1Ò,Ð,Ø -× 5Ò 5°bÑ 9Ô 9�à(×-Ò-¨aÑ0Ô0ˆMØ-×3Ò3°A°}ÑEÔEˆOØ)×/Ò/°°=ÑAÔAˆMå'°]ÐCÑCÔCˆHØ!˜ ,°Ñ@Ô@ˆJØ�x 
¨MÑ:Ô:ˆHØ$ xÑ/°1Ñ4ˆJå+ØØ%Ø!Ø!Ô/ØÔ)ð
ñ 
ô 
ð 	
r0   )NNNNNNN)r^   r_   r`   r$   r   r   rD   ra   rb   r   r   r¹   r»   r   rQ   rd   re   s   @r/   r  r  «  s"  ø€ € € € € ðð ð ð ð ð Øð .2Ø37Ø26Ø04Ø26Ø37Ø15ð>
ð >
àÔ# dÑ*ð>
ð Ô)¨DÑ0ð>
ð Ô(¨4Ñ/ð	>
ð
 Ô&¨Ñ-ð>
ð Ô(¨4Ñ/ð>
ð Ô)¨DÑ0ð>
ð Ô'¨$Ñ.ð>
ð Ð+Ô,ð>
ð 
ˆuŒ|Ô	Ð;Ñ	;ð>
ð >
ð >
ñ „^ñ Ôð>
ð >
ð >
ð >
ð >
r0   r  )rŽ   r½   rç   r  r×   rö   rˆ   rp   );rÕ   rD   Útorch.nnr'   r   r   r   Ú r   ry   Úactivationsr   Ú
generationr	   Úmodeling_outputsr
   r   r   r   r   r   r   Úmodeling_utilsr   Úprocessing_utilsr   Úutilsr   r   r   Úutils.genericr   Úbert.modeling_bertr   r   r   r   r   Úconfiguration_robertar   Ú
get_loggerr^   r–   r    rg   rl   rn   rp   rˆ   rŽ   r½   ÚModulerx   r×   rç   rö   rÛ   r  Ú__all__ri   r0   r/   ú<module>r      s°  ðð Ð à €€€Ø Ð Ð Ð Ð Ð Ø AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ Aà &Ð &Ð &Ð &Ð &Ð &Ø Ð Ð Ð Ð Ð Ø )Ð )Ð )Ð )Ð )Ð )ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð .Ð -Ð -Ð -Ð -Ð -Ø &Ð &Ð &Ð &Ð &Ð &Ø @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ø -Ð -Ð -Ð -Ð -Ð -Ø lÐ lÐ lÐ lÐ lÐ lÐ lÐ lÐ lÐ lÐ lÐ lÐ lÐ lØ 0Ð 0Ð 0Ð 0Ð 0Ð 0ð 
ˆÔ	˜HÑ	%Ô	%€ð\8ð \8ð \8ð \8ð \8˜ñ \8ô \8ð \8ð~	ð 	ð 	ð 	ð 	Ð,ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	Ð.ñ 	ô 	ð 	ð	ð 	ð 	ð 	ð 	�9ñ 	ô 	ð 	ð ð/ð /ð /ð /ð /˜_ñ /ô /ñ „ð/ð2'ð 'ð 'ð 'ð '�9ñ 'ô 'ð 'ð
 €ððñ ô ð
i
ð i
ð i
ð i
ð i
Ð/°ñ i
ô i
ñô ð
i
ðX ðR
ð R
ð R
ð R
ð R
Ð/ñ R
ô R
ñ „ðR
ðjð ð ð ð �B”Iñ ô ð ð, €ððñ ô ðQ
ð Q
ð Q
ð Q
ð Q
Ð'=ñ Q
ô Q
ñô ðQ
ðh ð]
ð ]
ð ]
ð ]
ð ]
Ð5ñ ]
ô ]
ñ „ð]
ð@ ðC
ð C
ð C
ð C
ð C
Ð$:ñ C
ô C
ñ „ðC
ðLð ð ð ð  ¤	ñ ô ð ð, ðK
ð K
ð K
ð K
ð K
Ð"8ñ K
ô K
ñ „ðK
ð\	ð 	ð 	€€€r0   