§
    ‚Štj‰ï  ã                   óJ  — d Z ddlZddlmZ ddlZddlmZ ddlmZmZm	Z	 ddl
mZ ddlmZ dd	lmZmZmZ dd
lmZ ddlmZmZ ddlmZ ddlmZ ddlmZmZmZm Z m!Z!m"Z"m#Z# ddl$m%Z%m&Z& ddl'm(Z( ddl)m*Z*m+Z+m,Z,m-Z-m.Z.m/Z/m0Z0 ddl1m2Z2 ddl3m4Z4m5Z5 ddl6m7Z7  e-¦   «         r	  e/j8        e9¦  «        Z:dej;        de<fd„Z= G d„ dej>        ¦  «        Z? G d„ dej>        ¦  «        Z@	 	 dEdejA        dej;        d ej;        d!ej;        d"ej;        dz  d#eBdz  d$eBd%e(e*         fd&„ZC G d'„ d(ejA        ¦  «        ZD G d)„ d*e¦  «        ZE G d+„ d,e¦  «        ZF G d-„ d.ejA        ¦  «        ZGe+ G d/„ d0e&¦  «        ¦   «         ZH G d1„ d2eH¦  «        ZI G d3„ d4eH¦  «        ZJe+ G d5„ d6eH¦  «        ¦   «         ZK e+d7¬8¦  «         G d9„ d:eHe¦  «        ¦   «         ZL e+d;¬8¦  «         G d<„ d=eH¦  «        ¦   «         ZMe+ G d>„ d?eH¦  «        ¦   «         ZN G d@„ dAeH¦  «        ZO G dB„ dCeHe¦  «        ZPg dD¢ZQdS )FzPyTorch MBART model.é    N)ÚCallable)Únn)ÚBCEWithLogitsLossÚCrossEntropyLossÚMSELossé   )Úinitialization)ÚACT2FN)ÚCacheÚDynamicCacheÚEncoderDecoderCache)ÚGenerationMixin)Úcreate_bidirectional_maskÚcreate_causal_mask)ÚFlashAttentionKwargs)ÚGradientCheckpointingLayer)ÚBaseModelOutputÚ)BaseModelOutputWithPastAndCrossAttentionsÚ!CausalLMOutputWithCrossAttentionsÚSeq2SeqLMOutputÚSeq2SeqModelOutputÚ#Seq2SeqQuestionAnsweringModelOutputÚSeq2SeqSequenceClassifierOutput)ÚALL_ATTENTION_FUNCTIONSÚPreTrainedModel)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚcan_return_tupleÚis_torch_flex_attn_availableÚis_torchdynamo_compilingÚloggingÚtorch_compilable_check)Úmerge_with_config_defaults)ÚOutputRecorderÚcapture_outputsé   )ÚMBartConfigÚ	input_idsÚpad_token_idc                 ó¶  — |                       ¦   «         }|€t          d¦  «        ‚|                     |dk    |¦  «         |                     |¦  «                             d¬¦  «        dz
                       d¦  «        }|                     d|¦  «                             ¦   «         }|dd…dd…f                               ¦   «         |dd…dd…f<   ||dd…df<   |S )zÎ
    Shift input ids one token to the right, and wrap the last non pad token (the <LID> token) Note that MBart does not
    have a single `decoder_start_token_id` in contrast to other Bart-like models.
    Nz1self.model.config.pad_token_id has to be defined.iœÿÿÿr'   ©Údiméÿÿÿÿr   )ÚcloneÚ
ValueErrorÚmasked_fill_ÚneÚsumÚ	unsqueezeÚgatherÚsqueeze)r)   r*   Úprev_output_tokensÚindex_of_eosÚdecoder_start_tokenss        úf/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/mbart/modeling_mbart.pyÚshift_tokens_rightr;   @   sì   € ð
 #ŸšÑ*Ô*ÐàÐÝÐLÑMÔMÐMà×#Ò#Ð$6¸$Ò$>ÀÑMÔMÐMà&×)Ò)¨,Ñ7Ô7×;Ò;ÀÐ;ÑBÔBÀQÑF×QÒQÐRTÑUÔU€LØ-×4Ò4°Q¸ÑEÔE×MÒMÑOÔOÐØ 2°1°1°1°c°r°c°6Ô :× @Ò @Ñ BÔ BÐ�q�q�q˜!˜"˜"�uÑØ3Ð�q�q�q˜!�tÑàÐó    c                   ób   ‡ — e Zd ZdZdedefˆ fd„Z	 ddej        ded	ej        dz  fˆ fd
„Zˆ xZ	S )ÚMBartLearnedPositionalEmbeddingzN
    This module learns positional embeddings up to a fixed maximum size.
    Únum_embeddingsÚembedding_dimc                 ój   •— d| _         t          ¦   «                              || j         z   |¦  «         d S ©Né   )ÚoffsetÚsuperÚ__init__)Úselfr?   r@   Ú	__class__s      €r:   rF   z(MBartLearnedPositionalEmbedding.__init__Z   s3   ø€ ð ˆŒÝ‰Œ×Ò˜¨$¬+Ñ5°}ÑEÔEÐEÐEÐEr<   r   Nr)   Úpast_key_values_lengthÚposition_idsc                 ó0  •— |€V|j         dd…         \  }}t          j        |||z   t          j        | j        j        ¬¦  «                             |d¦  «        }n|                     d¦  «        }t          ¦   «          	                    || j
        z   ¦  «        S )z3`input_ids' shape is expected to be [bsz x seqlen].NrC   )ÚdtypeÚdevicer.   r   )ÚshapeÚtorchÚarangeÚlongÚweightrM   Úexpandr4   rE   ÚforwardrD   )rG   r)   rI   rJ   ÚbszÚseq_lenrH   s         €r:   rT   z'MBartLearnedPositionalEmbedding.forward`   s“   ø€ ð
 ÐØ$œ?¨2¨A¨2Ô.‰LˆC�Ý œ<Ø&Ð(>ÀÑ(HÕPUÔPZÐcgÔcnÔcuðñ ô çŠf�S˜"‰oŒoð ˆLð (×1Ò1°!Ñ4Ô4ˆLå‰wŒw�Š˜|¨d¬kÑ9Ñ:Ô:Ð:r<   )r   N)
Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚintrF   rO   ÚTensorrT   Ú__classcell__©rH   s   @r:   r>   r>   U   sª   ø€ € € € € ðð ðF sð F¸3ð Fð Fð Fð Fð Fð Fð mqð;ð ;Øœð;Ø?Bð;ØV[ÔVbÐeiÑVið;ð ;ð ;ð ;ð ;ð ;ð ;ð ;ð ;ð ;r<   r>   c            
       óV   ‡ — e Zd ZdZddededededz  fˆ fd„Zd	ej        fˆ fd
„Z	ˆ xZ
S )ÚMBartScaledWordEmbeddingz\
    This module overrides nn.Embeddings' forward by multiplying with embeddings scale.
    ç      ð?r?   r@   Úpadding_idxÚembed_scaleNc                 ó\   •— t          ¦   «                              |||¦  «         || _        d S ©N)rE   rF   rc   )rG   r?   r@   rb   rc   rH   s        €r:   rF   z!MBartScaledWordEmbedding.__init__v   s-   ø€ Ý‰Œ×Ò˜¨¸ÑDÔDÐDØ&ˆÔÐÐr<   r)   c                 óV   •— t          ¦   «                              |¦  «        | j        z  S re   )rE   rT   rc   )rG   r)   rH   s     €r:   rT   z MBartScaledWordEmbedding.forwardz   s!   ø€ Ý‰wŒw�Š˜yÑ)Ô)¨DÔ,<Ñ<Ð<r<   )ra   ©rW   rX   rY   rZ   r[   ÚfloatrF   rO   r\   rT   r]   r^   s   @r:   r`   r`   q   s–   ø€ € € € € ðð ð'ð ' sð '¸3ð 'ÈSð 'Ð_dÐgkÑ_kð 'ð 'ð 'ð 'ð 'ð 'ð= ¤ð =ð =ð =ð =ð =ð =ð =ð =ð =ð =r<   r`   ç        ÚmoduleÚqueryÚkeyÚvalueÚattention_maskÚscalingÚdropoutÚkwargsc                 ó®  — |€|                      d¦  «        dz  }t          j        ||                     dd¦  «        ¦  «        |z  }|�||z   }t          j                             |d¬¦  «        }t          j                             ||| j        ¬¦  «        }t          j        ||¦  «        }	|	                     dd¦  «         	                    ¦   «         }	|	|fS )Nr.   ç      à¿rC   r   r,   ©ÚpÚtrainingr'   )
ÚsizerO   ÚmatmulÚ	transposer   Ú
functionalÚsoftmaxrp   rv   Ú
contiguous)
rj   rk   rl   rm   rn   ro   rp   rq   Úattn_weightsÚattn_outputs
             r:   Úeager_attention_forwardr      sÈ   € ð €Ø—*’*˜R‘.”. DÑ(ˆõ ”<  s§}¢}°Q¸Ñ':Ô':Ñ;Ô;¸gÑE€LàÐ!Ø# nÑ4ˆå”=×(Ò(¨¸2Ð(Ñ>Ô>€LÝ”=×(Ò(¨¸È6Ì?Ð(Ñ[Ô[€Lå”,˜|¨UÑ3Ô3€KØ×'Ò'¨¨1Ñ-Ô-×8Ò8Ñ:Ô:€Kà˜Ð$Ð$r<   c                   óì   ‡ — e Zd ZdZ	 	 	 	 	 	 ddededed	ed
edededz  dedz  fˆ fd„Z	 	 	 dde	j
        de	j
        dz  dedz  de	j
        dz  dee         dee	j
        e	j
        dz  f         fd„Zˆ xZS )ÚMBartAttentionz=Multi-headed attention from 'Attention Is All You Need' paperri   FTNÚ	embed_dimÚ	num_headsrp   Ú
is_decoderÚbiasÚ	is_causalÚconfigÚ	layer_idxc	                 óz  •— t          ¦   «                              ¦   «          || _        || _        || _        ||z  | _        || _        | j        |z  | j        k    rt          d| j        › d|› d�¦  «        ‚| j        dz  | _        || _	        || _
        || _        |€/| j	        r(t                               d| j        j        › d�¦  «         t!          j        |||¬¦  «        | _        t!          j        |||¬¦  «        | _        t!          j        |||¬¦  «        | _        t!          j        |||¬¦  «        | _        d S )Nz;embed_dim must be divisible by num_heads (got `embed_dim`: z and `num_heads`: z).rs   zInstantiating a decoder z¸ without passing `layer_idx` is not recommended and will lead to errors during the forward call, if caching is used. Please make sure to provide a `layer_idx` when creating this class.©r…   )rE   rF   r‚   rƒ   rp   Úhead_dimr‡   r0   ro   r„   r†   rˆ   ÚloggerÚwarning_oncerH   rW   r   ÚLinearÚk_projÚv_projÚq_projÚout_proj)
rG   r‚   rƒ   rp   r„   r…   r†   r‡   rˆ   rH   s
            €r:   rF   zMBartAttention.__init__Ÿ   sY  ø€ õ 	‰Œ×ÒÑÔÐØ"ˆŒØ"ˆŒØˆŒØ! YÑ.ˆŒØˆŒàŒM˜IÑ%¨$¬.Ò8Ð8Ýð3ÈdÌnð 3ð 3Ø%.ð3ð 3ð 3ñô ð ð ”} dÑ*ˆŒØ$ˆŒØ"ˆŒØ"ˆŒØÐ ¤ÐÝ×Òð,¨4¬>Ô+Bð ,ð ,ð ,ñô ð õ ”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝ”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝ”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝœ	 )¨Y¸TÐBÑBÔBˆŒˆˆr<   Úhidden_statesÚkey_value_statesÚpast_key_valuesrn   rq   Úreturnc                 óŽ  — |du}|j         dd…         }g |¢d‘| j        ‘R }|                      |¦  «                             |¦  «                             dd¦  «        }	d}
|�Ht          |t          ¦  «        r1|j                             | j	        ¦  «        }
|r|j
        }n
|j        }n|}|r|n|}|r3|�1|
r/|j        | j	                 j        }|j        | j	                 j        }nÞ|                      |¦  «        }|                      |¦  «        }g |j         dd…         ¢d‘| j        ‘R }|                     |¦  «                             dd¦  «        }|                     |¦  «                             dd¦  «        }|�E|                     ||| j	        ¦  «        \  }}|r$t          |t          ¦  «        rd|j        | j	        <   t%          j        | j        j        t,          ¦  «        } || |	|||f| j        sdn| j        | j        dœ|¤Ž\  }} |j        g |¢d‘R Ž                      ¦   «         }|                      |¦  «        }||fS )	z#Input shape: Batch x Time x ChannelNr.   r'   rC   FTri   )rp   ro   )rN   r‹   r‘   Úviewry   Ú
isinstancer   Ú
is_updatedÚgetrˆ   Úcross_attention_cacheÚself_attention_cacheÚlayersÚkeysÚvaluesr�   r�   Úupdater   Úget_interfacer‡   Ú_attn_implementationr   rv   rp   ro   Úreshaper|   r’   )rG   r“   r”   r•   rn   rq   Úis_cross_attentionÚinput_shapeÚhidden_shapeÚquery_statesrš   Úcurr_past_key_valuesÚcurrent_statesÚ
key_statesÚvalue_statesÚkv_shapeÚattention_interfacer~   r}   s                      r:   rT   zMBartAttention.forwardÆ   s¨  € ð .°TÐ9Ðð $Ô)¨#¨2¨#Ô.ˆà8˜Ð8 bÐ8¨$¬-Ð8Ð8ˆð —{’{ =Ñ1Ô1×6Ò6°|ÑDÔD×NÒNÈqÐRSÑTÔTˆàˆ
ØÐ&Ý˜/Õ+>Ñ?Ô?ð 7Ø,Ô7×;Ò;¸D¼NÑKÔK�
Ø%ð Pà+:Ô+PÐ(Ð(à+:Ô+OÐ(Ð(à'6Ð$à-?ÐRÐ)Ð)À]ˆØð 	F /Ð"=À*Ð"=à-Ô4°T´^ÔDÔIˆJØ/Ô6°t´~ÔFÔMˆLˆLàŸš ^Ñ4Ô4ˆJØŸ;š; ~Ñ6Ô6ˆLØF˜Ô-¨c¨r¨cÔ2ÐF°BÐF¸¼ÐFÐFˆHØ#Ÿš¨Ñ2Ô2×<Ò<¸QÀÑBÔBˆJØ'×,Ò,¨XÑ6Ô6×@Ò@ÀÀAÑFÔFˆLàÐ*Ø+?×+FÒ+FÀzÐS_ÐaeÔaoÑ+pÔ+pÑ(�
˜Là%ð F­*°_ÕFYÑ*ZÔ*Zð FØAE�OÔ.¨t¬~Ñ>å(?Ô(MØŒKÔ,Õ.Eñ)
ô )
Ðð %8Ð$7ØØØØØð	%
ð  $œ}Ð>�C�C°$´,Ø”Lð	%
ð 	%
ð ð	%
ð 	%
Ñ!ˆ�\ð *�kÔ)Ð;¨;Ð;¸Ð;Ð;Ð;×FÒFÑHÔHˆØ—m’m KÑ0Ô0ˆà˜LÐ(Ð(r<   )ri   FTFNN©NNN)rW   rX   rY   rZ   r[   rh   Úboolr(   rF   rO   r\   r   r   r   ÚtuplerT   r]   r^   s   @r:   r�   r�   œ   s]  ø€ € € € € ØGÐGð Ø ØØØ%)Ø $ð%Cð %Càð%Cð ð%Cð ð	%Cð
 ð%Cð ð%Cð ð%Cð ˜dÑ"ð%Cð ˜‘:ð%Cð %Cð %Cð %Cð %Cð %CðT 15Ø(,Ø.2ðH)ð H)à”|ðH)ð  œ,¨Ñ-ðH)ð  ™ð	H)ð
 œ tÑ+ðH)ð Ð-Ô.ðH)ð 
ˆuŒ|˜Uœ\¨DÑ0Ð0Ô	1ðH)ð H)ð H)ð H)ð H)ð H)ð H)ð H)r<   r�   c                   óf   ‡ — e Zd Zdefˆ fd„Zdej        dej        dee         dej        fd„Z	ˆ xZ
S )ÚMBartEncoderLayerr‡   c                 ó  •— t          ¦   «                              ¦   «          |j        | _        t	          | j        |j        |j        |¬¦  «        | _        t          j	        | j        ¦  «        | _
        |j        | _        t          |j                 | _        |j        | _        t          j        | j        |j        ¦  «        | _        t          j        |j        | j        ¦  «        | _        t          j	        | j        ¦  «        | _        d S )N)r‚   rƒ   rp   r‡   )rE   rF   Úd_modelr‚   r�   Úencoder_attention_headsÚattention_dropoutÚ	self_attnr   Ú	LayerNormÚself_attn_layer_normrp   r
   Úactivation_functionÚactivation_fnÚactivation_dropoutrŽ   Úencoder_ffn_dimÚfc1Úfc2Úfinal_layer_norm©rG   r‡   rH   s     €r:   rF   zMBartEncoderLayer.__init__  sÐ   ø€ Ý‰Œ×ÒÑÔÐØœˆŒå'Ø”nØÔ4ØÔ,Øð	
ñ 
ô 
ˆŒõ %'¤L°´Ñ$@Ô$@ˆÔ!Ø”~ˆŒÝ# FÔ$>Ô?ˆÔØ"(Ô";ˆÔÝ”9˜Tœ^¨VÔ-CÑDÔDˆŒÝ”9˜VÔ3°T´^ÑDÔDˆŒÝ "¤¨T¬^Ñ <Ô <ˆÔÐÐr<   r“   rn   rq   r–   c                 óº  — |}|                       |¦  «        } | j        d||dœ|¤Ž\  }}t          j                             || j        | j        ¬¦  «        }||z   }|}|                      |¦  «        }|                      |                      |¦  «        ¦  «        }t          j                             || j	        | j        ¬¦  «        }|  
                    |¦  «        }t          j                             || j        | j        ¬¦  «        }||z   }|j        t          j        k    r9t          j        |j        ¦  «        j        dz
  }t          j        || |¬¦  «        }|S )a>  
        Args:
            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
            attention_mask (`torch.FloatTensor`): attention mask of size
                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
        )r“   rn   rt   iè  )ÚminÚmax© )rº   r¸   r   rz   rp   rv   rÁ   r¼   r¿   r½   rÀ   rL   rO   Úfloat16ÚfinforÅ   Úclamp)rG   r“   rn   rq   ÚresidualÚ_Úclamp_values          r:   rT   zMBartEncoderLayer.forward$  s[  € ð !ˆØ×1Ò1°-Ñ@Ô@ˆØ)˜4œ>ð 
Ø'Ø)ð
ð 
ð ð
ð 
Ñˆ�qõ
 œ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆà ˆØ×-Ò-¨mÑ<Ô<ˆØ×*Ò*¨4¯8ª8°MÑ+BÔ+BÑCÔCˆÝœ×-Ò-¨m¸tÔ?VÐaeÔanÐ-ÑoÔoˆØŸš Ñ/Ô/ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆàÔ¥%¤-Ò/Ð/Ýœ+ mÔ&9Ñ:Ô:Ô>ÀÑEˆKÝ!œK¨¸K¸<È[ÐYÑYÔYˆMàÐr<   )rW   rX   rY   r(   rF   rO   r\   r   r   rT   r]   r^   s   @r:   r³   r³     sŠ   ø€ € € € € ð=˜{ð =ð =ð =ð =ð =ð =ð$"à”|ð"ð œð"ð Ð+Ô,ð	"ð
 
Œð"ð "ð "ð "ð "ð "ð "ð "r<   r³   c                   óÀ   ‡ — e Zd Zddededz  fˆ fd„Z	 	 	 	 	 ddej        dej        dz  dej        dz  d	ej        dz  d
edz  de	dz  de
e         dej        fd„Zˆ xZS )ÚMBartDecoderLayerNr‡   rˆ   c           	      ó¨  •— t          ¦   «                              ¦   «          |j        | _        t	          | j        |j        |j        dd||¬¦  «        | _        |j        | _        t          |j
                 | _        |j        | _        t          j        | j        ¦  «        | _        t	          | j        |j        |j        d||¬¦  «        | _        t          j        | j        ¦  «        | _        t          j        | j        |j        ¦  «        | _        t          j        |j        | j        ¦  «        | _        t          j        | j        ¦  «        | _        d S )NT)r‚   rƒ   rp   r„   r†   r‡   rˆ   )rp   r„   r‡   rˆ   )rE   rF   rµ   r‚   r�   Údecoder_attention_headsr·   r¸   rp   r
   r»   r¼   r½   r   r¹   rº   Úencoder_attnÚencoder_attn_layer_normrŽ   Údecoder_ffn_dimr¿   rÀ   rÁ   )rG   r‡   rˆ   rH   s      €r:   rF   zMBartDecoderLayer.__init__J  s   ø€ Ý‰Œ×ÒÑÔÐØœˆŒå'Ø”nØÔ4ØÔ,ØØØØð
ñ 
ô 
ˆŒð ”~ˆŒÝ# FÔ$>Ô?ˆÔØ"(Ô";ˆÔå$&¤L°´Ñ$@Ô$@ˆÔ!Ý*ØŒNØÔ*ØÔ,ØØØð
ñ 
ô 
ˆÔõ (*¤|°D´NÑ'CÔ'CˆÔ$Ý”9˜Tœ^¨VÔ-CÑDÔDˆŒÝ”9˜VÔ3°T´^ÑDÔDˆŒÝ "¤¨T¬^Ñ <Ô <ˆÔÐÐr<   Tr“   rn   Úencoder_hidden_statesÚencoder_attention_maskr•   Ú	use_cacherq   r–   c                 óÞ  — |}|                       |¦  «        } | j        d|||dœ|¤Ž\  }}	t          j                             || j        | j        ¬¦  «        }||z   }|�]|}|                      |¦  «        } | j        d||||dœ|¤Ž\  }}	t          j                             || j        | j        ¬¦  «        }||z   }|}|                      |¦  «        }|  	                    |  
                    |¦  «        ¦  «        }t          j                             || j        | j        ¬¦  «        }|                      |¦  «        }t          j                             || j        | j        ¬¦  «        }||z   }|S )að  
        Args:
            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
            attention_mask (`torch.FloatTensor`): attention mask of size
                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
            encoder_hidden_states (`torch.FloatTensor`):
                cross attention input to the layer of shape `(batch, seq_len, embed_dim)`
            encoder_attention_mask (`torch.FloatTensor`): encoder attention mask of size
                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
            past_key_values (`Cache`): cached past key and value projection states
        )r“   r•   rn   rt   N)r“   r”   rn   r•   rÆ   )rº   r¸   r   rz   rp   rv   rÒ   rÑ   rÁ   r¼   r¿   r½   rÀ   )
rG   r“   rn   rÔ   rÕ   r•   rÖ   rq   rÊ   rË   s
             r:   rT   zMBartDecoderLayer.forwardi  s§  € ð* !ˆØ×1Ò1°-Ñ@Ô@ˆð *˜4œ>ð 
Ø'Ø+Ø)ð
ð 
ð ð	
ð 
Ñˆ�qõ œ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆð !Ð,Ø$ˆHØ ×8Ò8¸ÑGÔGˆMà0˜tÔ0ð  Ø+Ø!6Ø5Ø /ð	 ð  ð
 ð ð  ÑˆM˜1õ œM×1Ò1°-À4Ä<ÐZ^ÔZgÐ1ÑhÔhˆMØ$ }Ñ4ˆMð !ˆØ×-Ò-¨mÑ<Ô<ˆØ×*Ò*¨4¯8ª8°MÑ+BÔ+BÑCÔCˆÝœ×-Ò-¨m¸tÔ?VÐaeÔanÐ-ÑoÔoˆØŸš Ñ/Ô/ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆàÐr<   re   )NNNNT)rW   rX   rY   r(   r[   rF   rO   r\   r   r°   r   r   rT   r]   r^   s   @r:   rÎ   rÎ   I  sô   ø€ € € € € ð=ð =˜{ð =°s¸T±zð =ð =ð =ð =ð =ð =ðD /3Ø59Ø6:Ø(,Ø!%ð:ð :à”|ð:ð œ tÑ+ð:ð  %œ|¨dÑ2ð	:ð
 !&¤¨tÑ 3ð:ð  ™ð:ð ˜$‘;ð:ð Ð+Ô,ð:ð 
Œð:ð :ð :ð :ð :ð :ð :ð :r<   rÎ   c                   óX   ‡ — e Zd ZdZdedededefˆ fd„Zdej        dej        fd	„Z	ˆ xZ
S )
ÚMBartClassificationHeadz-Head for sentence-level classification tasks.Ú	input_dimÚ	inner_dimÚnum_classesÚpooler_dropoutc                 óä   •— t          ¦   «                              ¦   «          t          j        ||¦  «        | _        t          j        |¬¦  «        | _        t          j        ||¦  «        | _        d S )N)ru   )rE   rF   r   rŽ   ÚdenseÚDropoutrp   r’   )rG   rÚ   rÛ   rÜ   rÝ   rH   s        €r:   rF   z MBartClassificationHead.__init__ª  sY   ø€ õ 	‰Œ×ÒÑÔÐÝ”Y˜y¨)Ñ4Ô4ˆŒ
Ý”z NÐ3Ñ3Ô3ˆŒÝœ	 )¨[Ñ9Ô9ˆŒˆˆr<   r“   r–   c                 óÖ   — |                       |¦  «        }|                      |¦  «        }t          j        |¦  «        }|                       |¦  «        }|                      |¦  «        }|S re   )rp   rß   rO   Útanhr’   )rG   r“   s     r:   rT   zMBartClassificationHead.forward¶  s[   € ØŸš ]Ñ3Ô3ˆØŸ
š
 =Ñ1Ô1ˆÝœ
 =Ñ1Ô1ˆØŸš ]Ñ3Ô3ˆØŸš mÑ4Ô4ˆØÐr<   rg   r^   s   @r:   rÙ   rÙ   §  s�   ø€ € € € € Ø7Ð7ð
:àð
:ð ð
:ð ð	
:ð
 ð
:ð 
:ð 
:ð 
:ð 
:ð 
:ð U¤\ð °e´lð ð ð ð ð ð ð ð r<   rÙ   c                   ó`   ‡ — e Zd ZU eed<   dZdZg d¢ZdZdZ	dZ
dZˆ fd„Zed„ ¦   «         Zˆ xZS )ÚMBartPreTrainedModelr‡   ÚmodelT)rÎ   r³   r�   c                 óª   •— t          ¦   «                              |¦  «         t          |t          ¦  «        rt	          j        |j        ¦  «         d S d S re   )rE   Ú_init_weightsr™   ÚMBartForConditionalGenerationÚinitÚzeros_Úfinal_logits_bias)rG   rj   rH   s     €r:   rç   z"MBartPreTrainedModel._init_weightsÊ  sQ   ø€ Ý‰Œ×Ò˜fÑ%Ô%Ð%Ý�fÕ;Ñ<Ô<ð 	2ÝŒK˜Ô0Ñ1Ô1Ð1Ð1Ð1ð	2ð 	2r<   c                 ó–   — | j         j        }t          j        g d¢dddd|gg| j        ¬¦  «        }|                     |¦  «        |dœ}|S )N)r   é   é
   é   rC   r   é   é   rC   ©rM   )rn   r)   )r‡   r*   rO   ÚtensorrM   r2   )rG   Ú	pad_tokenr)   Údummy_inputss       r:   rõ   z!MBartPreTrainedModel.dummy_inputsÏ  sa   € à”KÔ,ˆ	Ý”LÐ"2Ð"2Ð"2°Q¸¸2¸qÀ)Ð4LÐ!MÐVZÔVaÐbÑbÔbˆ	à'Ÿlšl¨9Ñ5Ô5Ø"ð
ð 
ˆð Ðr<   )rW   rX   rY   r(   Ú__annotations__Úbase_model_prefixÚsupports_gradient_checkpointingÚ_no_split_modulesÚ_supports_flash_attnÚ_supports_sdpaÚ_supports_flex_attnÚ_can_compile_fullgraphrç   Úpropertyrõ   r]   r^   s   @r:   rä   rä   ¿  s�   ø€ € € € € € àÐÐÑØÐØ&*Ð#ØTÐTÐTÐØÐØ€NØÐØ!Ðð2ð 2ð 2ð 2ð 2ð
 ðð ñ „Xðð ð ð ð r<   rä   c                   óÖ   ‡ — e Zd ZdZe eedd¬¦  «        dœZdefˆ fd„Z	d„ Z
ee	 	 	 dd
ej        d	z  dej        d	z  dej        d	z  dee         deez  f
d„¦   «         ¦   «         Zˆ xZS )ÚMBartEncoderzâ
    Transformer encoder consisting of *config.encoder_layers* self attention layers. Each layer is a
    [`MBartEncoderLayer`].

    Args:
        config: MBartConfig
        embed_tokens (nn.Embedding): output embedding
    r'   r¸   ©ÚindexÚ
layer_name)r“   Ú
attentionsr‡   c                 óŒ  •‡— t          ¦   «                              ‰¦  «         ‰j        | _        ‰j        | _        ‰j        }‰j        | _        ‰j        | _	        ‰j
        rt          j        |¦  «        nd}t          ‰j        || j        |¬¦  «        | _        t!          ‰j        |¦  «        | _        t%          j        ˆfd„t)          ‰j        ¦  «        D ¦   «         ¦  «        | _        ‰| _        t%          j        |¦  «        | _        t%          j        ‰j        ¦  «        | _        d| _        |                      ¦   «          d S )Nra   ©rc   c                 ó.   •— g | ]}t          ‰¦  «        ‘ŒS rÆ   )r³   )Ú.0rË   r‡   s     €r:   ú
<listcomp>z)MBartEncoder.__init__.<locals>.<listcomp>ü  s"   ø€ Ð$eÐ$eÐ$eÀ1Õ%6°vÑ%>Ô%>Ð$eÐ$eÐ$er<   F)rE   rF   rp   Úencoder_layerdropÚ	layerdroprµ   r*   rb   Úmax_position_embeddingsÚmax_source_positionsÚscale_embeddingÚmathÚsqrtr`   Ú
vocab_sizeÚembed_tokensr>   Úembed_positionsr   Ú
ModuleListÚrangeÚencoder_layersrž   r‡   r¹   Úlayernorm_embeddingÚ
layer_normÚgradient_checkpointingÚ	post_init)rG   r‡   r‚   rc   rH   s    `  €r:   rF   zMBartEncoder.__init__é  s(  øø€ Ý‰Œ×Ò˜Ñ Ô Ð à”~ˆŒØÔ1ˆŒà”Nˆ	Ø!Ô.ˆÔØ$*Ô$BˆÔ!Ø.4Ô.DÐM•d”i 	Ñ*Ô*Ð*È#ˆå4ØÔ˜y¨$Ô*:Èð
ñ 
ô 
ˆÔõ  ?ØÔ*Øñ 
ô  
ˆÔõ ”mÐ$eÐ$eÐ$eÐ$eÍÈfÔNcÑHdÔHdÐ$eÑ$eÔ$eÑfÔfˆŒØˆŒÝ#%¤<°	Ñ#:Ô#:ˆÔ Ýœ, v¤~Ñ6Ô6ˆŒà&+ˆÔ#à�ŠÑÔÐÐÐr<   c                 óp   — | j         r,t          | j        dd¦  «        r|                      ¦   «          d S d S d S )Nr  F)rø   Úgetattrr‡   Úgradient_checkpointing_enable©rG   s    r:   Ú._backward_compatibility_gradient_checkpointingz;MBartEncoder._backward_compatibility_gradient_checkpointing  sP   € àÔ/ð 	1µG¸D¼KÐIaÐchÑ4iÔ4ið 	1Ø×.Ò.Ñ0Ô0Ð0Ð0Ð0ð	1ð 	1ð 	1ð 	1r<   Nr)   rn   Úinputs_embedsrq   r–   c                 ój  — |du |duz  rt          d¦  «        ‚|€|                      |¦  «        }|                      |d         ¦  «        }||                     |j        ¦  «        z   }|                      |¦  «        }t          j                             || j        | j	        ¬¦  «        }t          | j        ||¬¦  «        }t          | j        ¦  «        D ];\  }}d}	| j	        r!t          j        g ¦  «        }
|
| j        k     rd}	|	s
 |||fi |¤Ž}Œ<|                      |¦  «        }t%          |¬¦  «        S )	a  
        Args:
            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you
                provide it.

                Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
                [`PreTrainedTokenizer.__call__`] for details.

                [What are input IDs?](../glossary#input-ids)
            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
                Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation.
                This is useful if you want more control over how to convert `input_ids` indices into associated vectors
                than the model's internal embedding lookup matrix.
        Nz:You must specify exactly one of input_ids or inputs_embeds).r.   rt   )r‡   r   rn   FT)Úlast_hidden_state)r0   r  r  ÚtorM   r  r   rz   rp   rv   r   r‡   Ú	enumeraterž   rO   Úrandr  r  r   )rG   r)   rn   r   rq   Ú	embed_posr“   ÚidxÚencoder_layerÚto_dropÚdropout_probabilitys              r:   rT   zMBartEncoder.forward
  sk  € ð> ˜Ð -°tÐ";Ñ<ð 	[ÝÐYÑZÔZÐZàÐ Ø ×-Ò-¨iÑ8Ô8ˆMà×(Ò(¨°wÔ)?Ñ@Ô@ˆ	à%¨	¯ª°]Ô5IÑ(JÔ(JÑJˆØ×0Ò0°Ñ?Ô?ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆå2Ø”;Ø'Ø)ð
ñ 
ô 
ˆõ #,¨D¬KÑ"8Ô"8ð 	ð 	ÑˆC�àˆGØŒ}ð #Ý&+¤j°¡n¤nÐ#Ø&¨¬Ò7Ð7Ø"�Gàð Ø - Ø!Ø"ð!ð !ð ð!ð !�øð Ÿš¨Ñ6Ô6ˆå°Ð?Ñ?Ô?Ð?r<   r¯   )rW   rX   rY   rZ   r³   r%   r�   Ú_can_record_outputsr(   rF   r  r$   r&   rO   Ú
LongTensorr\   ÚFloatTensorr   r   r±   r   rT   r]   r^   s   @r:   r   r   Ú  s,  ø€ € € € € ðð ð +Ø$�n ^¸1ÈÐUÑUÔUðð Ðð
˜{ð ð ð ð ð ð ð81ð 1ð 1ð
  Øð .2Ø.2Ø26ð	@@ð @@àÔ# dÑ*ð@@ð œ tÑ+ð@@ð Ô(¨4Ñ/ð	@@ð
 Ð+Ô,ð@@ð 
�Ñ	 ð@@ð @@ð @@ñ „_ñ  Ôð@@ð @@ð @@ð @@ð @@r<   r   c                   ó.  ‡ — e Zd ZdZe eedd¬¦  «         eedd¬¦  «        dœZdefˆ fd„Z	e
e	 	 	 	 	 	 	 dd
ej        d	z  dej        d	z  dej        d	z  dej        d	z  ded	z  dej        d	z  ded	z  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )ÚMBartDecoderzÎ
    Transformer decoder consisting of *config.decoder_layers* layers. Each layer is a [`MBartDecoderLayer`]

    Args:
        config: MBartConfig
        embed_tokens (nn.Embedding): output embedding
    r'   r¸   r  rÑ   )r“   r  Úcross_attentionsr‡   c                 ó¦  •‡— t          ¦   «                              ‰¦  «         ‰j        | _        ‰j        | _        ‰j        | _        ‰j        | _        ‰j	        rt          j        ‰j        ¦  «        nd}t          ‰j        ‰j        | j        |¬¦  «        | _        t!          ‰j        ‰j        ¦  «        | _        t%          j        ˆfd„t)          ‰j        ¦  «        D ¦   «         ¦  «        | _        ‰| _        t%          j        ‰j        ¦  «        | _        t%          j        ‰j        ¦  «        | _        d| _        |                      ¦   «          d S )Nra   r  c                 ó2   •— g | ]}t          ‰|¬ ¦  «        ‘ŒS ))rˆ   )rÎ   )r  Úir‡   s     €r:   r	  z)MBartDecoder.__init__.<locals>.<listcomp>n  s(   ø€ Ð$rÐ$rÐ$rÐPQÕ%6°vÈÐ%KÑ%KÔ%KÐ$rÐ$rÐ$rr<   F)rE   rF   rp   Údecoder_layerdropr  r*   rb   r  Úmax_target_positionsr  r  r  rµ   r`   r  r  r>   r  r   r  r  Údecoder_layersrž   r‡   r¹   r  r  r  r  )rG   r‡   rc   rH   s    ` €r:   rF   zMBartDecoder.__init__^  s+  øø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø”~ˆŒØÔ1ˆŒØ!Ô.ˆÔØ$*Ô$BˆÔ!Ø39Ô3IÐR•d”i ¤Ñ/Ô/Ð/Èsˆå4ØÔ˜vœ~¨tÔ/?È[ð
ñ 
ô 
ˆÔõ  ?ØÔ*ØŒNñ 
ô  
ˆÔõ ”mÐ$rÐ$rÐ$rÐ$rÕUZÐ[aÔ[pÑUqÔUqÐ$rÑ$rÔ$rÑsÔsˆŒØˆŒå#%¤<°´Ñ#?Ô#?ˆÔ Ýœ, v¤~Ñ6Ô6ˆŒà&+ˆÔ#à�ŠÑÔÐÐÐr<   Nr)   rn   rÔ   rÕ   r•   r   rÖ   rq   r–   c                 óš  — |du |duz  rt          d¦  «        ‚|€|                      |¦  «        }|r[|€Y|€| j        j        r6t	          t          | j        ¬¦  «        t          | j        ¬¦  «        ¦  «        nt          | j        ¬¦  «        }|                     ¦   «         dd…         \  }	}
|�|                     ¦   «         nd}t          j	        |
|j
        ¬¦  «        |z   }|€/t          ¦   «         s!||
z   }t          j        |	||j
        ¬¦  «        }t          |t          ¦  «        r|j        n|}t          | j        |||¬¦  «        }t!          | j        |||¬¦  «        }|                      |||¬	¦  «        }||                     |j
        ¦  «        z   }|                      |¦  «        }t(          j                             || j        | j        ¬
¦  «        }t1          | j        ¦  «        D ];\  }}| j        r t          j        g ¦  «        }|| j        k     rŒ, ||||f|||dœ|¤Ž}Œ<|                      |¦  «        }t;          ||¬¦  «        S )a(  
        Args:
            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you
                provide it.

                Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
                [`PreTrainedTokenizer.__call__`] for details.

                [What are input IDs?](../glossary#input-ids)
            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            encoder_hidden_states (`torch.FloatTensor` of shape `(batch_size, encoder_sequence_length, hidden_size)`, *optional*):
                Sequence of hidden-states at the output of the last layer of the encoder. Used in the cross-attention
                of the decoder.
            encoder_attention_mask (`torch.LongTensor` of shape `(batch_size, encoder_sequence_length)`, *optional*):
                Mask to avoid performing cross-attention on padding tokens indices of encoder input_ids. Mask values
                selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
                It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

                Contains pre-computed hidden-states (key and values in the self-attention blocks and in the
                cross-attention blocks) that can be used (see `past_key_values` input) to speed up sequential decoding.

                If `past_key_values` are used, the user can optionally input only the last `decoder_input_ids` (those
                that don't have their past key value states given to this model) of shape `(batch_size, 1)` instead of
                all `decoder_input_ids` of shape `(batch_size, sequence_length)`.
            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
                Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation.
                This is useful if you want more control over how to convert `input_ids` indices into associated vectors
                than the model's internal embedding lookup matrix.
        NzTYou cannot specify both decoder_input_ids and decoder_inputs_embeds at the same time)r‡   r.   r   rò   )r‡   r   rn   r•   )r‡   r   rn   rÔ   )rJ   rt   )rÕ   r•   rÖ   )r"  r•   )r0   r  r‡   Úis_encoder_decoderr   r   rw   Úget_seq_lengthrO   rP   rM   r!   Úonesr™   r�   r   r   r  r#  r  r   rz   rp   rv   r$  rž   r%  r  r  r   )rG   r)   rn   rÔ   rÕ   r•   r   rÖ   rq   Ú
batch_sizeÚ
seq_lengthrI   rJ   Úmask_seq_lengthÚself_attn_cacheÚcausal_maskr“   r'  Údecoder_layerr*  s                       r:   rT   zMBartDecoder.forwardx  sÄ  € ðn ˜Ð -°tÐ";Ñ<ð 	uÝÐsÑtÔtÐtàÐ Ø ×-Ò-¨iÑ8Ô8ˆMð ð 	˜Ð0ð )Ð4¸¼Ô8VÐ4õ $¥L¸¼Ð$DÑ$DÔ$DÅlÐZ^ÔZeÐFfÑFfÔFfÑgÔgÐgå!¨¬Ð5Ñ5Ô5ð ð "/×!3Ò!3Ñ!5Ô!5°c°r°cÔ!:Ñˆ
�JØETÐE` ×!?Ò!?Ñ!AÔ!AÐ!AÐfgÐÝ”| J°}Ô7KÐLÑLÔLÐOeÑeˆàÐ!Õ*BÑ*DÔ*DÐ!à4°zÑAˆOÝ"œZ¨
°OÈMÔL`ÐaÑaÔaˆNõ ˜/Õ+>Ñ?Ô?ð!ˆOÔ0Ð0à ð 	õ )Ø”;Ø'Ø)Ø+ð	
ñ 
ô 
ˆõ ";Ø”;Ø'Ø1Ø"7ð	"
ñ "
ô "
Ðð ×+Ò+¨IÐ7MÐ\hÐ+ÑiÔiˆà%¨¯ª¸Ô8LÑ(MÔ(MÑMˆØ×0Ò0°Ñ?Ô?ˆåœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆå"+¨D¬KÑ"8Ô"8ð 	ð 	ÑˆC�àŒ}ð Ý&+¤j°¡n¤nÐ#Ø&¨¬Ò7Ð7Øà)˜MØØØ%ðð (>Ø /Ø#ðð ð ðð ˆMˆMð Ÿš¨Ñ6Ô6ˆå8Ø+Ø+ð
ñ 
ô 
ð 	
r<   )NNNNNNN)rW   rX   rY   rZ   rÎ   r%   r�   r+  r(   rF   r$   r&   rO   r,  r\   r-  r   r°   r   r   r±   r   rT   r]   r^   s   @r:   r/  r/  O  st  ø€ € € € € ðð ð +Ø$�n ^¸1ÈÐUÑUÔUØ*˜N¨>ÀÈ~Ð^Ñ^Ô^ðð Ðð˜{ð ð ð ð ð ð ð4  Øð .2Ø.2Ø:>Ø:>Ø(,Ø26Ø!%ð}
ð }
àÔ# dÑ*ð}
ð œ tÑ+ð}
ð  %Ô0°4Ñ7ð	}
ð
 !&Ô 0°4Ñ 7ð}
ð  ™ð}
ð Ô(¨4Ñ/ð}
ð ˜$‘;ð}
ð Ð+Ô,ð}
ð 
Ð:Ñ	:ð}
ð }
ð }
ñ „_ñ  Ôð}
ð }
ð }
ð }
ð }
r<   r/  c                   ój  ‡ — e Zd ZdddœZdefˆ fd„Zd„ Zd„ Zee		 	 	 	 	 	 	 	 	 	 dde
j        dz  d	e
j        dz  d
e
j        dz  de
j        dz  deee
j                          dz  dedz  de
j        dz  de
j        dz  dedz  dedz  dee         deee
j                 z  fd„¦   «         ¦   «         Zˆ xZS )Ú
MBartModelzshared.weight)zdecoder.embed_tokens.weightzencoder.embed_tokens.weightr‡   c                 ó\  •— t          ¦   «                              |¦  «         |j        |j        }}|j        rt          j        |j        ¦  «        nd}t          ||j        ||¬¦  «        | _	        t          |¦  «        | _        t          |¦  «        | _        |                      ¦   «          d S )Nra   r  )rE   rF   r*   r  r  r  r  rµ   r`   Úsharedr   Úencoderr/  Údecoderr  )rG   r‡   rb   r  rc   rH   s        €r:   rF   zMBartModel.__init__  s™   ø€ Ý‰Œ×Ò˜Ñ Ô Ð à"(Ô"5°vÔ7H�ZˆØ39Ô3IÐR•d”i ¤Ñ/Ô/Ð/ÈsˆÝ.¨z¸6¼>È;ÐdoÐpÑpÔpˆŒå# FÑ+Ô+ˆŒÝ# FÑ+Ô+ˆŒð 	�ŠÑÔÐÐÐr<   c                 ó   — | j         S re   )rD  r  s    r:   Úget_input_embeddingszMBartModel.get_input_embeddings  s
   € ØŒ{Ðr<   c                 óX   — || _         | j         | j        _        | j         | j        _        d S re   )rD  rE  r  rF  ©rG   rm   s     r:   Úset_input_embeddingszMBartModel.set_input_embeddings  s'   € ØˆŒØ$(¤KˆŒÔ!Ø$(¤KˆŒÔ!Ð!Ð!r<   Nr)   rn   Údecoder_input_idsÚdecoder_attention_maskÚencoder_outputsr•   r   Údecoder_inputs_embedsrÖ   Úreturn_dictrq   r–   c                 ó  — |
�|
n| j         j        }
|€|€t          || j         j        ¦  «        }|€ | j        d	||||
dœ|¤Ž}ne|
rct          |t          ¦  «        sNt          |d         t          |¦  «        dk    r|d         ndt          |¦  «        dk    r|d         nd¬¦  «        } | j        d	|||d         ||||	|
dœ|¤Ž}|
s||z   S t          |j
        |j        |j        |j        |j        |j
        |j        |j        ¬¦  «        S )
a5  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are decoder input IDs?](../glossary#decoder-input-ids)

            MBart uses a specific language id token as the starting token for `decoder_input_ids` generation that
            varies according to source and target language, *e.g.* 25004 for *en_XX*, and 25003 for *de_DE*. If
            `past_key_values` is used, optionally only the last `decoder_input_ids` have to be input (see
            `past_key_values`).

            For translation and summarization training, `decoder_input_ids` should be provided. If no
            `decoder_input_ids` is provided, the model will create this tensor by shifting the `input_ids` to the right
            for denoising pre-training following the paper.
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.
        N)r)   rn   r   rP  r   r'   rC   )r"  r“   r  )r)   rn   rÔ   rÕ   r•   r   rÖ   rP  )r"  r•   Údecoder_hidden_statesÚdecoder_attentionsr0  Úencoder_last_hidden_staterÔ   Úencoder_attentionsrÆ   )r‡   rP  r;   r*   rE  r™   r   ÚlenrF  r   r"  r•   r“   r  r0  )rG   r)   rn   rL  rM  rN  r•   r   rO  rÖ   rP  rq   Údecoder_outputss                r:   rT   zMBartModel.forward  sŽ  € ðJ &1Ð%<�k�kÀ$Ä+ÔBYˆð Ð$Ð)>Ð)FÝ 2°9¸d¼kÔ>VÑ WÔ WÐàÐ"Ø*˜dœlð Ø#Ø-Ø+Ø'ð	ð ð
 ðð ˆOˆOð ð 	¥¨O½_Ñ!MÔ!Mð 	Ý-Ø"1°!Ô"4Ý47¸Ñ4HÔ4HÈ1Ò4LÐ4L˜o¨aÔ0Ð0ÐRVÝ14°_Ñ1EÔ1EÈÒ1IÐ1I˜?¨1Ô-Ð-Ètðñ ô ˆOð '˜$œ,ð 

Ø'Ø1Ø"1°!Ô"4Ø#1Ø+Ø/ØØ#ð

ð 

ð ð

ð 

ˆð ð 	5Ø" _Ñ4Ð4å!Ø-Ô?Ø+Ô;Ø"1Ô"?Ø.Ô9Ø,Ô=Ø&5Ô&GØ"1Ô"?Ø.Ô9ð	
ñ 	
ô 	
ð 		
r<   ©
NNNNNNNNNN)rW   rX   rY   Ú_tied_weights_keysr(   rF   rH  rK  r   r   rO   r,  r\   r±   r-  r   r°   r   r   r   rT   r]   r^   s   @r:   rB  rB  ú  s²  ø€ € € € € ð (7Ø'6ðð Ðð
˜{ð ð ð ð ð ð ðð ð ð0ð 0ð 0ð
 Øð .2Ø.2Ø59Ø:>ØBFØ(,Ø26Ø:>Ø!%Ø#'ðS
ð S
àÔ# dÑ*ðS
ð œ tÑ+ðS
ð !Ô+¨dÑ2ð	S
ð
 !&Ô 0°4Ñ 7ðS
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðS
ð  ™ðS
ð Ô(¨4Ñ/ðS
ð  %Ô0°4Ñ7ðS
ð ˜$‘;ðS
ð ˜D‘[ðS
ð Ð+Ô,ðS
ð 
˜e EÔ$5Ô6Ñ	6ðS
ð S
ð S
ñ „^ñ ÔðS
ð S
ð S
ð S
ð S
r<   rB  z€
    The MBART Model with a language modeling head. Can be used for summarization, after fine-tuning the pretrained models.
    )Úcustom_introc                   óÂ  ‡ — e Zd ZdZdgZddiZdefˆ fd„Z	 dd	ed
edz  de	de
j        fˆ fd„Zd	eddfd„Ze	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  deeej                          dz  dedz  dej        dz  dej        dz  dej        dz  de	dz  de	dz  dee         deeej                 z  fd„¦   «         Zdej        fd„Zˆ xZS )rè   rå   rë   úlm_head.weightzmodel.shared.weightr‡   c                 ól  •— t          ¦   «                              |¦  «         t          |¦  «        | _        |                      dt          j        d| j        j        j        f¦  «        ¦  «         t          j
        |j        | j        j        j        d¬¦  «        | _        |                      ¦   «          d S )Nrë   r'   FrŠ   )rE   rF   rB  rå   Úregister_bufferrO   ÚzerosrD  r?   r   rŽ   rµ   Úlm_headr  rÂ   s     €r:   rF   z&MBartForConditionalGeneration.__init__x  s“   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý Ñ'Ô'ˆŒ
Ø×ÒÐ0µ%´+¸qÀ$Ä*ÔBSÔBbÐ>cÑ2dÔ2dÑeÔeÐeÝ”y ¤°´Ô1BÔ1QÐX]Ð^Ñ^Ô^ˆŒð 	�ŠÑÔÐÐÐr<   NTÚnew_num_tokensÚpad_to_multiple_ofÚmean_resizingr–   c                 ó˜   •— t          ¦   «                              |||¦  «        }|                      |j        j        d         ¦  «         |S )Nr   )rE   Úresize_token_embeddingsÚ_resize_final_logits_biasrR   rN   )rG   ra  rb  rc  Únew_embeddingsrH   s        €r:   re  z5MBartForConditionalGeneration.resize_token_embeddings�  sG   ø€ õ ™œ×8Ò8¸ÐI[Ð]jÑkÔkˆØ×&Ò& ~Ô'<Ô'BÀ1Ô'EÑFÔFÐFØÐr<   c                 ó  — | j         j        d         }||k    r| j         d d …d |…f         }nBt          j        d||z
  f| j         j        ¬¦  «        }t          j        | j         |gd¬¦  «        }|                      d|¦  «         d S )Nr.   r'   rò   r,   rë   )rë   rN   rO   r_  rM   Úcatr^  )rG   ra  Úold_num_tokensÚnew_biasÚ
extra_biass        r:   rf  z7MBartForConditionalGeneration._resize_final_logits_biasˆ  s—   € ØÔ/Ô5°bÔ9ˆØ˜^Ò+Ð+ØÔ-¨a¨a¨a°°.°Ð.@ÔAˆHˆHåœ a¨¸.Ñ)HÐ%IÐRVÔRhÔRoÐpÑpÔpˆJÝ”y $Ô"8¸*Ð!EÈ1ÐMÑMÔMˆHØ×ÒÐ0°(Ñ;Ô;Ð;Ð;Ð;r<   r)   rn   rL  rM  rN  r•   r   rO  ÚlabelsrÖ   rP  rq   c                 ó\  — |�|n| j         j        }|	�<|
rt                               d¦  «         d}
|€|€t	          |	| j         j        ¦  «        } | j        |f||||||||
|dœ	|¤Ž}|                      |d         ¦  «        | j        z   }d}|	�Kt          ¦   «         } || 
                    d| j         j        ¦  «        |	 
                    d¦  «        ¦  «        }|s|f|dd…         z   }|�|f|z   n|S t          |||j        |j        |j        |j        |j        |j        |j        ¬¦	  «	        S )	u6  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are decoder input IDs?](../glossary#decoder-input-ids)

            MBart uses a specific language id token as the starting token for `decoder_input_ids` generation that
            varies according to source and target language, *e.g.* 25004 for *en_XX*, and 25003 for *de_DE*. If
            `past_key_values` is used, optionally only the last `decoder_input_ids` have to be input (see
            `past_key_values`).

            For translation and summarization training, `decoder_input_ids` should be provided. If no
            `decoder_input_ids` is provided, the model will create this tensor by shifting the `input_ids` to the right
            for denoising pre-training following the paper.
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.

        Example Translation:

        ```python
        >>> from transformers import AutoTokenizer, MBartForConditionalGeneration

        >>> model = MBartForConditionalGeneration.from_pretrained("facebook/mbart-large-en-ro")
        >>> tokenizer = AutoTokenizer.from_pretrained("facebook/mbart-large-en-ro")

        >>> example_english_phrase = "42 is the answer"
        >>> inputs = tokenizer(example_english_phrase, return_tensors="pt")

        >>> # Translate
        >>> generated_ids = model.generate(**inputs, num_beams=4, max_length=5)
        >>> tokenizer.batch_decode(generated_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
        '42 este rÄƒspuns'
        ```

        Mask filling example:

        ```python
        >>> from transformers import AutoTokenizer, MBartForConditionalGeneration

        >>> model = MBartForConditionalGeneration.from_pretrained("facebook/mbart-large-cc25")
        >>> tokenizer = AutoTokenizer.from_pretrained("facebook/mbart-large-cc25")

        >>> # de_DE is the language symbol id <LID> for German
        >>> TXT = "</s> Meine Freunde sind <mask> nett aber sie essen zu viel Kuchen. </s> de_DE"

        >>> input_ids = tokenizer([TXT], add_special_tokens=False, return_tensors="pt")["input_ids"]
        >>> logits = model(input_ids).logits

        >>> masked_index = (input_ids[0] == tokenizer.mask_token_id).nonzero().item()
        >>> probs = logits[0, masked_index].softmax(dim=0)
        >>> values, predictions = probs.topk(5)

        >>> tokenizer.decode(predictions).split()
        ['nett', 'sehr', 'ganz', 'nicht', 'so']
        ```
        NzJThe `use_cache` argument is changed to `False` since `labels` is provided.F)	rn   rL  rN  rM  r•   r   rO  rÖ   rP  r   r.   r'   ©	ÚlossÚlogitsr•   rR  rS  r0  rT  rÔ   rU  )r‡   rP  rŒ   Úwarningr;   r*   rå   r`  rë   r   r˜   r  r   r•   rR  rS  r0  rT  rÔ   rU  )rG   r)   rn   rL  rM  rN  r•   r   rO  rm  rÖ   rP  rq   ÚoutputsÚ	lm_logitsÚmasked_lm_lossÚloss_fctÚoutputs                     r:   rT   z%MBartForConditionalGeneration.forward‘  s‹  € ð` &1Ð%<�k�kÀ$Ä+ÔBYˆàÐØð mÝ—’ÐkÑlÔlÐlØˆIØ Ð(Ð-BÐ-JÝ$6°v¸t¼{Ô?WÑ$XÔ$XÐ!à�$”*Øð
à)Ø/Ø+Ø#9Ø+Ø'Ø"7ØØ#ð
ð 
ð ð
ð 
ˆð —L’L ¨¤Ñ,Ô,¨tÔ/EÑEˆ	àˆØÐÝ'Ñ)Ô)ˆHØ%˜X i§n¢n°R¸¼Ô9OÑ&PÔ&PÐRX×R]ÒR]Ð^`ÑRaÔRaÑbÔbˆNàð 	ZØ�\ G¨A¨B¨B¤KÑ/ˆFØ3AÐ3M�^Ð%¨Ñ.Ð.ÐSYÐYåØØØ#Ô3Ø")Ô"?Ø&Ô9Ø$Ô5Ø&-Ô&GØ")Ô"?Ø&Ô9ð

ñ 

ô 

ð 
	
r<   c                 ó6   — t          || j        j        ¦  «        S re   )r;   r‡   r*   )rG   rm  s     r:   Ú%prepare_decoder_input_ids_from_labelszCMBartForConditionalGeneration.prepare_decoder_input_ids_from_labels  s   € Ý! &¨$¬+Ô*BÑCÔCÐCr<   )NT)NNNNNNNNNNN)rW   rX   rY   r÷   Ú_keys_to_ignore_on_load_missingrY  r(   rF   r[   r°   r   Ú	Embeddingre  rf  r   rO   r,  r\   r±   r-  r   r   r   r   rT   ry  r]   r^   s   @r:   rè   rè   n  s<  ø€ € € € € ð  ÐØ':Ð&;Ð#Ø*Ð,AÐBÐð˜{ð ð ð ð ð ð ð aeðð Ø!ðØ7:¸T±zðØY]ðà	Œðð ð ð ð ð ð<¸ð <Àð <ð <ð <ð <ð ð .2Ø.2Ø59Ø:>ØBFØ(,Ø26Ø:>Ø*.Ø!%Ø#'ðz
ð z
àÔ# dÑ*ðz
ð œ tÑ+ðz
ð !Ô+¨dÑ2ð	z
ð
 !&Ô 0°4Ñ 7ðz
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðz
ð  ™ðz
ð Ô(¨4Ñ/ðz
ð  %Ô0°4Ñ7ðz
ð Ô  4Ñ'ðz
ð ˜$‘;ðz
ð ˜D‘[ðz
ð Ð+Ô,ðz
ð 
˜5 Ô!2Ô3Ñ	3ðz
ð z
ð z
ñ „^ðz
ðxD¸E¼Lð Dð Dð Dð Dð Dð Dð Dð Dr<   rè   z†
    MBart model with a sequence classification/head on top (a linear layer on top of the pooled output) e.g. for GLUE
    tasks.
    c                   ó0  ‡ — e Zd Zdefˆ fd„Zee	 	 	 	 	 	 	 	 	 ddej        dz  dej	        dz  dej        dz  dej        dz  de
ej                 dz  d	ej        dz  d
ej        dz  dej        dz  dedz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )ÚMBartForSequenceClassificationr‡   c                 óâ   •—  t          ¦   «         j        |fi |¤Ž t          |¦  «        | _        t	          |j        |j        |j        |j        ¦  «        | _        |  	                    ¦   «          d S re   )
rE   rF   rB  rå   rÙ   rµ   Ú
num_labelsÚclassifier_dropoutÚclassification_headr  )rG   r‡   rq   rH   s      €r:   rF   z'MBartForSequenceClassification.__init__  sq   ø€ Ø�‰ŒÔ˜Ð*Ð* 6Ð*Ð*Ð*Ý Ñ'Ô'ˆŒ
Ý#:ØŒNØŒNØÔØÔ%ñ	$
ô $
ˆÔ ð 	�ŠÑÔÐÐÐr<   Nr)   rn   rL  rM  rN  r   rO  rm  rÖ   rq   r–   c
                 óJ  — |�d}	|€|�t          d| j        j        › �¦  «        ‚ | j        |f|||||||	dœ|
¤Ž}|d         }|                     | j        j        ¦  «                             |j        ¦  «        }t          t          j        |                     d¦  «        ¦  «                             ¦   «         dk    d¦  «         ||dd…f         }t          |j        d         |j        d         z  dk    d¦  «         |                     |                     d¦  «        d	|                     d	¦  «        ¦  «        dd…d	dd…f         }|                      |¦  «        }d}|��ˆ|                     |j        ¦  «        }| j        j        €p| j        j        dk    rd
| j        _        nS| j        j        dk    r7|j        t          j        k    s|j        t          j        k    rd| j        _        nd| j        _        | j        j        d
k    r\t/          ¦   «         }| j        j        dk    r1 ||                     ¦   «         |                     ¦   «         ¦  «        }n“ |||¦  «        }n†| j        j        dk    rLt3          ¦   «         } ||                     d	| j        j        ¦  «        |                     d	¦  «        ¦  «        }n*| j        j        dk    rt5          ¦   «         } |||¦  «        }t7          |||j        |j        |j        |j        |j         |j!        |j"        ¬¦	  «	        S )aõ  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are decoder input IDs?](../glossary#decoder-input-ids)

            Bart uses the `eos_token_id` as the starting token for `decoder_input_ids` generation. If `past_key_values`
            is used, optionally only the last `decoder_input_ids` have to be input (see `past_key_values`).

            For translation and summarization training, `decoder_input_ids` should be provided. If no
            `decoder_input_ids` is provided, the model will create this tensor by shifting the `input_ids` to the right
            for denoising pre-training following the paper.
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.

            If you want to change padding behavior, you should read [`modeling_bart._prepare_decoder_attention_mask`]
            and modify to your needs. See diagram 1 in [the paper](https://huggingface.co/papers/1910.13461) for more
            information on the default strategy.
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the sequence classification/regression loss. Indices should be in `[0, ...,
            config.num_labels - 1]`. If `config.num_labels > 1` a classification loss is computed (Cross-Entropy).
        NFz8Passing input embeddings is currently not supported for ©rn   rL  rM  rN  r   rO  rÖ   r   r'   z7All examples must have the same number of <eos> tokens.z3Each example must contain at least one <eos> token.r.   Ú
regressionÚsingle_label_classificationÚmulti_label_classificationro  )#ÚNotImplementedErrorrH   rW   rå   Úeqr‡   Úeos_token_idr#  rM   r#   rO   Úunique_consecutiver3   ÚnumelrN   r˜   rw   r�  Úproblem_typer  rL   rQ   r[   r   r6   r   r   r   r•   rR  rS  r0  rT  rÔ   rU  )rG   r)   rn   rL  rM  rN  r   rO  rm  rÖ   rq   rs  r“   Úeos_maskÚselectedÚsentence_representationrq  rp  rv  s                      r:   rT   z&MBartForSequenceClassification.forward&  sH  € ðT ÐØˆIàÐ Ð!:Ý%ØdÈ4Ì>ÔKbÐdÐdñô ð ð '1 d¤jØð
'
à)Ø/Ø#9Ø+Ø'Ø"7Øð
'
ð 
'
ð ð
'
ð 
'
ˆð   œ
ˆà—<’< ¤Ô 8Ñ9Ô9×<Ò<¸]Ô=QÑRÔRˆåÝÔ$ X§\¢\°!¡_¤_Ñ5Ô5×;Ò;Ñ=Ô=ÀÒBØEñ	
ô 	
ð 	
ð ! ¨1¨1¨1 Ô-ˆÝØŒN˜1Ô Ô!4°QÔ!7Ñ7¸1Ò<ØAñ	
ô 	
ð 	
ð #+§-¢-°×0BÒ0BÀ1Ñ0EÔ0EÀrÈ=×K]ÒK]Ð^`ÑKaÔKaÑ"bÔ"bÐcdÐcdÐcdÐfhÐjkÐjkÐjkÐckÔ"lÐØ×)Ò)Ð*AÑBÔBˆàˆØÑØ—Y’Y˜vœ}Ñ-Ô-ˆFØŒ{Ô'Ð/Ø”;Ô)¨QÒ.Ð.Ø/;�D”KÔ,Ð,Ø”[Ô+¨aÒ/Ð/°V´\ÅUÄZÒ5OÐ5OÐSYÔS_ÕchÔclÒSlÐSlØ/L�D”KÔ,Ð,à/K�D”KÔ,àŒ{Ô'¨<Ò7Ð7Ý"™9œ9�Ø”;Ô)¨QÒ.Ð.Ø#˜8 F§N¢NÑ$4Ô$4°f·n²nÑ6FÔ6FÑGÔG�D�Dà#˜8 F¨FÑ3Ô3�D�DØ”Ô)Ð-JÒJÐJÝ+Ñ-Ô-�Ø�x §¢¨B°´Ô0FÑ GÔ GÈÏÊÐUWÉÌÑYÔY��Ø”Ô)Ð-IÒIÐIÝ,Ñ.Ô.�Ø�x ¨Ñ/Ô/�å.ØØØ#Ô3Ø")Ô"?Ø&Ô9Ø$Ô5Ø&-Ô&GØ")Ô"?Ø&Ô9ð

ñ 

ô 

ð 
	
r<   )	NNNNNNNNN)rW   rX   rY   r(   rF   r   r   rO   r,  r\   Úlistr-  r°   r   r   r±   r   rT   r]   r^   s   @r:   r}  r}    se  ø€ € € € € ð˜{ð ð ð ð ð ð ð Øð .2Ø.2Ø59Ø:>Ø:>Ø26Ø:>Ø*.Ø!%ðl
ð l
àÔ# dÑ*ðl
ð œ tÑ+ðl
ð !Ô+¨dÑ2ð	l
ð
 !&Ô 0°4Ñ 7ðl
ð ˜eÔ/Ô0°4Ñ7ðl
ð Ô(¨4Ñ/ðl
ð  %Ô0°4Ñ7ðl
ð Ô  4Ñ'ðl
ð ˜$‘;ðl
ð Ð+Ô,ðl
ð 
Ð0Ñ	0ðl
ð l
ð l
ñ „^ñ Ôðl
ð l
ð l
ð l
ð l
r<   r}  c                   ó@  ‡ — e Zd Zˆ fd„Zee	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  de	ej
                 dz  dej        dz  d	ej        dz  d
ej
        dz  dej
        dz  dedz  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )ÚMBartForQuestionAnsweringc                 ó  •— t          ¦   «                              |¦  «         d|_        |j        | _        t          |¦  «        | _        t          j        |j        |j        ¦  «        | _        |  	                    ¦   «          d S rB   )
rE   rF   r  rB  rå   r   rŽ   Úhidden_sizeÚ
qa_outputsr  rÂ   s     €r:   rF   z"MBartForQuestionAnswering.__init__š  sm   ø€ Ý‰Œ×Ò˜Ñ Ô Ð àˆÔØ Ô+ˆŒå Ñ'Ô'ˆŒ
Ýœ) FÔ$6¸Ô8IÑJÔJˆŒð 	�ŠÑÔÐÐÐr<   Nr)   rn   rL  rM  rN  Ústart_positionsÚend_positionsr   rO  rÖ   rq   r–   c                 ó’  — |�|�d}
 | j         |f||||||	|
dœ|¤Ž}|d         }|                      |¦  «        }|                     dd¬¦  «        \  }}|                     d¦  «                             ¦   «         }|                     d¦  «                             ¦   «         }d}|�ç|�åt          |                     ¦   «         ¦  «        dk    r|                     d¦  «        }t          |                     ¦   «         ¦  «        dk    r|                     d¦  «        }|                     d¦  «        }|                     d|¦  «        }|                     d|¦  «        }t          |¬¦  «        } |||¦  «        } |||¦  «        }||z   d	z  }t          ||||j
        |j        |j        |j        |j        |j        |j        ¬
¦
  «
        S )aË  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are decoder input IDs?](../glossary#decoder-input-ids)

            Bart uses the `eos_token_id` as the starting token for `decoder_input_ids` generation. If `past_key_values`
            is used, optionally only the last `decoder_input_ids` have to be input (see `past_key_values`).

            For translation and summarization training, `decoder_input_ids` should be provided. If no
            `decoder_input_ids` is provided, the model will create this tensor by shifting the `input_ids` to the right
            for denoising pre-training following the paper.
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.

            If you want to change padding behavior, you should read [`modeling_bart._prepare_decoder_attention_mask`]
            and modify to your needs. See diagram 1 in [the paper](https://huggingface.co/papers/1910.13461) for more
            information on the default strategy.
        NFrƒ  r   r'   r.   r,   )Úignore_indexrC   )
rp  Ústart_logitsÚ
end_logitsr•   rR  rS  r0  rT  rÔ   rU  )rå   r•  Úsplitr6   r|   rV  rw   rÉ   r   r   r•   rR  rS  r0  rT  rÔ   rU  )rG   r)   rn   rL  rM  rN  r–  r—  r   rO  rÖ   rq   rs  Úsequence_outputrq  rš  r›  Ú
total_lossÚignored_indexrv  Ú
start_lossÚend_losss                         r:   rT   z!MBartForQuestionAnswering.forward¦  s  € ðP Ð&¨=Ð+DØˆIà&0 d¤jØð
'
à)Ø/Ø#9Ø+Ø'Ø"7Øð
'
ð 
'
ð ð
'
ð 
'
ˆð " !œ*ˆà—’ Ñ1Ô1ˆØ#)§<¢<°°r <Ñ#:Ô#:Ñ ˆ�jØ#×+Ò+¨BÑ/Ô/×:Ò:Ñ<Ô<ˆØ×'Ò'¨Ñ+Ô+×6Ò6Ñ8Ô8ˆ
àˆ
ØÐ&¨=Ð+Då�?×'Ò'Ñ)Ô)Ñ*Ô*¨QÒ.Ð.Ø"1×"9Ò"9¸"Ñ"=Ô"=�Ý�=×%Ò%Ñ'Ô'Ñ(Ô(¨1Ò,Ð,Ø -× 5Ò 5°bÑ 9Ô 9�à(×-Ò-¨aÑ0Ô0ˆMØ-×3Ò3°A°}ÑEÔEˆOØ)×/Ò/°°=ÑAÔAˆMå'°]ÐCÑCÔCˆHØ!˜ ,°Ñ@Ô@ˆJØ�x 
¨MÑ:Ô:ˆHØ$ xÑ/°1Ñ4ˆJå2ØØ%Ø!Ø#Ô3Ø")Ô"?Ø&Ô9Ø$Ô5Ø&-Ô&GØ")Ô"?Ø&Ô9ð
ñ 
ô 
ð 	
r<   rX  )rW   rX   rY   rF   r   r   rO   r\   r,  r�  r-  r°   r   r   r±   r   rT   r]   r^   s   @r:   r’  r’  ˜  sn  ø€ € € € € ð
ð 
ð 
ð 
ð 
ð Øð *.Ø.2Ø59Ø:>Ø:>Ø37Ø15Ø26Ø:>Ø!%ðW
ð W
à”< $Ñ&ðW
ð œ tÑ+ðW
ð !Ô+¨dÑ2ð	W
ð
 !&Ô 0°4Ñ 7ðW
ð ˜eÔ/Ô0°4Ñ7ðW
ð Ô)¨DÑ0ðW
ð Ô'¨$Ñ.ðW
ð Ô(¨4Ñ/ðW
ð  %Ô0°4Ñ7ðW
ð ˜$‘;ðW
ð Ð+Ô,ðW
ð 
Ð4Ñ	4ðW
ð W
ð W
ñ „^ñ ÔðW
ð W
ð W
ð W
ð W
r<   r’  c                   ó(   ‡ — e Zd ZdZˆ fd„Zd„ Zˆ xZS )ÚMBartDecoderWrapperz½
    This wrapper class is a helper class to correctly load pretrained checkpoints when the causal language model is
    used in combination with the [`EncoderDecoderModel`] framework.
    c                 óš   •— t          ¦   «                              |¦  «         t          |¦  «        | _        |                      ¦   «          d S re   )rE   rF   r/  rF  r  rÂ   s     €r:   rF   zMBartDecoderWrapper.__init__
  s@   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý# FÑ+Ô+ˆŒØ�ŠÑÔÐÐÐr<   c                 ó   —  | j         |i |¤ŽS re   )rF  )rG   Úargsrq   s      r:   rT   zMBartDecoderWrapper.forward  s   € ØˆtŒ|˜TÐ, VÐ,Ð,Ð,r<   )rW   rX   rY   rZ   rF   rT   r]   r^   s   @r:   r£  r£    sQ   ø€ € € € € ðð ð
ð ð ð ð ð
-ð -ð -ð -ð -ð -ð -r<   r£  c                   ó(  ‡ — e Zd ZddiZˆ fd„Zd„ Zd„ Zee	 	 	 	 	 	 	 	 	 dde	j
        dz  d	e	j        dz  d
e	j        dz  de	j        dz  dedz  de	j        dz  de	j
        dz  dedz  dee	j        z  dee         deez  fd„¦   «         ¦   «         Zˆ xZS )ÚMBartForCausalLMr\  z!model.decoder.embed_tokens.weightc                 ó  •— d|_         d|_        t          ¦   «                              |¦  «         t	          |¦  «        | _        t          j        |j        |j	        d¬¦  «        | _
        |                      ¦   «          d S )NTFrŠ   )r„   r8  rE   rF   r£  rå   r   rŽ   r”  r  r`  r  rÂ   s     €r:   rF   zMBartForCausalLM.__init__  sp   ø€ Ø ˆÔØ$)ˆÔ!Ý‰Œ×Ò˜Ñ Ô Ð Ý(¨Ñ0Ô0ˆŒ
å”y Ô!3°VÔ5FÈUÐSÑSÔSˆŒð 	�ŠÑÔÐÐÐr<   c                 ó$   — | j         j        j        S re   ©rå   rF  r  r  s    r:   rH  z%MBartForCausalLM.get_input_embeddings$  s   € ØŒzÔ!Ô.Ð.r<   c                 ó(   — || j         j        _        d S re   r«  rJ  s     r:   rK  z%MBartForCausalLM.set_input_embeddings'  s   € Ø*/ˆŒ
ÔÔ'Ð'Ð'r<   Nr   r)   rn   rÔ   rÕ   r•   r   rm  rÖ   Úlogits_to_keeprq   r–   c
                 óþ  —  | j         j        d|||||||dœ|
¤Ž}|d         }t          |	t          ¦  «        rt	          |	 d¦  «        n|	}|                      |dd…|dd…f         ¦  «        }d}|�e|                     |j        ¦  «        }t          ¦   «         } || 	                    d| j
        j        ¦  «        | 	                    d¦  «        ¦  «        }t          |||j        |j        |j        |j        ¬¦  «        S )aP  
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.

        Example:

        ```python
        >>> from transformers import AutoTokenizer, MBartForCausalLM

        >>> tokenizer = AutoTokenizer.from_pretrained("facebook/mbart-large-cc25")
        >>> model = MBartForCausalLM.from_pretrained("facebook/mbart-large-cc25")
        >>> assert model.config.is_decoder, f"{model.__class__} has to be configured as a decoder."
        >>> inputs = tokenizer("Hello, my dog is cute", return_tensors="pt")
        >>> outputs = model(**inputs)

        >>> logits = outputs.logits
        >>> expected_shape = [1, inputs.input_ids.shape[-1], model.config.vocab_size]
        >>> list(logits.shape) == expected_shape
        True
        ```)r)   rn   rÔ   rÕ   r•   r   rÖ   r   Nr.   )rp  rq  r•   r“   r  r0  rÆ   )rå   rF  r™   r[   Úslicer`  r#  rM   r   r˜   r‡   r  r   r•   r“   r  r0  )rG   r)   rn   rÔ   rÕ   r•   r   rm  rÖ   r­  rq   rs  r“   Úslice_indicesrq  rp  rv  s                    r:   rT   zMBartForCausalLM.forward*  s)  € ðL >P¸T¼ZÔ=Oð 	>
ØØ)Ø"7Ø#9Ø+Ø'Øð	>
ð 	>
ð ð	>
ð 	>
ˆð   œ
ˆå8BÀ>ÕSVÑ8WÔ8WÐk�˜~˜o¨tÑ4Ô4Ð4Ð]kˆØ—’˜m¨A¨A¨A¨}¸a¸a¸aÐ,?Ô@ÑAÔAˆàˆØÐØ—Y’Y˜vœ}Ñ-Ô-ˆFÝ'Ñ)Ô)ˆHØ�8˜FŸKšK¨¨D¬KÔ,BÑCÔCÀVÇ[Â[ÐQSÁ_Ä_ÑUÔUˆDå0ØØØ#Ô3Ø!Ô/ØÔ)Ø$Ô5ð
ñ 
ô 
ð 	
r<   )	NNNNNNNNr   )rW   rX   rY   rY  rF   rH  rK  r   r   rO   r,  r\   r-  r   r°   r[   r   r   r±   r   rT   r]   r^   s   @r:   r¨  r¨    s{  ø€ € € € € àÐ=ðÐð	ð 	ð 	ð 	ð 	ð/ð /ð /ð0ð 0ð 0ð Øð .2Ø.2Ø:>Ø;?Ø(,Ø26Ø*.Ø!%Ø-.ðA
ð A
àÔ# dÑ*ðA
ð œ tÑ+ðA
ð  %Ô0°4Ñ7ð	A
ð
 !&Ô 1°DÑ 8ðA
ð  ™ðA
ð Ô(¨4Ñ/ðA
ð Ô  4Ñ'ðA
ð ˜$‘;ðA
ð ˜eœlÑ*ðA
ð Ð+Ô,ðA
ð 
Ð2Ñ	2ðA
ð A
ð A
ñ „^ñ ÔðA
ð A
ð A
ð A
ð A
r<   r¨  )r¨  rè   r’  r}  rB  rä   )Nri   )RrZ   r  Úcollections.abcr   rO   r   Útorch.nnr   r   r   Ú r	   ré   Úactivationsr
   Úcache_utilsr   r   r   Ú
generationr   Úmasking_utilsr   r   Úmodeling_flash_attention_utilsr   Úmodeling_layersr   Úmodeling_outputsr   r   r   r   r   r   r   Úmodeling_utilsr   r   Úprocessing_utilsr   Úutilsr   r   r   r    r!   r"   r#   Úutils.genericr$   Úutils.output_capturingr%   r&   Úconfiguration_mbartr(   Ú
get_loggerrW   rŒ   r\   r[   r;   r{  r>   r`   ÚModulerh   r   r�   r³   rÎ   rÙ   rä   r   r/  rB  rè   r}  r’  r£  r¨  Ú__all__rÆ   r<   r:   ú<module>rÄ     sc  ðð Ð à €€€Ø $Ð $Ð $Ð $Ð $Ð $à €€€Ø Ð Ð Ð Ð Ð Ø AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ AÐ Aà &Ð &Ð &Ð &Ð &Ð &Ø !Ð !Ð !Ð !Ð !Ð !Ø CÐ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CØ )Ð )Ð )Ð )Ð )Ð )Ø JÐ JÐ JÐ JÐ JÐ JÐ JÐ Jðð ð ð ð ð ð :Ð 9Ð 9Ð 9Ð 9Ð 9ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð GÐ FÐ FÐ FÐ FÐ FÐ FÐ FØ &Ð &Ð &Ð &Ð &Ð &ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð 8Ð 7Ð 7Ð 7Ð 7Ð 7Ø EÐ EÐ EÐ EÐ EÐ EÐ EÐ EØ ,Ð ,Ð ,Ð ,Ð ,Ð ,ð  ÐÑ!Ô!ð 	Øð 
ˆÔ	˜HÑ	%Ô	%€ð %¤,ð ¸cð ð ð ð ð*;ð ;ð ;ð ;ð ; b¤lñ ;ô ;ð ;ð8
=ð 
=ð 
=ð 
=ð 
=˜rœ|ñ 
=ô 
=ð 
=ð( !Øð%ð %ØŒIð%àŒ<ð%ð 
Œð%ð Œ<ð	%ð
 ”L 4Ñ'ð%ð �T‰\ð%ð ð%ð Ð'Ô(ð%ð %ð %ð %ð:r)ð r)ð r)ð r)ð r)�R”Yñ r)ô r)ð r)ðj5ð 5ð 5ð 5ð 5Ð2ñ 5ô 5ð 5ðpZð Zð Zð Zð ZÐ2ñ Zô Zð Zð|ð ð ð ð ˜bœiñ ô ð ð0 ðð ð ð ð ˜?ñ ô ñ „ðð4r@ð r@ð r@ð r@ð r@Ð'ñ r@ô r@ð r@ðjh
ð h
ð h
ð h
ð h
Ð'ñ h
ô h
ð h
ðV ðp
ð p
ð p
ð p
ð p
Ð%ñ p
ô p
ñ „ðp
ðf €ððñ ô ð
\Dð \Dð \Dð \Dð \DÐ$8¸/ñ \Dô \Dñô ð
\Dð~ €ððñ ô ð}
ð }
ð }
ð }
ð }
Ð%9ñ }
ô }
ñô ð}
ð@ ðg
ð g
ð g
ð g
ð g
Ð 4ñ g
ô g
ñ „ðg
ðV-ð -ð -ð -ð -Ð.ñ -ô -ð -ð Y
ð Y
ð Y
ð Y
ð Y
Ð+¨_ñ Y
ô Y
ð Y
ðxð ð €€€r<   