§
    ‚Štj¦¡ ã                   óÄ  — d Z ddlZddlmZ ddlZddlmZ ddlmZ ddlm	Z
 ddlmZ dd	lmZmZmZ dd
lmZ ddlmZmZ ddlmZ ddlmZ ddlmZ ddlmZmZmZ ddl m!Z!  ej"        e#¦  «        Z$dej%        de&de&fd„Z'dFdej%        dej(        de&dz  fd„Z) G d„ dej*        ¦  «        Z+ G d„ dej,        ¦  «        Z- G d„ dej,        ¦  «        Z. G d „ d!ej,        ¦  «        Z/ G d"„ d#e¦  «        Z0 G d$„ d%e¦  «        Z1 G d&„ d'ej,        ¦  «        Z2e G d(„ d)e¦  «        ¦   «         Z3 ed*¬+¦  «        e G d,„ d-e¦  «        ¦   «         ¦   «         Z4 ed.¬+¦  «        e G d/„ d0e¦  «        ¦   «         ¦   «         Z5 ed1¬+¦  «        e G d2„ d3e¦  «        ¦   «         ¦   «         Z6 ed4¬+¦  «        e G d5„ d6e¦  «        ¦   «         ¦   «         Z7 ed7¬+¦  «        e G d8„ d9e¦  «        ¦   «         ¦   «         Z8 G d:„ d;e3¦  «        Z9 G d<„ d=e3¦  «        Z:e G d>„ d?e3¦  «        ¦   «         Z; ed@¬+¦  «         G dA„ dBe3e¦  «        ¦   «         Z<e G dC„ dDe3¦  «        ¦   «         Z=g dE¢Z>dS )GzPyTorch LED model.é    N)Ú	dataclass)Únn)ÚCrossEntropyLossé   )Úinitialization)ÚACT2FN)ÚCacheÚDynamicCacheÚEncoderDecoderCache)ÚGenerationMixin)Úcreate_bidirectional_maskÚcreate_causal_mask)ÚGradientCheckpointingLayer)Ú)BaseModelOutputWithPastAndCrossAttentions)ÚPreTrainedModel)ÚModelOutputÚauto_docstringÚloggingé   )Ú	LEDConfigÚ	input_idsÚpad_token_idÚdecoder_start_token_idc                 óô   — |                       | j        ¦  «        }| dd…dd…f                              ¦   «         |dd…dd…f<   ||dd…df<   |€t          d¦  «        ‚|                     |dk    |¦  «         |S )z1
    Shift input ids one token to the right.
    Néÿÿÿÿr   r   z&config.pad_token_id has to be defined.iœÿÿÿ)Ú	new_zerosÚshapeÚcloneÚ
ValueErrorÚmasked_fill_)r   r   r   Úshifted_input_idss       úb/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/led/modeling_led.pyÚshift_tokens_rightr#   &   s˜   € ð "×+Ò+¨I¬OÑ<Ô<ÐØ(¨¨¨¨C¨R¨C¨Ô0×6Ò6Ñ8Ô8Ð�a�a�a˜˜˜�eÑØ4Ð�a�a�a˜�dÑàÐÝÐAÑBÔBÐBà×"Ò"Ð#4¸Ò#<¸lÑKÔKÐKàÐó    ÚmaskÚdtypeÚtgt_lenc                 óD  — |                       ¦   «         \  }}|�|n|}| dd…dddd…f                              |d||¦  «                             |¦  «        }d|z
  }|                     |                     ¦   «         t          j        |¦  «        j        ¦  «        }||z  }|S )z_
    Expands attention_mask from `[bsz, seq_len]` to `[bsz, 1, tgt_seq_len, src_seq_len]`.
    Nr   g      ð?)ÚsizeÚexpandÚtoÚmasked_fillÚboolÚtorchÚfinfoÚmin)r%   r&   r'   ÚbszÚsrc_lenÚexpanded_maskÚinverted_maskÚexpanded_attention_masks           r"   Ú#_prepare_4d_attention_mask_invertedr6   6   s­   € ð —9’9‘;”;�L€CˆØ Ð,ˆgˆg°'€Gà˜˜˜˜D $¨¨¨Ð)Ô*×1Ò1°#°q¸'À7ÑKÔK×NÒNÈuÑUÔU€Mà˜-Ñ'€MØ+×7Ò7¸×8JÒ8JÑ8LÔ8LÍeÌkÐZ_ÑN`ÔN`ÔNdÑeÔeÐð 6¸ÑEÐà"Ð"r$   c                   óL   ‡ — e Zd ZdZdedefˆ fd„Zd	dej        defˆ fd„Zˆ xZ	S )
ÚLEDLearnedPositionalEmbeddingzN
    This module learns positional embeddings up to a fixed maximum size.
    Únum_embeddingsÚembedding_dimc                 óL   •— t          ¦   «                              ||¦  «         d S ©N)ÚsuperÚ__init__)Úselfr9   r:   Ú	__class__s      €r"   r>   z&LEDLearnedPositionalEmbedding.__init__M   s#   ø€ Ý‰Œ×Ò˜¨Ñ7Ô7Ð7Ð7Ð7r$   r   Úinput_ids_shapeÚpast_key_values_lengthc                 ó¾   •— |dd…         \  }}t          j        |||z   t           j        | j        j        ¬¦  «        }t          ¦   «                              |¦  «        S )z3`input_ids_shape` is expected to be [bsz x seqlen].Né   )r&   Údevice)r.   ÚarangeÚlongÚweightrE   r=   Úforward)r?   rA   rB   r1   Úseq_lenÚ	positionsr@   s         €r"   rI   z%LEDLearnedPositionalEmbedding.forwardP   s[   ø€ à& r¨ rÔ*‰ˆˆWÝ”LØ"Ð$:¸WÑ$DÍEÌJÐ_cÔ_jÔ_qð
ñ 
ô 
ˆ	õ ‰wŒw�Š˜yÑ)Ô)Ð)r$   )r   )
Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úintr>   r.   ÚSizerI   Ú__classcell__©r@   s   @r"   r8   r8   H   sˆ   ø€ € € € € ðð ð8 sð 8¸3ð 8ð 8ð 8ð 8ð 8ð 8ð*ð * u¤zð *È3ð *ð *ð *ð *ð *ð *ð *ð *ð *ð *r$   r8   c                   ó  ‡ — e Zd Zˆ fd„Z	 	 	 	 	 dd„Zed„ ¦   «         Zed„ ¦   «         Zeddefd„¦   «         Z	ed	e
j        fd
„¦   «         Zde
j        de
j        defd„Zde
j        de
j        defd„Zed„ ¦   «         Zd„ Zd„ Zd„ Zˆ xZS )ÚLEDEncoderSelfAttentionc                 ó®  •— t          ¦   «                              ¦   «          |j        |j        z  dk    r t	          d|j        › d|j        › d�¦  «        ‚|j        | _        t          |j        |j        z  ¦  «        | _        |j        | _        t          j
        |j        | j        ¦  «        | _        t          j
        |j        | j        ¦  «        | _        t          j
        |j        | j        ¦  «        | _        t          j
        |j        | j        ¦  «        | _        t          j
        |j        | j        ¦  «        | _        t          j
        |j        | j        ¦  «        | _        |j        | _        || _        |j        | j                 }|dz  dk    sJ d| j        › d|› �¦   «         ‚|dk    sJ d| j        › d|› �¦   «         ‚|dz  | _        || _        d S )	Nr   zThe hidden size (z6) is not a multiple of the number of attention heads (ú)rD   z`attention_window` for layer z  has to be an even value. Given z has to be positive. Given )r=   r>   Úhidden_sizeÚnum_attention_headsr   Ú	num_headsrP   Úhead_dimÚ	embed_dimr   ÚLinearÚqueryÚkeyÚvalueÚquery_globalÚ
key_globalÚvalue_globalÚattention_probs_dropout_probÚdropoutÚlayer_idÚattention_windowÚone_sided_attn_window_sizeÚconfig)r?   ri   rf   rg   r@   s       €r"   r>   z LEDEncoderSelfAttention.__init__[   sÎ  ø€ Ý‰Œ×ÒÑÔÐØÔ Ô :Ñ:¸aÒ?Ð?Ýð8 FÔ$6ð 8ð 8Ø Ô4ð8ð 8ð 8ñô ð ð  Ô3ˆŒÝ˜FÔ.°Ô1KÑKÑLÔLˆŒØÔ+ˆŒå”Y˜vÔ1°4´>ÑBÔBˆŒ
Ý”9˜VÔ/°´Ñ@Ô@ˆŒÝ”Y˜vÔ1°4´>ÑBÔBˆŒ
õ œI fÔ&8¸$¼.ÑIÔIˆÔÝœ) FÔ$6¸¼ÑGÔGˆŒÝœI fÔ&8¸$¼.ÑIÔIˆÔàÔ:ˆŒà ˆŒØ!Ô2°4´=ÔAÐØ !Ñ# qÒ(Ð(Ð(Øm¨D¬MÐmÐmÐ[kÐmÐmñ )Ô(Ð(ð   !Ò#Ð#Ð#Øh¨D¬MÐhÐhÐVfÐhÐhñ $Ô#Ð#ð +;¸aÑ*?ˆÔ'àˆŒˆˆr$   NFc                 óD	  — |                      dd¦  «        }|                      |¦  «        }|                      |¦  «        }|                      |¦  «        }	|                     ¦   «         \  }
}}|| j        k    sJ d| j        › d|› �¦   «         ‚|t          j        | j        ¦  «        z  }| 	                    |
|| j
        | j        ¦  «                              dd¦  «        }| 	                    |
|| j
        | j        ¦  «                              dd¦  «        }|                      ||| j        ¦  «        }|dk    dd…dd…ddf         }|                     |¦  «                             |t          j        |j        ¦  «        j        ¦  «        }|                      |                     |                     ¦   «         ¬¦  «        || j        ¦  «        }||z  }t)          |                     ¦   «         ¦  «        ||
| j
        | j        dz  dz   gk    s;J d|› d	|
› d	| j
        › d	| j        dz  dz   › d
|                     ¦   «         › �
¦   «         ‚|rN|                      |¦  «        \  }}}}|                      ||||||¬¦  «        }t          j        ||fd¬¦  «        }~t0          j                             |dt          j        ¬¦  «        }t          j        ||dd…dd…ddf         d¦  «        }|                     |¦  «        }~t0          j                             || j        | j        ¬¦  «        }|	 	                    |
|| j
        | j        ¦  «                              dd¦  «        }	|r|                      |	||||¬¦  «        }n|                      ||	| j        ¦  «        }|                     ¦   «         ||
| j
        | j        fk    s
J d¦   «         ‚|                      dd¦  «                              |
||¦  «         !                    ¦   «         }|rq|  "                    ||||||¬¦  «        \  }}||d         dd…|d         f         }| 	                    tG          |d         ¦  «        d¦  «        ||ddd…         <   d||<   |                      dd¦  «        f}|r||fz  }|r|r||fz   n|S )a®  
        [`LEDEncoderSelfAttention`] expects *len(hidden_states)* to be multiple of *attention_window*. Padding to
        *attention_window* happens in [`LEDEncoderModel.forward`] to avoid redoing the padding on each layer.

        The *attention_mask* is changed in [`LEDEncoderModel.forward`] from 0, 1, 2 to:

            - -10000: no attention
            - 0: local attention
            - +10000: global attention
        r   r   z&hidden_states should have embed_dim = z
, but has N)r)   rD   z$local_attn_probs should be of size (z, z), but is of size )Úquery_vectorsÚkey_vectorsÚmax_num_global_attn_indicesÚis_index_global_attn_nonzeroÚ"is_local_index_global_attn_nonzeroÚ%is_local_index_no_global_attn_nonzeror   ©Údim©rr   r&   ç        ©ÚpÚtraining)Úvalue_vectorsÚ
attn_probsrm   rn   ro   zUnexpected size)Úhidden_statesrm   ro   rn   rp   Úis_index_masked)$Ú	transposer^   r_   r`   r)   r\   ÚmathÚsqrtr[   ÚviewrZ   Ú _sliding_chunks_query_key_matmulrh   Útype_asr,   r.   r/   r&   r0   Únew_onesÚlistÚ_get_global_attn_indicesÚ"_concat_with_global_key_attn_probsÚcatr   Ú
functionalÚsoftmaxÚfloat32re   rw   Ú(_compute_attn_output_with_global_indicesÚ'_sliding_chunks_matmul_attn_probs_valueÚreshapeÚ
contiguousÚ'_compute_global_attn_output_from_hiddenÚlen)r?   rz   Úattention_maskr{   Úis_index_global_attnÚis_global_attnÚoutput_attentionsrk   rl   rx   rJ   Ú
batch_sizer\   Úattn_scoresÚ#remove_from_windowed_attention_maskÚ
float_maskÚdiagonal_maskrm   rn   ro   rp   Úglobal_key_attn_scoresry   Úattn_outputÚglobal_attn_outputÚglobal_attn_probsÚnonzero_global_attn_outputÚoutputss                               r"   rI   zLEDEncoderSelfAttention.forward~   s‰  € ð& &×/Ò/°°1Ñ5Ô5ˆð Ÿ
š
 =Ñ1Ô1ˆØ—h’h˜}Ñ-Ô-ˆØŸ
š
 =Ñ1Ô1ˆà)6×);Ò);Ñ)=Ô)=Ñ&ˆ�˜YØ˜DœNÒ*Ð*Ð*ØZ°T´^ÐZÐZÈyÐZÐZñ +Ô*Ð*ð
 	�œ 4¤=Ñ1Ô1Ñ1ˆà%×*Ò*¨7°JÀÄÐPTÔP]Ñ^Ô^×hÒhÐijÐlmÑnÔnˆØ!×&Ò& w°
¸D¼NÈDÌMÑZÔZ×dÒdÐefÐhiÑjÔjˆà×;Ò;Ø˜;¨Ô(Gñ
ô 
ˆð
 0>ÀÒ/BÀAÀAÀAÀqÀqÀqÈ$ÐPTÐDTÔ.UÐ+ð 9×@Ò@ÀÑOÔO×[Ò[Ø/µ´¸]Ô=PÑ1QÔ1QÔ1Uñ
ô 
ˆ
ð ×=Ò=Ø×Ò Z§_¢_Ñ%6Ô%6ÐÑ7Ô7¸ÀTÔEdñ
ô 
ˆð
 	�}Ñ$ˆå�K×$Ò$Ñ&Ô&Ñ'Ô'ØØØŒNØÔ+¨aÑ/°!Ñ3ð	,
ò 
ð 
ð 
ð`°:ð `ð `Àð `ð `ÈDÌNð `ð `ØÔ/°!Ñ3°aÑ7ð`ð `ØKV×K[ÒK[ÑK]ÔK]ð`ð `ñ
ô 
ð 
ð ð 	'ð ×-Ò-Ð.BÑCÔCñØ+Ø,Ø2Ø5ð &*×%LÒ%LØ+Ø'Ø,GØ-IØ3UØ6[ð &Mñ &ô &Ð"õ  œ)Ð%;¸[Ð$IÈrÐRÑRÔRˆKð 'å”]×*Ò*Ø˜R¥u¤}ð +ñ 
ô 
ˆ
õ
 Ô& z°?À1À1À1ÀaÀaÀaÈÈtÐCSÔ3TÐVYÑZÔZˆ
Ø×'Ò'¨Ñ4Ô4ˆ
ð õ ”]×*Ò*¨:¸¼ÐPTÔP]Ð*Ñ^Ô^ˆ
à%×*Ò*¨7°JÀÄÐPTÔP]Ñ^Ô^×hÒhÐijÐlmÑnÔnˆð ð 	à×GÒGØ+Ø%Ø,GØ-IØ3Uð Hñ ô ˆKˆKð ×FÒFØ˜M¨4Ô+Jñô ˆKð ×ÒÑ!Ô! j°'¸4¼>È4Ì=Ð%YÒYÐYÐYÐ[lÑYÔYÐYØ!×+Ò+¨A¨qÑ1Ô1×9Ò9¸'À:ÈyÑYÔY×dÒdÑfÔfˆð ð 	9Ø48×4`Ò4`Ø+Ø,GØ3UØ-IØ6[Ø /ð 5añ 5ô 5Ñ1ÐÐ 1ð *<Ø2°1Ô5°q°q°qÐ:\Ð]^Ô:_Ð_ô*Ð&ð
 ?Y×>]Ò>]ÝÐ6°qÔ9Ñ:Ô:¸Bñ?ô ?ˆKÐ4°T°T°r°TÔ:Ñ;ð 89ˆJÐ3Ñ4à×(Ò(¨¨AÑ.Ô.Ð0ˆàð 	%Ø˜
�}Ñ$ˆGà2@ÐdÐEVÐdˆwÐ+Ð-Ñ-Ð-Ð]dÐdr$   c                 óè   — t           j                             | |¦  «        }  | j        g |                      ¦   «         dd…         ¢|                      d¦  «        ‘|                      d¦  «        ‘R Ž } | S )z)pads rows and then flips rows and columnsNéþÿÿÿr   )r   r‡   Úpadr   r)   )Úhidden_states_paddedÚpaddings     r"   Ú _pad_and_transpose_last_two_dimsz8LEDEncoderSelfAttention._pad_and_transpose_last_two_dims  s�   € õ  "œ}×0Ò0Ø  'ñ 
ô  
Ðð  9Ð3Ô8ð  
Ø!×&Ò&Ñ(Ô(¨¨"¨Ô-ð 
Ø/C×/HÒ/HÈÑ/LÔ/Lð 
ØNb×NgÒNgÐhjÑNkÔNkð 
ð  
ð  
Ðð $Ð#r$   c                 ó2  — |                       ¦   «         \  }}}}t          j                             | d|dz   f¦  «        } |                      ||d¦  «        } | dd…dd…d| …f         } |                      |||||z   ¦  «        } | dd…dd…dd…dd…f         } | S )aY  
        shift every row 1 step right, converting columns into diagonals.

        Example:

        ```python
        chunked_hidden_states: [
            0.4983,
            2.6918,
            -0.0071,
            1.0492,
            -1.8348,
            0.7672,
            0.2986,
            0.0285,
            -0.7584,
            0.4206,
            -0.0405,
            0.1599,
            2.0514,
            -1.1600,
            0.5372,
            0.2629,
        ]
        window_overlap = num_rows = 4
        ```

                     (pad & diagonalize) => [ 0.4983, 2.6918, -0.0071, 1.0492, 0.0000, 0.0000, 0.0000
                       0.0000, -1.8348, 0.7672, 0.2986, 0.0285, 0.0000, 0.0000 0.0000, 0.0000, -0.7584, 0.4206,
                       -0.0405, 0.1599, 0.0000 0.0000, 0.0000, 0.0000, 2.0514, -1.1600, 0.5372, 0.2629 ]
        r   r   r   N)r)   r   r‡   r¡   r   )Úchunked_hidden_statesÚtotal_num_headsÚ
num_chunksÚwindow_overlapÚ
hidden_dims        r"   Ú_pad_and_diagonalizez,LEDEncoderSelfAttention._pad_and_diagonalize)  sß   € ðB CX×B\ÒB\ÑB^ÔB^Ñ?ˆ˜ ^°ZÝ "¤× 1Ò 1Ø! A ~¸Ñ'9Ð#:ñ!
ô !
Ðð !6× :Ò :Ø˜Z¨ñ!
ô !
Ðð !6ØˆAˆAˆqˆqˆqÐ"�N�?Ð"Ð"ô!
Ðð !6× :Ò :Ø˜Z¨¸È*Ñ9Tñ!
ô !
Ðð !6°a°a°a¸¸¸¸A¸A¸A¸sÀ¸s°lÔ CÐØ$Ð$r$   Úonnx_exportc                 ó@  — |sä|                       |                      d¦  «        t          j        |                      d¦  «        |dz  d¬¦  «        |dz  |                      d¦  «        ¦  «        } t	          |                      ¦   «         ¦  «        }|d         dz  dz
  |d<   t	          |                      ¦   «         ¦  «        }|d         dz  |d<   |                      ||¬¦  «        S |                      d¦  «        t          j        |                      d¦  «        |d¬¦  «        dz
  |dz  |                      d¦  «        g}t          j        || j        ¬¦  «        }t          |d         ¦  «        D ],}| dd…||z  ||z  d|z  z   …dd…f         |dd…|dd…dd…f<   Œ-|S )	zBconvert into overlapping chunks. Chunk size = 2w, overlap size = wr   r   rD   Útrunc©Úrounding_mode©r)   Ústride©rE   N)
r   r)   r.   Údivrƒ   r²   Ú
as_stridedÚemptyrE   Úrange)rz   r©   r¬   Ú
chunk_sizeÚchunk_strideÚoverlapping_chunksÚchunks          r"   Ú_chunkzLEDEncoderSelfAttention._chunkZ  s×  € ð ð 	Rà)×.Ò.Ø×"Ò" 1Ñ%Ô%Ý”	˜-×,Ò,¨QÑ/Ô/°.À1Ñ2DÐU\Ð]Ñ]Ô]Ø Ñ"Ø×"Ò" 1Ñ%Ô%ñ	ô ˆMõ ˜m×0Ò0Ñ2Ô2Ñ3Ô3ˆJØ& qœM¨AÑ-°Ñ1ˆJ�q‰Må × 4Ò 4Ñ 6Ô 6Ñ7Ô7ˆLØ*¨1œo°Ñ2ˆL˜‰OØ ×+Ò+°ÀLÐ+ÑQÔQÐQð ×Ò˜qÑ!Ô!ÝŒI�m×(Ò(¨Ñ+Ô+¨^È7ÐSÑSÔSÐVWÑWØ˜QÑØ×Ò˜qÑ!Ô!ð	
ˆ
õ #œ[¨¸MÔ<PÐQÑQÔQÐÝ˜: aœ=Ñ)Ô)ð 	ð 	ˆEØ1>Ø���5˜>Ñ)¨E°NÑ,BÀQÈÑEWÑ,WÐWÐYZÐYZÐYZÐZô2Ð˜q˜q˜q %¨¨¨¨A¨A¨A˜~Ñ.Ð.ð "Ð!r$   Úreturnc                 ó>  — |                       ||dz   ¦  «                             ¦   «                              dg¬¦  «        }|d d d …d d d …f         }|                     d¬¦  «        }| d d …d |…d d …d |dz   …f         }|                     |                     ¦   «         ¦  «        }t          j        |t          d¦  «         ¦  «                             | 	                    ¦   «         |¦  «        | d d …d |…d d …d |dz   …f<   | d d …| d …d d …|dz    d …f         }|                     |                     ¦   «         ¦  «        }t          j        |t          d¦  «         ¦  «                             | 	                    ¦   «         |¦  «        | d d …| d …d d …|dz    d …f<   d S )Nr   r   )Údims)r   r   Úinf)
r‚   ÚtrilÚflipr*   r)   r.   Ú	full_likeÚfloatÚwherer-   )Úinput_tensorÚaffected_seq_lenÚbeginning_mask_2dÚbeginning_maskÚending_maskÚbeginning_inputÚending_inputs          r"   Ú_mask_invalid_locationsz/LEDEncoderSelfAttention._mask_invalid_locationsƒ  sð  € à(×1Ò1Ð2BÐDTÐWXÑDXÑYÔY×^Ò^Ñ`Ô`×eÒeÐlmÐknÐeÑoÔoÐØ*¨4°°°°D¸!¸!¸!Ð+;Ô<ˆØ$×)Ò)¨vÐ)Ñ6Ô6ˆØ& q q qÐ*;Ð+;Ð*;¸Q¸Q¸QÐ@VÐBRÐUVÑBVÐ@VÐ'VÔWˆØ'×.Ò.¨×/CÒ/CÑ/EÔ/EÑFÔFˆÝHMÌØ�e E™lœl˜]ñI
ô I
ç
Š%�×#Ò#Ñ%Ô% Ñ
7Ô
7ð 	�Q�Q�QÐ)Ð)Ð)¨1¨1¨1Ð.DÐ0@À1Ñ0DÐ.DÐDÑEð $ A A AÐ(8Ð'8Ð'9Ð'9¸1¸1¸1Ð@PÐSTÑ@TÐ>UÐ>WÐ>WÐ$WÔXˆØ!×(Ò(¨×):Ò):Ñ)<Ô)<Ñ=Ô=ˆÝLQÌOØ�5 ™<œ<˜-ñM
ô M
ç
Š%�× Ò Ñ"Ô" LÑ
1Ô
1ð 	�Q�Q�QÐ)Ð)Ð*Ð*¨A¨A¨AÐ1AÀAÑ1EÐ/FÐ/HÐ/HÐHÑIÐIÐIr$   r^   r_   r©   c           	      óÊ  — |                      ¦   «         \  }}}}||dz  z  dk    sJ d|dz  › d|› �¦   «         ‚|                      ¦   «         |                      ¦   «         k    sJ ‚t          j        ||d¬¦  «        dz
  }|                     dd¦  «                             ||z  ||¦  «        }|                     dd¦  «                             ||z  ||¦  «        }|                      ||t          | j        dd	¦  «        ¦  «        }|                      ||t          | j        dd	¦  «        ¦  «        }t          j        d
||f¦  «        }	|  	                    |	d¬¦  «        }	|	 
                    ||z  |dz   ||dz  dz   f¦  «        }
|	dd…dd…d|…d|dz   …f         |
dd…dd…dd…|d…f<   |	dd…d|d…d|dz   …f         |
dd…ddd…|d…f<   |	dd…dd…|dz    d…|dz   d…f         |
dd…dd…dd…d|…f<   |	dd…dd|dz
  …d|z
  d…f         |
dd…dd|…d|…f<   |
                     |||d|z  dz   ¦  «                             dd¦  «        }
|                      |
|¦  «         |
S )a  
        Matrix multiplication of query and key tensors using with a sliding window attention pattern. This
        implementation splits the input into overlapping chunks of size 2w (e.g. 512 for pretrained LEDEncoder) with an
        overlap of size window_overlap
        rD   r   z&Sequence length should be multiple of z. Given r®   r¯   r   r¬   Fzbcxd,bcyd->bcxy)r   r   r   r   )r£   Nr   )r)   r.   r´   r|   rŒ   r¼   Úgetattrri   Úeinsumr¤   r   r   rÍ   )r?   r^   r_   r©   r”   rJ   rZ   r[   Úchunks_countÚ!diagonal_chunked_attention_scoresÚdiagonal_attention_scoress              r"   r€   z8LEDEncoderSelfAttention._sliding_chunks_query_key_matmul“  sJ  € ð 49·:²:±<´<Ñ0ˆ
�G˜Y¨Ø˜.¨1Ñ,Ñ-°Ò2Ð2Ð2ØZ°^ÀaÑ5GÐZÐZÐQXÐZÐZñ 3Ô2Ð2ð �zŠz‰|Œ|˜sŸxšx™zœzÒ)Ð)Ð)Ð)å”y ¨.ÈÐPÑPÔPÐSTÑTˆð —’  1Ñ%Ô%×-Ò-¨j¸9Ñ.DÀgÈxÑXÔXˆØ�mŠm˜A˜qÑ!Ô!×)Ò)¨*°yÑ*@À'È8ÑTÔTˆà—’˜E >µ7¸4¼;ÈÐW\Ñ3]Ô3]Ñ^Ô^ˆØ�kŠk˜#˜~­w°t´{ÀMÐSXÑ/YÔ/YÑZÔZˆõ -2¬LÐ9JÈUÐTWÈLÑ,YÔ,YÐ)ð -1×,QÒ,QØ-°|ð -Rñ -
ô -
Ð)ð %F×$OÒ$OØ˜)Ñ# \°AÑ%5°~À~ÐXYÑGYÐ\]ÑG]Ð^ñ%
ô %
Ð!ð AbØˆAˆAˆqˆqˆq�/�>�/Ð#7 ^°aÑ%7Ð#7Ð7ôA
Ð! ! ! ! S b S¨!¨!¨!¨^¨_¨_Ð"<Ñ=ð @aØˆAˆAˆr�>�?�?Ð$8 n°qÑ&8Ð$8Ð8ô@
Ð! ! ! ! R¨¨¨¨N¨O¨OÐ";Ñ<ð @aØˆAˆAˆqˆqˆq�N QÑ&Ð'¨"Ð,¨n¸qÑ.@Ð.BÐ.BÐBô@
Ð! ! ! ! Q R R¨¨¨¨O¨^¨OÐ";Ñ<ð OpØˆAˆAˆqÐ&�N QÑ&Ð&¨¨NÑ(:Ð(<Ð(<Ð<ôO
Ð! ! ! ! Q¨¨.Ð(8¸!¸NÐ:JÐ"JÑKð
 %>×$BÒ$BØ˜	 7¨A°Ñ,>ÀÑ,Bñ%
ô %
ç
Š)�A�q‰/Œ/ð 	"ð 	×$Ò$Ð%>ÀÑOÔOÐOØ(Ð(r$   ry   r`   c                 óà  — |                      ¦   «         \  }}}}||dz  z  dk    sJ ‚|                      ¦   «         dd…         |                      ¦   «         dd…         k    sJ ‚|                      d¦  «        d|z  dz   k    sJ ‚t          j        ||d¬¦  «        dz
  }|                     dd¦  «                             ||z  t          j        ||d¬¦  «        |d|z  dz   ¦  «        }	|                     dd¦  «                             ||z  ||¦  «        }t
          j                             |dd||fd¬	¦  «        }
||z  |dz   d|z  |f}|
                     ¦   «         }|d         ||d         z  |d         |d         f}|
 	                    ||¬
¦  «        }|  
                    |	¦  «        }	t          j        d|	|f¦  «        }|                     ||||¦  «                             dd¦  «        S )z¢
        Same as _sliding_chunks_query_key_matmul but for attn_probs and value tensors. Returned tensor will be of the
        same shape as `attn_probs`
        rD   r   Nr   r   r®   r¯   r   ©r`   r±   zbcwd,bcdh->bcwh)r)   r.   r´   r|   rŒ   r   r‡   r¡   r²   rµ   r«   rÐ   r   )r?   ry   r`   r©   r”   rJ   rZ   r[   rÑ   Úchunked_attn_probsÚpadded_valueÚchunked_value_sizeÚchunked_value_strideÚchunked_valueÚcontexts                  r"   r‹   z?LEDEncoderSelfAttention._sliding_chunks_matmul_attn_probs_valueÕ  s)  € ð 49·:²:±<´<Ñ0ˆ
�G˜Y¨à˜.¨1Ñ,Ñ-°Ò2Ð2Ð2Ð2Ø�ŠÑ Ô   ! Ô$¨¯
ª
©¬°R°a°RÔ(8Ò8Ð8Ð8Ð8Ø�Š˜qÑ!Ô! Q¨Ñ%7¸!Ñ%;Ò;Ð;Ð;Ð;Ý”y ¨.ÈÐPÑPÔPÐSTÑTˆð (×1Ò1°!°QÑ7Ô7×?Ò?Ø˜Ñ"ÝŒI�g˜~¸WÐEÑEÔEØØ�Ñ Ñ"ñ	
ô 
Ðð —’  1Ñ%Ô%×-Ò-¨j¸9Ñ.DÀgÈxÑXÔXˆõ ”}×(Ò(¨°°A°~À~Ð0VÐ^`Ð(ÑaÔaˆð )¨9Ñ4°lÀQÑ6FÈÈNÑHZÐ\dÐeÐØ+×2Ò2Ñ4Ô4Ðà  Ô#ØÐ1°!Ô4Ñ4Ø  Ô#Ø  Ô#ð	 
Ðð %×/Ò/Ð5GÐPdÐ/ÑeÔeˆà!×6Ò6Ð7IÑJÔJÐå”,Ð0Ð3EÀ}Ð2UÑVÔVˆØ�|Š|˜J¨	°7¸HÑEÔE×OÒOÐPQÐSTÑUÔUÐUr$   c                 óx  — |                       ¦   «                              d¬¦  «        }|                     ¦   «         }|                      d¬¦  «        }t	          j        || j        ¬¦  «        |                     d¬¦  «        k     }|                     d¬¦  «        }|dk                         d¬¦  «        }||||fS )z<compute global attn indices required throughout forward passr   rq   T)Úas_tupler³   r   r   )rG   ÚsumÚmaxÚnonzeror.   rF   rE   Ú	unsqueeze)r‘   Únum_global_attn_indicesrm   rn   Úis_local_index_global_attnro   rp   s          r"   r„   z0LEDEncoderSelfAttention._get_global_attn_indices  sà   € ð #7×";Ò";Ñ"=Ô"=×"AÒ"AÀaÐ"AÑ"HÔ"HÐð '>×&AÒ&AÑ&CÔ&CÐ#ð (<×'CÒ'CÈTÐ'CÑ'RÔ'RÐ$õ &+¤\Ø'Ð0DÔ0Kð&
ñ &
ô &
à#×-Ò-°"Ð-Ñ5Ô5ò&6Ð"ð
 .H×-OÒ-OÐY]Ð-OÑ-^Ô-^Ð*ð 2LÈqÒ1P×0YÒ0YÐcgÐ0YÑ0hÔ0hÐ-à'Ø(Ø.Ø1ð	
ð 	
r$   c                 ój  — |j         d         }|                     ||| j        | j        ¦  «        }||         ||<   t	          j        d||f¦  «        }	|	                     dd¦  «        }	t	          j        |	j        ¦  «        j	        |	|d         |d         d d …d d …f<   |	                     dd¦  «        }	|	S )Nr   zblhd,bshd->blhsr   r   )
r   r   rZ   r[   r.   rÐ   r|   r/   r&   r0   )
r?   rl   rk   rm   rn   ro   rp   r”   Úkey_vectors_only_globalÚattn_probs_from_global_keys
             r"   r…   z:LEDEncoderSelfAttention._concat_with_global_key_attn_probs  sæ   € ð !Ô& qÔ)ˆ
ð #.×"7Ò"7ØÐ3°T´^ÀTÄ]ñ#
ô #
Ðð GRÐRnÔFoÐÐ BÑCõ &+¤\Ð2CÀmÐUlÐEmÑ%nÔ%nÐ"ð &@×%IÒ%IÈ!ÈQÑ%OÔ%OÐ"õ ŒKÐ2Ô8Ñ9Ô9Ô=ð 	#Ø1°!Ô4Ð6[Ð\]Ô6^Ð`aÐ`aÐ`aÐcdÐcdÐcdÐdñ	
ð &@×%IÒ%IÈ!ÈQÑ%OÔ%OÐ"à)Ð)r$   c                 óN  — |j         d         }|                     dd|¦  «        }|                     ||| j        | j        ¦  «        }||         ||<   t          j        |                     dd¦  «                             ¦   «         |                     dd¦  «                             ¦   «         ¦  «                             dd¦  «        }	|                     d|| 	                    d¦  «        |z
  ¦  «         
                    ¦   «         }
|                      |
|| j        ¦  «        }|	|z   S )Nr   r   r   rD   )r   Únarrowr   rZ   r[   r.   Úmatmulr|   r   r)   r�   r‹   rh   )r?   rx   ry   rm   rn   ro   r”   Úattn_probs_only_globalÚvalue_vectors_only_globalÚattn_output_only_globalÚattn_probs_without_globalÚattn_output_without_globals               r"   rŠ   z@LEDEncoderSelfAttention._compute_attn_output_with_global_indices<  s6  € ð  Ô% aÔ(ˆ
ð ",×!2Ò!2°2°qÐ:UÑ!VÔ!VÐà$1×$;Ò$;ØÐ3°T´^ÀTÄ]ñ%
ô %
Ð!ð IVÐVrÔHsÐ!Ð"DÑEõ
 #(¤,Ø"×,Ò,¨Q°Ñ2Ô2×8Ò8Ñ:Ô:Ð<U×<_Ò<_Ð`aÐcdÑ<eÔ<e×<kÒ<kÑ<mÔ<mñ#
ô #
ç
Š)�A�q‰/Œ/ð 	 ð
 %/×$5Ò$5ØÐ+¨Z¯_ª_¸RÑ-@Ô-@ÐC^Ñ-^ñ%
ô %
ç
Š*‰,Œ,ð 	"ð
 &*×%QÒ%QØ% }°dÔ6Uñ&
ô &
Ð"ð 'Ð)CÑCÐCr$   c                 ó(  — |j         d d…         \  }}|                     ||| j        ¦  «        }	||d d d…                  |	|d d d…         <   |                      |	¦  «        }
|                      |¦  «        }|                      |¦  «        }|
t          j        | j        ¦  «        z  }
|
 	                    ¦   «          
                    ||| j        z  | j        ¦  «                             dd¦  «        }
| 	                    ¦   «          
                    d|| j        z  | j        ¦  «                             dd¦  «        }| 	                    ¦   «          
                    d|| j        z  | j        ¦  «                             dd¦  «        }t          j        |
|                     dd¦  «        ¦  «        }t          |                     ¦   «         ¦  «        || j        z  ||gk    s.J d|| j        z  ||f› d|                     ¦   «         › d�¦   «         ‚| 
                    || j        ||¦  «        }|                     dd¦  «        }t          j        |j        ¦  «        j        ||d         |d         d d …d d …f<   |                     dd¦  «        }|                     |d d …d d d d …f         t          j        |j        ¦  «        j        ¦  «        }| 
                    || j        z  ||¦  «        }t*          j                             |dt          j        ¬¦  «        }t*          j                             |                     |¦  «        | j        | j        ¬	¦  «        }t          j        ||¦  «        }t          |                     ¦   «         ¦  «        || j        z  || j        gk    s3J d
|| j        z  || j        f› d|                     ¦   «         › d�¦   «         ‚| 
                    || j        ||¦  «        }| 
                    || j        || j        ¦  «        }||fS )NrD   r   r   r   z7global_attn_scores have the wrong size. Size should be ú	, but is ú.rs   ru   z=global_attn_output tensor has the wrong size. Size should be )r   r   r\   ra   rb   rc   r}   r~   r[   r�   r   rZ   r|   r.   Úbmmrƒ   r)   r/   r&   r0   r,   r   r‡   rˆ   r‰   re   r�   rw   )r?   rz   rm   ro   rn   rp   r{   rJ   r”   Úglobal_attn_hidden_statesÚ global_query_vectors_only_globalÚglobal_key_vectorsÚglobal_value_vectorsÚglobal_attn_scoresÚglobal_attn_probs_floatrœ   r›   s                    r"   rŽ   z?LEDEncoderSelfAttention._compute_global_attn_output_from_hidden`  sŠ  € ð ,Ô1°"°1°"Ô5Ñˆ�ð %2×$;Ò$;Ð<WÐYcÐeiÔesÑ$tÔ$tÐ!ØN[Ø(¨¨¨2¨Ô.ôO
Ð!Ð"DÀTÀTÀrÀTÔ"JÑKð
 ,0×+<Ò+<Ð=VÑ+WÔ+WÐ(Ø!Ÿ_š_¨]Ñ;Ô;ÐØ#×0Ò0°Ñ?Ô?Ðð 	)­D¬I°d´mÑ,DÔ,DÑDÐ(ð -×7Ò7Ñ9Ô9ßŠTÐ-¨z¸D¼NÑ/JÈDÌMÑZÔZßŠY�q˜!‰_Œ_ð 	)ð ×)Ò)Ñ+Ô+×0Ò0°°ZÀ$Ä.Ñ5PÐRVÔR_Ñ`Ô`×jÒjÐklÐnoÑpÔpð 	ð !×+Ò+Ñ-Ô-×2Ò2°2°zÀDÄNÑ7RÐTXÔTaÑbÔb×lÒlÐmnÐpqÑrÔrð 	õ
 #œYÐ'GÐI[×IeÒIeÐfgÐijÑIkÔIkÑlÔlÐåÐ&×+Ò+Ñ-Ô-Ñ.Ô.Ø˜œÑ'Ø'Øð3
ò 
ð 
ð 
ð
-Ø˜dœnÑ,Ð.IÈ7ÐSð-ð -à"×'Ò'Ñ)Ô)ð-ð -ð -ñ
ô 
ð 
ð 0×4Ò4°ZÀÄÐQlÐnuÑvÔvÐð 0×9Ò9¸!¸QÑ?Ô?Ðõ ŒKÐ*Ô0Ñ1Ô1Ô5ð 	Ø1°!Ô4Ð6[Ð\]Ô6^Ð`aÐ`aÐ`aÐcdÐcdÐcdÐdñ	
ð 0×9Ò9¸!¸QÑ?Ô?Ðà/×;Ò;Ø˜A˜A˜A˜t T¨1¨1¨1Ð,Ô-ÝŒKÐ*Ô0Ñ1Ô1Ô5ñ
ô 
Ðð
 0×4Ò4°ZÀ$Ä.Ñ5PÐRmÐovÑwÔwÐõ #%¤-×"7Ò"7Ø B­e¬mð #8ñ #
ô #
Ðõ œM×1Ò1Ø#×+Ò+Ð,>Ñ?Ô?À4Ä<ÐZ^ÔZgð 2ñ 
ô 
Ðõ
 #œYÐ'8Ð:NÑOÔOÐåÐ&×+Ò+Ñ-Ô-Ñ.Ô.Ø˜œÑ'Ø'ØŒMð3
ò 
ð 
ð 
ð
-Ø˜dœnÑ,Ð.IÈ4Ì=ÐYð-ð -à"×'Ò'Ñ)Ô)ð-ð -ð -ñ
ô 
ð 
ð .×2Ò2°:¸t¼~ÐOjÐlsÑtÔtÐØ/×4Ò4Ø˜œÐ(CÀTÄ]ñ
ô 
Ðð "Ð#4Ð4Ð4r$   ©NNNNF)F)rL   rM   rN   r>   rI   Ústaticmethodr¤   r«   r-   r¼   r.   ÚTensorrÍ   rP   r€   r‹   r„   r…   rŠ   rŽ   rR   rS   s   @r"   rU   rU   Z   sµ  ø€ € € € € ð!ð !ð !ð !ð !ðL ØØ!ØØð^eð ^eð ^eð ^eð@ ð$ð $ñ „\ð$ð ð.%ð .%ñ „\ð.%ð` ð&"ð &"¸4ð &"ð &"ð &"ñ „\ð&"ðP ð2À5Ä<ð 2ð 2ð 2ñ „\ð2ð@)°e´lð @)ÈÌð @)Ðgjð @)ð @)ð @)ð @)ðD*VØœ,ð*VØ/4¬|ð*VØMPð*Vð *Vð *Vð *VðX ð
ð 
ñ „\ð
ð8*ð *ð *ð<"Dð "Dð "DðH]5ð ]5ð ]5ð ]5ð ]5ð ]5ð ]5r$   rU   c                   óÖ   ‡ — e Zd Zˆ fd„Z	 	 	 	 	 ddej        dej        dz  dej        dz  dej        dz  dedz  d	ed
eej        ej        dz  eej                 dz  f         fd„Zˆ xZ	S )ÚLEDEncoderAttentionc                 ó¼   •— t          ¦   «                              ¦   «          t          ||¬¦  «        | _        t	          j        |j        |j        ¦  «        | _        d S )N)rf   )r=   r>   rU   Úlongformer_self_attnr   r]   Úd_modelÚoutput©r?   ri   rf   r@   s      €r"   r>   zLEDEncoderAttention.__init__Á  sI   ø€ Ý‰Œ×ÒÑÔÐÝ$;¸FÈXÐ$VÑ$VÔ$VˆÔ!Ý”i ¤°´Ñ?Ô?ˆŒˆˆr$   NFrz   r�   r{   r‘   r’   r“   r½   c                 óŽ   — |                       ||||||¬¦  «        }|                      |d         ¦  «        }|f|dd…         z   }	|	S )ú#Input shape: Batch x Time x Channel©rz   r�   r{   r‘   r’   r“   r   r   N)rÿ   r  )
r?   rz   r�   r{   r‘   r’   r“   Úself_outputsrš   rž   s
             r"   rI   zLEDEncoderAttention.forwardÆ  sa   € ð ×0Ò0Ø'Ø)Ø+Ø!5Ø)Ø/ð 1ñ 
ô 
ˆð —k’k ,¨q¤/Ñ2Ô2ˆØ�. <°°°Ô#3Ñ3ˆàˆr$   rù   )
rL   rM   rN   r>   r.   rû   r-   ÚtuplerI   rR   rS   s   @r"   rý   rý   À  sç   ø€ € € € € ð@ð @ð @ð @ð @ð /3Ø/3Ø48Ø&*Ø"'ðð à”|ðð œ tÑ+ðð œ¨Ñ,ð	ð
 $œl¨TÑ1ðð ˜t™ðð  ðð 
ˆuŒ|˜Uœ\¨DÑ0°%¸¼Ô2EÈÑ2LÐLÔ	Mðð ð ð ð ð ð ð r$   rý   c                   óê   ‡ — e Zd ZdZ	 	 	 	 ddedededz  d	edz  d
edz  dedz  fˆ fd„Z	 	 	 	 ddej	        dej	        dz  de
dz  dej	        dz  dedeej	        ej	        dz  e
dz  f         fd„Zˆ xZS )ÚLEDDecoderAttentionz=Multi-headed attention from 'Attention Is All You Need' paperrt   FTNr\   rZ   re   Ú
is_decoderÚbiasÚ	layer_idxc                 óü  •— t          ¦   «                              ¦   «          || _        || _        || _        ||z  | _        | j        |z  | j        k    rt          d| j        › d|› d�¦  «        ‚| j        dz  | _        || _        || _	        t          j        |||¬¦  «        | _        t          j        |||¬¦  «        | _        t          j        |||¬¦  «        | _        t          j        |||¬¦  «        | _        d S )Nz;embed_dim must be divisible by num_heads (got `embed_dim`: z and `num_heads`: z).g      à¿©r  )r=   r>   r\   rZ   re   r[   r   Úscalingr
  r  r   r]   Úk_projÚv_projÚq_projÚout_proj)r?   r\   rZ   re   r
  r  r  r@   s          €r"   r>   zLEDDecoderAttention.__init__ã  s	  ø€ õ 	‰Œ×ÒÑÔÐØ"ˆŒØ"ˆŒØˆŒØ! YÑ.ˆŒØŒ=˜9Ñ$¨¬Ò6Ð6Ýð"ÈdÌnð "ð "Øð"ð "ð "ñô ð ð ”} dÑ*ˆŒØ$ˆŒØ"ˆŒå”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝ”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝ”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝœ	 )¨Y¸TÐBÑBÔBˆŒˆˆr$   rz   Úkey_value_statesÚpast_key_valuesr�   r“   r½   c                 ó	  — |du}|                      ¦   «         \  }}}	|                      |¦  «        | j        z  }
d}|�Ht          |t          ¦  «        r1|j                             | j        ¦  «        }|r|j        }n
|j	        }n|}|r|n|}|r3|�1|r/|j
        | j                 j        }|j
        | j                 j        }nÝ|                      |¦  «        }|                      |¦  «        }|                     |d| j        | j        ¦  «                             dd¦  «        }|                     |d| j        | j        ¦  «                             dd¦  «        }|�E|                     ||| j        ¦  «        \  }}|r$t          |t          ¦  «        rd|j        | j        <   || j        z  d| j        f}|
                     ||| j        | j        ¦  «                             dd¦  «        }
 |
j        |Ž }
 |j        |Ž } |j        |Ž }|                      d¦  «        }t+          j        |
|                     dd¦  «        ¦  «        }|                      ¦   «         || j        z  ||fk    r2t/          d|| j        z  ||f› d|                      ¦   «         › �¦  «        ‚|�†|                      ¦   «         |d||fk    r+t/          d	|d||f› d|                      ¦   «         › �¦  «        ‚|                     || j        ||¦  «        |z   }|                     || j        z  ||¦  «        }t0          j                             |d¬
¦  «        }|r=|                     || j        ||¦  «        }|                     || j        z  ||¦  «        }nd}t0          j                             || j        | j        ¬¦  «        }t+          j        ||¦  «        }|                      ¦   «         || j        z  || j        fk    r5t/          d|| j        || j        f› d|                      ¦   «         › �¦  «        ‚|                     || j        || j        ¦  «                             dd¦  «                             |||	¦  «        }|                      |¦  «        }|||fS )r  NFr   r   rD   Tz$Attention weights should be of size rð   z!Attention mask should be of size rq   ru   z `attn_output` should be of size )r)   r  r  Ú
isinstancer   Ú
is_updatedÚgetr  Úcross_attention_cacheÚself_attention_cacheÚlayersÚkeysÚvaluesr  r  r   rZ   r[   r|   ÚupdaterŒ   r.   rò   r   r   r‡   rˆ   re   rw   r  )r?   rz   r  r  r�   r“   Úis_cross_attentionr1   r'   r\   Úquery_statesr  Úcurr_past_key_valuesÚcurrent_statesÚ
key_statesÚvalue_statesÚ
proj_shaper2   Úattn_weightsÚattn_weights_reshapedry   rš   s                         r"   rI   zLEDDecoderAttention.forwardÿ  sÜ  € ð .°TÐ9ÐØ"/×"4Ò"4Ñ"6Ô"6ÑˆˆW�ið —{’{ =Ñ1Ô1°D´LÑ@ˆàˆ
ØÐ&Ý˜/Õ+>Ñ?Ô?ð 7Ø,Ô7×;Ò;¸D¼NÑKÔK�
Ø%ð Pà+:Ô+PÐ(Ð(à+:Ô+OÐ(Ð(à'6Ð$à-?ÐRÐ)Ð)À]ˆØð 	F /Ð"=À*Ð"=à-Ô4°T´^ÔDÔIˆJØ/Ô6°t´~ÔFÔMˆLˆLàŸš ^Ñ4Ô4ˆJØŸ;š; ~Ñ6Ô6ˆLØ#Ÿš¨¨b°$´.À$Ä-ÑPÔP×ZÒZÐ[\Ð^_Ñ`Ô`ˆJØ'×,Ò,¨S°"°d´nÀdÄmÑTÔT×^Ò^Ð_`ÐbcÑdÔdˆLàÐ*à+?×+FÒ+FÀzÐS_ÐaeÔaoÑ+pÔ+pÑ(�
˜Là%ð F­*°_ÕFYÑ*ZÔ*Zð FØAE�OÔ.¨t¬~Ñ>à˜DœNÑ*¨B°´Ð>ˆ
Ø#×(Ò(¨¨g°t´~ÀtÄ}ÑUÔU×_Ò_Ð`aÐcdÑeÔeˆØ+�|Ô+¨ZÐ8ˆØ'�ZÔ'¨Ð4ˆ
Ø+�|Ô+¨ZÐ8ˆà—/’/ !Ñ$Ô$ˆÝ”y ¨z×/CÒ/CÀAÀqÑ/IÔ/IÑJÔJˆà×ÒÑÔ 3¨¬Ñ#7¸À'Ð"JÒJÐJÝð*¸¸d¼nÑ8LÈgÐW^Ð7_ð *ð *Ø ×%Ò%Ñ'Ô'ð*ð *ñô ð ð
 Ð%Ø×"Ò"Ñ$Ô$¨¨a°¸'Ð(BÒBÐBÝ Øt¸¸aÀÈ'Ð8RÐtÐtÐ]k×]pÒ]pÑ]rÔ]rÐtÐtñô ð ð (×,Ò,¨S°$´.À'È7ÑSÔSÐVdÑdˆLØ'×,Ò,¨S°4´>Ñ-AÀ7ÈGÑTÔTˆLå”}×,Ò,¨\¸rÐ,ÑBÔBˆàð 	)ð
 %1×$5Ò$5°c¸4¼>È7ÐT[Ñ$\Ô$\Ð!Ø0×5Ò5°c¸D¼NÑ6JÈGÐU\Ñ]Ô]ˆLˆLà$(Ð!å”]×*Ò*¨<¸4¼<ÐRVÔR_Ð*Ñ`Ô`ˆ
å”i 
¨LÑ9Ô9ˆà×ÒÑÔ #¨¬Ñ"6¸ÀÄÐ!OÒOÐOÝð)°C¸¼ÈÐRVÔR_Ð3`ð )ð )Ø×$Ò$Ñ&Ô&ð)ð )ñô ð ð ×Ò˜S $¤.°'¸4¼=ÑIÔIßŠY�q˜!‰_Œ_ßŠW�S˜' 9Ñ-Ô-ð 	ð —m’m KÑ0Ô0ˆàÐ1°?ÐBÐBr$   )rt   FTN©NNNF)rL   rM   rN   rO   rP   rÄ   r-   r>   r.   rû   r	   r  rI   rR   rS   s   @r"   r	  r	  à  sY  ø€ € € € € ØGÐGð !$Ø"'Ø Ø!%ðCð CàðCð ðCð ˜‘ð	Cð
 ˜4‘KðCð �T‰kðCð ˜$‘;ðCð Cð Cð Cð Cð Cð> 15Ø(,Ø.2Ø"'ðeCð eCà”|ðeCð  œ,¨Ñ-ðeCð  ™ð	eCð
 œ tÑ+ðeCð  ðeCð 
ˆuŒ|˜Uœ\¨DÑ0°%¸$±,Ð>Ô	?ðeCð eCð eCð eCð eCð eCð eCð eCr$   r	  c                   óV   ‡ — e Zd Zdedefˆ fd„Z	 	 	 	 d	dej        dej        fd„Zˆ xZ	S )
ÚLEDEncoderLayerri   rf   c                 óð  •— t          ¦   «                              ¦   «          |j        | _        t	          ||¦  «        | _        t          j        | j        ¦  «        | _        |j	        | _	        t          |j                 | _        |j        | _        t          j        | j        |j        ¦  «        | _        t          j        |j        | j        ¦  «        | _        t          j        | j        ¦  «        | _        d S r<   )r=   r>   r   r\   rý   Ú	self_attnr   Ú	LayerNormÚself_attn_layer_normre   r   Úactivation_functionÚactivation_fnÚactivation_dropoutr]   Úencoder_ffn_dimÚfc1Úfc2Úfinal_layer_normr  s      €r"   r>   zLEDEncoderLayer.__init__h  sµ   ø€ Ý‰Œ×ÒÑÔÐØœˆŒÝ,¨V°XÑ>Ô>ˆŒÝ$&¤L°´Ñ$@Ô$@ˆÔ!Ø”~ˆŒÝ# FÔ$>Ô?ˆÔØ"(Ô";ˆÔÝ”9˜Tœ^¨VÔ-CÑDÔDˆŒÝ”9˜VÔ3°T´^ÑDÔDˆŒÝ "¤¨T¬^Ñ <Ô <ˆÔÐÐr$   NFrz   r�   c                 ó>  — |}|                       ||||||¬¦  «        }|d         }t          j                             || j        | j        ¬¦  «        }||z   }|                      |¦  «        }|}|                      |                      |¦  «        ¦  «        }t          j                             || j        | j        ¬¦  «        }|  	                    |¦  «        }t          j                             || j        | j        ¬¦  «        }||z   }|  
                    |¦  «        }|j        t          j        k    r_t          j        |¦  «                             ¦   «         s9t          j        |j        ¦  «        j        dz
  }	t          j        ||	 |	¬¦  «        }|f|dd…         z   S )a>  
        Args:
            hidden_states (`torch.FloatTensor`): input to the layer of shape *(batch, seq_len, embed_dim)*
            attention_mask (`torch.FloatTensor`): attention mask of size
                *(batch, 1, tgt_len, src_len)* where padding elements are indicated by very large negative values.
        r  r   ru   iè  )r0   rß   r   N)r-  r   r‡   re   rw   r/  r1  r4  r2  r5  r6  r&   r.   Úfloat16ÚisfiniteÚallr/   rß   Úclamp)
r?   rz   r�   r{   r‘   r’   r“   ÚresidualÚattn_outputsÚclamp_values
             r"   rI   zLEDEncoderLayer.forwardt  sˆ  € ð !ˆØ—~’~Ø'Ø)Ø+Ø!5Ø)Ø/ð &ñ 
ô 
ˆð % QœˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆØ×1Ò1°-Ñ@Ô@ˆà ˆØ×*Ò*¨4¯8ª8°MÑ+BÔ+BÑCÔCˆÝœ×-Ò-¨m¸tÔ?VÐaeÔanÐ-ÑoÔoˆØŸš Ñ/Ô/ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆØ×-Ò-¨mÑ<Ô<ˆàÔ¥%¤-Ò/Ð/½¼À}Ñ8UÔ8U×8YÒ8YÑ8[Ô8[Ð/Ýœ+ mÔ&9Ñ:Ô:Ô>ÀÑEˆKÝ!œK¨¸K¸<È[ÐYÑYÔYˆMØÐ ,¨q¨r¨rÔ"2Ñ2Ð2r$   r)  )
rL   rM   rN   r   rP   r>   r.   rû   rI   rR   rS   s   @r"   r+  r+  g  sˆ   ø€ € € € € ð
=˜yð 
=°Cð 
=ð 
=ð 
=ð 
=ð 
=ð 
=ð  Ø!ØØð(3ð (3à”|ð(3ð œð(3ð (3ð (3ð (3ð (3ð (3ð (3ð (3r$   r+  c                   ó¤   ‡ — e Zd Zddefˆ fd„Z	 	 	 	 	 	 ddej        dej        dz  dej        dz  d	ej        dz  d
edz  dedz  dedz  fd„Z	ˆ xZ
S )ÚLEDDecoderLayerNri   c                 ó¢  •— t          ¦   «                              ¦   «          |j        | _        t	          | j        |j        |j        d|¬¦  «        | _        |j        | _        t          |j
                 | _        |j        | _        t          j        | j        ¦  «        | _        t	          | j        |j        |j        d|¬¦  «        | _        t          j        | j        ¦  «        | _        t          j        | j        |j        ¦  «        | _        t          j        |j        | j        ¦  «        | _        t          j        | j        ¦  «        | _        d S )NT)r\   rZ   re   r
  r  )re   r
  r  )r=   r>   r   r\   r	  Údecoder_attention_headsÚattention_dropoutr-  re   r   r0  r1  r2  r   r.  r/  Úencoder_attnÚencoder_attn_layer_normr]   Údecoder_ffn_dimr4  r5  r6  )r?   ri   r  r@   s      €r"   r>   zLEDDecoderLayer.__init__   s  ø€ Ý‰Œ×ÒÑÔÐØœˆŒå,Ø”nØÔ4ØÔ,ØØð
ñ 
ô 
ˆŒð ”~ˆŒÝ# FÔ$>Ô?ˆÔØ"(Ô";ˆÔå$&¤L°´Ñ$@Ô$@ˆÔ!Ý/ØŒNØÔ*ØÔ,ØØð
ñ 
ô 
ˆÔõ (*¤|°D´NÑ'CÔ'CˆÔ$Ý”9˜Tœ^¨VÔ-CÑDÔDˆŒÝ”9˜VÔ3°T´^ÑDÔDˆŒÝ "¤¨T¬^Ñ <Ô <ˆÔÐÐr$   FTrz   r�   Úencoder_hidden_statesÚencoder_attention_maskr  r“   Ú	use_cachec                 ó2  — |}	|                       ||||¬¦  «        \  }}
}t          j                             || j        | j        ¬¦  «        }|	|z   }|                      |¦  «        }d}d}|�f|}	|                      |||||¬¦  «        \  }}}t          j                             || j        | j        ¬¦  «        }|	|z   }|                      |¦  «        }|}	|                      |  	                    |¦  «        ¦  «        }t          j                             || j
        | j        ¬¦  «        }|                      |¦  «        }t          j                             || j        | j        ¬¦  «        }|	|z   }|                      |¦  «        }|f}|r||
|fz  }|r||fz  }|S )a˜  
        Args:
            hidden_states (`torch.FloatTensor`): input to the layer of shape *(batch, seq_len, embed_dim)*
            attention_mask (`torch.FloatTensor`): attention mask of size
                *(batch, 1, tgt_len, src_len)* where padding elements are indicated by very large negative values.
            encoder_hidden_states (`torch.FloatTensor`):
                cross attention input to the layer of shape *(batch, seq_len, embed_dim)*
            encoder_attention_mask (`torch.FloatTensor`): encoder attention mask of size
                *(batch, 1, tgt_len, src_len)* where padding elements are indicated by very large negative values.
            past_key_values (`Cache`): cached past key and value projection states
            output_attentions (`bool`): Whether the base model outputs attentions.
                This requires the attentions tensor to be reshaped in this function.
        )rz   r  r�   r“   ru   N)rz   r  r�   r  r“   )r-  r   r‡   re   rw   r/  rD  rE  r1  r4  r2  r5  r6  )r?   rz   r�   rG  rH  r  r“   rI  Úkwargsr<  Úself_attn_weightsÚpresent_key_valueÚcross_attn_present_key_valueÚcross_attn_weightsrž   s                  r"   rI   zLEDDecoderLayer.forward¼  sß  € ð0 !ˆð ?C¿nºnØ'Ø+Ø)Ø/ð	 ?Mñ ?
ô ?
Ñ;ˆÐ(Ð*;õ œ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆØ×1Ò1°-Ñ@Ô@ˆð (,Ð$Ø!ÐØ Ð,Ø$ˆHàNR×N_ÒN_Ø+Ø!6Ø5Ø /Ø"3ð O`ñ Oô OÑKˆMÐ-Ð/Kõ œM×1Ò1°-À4Ä<ÐZ^ÔZgÐ1ÑhÔhˆMØ$ }Ñ4ˆMØ ×8Ò8¸ÑGÔGˆMð !ˆØ×*Ò*¨4¯8ª8°MÑ+BÔ+BÑCÔCˆÝœ×-Ò-¨m¸tÔ?VÐaeÔanÐ-ÑoÔoˆØŸš Ñ/Ô/ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆØ×-Ò-¨mÑ<Ô<ˆà Ð"ˆàð 	?ØÐ)Ð+=Ð>Ñ>ˆGàð 	*Ø˜Ð)Ñ)ˆGàˆr$   r<   )NNNNFT)rL   rM   rN   r   r>   r.   rû   r	   r-   rI   rR   rS   s   @r"   r@  r@  Ÿ  sí   ø€ € € € € ð=ð =˜yð =ð =ð =ð =ð =ð =ð> /3Ø59Ø6:Ø(,Ø).Ø!%ðGð Gà”|ðGð œ tÑ+ðGð  %œ|¨dÑ2ð	Gð
 !&¤¨tÑ 3ðGð  ™ðGð   $™;ðGð ˜$‘;ðGð Gð Gð Gð Gð Gð Gð Gr$   r@  c                   óJ   ‡ — e Zd ZdZdedededefˆ fd„Zdej        fd„Z	ˆ xZ
S )	ÚLEDClassificationHeadz-Head for sentence-level classification tasks.Ú	input_dimÚ	inner_dimÚnum_classesÚpooler_dropoutc                 óä   •— t          ¦   «                              ¦   «          t          j        ||¦  «        | _        t          j        |¬¦  «        | _        t          j        ||¦  «        | _        d S )N)rv   )r=   r>   r   r]   ÚdenseÚDropoutre   r  )r?   rR  rS  rT  rU  r@   s        €r"   r>   zLEDClassificationHead.__init__	  sY   ø€ õ 	‰Œ×ÒÑÔÐÝ”Y˜y¨)Ñ4Ô4ˆŒ
Ý”z NÐ3Ñ3Ô3ˆŒÝœ	 )¨[Ñ9Ô9ˆŒˆˆr$   rz   c                 óÖ   — |                       |¦  «        }|                      |¦  «        }t          j        |¦  «        }|                       |¦  «        }|                      |¦  «        }|S r<   )re   rW  r.   Útanhr  )r?   rz   s     r"   rI   zLEDClassificationHead.forward  s[   € ØŸš ]Ñ3Ô3ˆØŸ
š
 =Ñ1Ô1ˆÝœ
 =Ñ1Ô1ˆØŸš ]Ñ3Ô3ˆØŸš mÑ4Ô4ˆØÐr$   )rL   rM   rN   rO   rP   rÄ   r>   r.   rû   rI   rR   rS   s   @r"   rQ  rQ    s†   ø€ € € € € Ø7Ð7ð
:àð
:ð ð
:ð ð	
:ð
 ð
:ð 
:ð 
:ð 
:ð 
:ð 
:ð U¤\ð ð ð ð ð ð ð ð r$   rQ  c                   óH   ‡ — e Zd ZU eed<   dZdZed„ ¦   «         Zˆ fd„Z	ˆ xZ
S )ÚLEDPreTrainedModelri   ÚledTc                 ó–   — | j         j        }t          j        g d¢dddd|gg| j        ¬¦  «        }|                     |¦  «        |dœ}|S )N)r   é   é
   é   rD   r   é   é   rD   r³   )r�   r   )ri   r   r.   ÚtensorrE   Úne)r?   Ú	pad_tokenr   Údummy_inputss       r"   rg  zLEDPreTrainedModel.dummy_inputs$  sa   € à”KÔ,ˆ	Ý”LÐ"2Ð"2Ð"2°Q¸¸2¸qÀ)Ð4LÐ!MÐVZÔVaÐbÑbÔbˆ	à'Ÿlšl¨9Ñ5Ô5Ø"ð
ð 
ˆð Ðr$   c                 óª   •— t          ¦   «                              |¦  «         t          |t          ¦  «        rt	          j        |j        ¦  «         d S d S r<   )r=   Ú_init_weightsr  ÚLEDForConditionalGenerationÚinitÚzeros_Úfinal_logits_bias)r?   Úmoduler@   s     €r"   ri  z LEDPreTrainedModel._init_weights.  sQ   ø€ Ý‰Œ×Ò˜fÑ%Ô%Ð%Ý�fÕ9Ñ:Ô:ð 	2ÝŒK˜Ô0Ñ1Ô1Ð1Ð1Ð1ð	2ð 	2r$   )rL   rM   rN   r   Ú__annotations__Úbase_model_prefixÚsupports_gradient_checkpointingÚpropertyrg  ri  rR   rS   s   @r"   r\  r\    sk   ø€ € € € € € àÐÐÑØÐØ&*Ð#àðð ñ „Xðð2ð 2ð 2ð 2ð 2ð 2ð 2ð 2ð 2r$   r\  zi
    Base class for LEDEncoder's outputs, with potential hidden states, local and global attentions.
    )Úcustom_introc                   ó²   — e Zd ZU dZej        ed<   dZeej        df         dz  ed<   dZ	eej        df         dz  ed<   dZ
eej        df         dz  ed<   dS )ÚLEDEncoderBaseModelOutputaI  
    attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
        Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length, x +
        attention_window + 1)`, where `x` is the number of tokens with global attention mask.

        Local attentions weights after the attention softmax, used to compute the weighted average in the
        self-attention heads. Those are the attention weights from every token in the sequence to every token with
        global attention (first `x` values) and to every token in the attention window (remaining `attention_window
        + 1` values). Note that the first `x` values refer to tokens with fixed positions in the text, but the
        remaining `attention_window + 1` values refer to tokens with relative positions: the attention weight of a
        token to itself is located at index `x + attention_window / 2` and the `attention_window / 2` preceding
        (succeeding) values are the attention weights to the `attention_window / 2` preceding (succeeding) tokens.
        If the attention window contains a token with global attention, the attention weight at the corresponding
        index is set to 0; the value should be accessed from the first `x` attention weights. If a token has global
        attention, the attention weights to all other tokens in `attentions` is set to 0, the values should be
        accessed from `global_attentions`.
    global_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
        Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length, x)`,
        where `x` is the number of tokens with global attention mask.

        Global attentions weights after the attention softmax, used to compute the weighted average in the
        self-attention heads. Those are the attention weights from every token with global attention to every token
        in the sequence.
    Úlast_hidden_stateN.rz   Ú
attentionsÚglobal_attentions)rL   rM   rN   rO   r.   ÚFloatTensorro  rz   r  rw  rx  © r$   r"   ru  ru  4  s”   € € € € € € ðð ð2 Ô(Ð(Ð(Ñ(Ø:>€M�5˜Ô*¨CÐ/Ô0°4Ñ7Ð>Ð>Ñ>Ø7;€J��eÔ'¨Ð,Ô-°Ñ4Ð;Ð;Ñ;Ø>BÐ�u˜UÔ.°Ð3Ô4°tÑ;ÐBÐBÑBÐBÐBr$   ru  z‹
    Base class for model encoder's outputs that also contains : pre-computed hidden states that can speed up sequential
    decoding.
    c                   óx  — e Zd ZU dZdZej        dz  ed<   dZe	dz  ed<   dZ
eej        df         dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dZej        dz  ed	<   dZeej        df         dz  ed
<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dS )ÚLEDSeq2SeqModelOutputal  
    last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
        Sequence of hidden-states at the output of the last layer of the decoder of the model.

        If `past_key_values` is used only the last hidden-state of the sequences of shape `(batch_size, 1,
        hidden_size)` is output.
    past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
        It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

        Contains pre-computed hidden-states (key and values in the attention blocks) of the decoder that can be
        used (see `past_key_values` input) to speed up sequential decoding.
    encoder_global_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
        Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length, x)`,
        where `x` is the number of tokens with global attention mask.

        Global attentions weights after the attention softmax, used to compute the weighted average in the
        self-attention heads. Those are the attention weights from every token with global attention to every token
        in the sequence.
    Nrv  r  .Údecoder_hidden_statesÚdecoder_attentionsÚcross_attentionsÚencoder_last_hidden_staterG  Úencoder_attentionsÚencoder_global_attentions)rL   rM   rN   rO   rv  r.   ry  ro  r  r	   r}  r  r~  r  r€  rG  r�  r‚  rz  r$   r"   r|  r|  [  s6  € € € € € € ðð ð( 37Ð�uÔ(¨4Ñ/Ð6Ð6Ñ6Ø$(€O�U˜T‘\Ð(Ð(Ñ(ØBFÐ˜5 Ô!2°CÐ!7Ô8¸4Ñ?ÐFÐFÑFØ?CÐ˜˜eÔ/°Ð4Ô5¸Ñ<ÐCÐCÑCØ=AÐ�e˜EÔ-¨sÐ2Ô3°dÑ:ÐAÐAÑAØ:>Ð˜uÔ0°4Ñ7Ð>Ð>Ñ>ØBFÐ˜5 Ô!2°CÐ!7Ô8¸4Ñ?ÐFÐFÑFØ?CÐ˜˜eÔ/°Ð4Ô5¸Ñ<ÐCÐCÑCØFJÐ˜u UÔ%6¸Ð%;Ô<¸tÑCÐJÐJÑJÐJÐJr$   r|  zF
    Base class for sequence-to-sequence language models outputs.
    c                   ó–  — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dZ	e
dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed	<   dZej        dz  ed
<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dS )ÚLEDSeq2SeqLMOutputaf  
    loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
        Language modeling loss.
    logits (`torch.FloatTensor` of shape `(batch_size, sequence_length, config.vocab_size)`):
        Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
    past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
        It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

        Contains pre-computed hidden-states (key and values in the attention blocks) of the decoder that can be
        used (see `past_key_values` input) to speed up sequential decoding.
    encoder_global_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
        Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length, x)`,
        where `x` is the number of tokens with global attention mask.

        Global attentions weights after the attention softmax, used to compute the weighted average in the
        self-attention heads. Those are the attention weights from every token with global attention to every token
        in the sequence.
    NÚlossÚlogitsr  .r}  r~  r  r€  rG  r�  r‚  ©rL   rM   rN   rO   r…  r.   ry  ro  r†  r  r	   r}  r  r~  r  r€  rG  r�  r‚  rz  r$   r"   r„  r„  ‚  óM  € € € € € € ðð ð& &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø'+€FˆEÔ Ñ$Ð+Ð+Ñ+Ø$(€O�U˜T‘\Ð(Ð(Ñ(ØBFÐ˜5 Ô!2°CÐ!7Ô8¸4Ñ?ÐFÐFÑFØ?CÐ˜˜eÔ/°Ð4Ô5¸Ñ<ÐCÐCÑCØ=AÐ�e˜EÔ-¨sÐ2Ô3°dÑ:ÐAÐAÑAØ:>Ð˜uÔ0°4Ñ7Ð>Ð>Ñ>ØBFÐ˜5 Ô!2°CÐ!7Ô8¸4Ñ?ÐFÐFÑFØ?CÐ˜˜eÔ/°Ð4Ô5¸Ñ<ÐCÐCÑCØFJÐ˜u UÔ%6¸Ð%;Ô<¸tÑCÐJÐJÑJÐJÐJr$   r„  zX
    Base class for outputs of sequence-to-sequence sentence classification models.
    c                   ó–  — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dZ	e
dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed	<   dZej        dz  ed
<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dS )Ú"LEDSeq2SeqSequenceClassifierOutputaf  
    loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `label` is provided):
        Classification (or regression if config.num_labels==1) loss.
    logits (`torch.FloatTensor` of shape `(batch_size, config.num_labels)`):
        Classification (or regression if config.num_labels==1) scores (before SoftMax).
    past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
        It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

        Contains pre-computed hidden-states (key and values in the attention blocks) of the decoder that can be
        used (see `past_key_values` input) to speed up sequential decoding.
    encoder_global_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
        Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length, x)`,
        where `x` is the number of tokens with global attention mask.

        Global attentions weights after the attention softmax, used to compute the weighted average in the
        self-attention heads. Those are the attention weights from every token with global attention to every token
        in the sequence.
    Nr…  r†  r  .r}  r~  r  r€  rG  r�  r‚  r‡  rz  r$   r"   rŠ  rŠ  ¨  rˆ  r$   rŠ  zS
    Base class for outputs of sequence-to-sequence question answering models.
    c                   ó´  — e Zd ZU dZdZej        dz  ed<   dZej        dz  ed<   dZ	ej        dz  ed<   dZ
edz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed	<   dZeej        df         dz  ed
<   dZej        dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dZeej        df         dz  ed<   dS )Ú&LEDSeq2SeqQuestionAnsweringModelOutputaß  
    loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
        Total span extraction loss is the sum of a Cross-Entropy for the start and end positions.
    past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
        It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

        Contains pre-computed hidden-states (key and values in the attention blocks) of the decoder that can be
        used (see `past_key_values` input) to speed up sequential decoding.
    encoder_global_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
        Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length, x)`,
        where `x` is the number of tokens with global attention mask.

        Global attentions weights after the attention softmax, used to compute the weighted average in the
        self-attention heads. Those are the attention weights from every token with global attention to every token
        in the sequence.
    Nr…  Ústart_logitsÚ
end_logitsr  .r}  r~  r  r€  rG  r�  r‚  )rL   rM   rN   rO   r…  r.   ry  ro  r�  rŽ  r  r	   r}  r  r~  r  r€  rG  r�  r‚  rz  r$   r"   rŒ  rŒ  Î  se  € € € € € € ðð ð" &*€Dˆ%Ô
˜dÑ
"Ð)Ð)Ñ)Ø-1€L�%Ô# dÑ*Ð1Ð1Ñ1Ø+/€J�Ô! DÑ(Ð/Ð/Ñ/Ø$(€O�U˜T‘\Ð(Ð(Ñ(ØBFÐ˜5 Ô!2°CÐ!7Ô8¸4Ñ?ÐFÐFÑFØ?CÐ˜˜eÔ/°Ð4Ô5¸Ñ<ÐCÐCÑCØ=AÐ�e˜EÔ-¨sÐ2Ô3°dÑ:ÐAÐAÑAØ:>Ð˜uÔ0°4Ñ7Ð>Ð>Ñ>ØBFÐ˜5 Ô!2°CÐ!7Ô8¸4Ñ?ÐFÐFÑFØ?CÐ˜˜eÔ/°Ð4Ô5¸Ñ<ÐCÐCÑCØFJÐ˜u UÔ%6¸Ð%;Ô<¸tÑCÐJÐJÑJÐJÐJr$   rŒ  c                   ó˜   ‡ — e Zd ZdZdefˆ fd„Zdej        dej        fd„Zdej        dej        dej        d	e	fd
„Z
	 	 	 	 	 	 	 dd„Zˆ xZS )Ú
LEDEncoderzÞ
    Transformer encoder consisting of *config.encoder_layers* self-attention layers. Each layer is a
    [`LEDEncoderLayer`].

    Args:
        config: LEDConfig
        embed_tokens (nn.Embedding): output embedding
    ri   c                 ón  •‡— t          ¦   «                              ‰¦  «         ‰j        | _        ‰j        | _        ‰j        }‰j        | _        ‰j        | _	        t          ‰j        t          ¦  «        rM‰j        dz  dk    rt          d¦  «        ‚‰j        dk    rt          d¦  «        ‚‰j        g‰j        z  ‰_        nIt          ‰j        ¦  «        ‰j        k    r,t          d‰j        › dt          ‰j        ¦  «        › �¦  «        ‚t!          j        ‰j        || j        ¦  «        | _        t)          | j	        |¦  «        | _        t!          j        ˆfd„t/          ‰j        ¦  «        D ¦   «         ¦  «        | _        t!          j        |¦  «        | _        d| _        |                      ¦   «          d S )	NrD   r   z1`config.attention_window` has to be an even valuez,`config.attention_window` has to be positivezQ`len(config.attention_window)` should equal `config.num_hidden_layers`. Expected z, given c                 ó0   •— g | ]}t          ‰|¦  «        ‘ŒS rz  )r+  ©Ú.0Úiri   s     €r"   ú
<listcomp>z'LEDEncoder.__init__.<locals>.<listcomp>  s#   ø€ Ð$fÐ$fÐ$fÀA¥_°V¸QÑ%?Ô%?Ð$fÐ$fÐ$fr$   F)r=   r>   re   Úencoder_layerdropÚ	layerdropr   r   Úpadding_idxÚmax_encoder_position_embeddingsÚmax_source_positionsr  rg   rP   r   Únum_hidden_layersr�   r   Ú	EmbeddingÚ
vocab_sizeÚembed_tokensr8   Úembed_positionsÚ
ModuleListr·   Úencoder_layersr  r.  Úlayernorm_embeddingÚgradient_checkpointingÚ	post_init)r?   ri   r\   r@   s    ` €r"   r>   zLEDEncoder.__init__ý  s²  øø€ Ý‰Œ×Ò˜Ñ Ô Ð à”~ˆŒØÔ1ˆŒà”Nˆ	Ø!Ô.ˆÔØ$*Ô$JˆÔ!å�fÔ-­sÑ3Ô3ð 	ØÔ&¨Ñ*¨aÒ/Ð/Ý Ð!TÑUÔUÐUØÔ&¨!Ò+Ð+Ý Ð!OÑPÔPÐPØ'-Ô'>Ð&?À&ÔBZÑ&ZˆFÔ#Ð#å�6Ô*Ñ+Ô+¨vÔ/GÒGÐGÝ ðaØ &Ô 8ðað aÝBEÀfÔF]ÑB^ÔB^ðað añô ð õ
 œL¨Ô):¸IÀtÔGWÑXÔXˆÔå<ØÔ%Øñ 
ô  
ˆÔõ ”mÐ$fÐ$fÐ$fÐ$fÍÈvÔOdÑIeÔIeÐ$fÑ$fÔ$fÑgÔgˆŒÝ#%¤<°	Ñ#:Ô#:ˆÔ à&+ˆÔ#à�ŠÑÔÐÐÐr$   r�   Úglobal_attention_maskc                 ó&   — |�	||dz   z  }n|dz   }|S )Nr   rz  )r?   r�   r¦  s      r"   Ú_merge_to_attention_maskz#LEDEncoder._merge_to_attention_mask!  s.   € ð Ð%Ø+Ð/DÀqÑ/HÑIˆNˆNð 3°QÑ6ˆNØÐr$   r   Úinputs_embedsr   c                 óÂ  — t          | j        j        t          ¦  «        r| j        j        nt	          | j        j        ¦  «        }|dz  dk    rt          d|› �¦  «        ‚|�|j        n|j        }|dd…         \  }}|||z  z
  |z  }	|	dk    rÍt                               d|› d||	z   › d|› �¦  «         |�$t          j
                             |d|	f|¬¦  «        }|�[|                     ||	f| j        j        t          j        ¬	¦  «        }
|                      |
¦  «        }t          j        ||gd
¬¦  «        }t          j
                             |d|	fd¬¦  «        }|	|||fS )zbA helper function to pad tokens and mask to work with implementation of Longformer self-attention.rD   r   z2`attention_window` should be an even value. Given Nz(Input ids are automatically padded from z to z0 to be a multiple of `config.attention_window`: rÕ   )r&   r    rq   F)r  ri   rg   rP   rß   r   r   ÚloggerÚwarning_oncer   r‡   r¡   Únew_fullr   r.   rG   rŸ  r†   )r?   r   r�   r©  r   rg   Úinput_shaper”   rJ   Úpadding_lenÚinput_ids_paddingÚinputs_embeds_paddings               r"   Ú_pad_to_window_sizezLEDEncoder._pad_to_window_size-  sÂ  € õ ˜$œ+Ô6½Ñ<Ô<ð3ˆDŒKÔ(Ð(å�T”[Ô1Ñ2Ô2ð 	ð ˜aÑ 1Ò$Ð$ÝÐdÐRbÐdÐdÑeÔeÐeØ)2Ð)>�i”o�oÀMÔDWˆØ)¨"¨1¨"œoÑˆ
�Gà'¨'Ð4DÑ*DÑDÐHXÑXˆØ˜Š?ˆ?Ý×ÒðA¸7ð Að AÈÐR]ÑH]ð Að AØ.>ðAð Añô ð ð Ð$ÝœM×-Ò-¨i¸!¸[Ð9IÐQ]Ð-Ñ^Ô^�	ØÐ(Ø$1×$:Ò$:Ø Ð-Ø”KÔ,Ýœ*ð %;ñ %ô %Ð!ð
 )-×(9Ò(9Ð:KÑ(LÔ(LÐ%Ý %¤	¨=Ð:OÐ*PÐVXÐ YÑ YÔ Y�åœ]×.Ò.Ø  KÐ 0¸ð /ñ ô ˆNð ˜I ~°}ÐDÐDr$   Nc           	      ó–  ‡— |�|n| j         j        }|�|n| j         j        }|�|n| j         j        }|�|�t	          d¦  «        ‚|€|€t	          d¦  «        ‚|€|                      |¦  «        }|€@t          j        |                     ¦   «         dd…         |j	        t          j
        ¬¦  «        }|�|                      ||¦  «        }|                      |||| j         j        ¬¦  «        \  Š}}}|�1|                     ¦   «         }	|                     d|	d         ¦  «        }n|�|                     ¦   «         dd…         }	|�#t          ||j        ¦  «        dd…dddd…f         }|dk     }
|dk    }|                     ¦   «                              ¦   «                              ¦   «         }|                      |	¦  «        }||z   }|                      |¦  «        }t,          j                             || j        | j        ¬¦  «        }|rd	nd}|rd	nd}|r|rd	nd}t5          | j        ¦  «        D ]“\  }}|r||fz   }t          j        g ¦  «        }| j        r|| j        k     rd
}n ||||
|||¬¦  «        }|d         }|rB||d                              dd¦  «        fz   }|r ||d                              dd¦  «        fz   }Œ”|r||fz   }‰dk    rI|dd…d‰ …f         }|rt?          ˆfd„|D ¦   «         ¦  «        }|rt?          ˆfd„|D ¦   «         ¦  «        }|st?          d„ ||||fD ¦   «         ¦  «        S tA          ||||¬¦  «        S )a  
        Args:
            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you
                provide it.

                Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
                [`PreTrainedTokenizer.__call__`] for details.

                [What are input IDs?](../glossary#input-ids)
            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            global_attention_mask (`torch.FloatTensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to decide the attention given on each token, local attention or global attention for the encoder.
                Tokens with global attention attends to all other tokens, and all other tokens attend to them. This is
                important for task-specific finetuning because it makes the model more flexible at representing the
                task. For example, for classification, the <s> token should be given global attention. For QA, all
                question tokens should also have global attention. Please refer to the [Longformer
                paper](https://huggingface.co/papers/2004.05150) for more details. Mask values selected in `[0, 1]`:

                - 0 for local attention (a sliding window attention),
                - 1 for global attention (tokens that attend to all other tokens, and all other tokens attend to them).
            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
                Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation.
                This is useful if you want more control over how to convert `input_ids` indices into associated vectors
                than the model's internal embedding lookup matrix.
            output_attentions (`bool`, *optional*):
                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
                returned tensors for more detail.
            output_hidden_states (`bool`, *optional*):
                Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors
                for more detail.
            return_dict (`bool`, *optional*):
                Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
        NzDYou cannot specify both input_ids and inputs_embeds at the same timez5You have to specify either input_ids or inputs_embedsr   )rE   r&   )r   r�   r©  r   r   ru   rz  )NNN)r�   r{   r‘   r’   r“   r   rD   r   c              3   ó6   •K  — | ]}|d d …d ‰ …f         V — Œd S r<   rz  ©r”  Ústater¯  s     €r"   ú	<genexpr>z%LEDEncoder.forward.<locals>.<genexpr>è  s7   øè è € Ð&[Ð&[À5 u¨Q¨Q¨Q°°+°°Ð-=Ô'>Ð&[Ð&[Ð&[Ð&[Ð&[Ð&[r$   c              3   óB   •K  — | ]}|d d …d d …d ‰ …d d …f         V — Œd S r<   rz  rµ  s     €r"   r·  z%LEDEncoder.forward.<locals>.<genexpr>ë  sC   øè è € Ð&aÐ&aÈ u¨Q¨Q¨Q°°°°=°[°L°=À!À!À!Ð-CÔ'DÐ&aÐ&aÐ&aÐ&aÐ&aÐ&ar$   c              3   ó   K  — | ]}|®|V — Œ	d S r<   rz  ©r”  Úvs     r"   r·  z%LEDEncoder.forward.<locals>.<genexpr>î  s1   è è € ð ð ØÐefÐer�ÐerÐerÐerÐerðð r$   ©rv  rz   rw  rx  )!ri   r“   Úoutput_hidden_statesÚreturn_dictr   rŸ  r.   Úonesr)   rE   rG   r¨  r²  r   r   r6   r&   ÚflattenÚanyÚitemr   r£  r   r‡   re   rw   Ú	enumerater  Úrandr˜  r|   r  ru  )r?   r   r�   r¦  r©  r“   r½  r¾  rK  r®  r{   r‘   r’   Ú	embed_posrz   Úencoder_statesÚall_attentionsÚall_global_attentionsÚidxÚencoder_layerÚdropout_probabilityÚlayer_outputsr¯  s                         @r"   rI   zLEDEncoder.forwardX  sb  ø€ ðf 2CÐ1NÐ-Ð-ÐTXÔT_ÔTqÐà$8Ð$DÐ Ð È$Ì+ÔJjð 	ð &1Ð%<�k�kÀ$Ä+ÔBYˆð Ð  ]Ð%>ÝÐcÑdÔdÐdØÐ =Ð#8ÝÐTÑUÔUÐUàÐ Ø ×-Ò-¨iÑ8Ô8ˆMð Ð!Ý"œZ¨×(:Ò(:Ñ(<Ô(<¸S¸b¸SÔ(AÈ-ÔJ^ÕfkÔfpÐqÑqÔqˆNð !Ð,Ø!×:Ò:¸>ÐK`ÑaÔaˆNð AE×@XÒ@XØØ)Ø'ØœÔ1ð	 AYñ A
ô A
Ñ=ˆ�Y °ð Ð Ø#Ÿ.š.Ñ*Ô*ˆKØ!Ÿš r¨;°r¬?Ñ;Ô;ˆIˆIØÐ&Ø'×,Ò,Ñ.Ô.¨s°¨sÔ3ˆKð Ð%å@ÀÐQ^ÔQdÑeÔeÐfgÐfgÐfgÐijÐlmÐopÐopÐopÐfpÔqˆNð )¨1Ò,ˆØ-°Ò1ÐØ-×5Ò5Ñ7Ô7×;Ò;Ñ=Ô=×BÒBÑDÔDˆà×(Ò(¨Ñ5Ô5ˆ	à%¨	Ñ1ˆØ×0Ò0°Ñ?Ô?ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆà3Ð=˜˜¸ˆØ0Ð:˜˜°dˆØ'8Ð V¸^Ð V  ÐRVÐå"+¨D¬KÑ"8Ô"8ð 	hð 	hÑˆC�Ø#ð CØ!/°=Ð2BÑ!B�å"'¤*¨R¡.¤.ÐàŒ}ð 1Ð"5¸¼Ò"FÐ"FØ 2��à - Ø!Ø#1Ø$3Ø)=Ø#1Ø&7ð!ñ !ô !�ð !.¨aÔ 0�à ð hà!/°=ÀÔ3C×3MÒ3MÈaÐQRÑ3SÔ3SÐ2UÑ!U�à!ð hà,AÀ]ÐSTÔEU×E_ÒE_Ð`aÐcdÑEeÔEeÐDgÑ,gÐ)øàð 	?Ø+¨}Ð.>Ñ>ˆNð ˜Š?ˆ?à)¨!¨!¨!¨]¨{¨l¨]Ð*:Ô;ˆMØ#ð \Ý!&Ð&[Ð&[Ð&[Ð&[ÈNÐ&[Ñ&[Ô&[Ñ![Ô![�à ð bÝ!&Ð&aÐ&aÐ&aÐ&aÐR`Ð&aÑ&aÔ&aÑ!aÔ!a�àð 	Ýð ð Ø)¨>¸>ÐK`Ðaðñ ô ñ ô ð õ )Ø+Ø(Ø%Ø3ð	
ñ 
ô 
ð 	
r$   )NNNNNNN)rL   rM   rN   rO   r   r>   r.   rû   r¨  rP   r²  rI   rR   rS   s   @r"   r�  r�  ó  só   ø€ € € € € ðð ð"˜yð "ð "ð "ð "ð "ð "ðH
°u´|ð 
Ð\aÔ\hð 
ð 
ð 
ð 
ð)Eà”<ð)Eð œð)Eð ”|ð	)Eð
 ð)Eð )Eð )Eð )EðZ ØØ"ØØØ!Øð^
ð ^
ð ^
ð ^
ð ^
ð ^
ð ^
ð ^
r$   r�  c                   óF   ‡ — e Zd ZdZdefˆ fd„Z	 	 	 	 	 	 	 	 	 	 	 dd„Zˆ xZS )Ú
LEDDecoderzÊ
    Transformer decoder consisting of *config.decoder_layers* layers. Each layer is a [`LEDDecoderLayer`]

    Args:
        config: LEDConfig
        embed_tokens (nn.Embedding): output embedding
    ri   c                 ó  •‡— t          ¦   «                              ‰¦  «         ‰j        | _        ‰j        | _        ‰j        | _        ‰j        | _        t          j
        ‰j        ‰j        | j        ¦  «        | _        t          | j        ‰j        ¦  «        | _        t          j        ˆfd„t#          ‰j        ¦  «        D ¦   «         ¦  «        | _        t          j        ‰j        ¦  «        | _        d| _        |                      ¦   «          d S )Nc                 ó2   •— g | ]}t          ‰|¬ ¦  «        ‘ŒS ))r  )r@  r“  s     €r"   r–  z'LEDDecoder.__init__.<locals>.<listcomp>  s&   ø€ Ð$pÐ$pÐ$pÈa¥_°VÀqÐ%IÑ%IÔ%IÐ$pÐ$pÐ$pr$   F)r=   r>   re   Údecoder_layerdropr˜  r   r™  Úmax_decoder_position_embeddingsÚmax_target_positionsr   r�  rž  r   rŸ  r8   r   r¡  r·   Údecoder_layersr  r.  r£  r¤  r¥  ©r?   ri   r@   s    `€r"   r>   zLEDDecoder.__init__  sç   øø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø”~ˆŒØÔ1ˆŒØ!Ô.ˆÔØ$*Ô$JˆÔ!åœL¨Ô):¸F¼NÈDÔL\Ñ]Ô]ˆÔå<ØÔ%ØŒNñ 
ô  
ˆÔõ ”mÐ$pÐ$pÐ$pÐ$pÕSXÐY_ÔYnÑSoÔSoÐ$pÑ$pÔ$pÑqÔqˆŒÝ#%¤<°´Ñ#?Ô#?ˆÔ à&+ˆÔ#à�ŠÑÔÐÐÐr$   Nc           
      ó^  — |	�|	n| j         j        }	|
�|
n| j         j        }
|�|n| j         j        }|�|n| j         j        }|�|�t          d¦  «        ‚|�1|                     ¦   «         }|                     d|d         ¦  «        }n.|�|                     ¦   «         dd…         }nt          d¦  «        ‚|€|                      |¦  «        }| j	        r%| j
        r|rt                               d¦  «         d}|r8|€6t          t          | j         ¬¦  «        t          | j         ¬¦  «        ¦  «        }|�|                     ¦   «         nd}d}|d         d	k    rt!          | j         |||¬
¦  «        }t#          | j         |||¬¦  «        }|                      ||¦  «        }||z   }|                      |¦  «        }t(          j                             || j        | j
        ¬¦  «        }|
rdnd}|	rdnd}|	rdnd}t/          | j        ¦  «        D ]h\  }}|
r||fz  }| j
        r t3          j        g ¦  «        }|| j        k     rŒ4 |||||||	|¬¦  «        }|d         }|	r||d	         fz  }||d         fz  }Œi|
r||fz  }|st9          d„ |||||fD ¦   «         ¦  «        S t;          |||||¬¦  «        S )a  
        Args:
            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you
                provide it.

                Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
                [`PreTrainedTokenizer.__call__`] for details.

                [What are input IDs?](../glossary#input-ids)
            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            global_attention_mask (`torch.FloatTensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to decide the attention given on each token, local attention or global attention. Tokens with
                global attention attends to all other tokens, and all other tokens attend to them. This is important
                for task-specific finetuning because it makes the model more flexible at representing the task. For
                example, for classification, the <s> token should be given global attention. For QA, all question
                tokens should also have global attention. Please refer to the [Longformer
                paper](https://huggingface.co/papers/2004.05150) for more details. Mask values selected in `[0, 1]`:

                - 0 for local attention (a sliding window attention),
                - 1 for global attention (tokens that attend to all other tokens, and all other tokens attend to them).
            encoder_hidden_states (`torch.FloatTensor` of shape `(batch_size, encoder_sequence_length, hidden_size)`, *optional*):
                Sequence of hidden-states at the output of the last layer of the encoder. Used in the cross-attention
                of the decoder.
            encoder_attention_mask (`torch.LongTensor` of shape `(batch_size, encoder_sequence_length)`, *optional*):
                Mask to avoid performing cross-attention on padding tokens indices of encoder input_ids. Mask values
                selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
                It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

                Contains pre-computed hidden-states (key and values in the self-attention blocks and in the
                cross-attention blocks) that can be used (see `past_key_values` input) to speed up sequential decoding.

                If `past_key_values` are used, the user can optionally input only the last `decoder_input_ids` (those
                that don't have their past key value states given to this model) of shape `(batch_size, 1)` instead of
                all `decoder_input_ids` of shape `(batch_size, sequence_length)`.
            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
                Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation.
                This is useful if you want more control over how to convert `input_ids` indices into associated vectors
                than the model's internal embedding lookup matrix.
            output_attentions (`bool`, *optional*):
                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
                returned tensors for more detail.
            output_hidden_states (`bool`, *optional*):
                Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors
                for more detail.
            return_dict (`bool`, *optional*):
                Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
        NzTYou cannot specify both decoder_input_ids and decoder_inputs_embeds at the same timer   zEYou have to specify either decoder_input_ids or decoder_inputs_embedszZ`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`...F)ri   r   r   )ri   r©  r�   r  )ri   r©  r�   rG  ru   rz  )rH  r  r“   rI  rD   c              3   ó   K  — | ]}|®|V — Œ	d S r<   rz  rº  s     r"   r·  z%LEDDecoder.forward.<locals>.<genexpr>¼  s0   è è € ð ð àØ�=ð à �=�=�=ðð r$   )rv  r  rz   rw  r  )ri   r“   r½  rI  r¾  r   r)   r   rŸ  r¤  rw   r«  r¬  r   r
   Úget_seq_lengthr   r   r   r£  r   r‡   re   rÃ  r  r.   rÄ  r˜  r  r   )r?   r   r�   r¦  rG  rH  r  r©  rI  r“   r½  r¾  rK  r®  rB   Úcombined_attention_maskrK   rz   Úall_hidden_statesÚall_self_attnsÚall_cross_attentionsrÉ  Údecoder_layerrË  rÌ  s                            r"   rI   zLEDDecoder.forward  s­  € ðV 2CÐ1NÐ-Ð-ÐTXÔT_ÔTqÐà$8Ð$DÐ Ð È$Ì+ÔJjð 	ð "+Ð!6�I�I¸D¼KÔ<Qˆ	Ø%0Ð%<�k�kÀ$Ä+ÔBYˆð Ð  ]Ð%>ÝÐsÑtÔtÐtØÐ"Ø#Ÿ.š.Ñ*Ô*ˆKØ!Ÿš r¨;°r¬?Ñ;Ô;ˆIˆIØÐ&Ø'×,Ò,Ñ.Ô.¨s°¨sÔ3ˆKˆKåÐdÑeÔeÐeàÐ Ø ×-Ò-¨iÑ8Ô8ˆMàÔ&ð 	"¨4¬=ð 	"Øð "Ý×#Ò#Øpñô ð ð "�	àð 	v˜Ð0Ý1µ,ÀdÄkÐ2RÑ2RÔ2RÕT`ÐhlÔhsÐTtÑTtÔTtÑuÔuˆOàETÐE` ×!?Ò!?Ñ!AÔ!AÐ!AÐfgÐà"&ÐØ�rŒ?˜QÒÐÝ&8Ø”{Ø+Ø-Ø /ð	'ñ 'ô 'Ð#õ ";Ø”;Ø'Ø1Ø"7ð	"
ñ "
ô "
Ðð ×(Ò(¨Ð6LÑMÔMˆ	à%¨	Ñ1ˆØ×0Ò0°Ñ?Ô?ˆåœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆð #7Ð@˜B˜B¸DÐØ0Ð:˜˜°dˆØ%6Ð@˜r˜r¸DÐå"+¨D¬KÑ"8Ô"8ð 	<ð 	<ÑˆC�à#ð 6Ø! mÐ%5Ñ5Ð!ØŒ}ð Ý&+¤j°¡n¤nÐ#Ø&¨¬Ò7Ð7Øà)˜MØØ'Ø%Ø'=Ø /Ø"3Ø#ðñ ô ˆMð *¨!Ô,ˆMØ ð <Ø =°Ô#3Ð"5Ñ5�Ø$¨°qÔ)9Ð(;Ñ;Ð$øð  ð 	2Ø -Ð!1Ñ1Ðàð 	Ýð ð à'¨Ð:KÈ^Ð]qÐrðñ ô ñ ô ð õ
 9Ø+Ø+Ø+Ø%Ø1ð
ñ 
ô 
ð 	
r$   )NNNNNNNNNNN)rL   rM   rN   rO   r   r>   rI   rR   rS   s   @r"   rÎ  rÎ  ù  s�   ø€ € € € € ðð ð˜yð ð ð ð ð ð ð, ØØ"Ø"Ø#ØØØØØ!Øðq
ð q
ð q
ð q
ð q
ð q
ð q
ð q
r$   rÎ  c                   óx  ‡ — e Zd ZdddœZdefˆ fd„Zd„ Zd„ Ze	 	 	 	 	 	 	 	 	 	 	 	 	 dde	j
        dz  d	e	j        dz  d
e	j
        dz  de	j
        dz  deee	j                          dz  de	j        dz  dedz  de	j        dz  de	j        dz  dedz  dedz  dedz  dedz  dee	j                 ez  fd„¦   «         Zˆ xZS )ÚLEDModelzshared.weight)zencoder.embed_tokens.weightzdecoder.embed_tokens.weightri   c                 ó  •— t          ¦   «                              |¦  «         |j        |j        }}t	          j        ||j        |¦  «        | _        t          |¦  «        | _	        t          |¦  «        | _        |                      ¦   «          d S r<   )r=   r>   r   rž  r   r�  r   Úsharedr�  ÚencoderrÎ  Údecoderr¥  )r?   ri   r™  rž  r@   s       €r"   r>   zLEDModel.__init__Ñ  sw   ø€ Ý‰Œ×Ò˜Ñ Ô Ð à"(Ô"5°vÔ7H�ZˆÝ”l :¨v¬~¸{ÑKÔKˆŒå! &Ñ)Ô)ˆŒÝ! &Ñ)Ô)ˆŒð 	�ŠÑÔÐÐÐr$   c                 ó   — | j         S r<   )rá  )r?   s    r"   Úget_input_embeddingszLEDModel.get_input_embeddingsÝ  s
   € ØŒ{Ðr$   c                 óX   — || _         | j         | j        _        | j         | j        _        d S r<   )rá  râ  rŸ  rã  )r?   r`   s     r"   Úset_input_embeddingszLEDModel.set_input_embeddingsà  s'   € ØˆŒØ$(¤KˆŒÔ!Ø$(¤KˆŒÔ!Ð!Ð!r$   Nr   r�   Údecoder_input_idsÚdecoder_attention_maskÚencoder_outputsr¦  r  r©  Údecoder_inputs_embedsrI  r“   r½  r¾  r½   c                 óö  — |�|n| j         j        }|�|n| j         j        }|
�|
n| j         j        }
|�|n| j         j        }|€'|	€%t          || j         j        | j         j        ¦  «        }|€|                      |||||||¬¦  «        }n�|rt          |t          ¦  «        sjt          |d         t          |¦  «        dk    r|d         ndt          |¦  «        dk    r|d         ndt          |¦  «        dk    r|d         nd¬¦  «        }|                      |||d         |||	|
|||¬¦
  «
        }|s||z   S t          |j        |j        |j        |j        |j        |j        |j        |j        |j        ¬	¦	  «	        S )
áK  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`LedTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are input IDs?](../glossary#input-ids)

            LED uses the `eos_token_id` as the starting token for `decoder_input_ids` generation. If `past_key_values`
            is used, optionally only the last `decoder_input_ids` have to be input (see `past_key_values`).
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.

            If you want to change padding behavior, you should read [`modeling_led._prepare_decoder_inputs`] and modify
            to your needs. See diagram 1 in [the paper](https://huggingface.co/papers/1910.13461) for more information on the
            default strategy.
        global_attention_mask (`torch.FloatTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Mask to decide the attention given on each token, local attention or global attention for the encoder.
            Tokens with global attention attends to all other tokens, and all other tokens attend to them. This is
            important for task-specific finetuning because it makes the model more flexible at representing the task.
            For example, for classification, the <s> token should be given global attention. For QA, all question
            tokens should also have global attention. Please refer to the [Longformer
            paper](https://huggingface.co/papers/2004.05150) for more details. Mask values selected in `[0, 1]`:

            - 0 for local attention (a sliding window attention),
            - 1 for global attention (tokens that attend to all other tokens, and all other tokens attend to them).
        N)r   r�   r¦  r©  r“   r½  r¾  r   r   rD   r   r¼  )
r   r�   rG  rH  r  r©  rI  r“   r½  r¾  )	rv  r  r}  r~  r  r€  rG  r�  r‚  )ri   r“   r½  rI  r¾  r#   r   r   râ  r  ru  r�   rã  r|  rv  r  rz   rw  r  rx  )r?   r   r�   rè  ré  rê  r¦  r  r©  rë  rI  r“   r½  r¾  rK  Údecoder_outputss                   r"   rI   zLEDModel.forwardå  sú  € ð^ 2CÐ1NÐ-Ð-ÐTXÔT_ÔTqÐà$8Ð$DÐ Ð È$Ì+ÔJjð 	ð "+Ð!6�I�I¸D¼KÔ<Qˆ	Ø%0Ð%<�k�kÀ$Ä+ÔBYˆð
 Ð$Ð)>Ð)FÝ 2Ø˜4œ;Ô3°T´[Ô5Wñ!ô !Ðð Ð"Ø"ŸlšlØ#Ø-Ø&;Ø+Ø"3Ø%9Ø'ð +ñ ô ˆOˆOð ð 	¥¨OÕ=VÑ!WÔ!Wð 	Ý7Ø"1°!Ô"4Ý47¸Ñ4HÔ4HÈ1Ò4LÐ4L˜o¨aÔ0Ð0ÐRVÝ14°_Ñ1EÔ1EÈÒ1IÐ1I˜?¨1Ô-Ð-ÈtÝ8;¸OÑ8LÔ8LÈqÒ8PÐ8P /°!Ô"4Ð"4ÐVZð	ñ ô ˆOð Ÿ,š,Ø'Ø1Ø"1°!Ô"4Ø#1Ø+Ø/ØØ/Ø!5Ø#ð 'ñ 
ô 
ˆð ð 	5Ø" _Ñ4Ð4å$Ø-Ô?Ø+Ô;Ø"1Ô"?Ø.Ô9Ø,Ô=Ø&5Ô&GØ"1Ô"?Ø.Ô9Ø&5Ô&Gð

ñ 

ô 

ð 
	
r$   )NNNNNNNNNNNNN)rL   rM   rN   Ú_tied_weights_keysr   r>   rå  rç  r   r.   Ú
LongTensorrû   r  ry  r	   r-   r|  rI   rR   rS   s   @r"   rß  rß  Ê  sÐ  ø€ € € € € ð (7Ø'6ðð Ðð

˜yð 
ð 
ð 
ð 
ð 
ð 
ðð ð ð0ð 0ð 0ð
 ð .2Ø.2Ø59Ø:>ØBFØ:>Ø(,Ø26Ø:>Ø!%Ø)-Ø,0Ø#'ðk
ð k
àÔ# dÑ*ðk
ð œ tÑ+ðk
ð !Ô+¨dÑ2ð	k
ð
 !&Ô 0°4Ñ 7ðk
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðk
ð  %Ô0°4Ñ7ðk
ð  ™ðk
ð Ô(¨4Ñ/ðk
ð  %Ô0°4Ñ7ðk
ð ˜$‘;ðk
ð   $™;ðk
ð # T™kðk
ð ˜D‘[ðk
ð  
ˆuŒ|Ô	Ð4Ñ	4ð!k
ð k
ð k
ñ „^ðk
ð k
ð k
ð k
ð k
r$   rß  zU
    The LED Model with a language modeling head. Can be used for summarization.
    c            !       óà  ‡ — e Zd ZdZdgZddiZdefˆ fd„Z	 dd	ed
edz  de	de
j        fˆ fd„Zd	eddfd„Ze	 	 	 	 	 	 	 	 	 	 	 	 	 	 d dej        dz  dej        dz  dej        dz  dej        dz  deeej                          dz  dej        dz  dedz  dej        dz  dej        dz  dej        dz  de	dz  de	dz  de	dz  de	dz  deej                 ez  fd„¦   «         Zdej        fd„Zˆ xZS )!rj  r]  rm  zlm_head.weightzled.shared.weightri   c                 ól  •— t          ¦   «                              |¦  «         t          |¦  «        | _        |                      dt          j        d| j        j        j        f¦  «        ¦  «         t          j
        |j        | j        j        j        d¬¦  «        | _        |                      ¦   «          d S )Nrm  r   Fr  )r=   r>   rß  r]  Úregister_bufferr.   Úzerosrá  r9   r   r]   r   Úlm_headr¥  rÕ  s     €r"   r>   z$LEDForConditionalGeneration.__init__`  s‘   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý˜FÑ#Ô#ˆŒØ×ÒÐ0µ%´+¸qÀ$Ä(Ä/ÔB`Ð>aÑ2bÔ2bÑcÔcÐcÝ”y ¤°´´Ô1OÐV[Ð\Ñ\Ô\ˆŒð 	�ŠÑÔÐÐÐr$   NTÚnew_num_tokensÚpad_to_multiple_ofÚmean_resizingr½   c                 ó˜   •— t          ¦   «                              |||¦  «        }|                      |j        j        d         ¦  «         |S )Nr   )r=   Úresize_token_embeddingsÚ_resize_final_logits_biasrH   r   )r?   rö  r÷  rø  Únew_embeddingsr@   s        €r"   rú  z3LEDForConditionalGeneration.resize_token_embeddingsi  sG   ø€ õ ™œ×8Ò8¸ÐI[Ð]jÑkÔkˆØ×&Ò& ~Ô'<Ô'BÀ1Ô'EÑFÔFÐFØÐr$   c                 ó  — | j         j        d         }||k    r| j         d d …d |…f         }nBt          j        d||z
  f| j         j        ¬¦  «        }t          j        | j         |gd¬¦  «        }|                      d|¦  «         d S )Nr   r   r³   rq   rm  )rm  r   r.   rô  rE   r†   ró  )r?   rö  Úold_num_tokensÚnew_biasÚ
extra_biass        r"   rû  z5LEDForConditionalGeneration._resize_final_logits_biasp  s—   € ØÔ/Ô5°bÔ9ˆØ˜^Ò+Ð+ØÔ-¨a¨a¨a°°.°Ð.@ÔAˆHˆHåœ a¨¸.Ñ)HÐ%IÐRVÔRhÔRoÐpÑpÔpˆJÝ”y $Ô"8¸*Ð!EÈ1ÐMÑMÔMˆHØ×ÒÐ0°(Ñ;Ô;Ð;Ð;Ð;r$   r   r�   rè  ré  rê  r¦  r  r©  rë  ÚlabelsrI  r“   r½  r¾  c                 ó’  — |�|n| j         j        }|
�G|rt                               d¦  «         d}|€'|	€%t	          |
| j         j        | j         j        ¦  «        }|                      |||||||||	||||¬¦  «        }|                      |d         ¦  «        | j	        z   }d}|
�Kt          ¦   «         } ||                     d| j         j        ¦  «        |
                     d¦  «        ¦  «        }|s|f|dd…         z   }|�|f|z   n|S t          |||j        |j        |j        |j        |j        |j        |j        |j        ¬¦
  «
        S )	aË  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`LedTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are input IDs?](../glossary#input-ids)

            LED uses the `eos_token_id` as the starting token for `decoder_input_ids` generation. If `past_key_values`
            is used, optionally only the last `decoder_input_ids` have to be input (see `past_key_values`).
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.

            If you want to change padding behavior, you should read [`modeling_led._prepare_decoder_inputs`] and modify
            to your needs. See diagram 1 in [the paper](https://huggingface.co/papers/1910.13461) for more information on the
            default strategy.
        global_attention_mask (`torch.FloatTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Mask to decide the attention given on each token, local attention or global attention for the encoder.
            Tokens with global attention attends to all other tokens, and all other tokens attend to them. This is
            important for task-specific finetuning because it makes the model more flexible at representing the task.
            For example, for classification, the <s> token should be given global attention. For QA, all question
            tokens should also have global attention. Please refer to the [Longformer
            paper](https://huggingface.co/papers/2004.05150) for more details. Mask values selected in `[0, 1]`:

            - 0 for local attention (a sliding window attention),
            - 1 for global attention (tokens that attend to all other tokens, and all other tokens attend to them).
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.

        Example Summarization:

        ```python
        >>> import torch
        >>> from transformers import AutoTokenizer, LEDForConditionalGeneration

        >>> model = LEDForConditionalGeneration.from_pretrained("allenai/led-large-16384-arxiv")
        >>> tokenizer = AutoTokenizer.from_pretrained("allenai/led-large-16384-arxiv")

        >>> ARTICLE_TO_SUMMARIZE = '''Transformers (Vaswani et al., 2017) have achieved state-of-the-art
        ...     results in a wide range of natural language tasks including generative language modeling
        ...     (Dai et al., 2019; Radford et al., 2019) and discriminative ... language understanding (Devlin et al., 2019).
        ...     This success is partly due to the self-attention component which enables the network to capture contextual
        ...     information from the entire sequence. While powerful, the memory and computational requirements of
        ...     self-attention grow quadratically with sequence length, making it infeasible (or very expensive) to
        ...     process long sequences. To address this limitation, we present Longformer, a modified Transformer
        ...     architecture with a self-attention operation that scales linearly with the sequence length, making it
        ...     versatile for processing long documents (Fig 1). This is an advantage for natural language tasks such as
        ...     long document classification, question answering (QA), and coreference resolution, where existing approaches
        ...     partition or shorten the long context into smaller sequences that fall within the typical 512 token limit
        ...     of BERT-style pretrained models. Such partitioning could potentially result in loss of important
        ...     cross-partition information, and to mitigate this problem, existing methods often rely on complex
        ...     architectures to address such interactions. On the other hand, our proposed Longformer is able to build
        ...     contextual representations of the entire context using multiple layers of attention, reducing the need for
        ...     task-specific architectures.'''
        >>> inputs = tokenizer.encode(ARTICLE_TO_SUMMARIZE, return_tensors="pt")

        >>> # Global attention on the first token (cf. Beltagy et al. 2020)
        >>> global_attention_mask = torch.zeros_like(inputs)
        >>> global_attention_mask[:, 0] = 1

        >>> # Generate Summary
        >>> summary_ids = model.generate(inputs, global_attention_mask=global_attention_mask, num_beams=3, max_length=32)
        >>> print(tokenizer.decode(summary_ids[0], skip_special_tokens=True, clean_up_tokenization_spaces=True))
        ```

        Example Conditional generation :

        ```python
        >>> from transformers import AutoTokenizer, LEDForConditionalGeneration

        >>> tokenizer = AutoTokenizer.from_pretrained("allenai/led-base-16384")
        >>> TXT = "My friends are <mask> but they eat too many carbs."

        >>> model = LEDForConditionalGeneration.from_pretrained("allenai/led-base-16384")
        >>> input_ids = tokenizer([TXT], return_tensors="pt")["input_ids"]

        >>> prediction = model.generate(input_ids)[0]
        >>> print(tokenizer.decode(prediction, skip_special_tokens=True))
        ```
        NzJThe `use_cache` argument is changed to `False` since `labels` is provided.F)r�   rè  ré  rê  r¦  r  r©  rë  rI  r“   r½  r¾  r   r   r   )
r…  r†  r  r}  r~  r  r€  rG  r�  r‚  )ri   r¾  r«  Úwarningr#   r   r   r]  rõ  rm  r   r   rž  r„  r  r}  r~  r  r€  rG  r�  r‚  )r?   r   r�   rè  ré  rê  r¦  r  r©  rë  r  rI  r“   r½  r¾  rK  rž   Ú	lm_logitsÚmasked_lm_lossÚloss_fctr  s                        r"   rI   z#LEDForConditionalGeneration.forwardy  s™  € ðN &1Ð%<�k�kÀ$Ä+ÔBYˆàÐØð mÝ—’ÐkÑlÔlÐlØˆIØ Ð(Ð-BÐ-JÝ$6Ø˜DœKÔ4°d´kÔ6Xñ%ô %Ð!ð —(’(ØØ)Ø/Ø#9Ø+Ø"7Ø+Ø'Ø"7ØØ/Ø!5Ø#ð ñ 
ô 
ˆð —L’L ¨¤Ñ,Ô,¨tÔ/EÑEˆ	àˆØÐÝ'Ñ)Ô)ˆHØ%˜X i§n¢n°R¸¼Ô9OÑ&PÔ&PÐRX×R]ÒR]Ð^`ÑRaÔRaÑbÔbˆNàð 	ZØ�\ G¨A¨B¨B¤KÑ/ˆFØ3AÐ3M�^Ð%¨Ñ.Ð.ÐSYÐYå!ØØØ#Ô3Ø")Ô"?Ø&Ô9Ø$Ô5Ø&-Ô&GØ")Ô"?Ø&Ô9Ø&-Ô&Gð
ñ 
ô 
ð 	
r$   c                 óL   — t          || j        j        | j        j        ¦  «        S r<   )r#   ri   r   r   )r?   r  s     r"   Ú%prepare_decoder_input_ids_from_labelszALEDForConditionalGeneration.prepare_decoder_input_ids_from_labels  s   € Ý! &¨$¬+Ô*BÀDÄKÔDfÑgÔgÐgr$   )NT©NNNNNNNNNNNNNN)rL   rM   rN   rp  Ú_keys_to_ignore_on_load_missingrï  r   r>   rP   r-   r   r�  rú  rû  r   r.   rð  rû   r  ry  r	   r„  rI   r  rR   rS   s   @r"   rj  rj  T  sg  ø€ € € € € ð ÐØ':Ð&;Ð#àÐ-ðÐð˜yð ð ð ð ð ð ð aeðð Ø!ðØ7:¸T±zðØY]ðà	Œðð ð ð ð ð ð<¸ð <Àð <ð <ð <ð <ð ð .2Ø.2Ø59Ø:>ØBFØ:>Ø(,Ø26Ø:>Ø*.Ø!%Ø)-Ø,0Ø#'ðV
ð V
àÔ# dÑ*ðV
ð œ tÑ+ðV
ð !Ô+¨dÑ2ð	V
ð
 !&Ô 0°4Ñ 7ðV
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðV
ð  %Ô0°4Ñ7ðV
ð  ™ðV
ð Ô(¨4Ñ/ðV
ð  %Ô0°4Ñ7ðV
ð Ô  4Ñ'ðV
ð ˜$‘;ðV
ð   $™;ðV
ð # T™kðV
ð ˜D‘[ðV
ð" 
ˆuŒ|Ô	Ð1Ñ	1ð#V
ð V
ð V
ñ „^ðV
ðph¸E¼Lð hð hð hð hð hð hð hð hr$   rj  c            !       ó|  ‡ — e Zd Zˆ fd„Ze	 	 	 	 	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  deeej	                          dz  dej	        dz  d	ej        dz  d
ej        dz  dej	        dz  dej	        dz  de
dz  de
dz  de
dz  de
dz  deej                 ez  fd„¦   «         Zˆ xZS )ÚLEDForQuestionAnsweringc                 ó  •— t          ¦   «                              |¦  «         d|_        |j        | _        t          |¦  «        | _        t          j        |j        |j        ¦  «        | _        |  	                    ¦   «          d S )NrD   )
r=   r>   Ú
num_labelsrß  r]  r   r]   rX   Ú
qa_outputsr¥  rÕ  s     €r"   r>   z LEDForQuestionAnswering.__init__  sm   ø€ Ý‰Œ×Ò˜Ñ Ô Ð àˆÔØ Ô+ˆŒå˜FÑ#Ô#ˆŒÝœ) FÔ$6¸Ô8IÑJÔJˆŒð 	�ŠÑÔÐÐÐr$   Nr   r�   rè  ré  rê  r¦  Ústart_positionsÚend_positionsr©  rë  rI  r“   r½  r¾  r½   c                 ó
  — |�|n| j         j        }|�|�d}|                      |||||||	|
||||¬¦  «        }|d         }|                      |¦  «        }|                     dd¬¦  «        \  }}|                     d¦  «                             ¦   «         }|                     d¦  «                             ¦   «         }d}|�ç|�åt          |                     ¦   «         ¦  «        dk    r|                     d¦  «        }t          |                     ¦   «         ¦  «        dk    r|                     d¦  «        }|                     d¦  «        }| 	                    d|¦  «        }| 	                    d|¦  «        }t          |¬¦  «        } |||¦  «        } |||¦  «        }||z   d	z  }|s||f|dd…         z   }|�|f|z   n|S t          ||||j        |j        |j        |j        |j        |j        |j        |j        ¬
¦  «        S )rí  NF)r�   rè  ré  r¦  rê  r©  rë  rI  r“   r½  r¾  r   r   r   rq   )Úignore_indexrD   )r…  r�  rŽ  r  r}  r~  r  r€  rG  r�  r‚  )ri   r¾  r]  r  ÚsplitÚsqueezer�   r�   r)   r;  r   rŒ  r  r}  r~  r  r€  rG  r�  r‚  )r?   r   r�   rè  ré  rê  r¦  r  r  r©  rë  rI  r“   r½  r¾  rK  rž   Úsequence_outputr†  r�  rŽ  Ú
total_lossÚignored_indexr  Ú
start_lossÚend_lossr  s                              r"   rI   zLEDForQuestionAnswering.forward$  s[  € ð` &1Ð%<�k�kÀ$Ä+ÔBYˆØÐ&¨=Ð+DØˆIà—(’(ØØ)Ø/Ø#9Ø"7Ø+Ø'Ø"7ØØ/Ø!5Ø#ð ñ 
ô 
ˆð " !œ*ˆà—’ Ñ1Ô1ˆØ#)§<¢<°°r <Ñ#:Ô#:Ñ ˆ�jØ#×+Ò+¨BÑ/Ô/×:Ò:Ñ<Ô<ˆØ×'Ò'¨Ñ+Ô+×6Ò6Ñ8Ô8ˆ
àˆ
ØÐ&¨=Ð+Då�?×'Ò'Ñ)Ô)Ñ*Ô*¨QÒ.Ð.Ø"1×"9Ò"9¸"Ñ"=Ô"=�Ý�=×%Ò%Ñ'Ô'Ñ(Ô(¨1Ò,Ð,Ø -× 5Ò 5°bÑ 9Ô 9�à(×-Ò-¨aÑ0Ô0ˆMØ-×3Ò3°A°}ÑEÔEˆOØ)×/Ò/°°=ÑAÔAˆMå'°]ÐCÑCÔCˆHØ!˜ ,°Ñ@Ô@ˆJØ�x 
¨MÑ:Ô:ˆHØ$ xÑ/°1Ñ4ˆJàð 	RàØðð ˜˜˜”ñˆFð 0:Ð/E�Z�M FÑ*Ð*È6ÐQå5ØØ%Ø!Ø#Ô3Ø")Ô"?Ø&Ô9Ø$Ô5Ø&-Ô&GØ")Ô"?Ø&Ô9Ø&-Ô&Gð
ñ 
ô 
ð 	
r$   r	  )rL   rM   rN   r>   r   r.   rð  rû   r  ry  r-   rŒ  rI   rR   rS   s   @r"   r  r    s«  ø€ € € € € ð
ð 
ð 
ð 
ð 
ð ð .2Ø.2Ø59Ø:>ØBFØ:>Ø37Ø15Ø26Ø:>Ø!%Ø)-Ø,0Ø#'ðm
ð m
àÔ# dÑ*ðm
ð œ tÑ+ðm
ð !Ô+¨dÑ2ð	m
ð
 !&Ô 0°4Ñ 7ðm
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðm
ð  %Ô0°4Ñ7ðm
ð Ô)¨DÑ0ðm
ð Ô'¨$Ñ.ðm
ð Ô(¨4Ñ/ðm
ð  %Ô0°4Ñ7ðm
ð ˜$‘;ðm
ð   $™;ðm
ð # T™kðm
ð ˜D‘[ðm
ð" 
ˆuŒ|Ô	ÐEÑ	Eð#m
ð m
ð m
ñ „^ðm
ð m
ð m
ð m
ð m
r$   r  )rj  r  rß  r\  r<   )?rO   r}   Údataclassesr   r.   r   Útorch.nnr   Ú r   rk  Úactivationsr   Úcache_utilsr	   r
   r   Ú
generationr   Úmasking_utilsr   r   Úmodeling_layersr   Úmodeling_outputsr   Úmodeling_utilsr   Úutilsr   r   r   Úconfiguration_ledr   Ú
get_loggerrL   r«  rû   rP   r#   r&   r6   r�  r8   ÚModulerU   rý   r	  r+  r@  rQ  r\  ru  r|  r„  rŠ  rŒ  r�  rÎ  rß  rj  r  Ú__all__rz  r$   r"   ú<module>r*     s€  ðð Ð à €€€Ø !Ð !Ð !Ð !Ð !Ð !à €€€Ø Ð Ð Ð Ð Ð Ø %Ð %Ð %Ð %Ð %Ð %à &Ð &Ð &Ð &Ð &Ð &Ø !Ð !Ð !Ð !Ð !Ð !Ø CÐ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CØ )Ð )Ð )Ð )Ð )Ð )Ø JÐ JÐ JÐ JÐ JÐ JÐ JÐ JØ 9Ð 9Ð 9Ð 9Ð 9Ð 9Ø IÐ IÐ IÐ IÐ IÐ IØ -Ð -Ð -Ð -Ð -Ð -Ø 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ø (Ð (Ð (Ð (Ð (Ð (ð 
ˆÔ	˜HÑ	%Ô	%€ð %¤,ð ¸cð Ð[^ð ð ð ð ð #ð #¨e¬lð #À5Ä;ð #ÐY\Ð_cÑYcð #ð #ð #ð #ð$*ð *ð *ð *ð * B¤Lñ *ô *ð *ð$c	5ð c	5ð c	5ð c	5ð c	5˜bœiñ c	5ô c	5ð c	5ðLð ð ð ð ˜"œ)ñ ô ð ð@DCð DCð DCð DCð DC˜"œ)ñ DCô DCð DCðN53ð 53ð 53ð 53ð 53Ð0ñ 53ô 53ð 53ðpdð dð dð dð dÐ0ñ dô dð dðNð ð ð ð ˜BœIñ ô ð ð0 ð2ð 2ð 2ð 2ð 2˜ñ 2ô 2ñ „ð2ð* €ððñ ô ð
 ðCð Cð Cð Cð C ñ Cô Cñ „ñô ðCð@ €ððñ ô ð ðKð Kð Kð Kð K˜Kñ Kô Kñ „ñô ðKð@ €ððñ ô ð
 ðKð Kð Kð Kð K˜ñ Kô Kñ „ñô ðKð@ €ððñ ô ð
 ðKð Kð Kð Kð K¨ñ Kô Kñ „ñô ðKð@ €ððñ ô ð
 ðKð Kð Kð Kð K¨[ñ Kô Kñ „ñô ðKð>C
ð C
ð C
ð C
ð C
Ð#ñ C
ô C
ð C
ðLN
ð N
ð N
ð N
ð N
Ð#ñ N
ô N
ð N
ðb ðF
ð F
ð F
ð F
ð F
Ð!ñ F
ô F
ñ „ðF
ðR €ððñ ô ð
zhð zhð zhð zhð zhÐ"4°oñ zhô zhñô ð
zhðz ð{
ð {
ð {
ð {
ð {
Ð0ñ {
ô {
ñ „ð{
ð|ð ð €€€r$   