§
    ‚Štj™1 ã                   óz
  — U d dl mZ d dlZd dlmc mZ ddlmZ ddl	m
Z
 ddlmZmZ ddlmZmZ ddlmZmZmZ  e¦   «         rd d	lmZ d d
lmZmZ nej        Z edd¬¦  «        Z edd¬¦  «        Z e¦   «         Zerd dlmZ  ej         e!¦  «        Z"dedefd„Z#dedefd„Z$de%de%de%de%de&f
d„Z'de%de%de%de%de&f
d„Z(de%defd„Z)de%dej        defd„Z*dej        defd „Z+de%defd!„Z,de%defd"„Z-de%defd#„Z.de%dej        defd$„Z/d%ej        defd&„Z0d'ej        defd(„Z1d)ed*e%d+e%defd,„Z2d-ej        dz  d.e%d+e%dej        dz  fd/„Z3dej        d-ej        dz  d.e%d+e%dej        f
d0„Z4d1ej5        dej5        fd2„Z6	 dcd%ej        dz  d3e%d.e%d*e%d+e%d4e%dz  de&fd5„Z7d%ej        dz  d.e%d4e%dz  de&fd6„Z8	 dcd%ej        dz  d.e%d4e%dz  de&fd7„Z9d)edefd8„Z:d9ej        d:ej        d;ej        d<ej        fd=„Z;d d e'dddd>dd>d?f
d@e%d3e%d.e%d*e%d+e%d)ed-ej        dz  dAe%dz  dBe&dCe&dDe&dEe&dFej<        e=z  dej        dz  fdG„Z>d d e'dej?        d>d>d?fd@e%d3e%d.e%d*e%d+e%d)ed-ej        dz  dHej@        dCe&dEe&dFej<        e=z  dej        fdI„ZAd d e'dfd@e%d3e%d.e%d*e%d+e%d)ed-ej        dz  fdJ„ZBd d e'dd?fd@e%d3e%d.e%d*e%d+e%d)ed-ej        dz  dFej<        e=z  defdK„ZC G dL„ dMe¦  «        ZD eD¦   «         ZEeDeFdN<   dOej        dej        dz  fdP„ZG	 dcdQe
dRej        d-ej        ez  dz  dSedz  dOej        dz  dTe%dz  dUej        dz  deHe&ej        ez  dz  e%e%f         fdV„ZI	 	 	 	 	 dddQe
dRej        d-ej        dz  dSedz  dOej        dz  dWedz  dXedz  dej        dz  dTe%dz  dej        ez  dz  fdY„ZJ	 	 	 	 dedQe
dRej        d-ej        dz  dUej        dz  dSedz  dWedz  dXedz  dej        ez  dz  fdZ„ZK	 	 	 	 	 dddQe
dRej        d-ej        dz  dSedz  dOej        dz  dWedz  dXedz  dej        dz  dTe%dz  dej        ez  dz  fd[„ZL	 	 	 	 dedQe
dRej        d-ej        dz  dUej        dz  dSedz  dWedz  dXedz  dej        ez  dz  fd\„ZM	 	 	 	 dedQe
dRej        d-ej        dz  dSedz  dOej        dz  dWedz  dXedz  dTe%dz  dej        ez  dz  fd]„ZN	 dcdQe
dRej        d-ej        dz  dSedz  dej        dz  f
d^„ZOeJeLeNeLeLeJeJeOeOeJeOd_œeJeOd`œdaœZP	 	 	 	 dedQe
dRej        d-ej        dz  dSedz  dOej        dz  dWedz  dXedz  dej        dz  fdb„ZQdS )fé    )ÚCallableNé   )ÚCache)ÚPreTrainedConfig)Úis_torch_xpu_availableÚlogging)ÚGeneralInterfaceÚis_flash_attention_requested)Úis_torch_flex_attn_availableÚis_torch_greater_or_equalÚ
is_tracing)Ú_DEFAULT_SPARSE_BLOCK_SIZE)Ú	BlockMaskÚcreate_block_maskz2.5T)Ú
accept_devz2.6)ÚTransformGetItemToIndexÚmask_functionsÚreturnc                  óh   ‡ — t          d„ ‰ D ¦   «         ¦  «        st          d‰ › �¦  «        ‚ˆ fd„}|S )zKReturns a mask function that is the intersection of provided mask functionsc              3   ó4   K  — | ]}t          |¦  «        V — Œd S ©N©Úcallable©Ú.0Úargs     úX/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/masking_utils.pyú	<genexpr>zand_masks.<locals>.<genexpr>2   ó(   è è € Ð7Ð7 �x˜‰}Œ}Ð7Ð7Ð7Ð7Ð7Ð7ó    ú.All inputs should be callable mask_functions: c                 ó¦   •— |                      dt          j        ¬¦  «        }‰D ]+}| || |||¦  «                             |j        ¦  «        z  }Œ,|S ©N© )Údtype)Únew_onesÚtorchÚboolÚtoÚdevice©Ú	batch_idxÚhead_idxÚq_idxÚkv_idxÚresultÚmaskr   s         €r   Úand_maskzand_masks.<locals>.and_mask5   s\   ø€ Ø—’ ­%¬*�Ñ5Ô5ˆØ"ð 	Yð 	YˆDØ˜d˜d 9¨h¸¸vÑFÔF×IÒIÈ&Ì-ÑXÔXÑXˆFˆFØˆr    ©ÚallÚRuntimeError)r   r2   s   ` r   Ú	and_masksr6   0   sZ   ø€ åÐ7Ð7¨Ð7Ñ7Ô7Ñ7Ô7ð ^ÝÐ\ÈNÐ\Ð\Ñ]Ô]Ð]ðð ð ð ð ð €Or    c                  óh   ‡ — t          d„ ‰ D ¦   «         ¦  «        st          d‰ › �¦  «        ‚ˆ fd„}|S )zDReturns a mask function that is the union of provided mask functionsc              3   ó4   K  — | ]}t          |¦  «        V — Œd S r   r   r   s     r   r   zor_masks.<locals>.<genexpr>@   r   r    r!   c                 ó¦   •— |                      dt          j        ¬¦  «        }‰D ]+}| || |||¦  «                             |j        ¦  «        z  }Œ,|S r#   )Ú	new_zerosr'   r(   r)   r*   r+   s         €r   Úor_maskzor_masks.<locals>.or_maskC   s\   ø€ Ø—’ ­5¬:�Ñ6Ô6ˆØ"ð 	Yð 	YˆDØ˜d˜d 9¨h¸¸vÑFÔF×IÒIÈ&Ì-ÑXÔXÑXˆFˆFØˆr    r3   )r   r;   s   ` r   Úor_masksr<   >   sZ   ø€ åÐ7Ð7¨Ð7Ñ7Ô7Ñ7Ô7ð ^ÝÐ\ÈNÐ\Ð\Ñ]Ô]Ð]ðð ð ð ð ð €Nr    r,   r-   r.   r/   c                 ó   — ||k    S )z:
    This creates a basic lower-diagonal causal mask.
    r$   ©r,   r-   r.   r/   s       r   Úcausal_mask_functionr?   L   s   € ð �UŠ?Ðr    c                 ó   — |dk    S )zƒ
    This creates a full bidirectional mask.

    NOTE: It is important to keep an index-based version for non-vmap expansion.
    r   r$   r>   s       r   Úbidirectional_mask_functionrA   S   s   € ð �AŠ:Ðr    Úsliding_windowc           
      óZ   ‡ — dt           dt           dt           dt           dt          f
ˆ fd„}|S )z…
    This is an overlay depicting a sliding window pattern. Add it on top of a causal mask for a proper sliding
    window mask.
    r,   r-   r.   r/   r   c                 ó   •— ||‰z
  k    S r   r$   ©r,   r-   r.   r/   rB   s       €r   Ú
inner_maskz*sliding_window_overlay.<locals>.inner_maskb   s   ø€ Ø˜ Ñ.Ò.Ð.r    ©Úintr(   ©rB   rF   s   ` r   Úsliding_window_overlayrJ   \   sL   ø€ ð/�cð /­Sð /½ð /Åcð /Ídð /ð /ð /ð /ð /ð /ð Ðr    Ú
chunk_sizeÚleft_paddingc           
      ó^   ‡ ‡— dt           dt           dt           dt           dt          f
ˆ ˆfd„}|S )z‹
    This is an overlay depicting a chunked attention pattern. Add it on top of a causal mask for a proper chunked
    attention mask.
    r,   r-   r.   r/   r   c                 ó@   •— |‰|          z
  ‰z  |‰|          z
  ‰z  k    S r   r$   )r,   r-   r.   r/   rK   rL   s       €€r   rF   z#chunked_overlay.<locals>.inner_maskn   s.   ø€ Ø˜ iÔ0Ñ0°ZÑ?ÀEÈLÐYbÔLcÑDcÐhrÑCrÒrÐrr    rG   )rK   rL   rF   s   `` r   Úchunked_overlayrO   h   s^   øø€ ðs�cð s­Sð s½ð sÅcð sÍdð sð sð sð sð sð sð sð Ðr    Úblock_sequence_idsc           
      óZ   ‡ — dt           dt           dt           dt           dt          f
ˆ fd„}|S )a‚  
    This is an overlay depicting a blockwise masking pattern. Instead of a single
    token, each block consists of arbitrary length tokens. In causal setup, each block
    can attend to prev block causally and can't attend to future blocks. Within one block
    the attention is always bidirectional.
    Mostly used in MLLMs when non-text data attends bidirectionally to itself.
    r,   r-   r.   r/   r   c                 óF   •— ‰| |f         }‰| |f         }||k    |dk    z  S )Nr   r$   )r,   r-   r.   r/   Úq_groupÚkv_grouprP   s         €r   rF   z%blockwise_overlay.<locals>.inner_mask}   s5   ø€ à$ Y°Ð%5Ô6ˆØ% i°Ð&7Ô8ˆØ˜8Ò#¨°1ªÑ5Ð5r    rG   )rP   rF   s   ` r   Úblockwise_overlayrU   t   sL   ø€ ð6�cð 6­Sð 6½ð 6Åcð 6Ídð 6ð 6ð 6ð 6ð 6ð 6ð Ðr    c                 óF   — t          t          | ¦  «        t          ¦  «        S )zQ
    This return the mask_function function to create a sliding window mask.
    )r6   rJ   r?   ©rB   s    r   Ú#sliding_window_causal_mask_functionrX   †   s   € õ Õ+¨NÑ;Ô;Õ=QÑRÔRÐRr    c           
      óZ   ‡ — dt           dt           dt           dt           dt          f
ˆ fd„}|S )zN
    This is an overlay depicting a bidirectional sliding window pattern.
    r,   r-   r.   r/   r   c                 ó0   •— t          ||z
  ¦  «        ‰k    S )z”A token can attend to any other token if their absolute distance is within
        the (inclusive) sliding window size (distance <= sliding_window).)ÚabsrE   s       €r   rF   z8sliding_window_bidirectional_overlay.<locals>.inner_mask’   s   ø€ õ �5˜6‘>Ñ"Ô" nÒ4Ð4r    rG   rI   s   ` r   Ú$sliding_window_bidirectional_overlayr\   �   sL   ø€ ð
5�cð 5­Sð 5½ð 5Åcð 5Ídð 5ð 5ð 5ð 5ð 5ð 5ð
 Ðr    c                 óF   — t          t          | ¦  «        t          ¦  «        S )z_
    This return the mask_function function to create a bidirectional sliding window mask.
    )r6   r\   rA   rW   s    r   Ú*sliding_window_bidirectional_mask_functionr^   š   s   € õ Õ9¸.ÑIÔIÕKfÑgÔgÐgr    c                 óH   — t          t          | |¦  «        t          ¦  «        S )zT
    This return the mask_function function to create a chunked attention mask.
    )r6   rO   r?   )rK   rL   s     r   Úchunked_causal_mask_functionr`   ¡   s   € õ •_ Z°Ñ>Ô>Õ@TÑUÔUÐUr    Úpadding_maskc           
      óZ   ‡ — dt           dt           dt           dt           dt          f
ˆ fd„}|S )zT
    This return the mask_function function corresponding to a 2D padding mask.
    r,   r-   r.   r/   r   c                 ó   •— ‰| |f         S r   r$   )r,   r-   r.   r/   ra   s       €r   rF   z)padding_mask_function.<locals>.inner_mask­   s   ø€ ð ˜I vÐ-Ô.Ð.r    rG   )ra   rF   s   ` r   Úpadding_mask_functionrd   ¨   sL   ø€ ð
/�cð /­Sð /½ð /Åcð /Ídð /ð /ð /ð /ð /ð /ð Ðr    Úpacked_sequence_maskc           
      óZ   ‡ — dt           dt           dt           dt           dt          f
ˆ fd„}|S )z\
    This return the mask_function function corresponding to a 2D packed sequence mask.
    r,   r-   r.   r/   r   c                 ó0   •— ‰| |f         ‰| |f         k    S r   r$   )r,   r-   r.   r/   re   s       €r   rF   z1packed_sequence_mask_function.<locals>.inner_mask»   s$   ø€ Ø# I¨uÐ$4Ô5Ð9MÈiÐY_ÐN_Ô9`Ò`Ð`r    rG   )re   rF   s   ` r   Úpacked_sequence_mask_functionrh   ¶   sW   ø€ ð
a�cð a­Sð a½ð aÅcð aÍdð að að að að að að Ðr    Úmask_functionÚq_offsetÚ	kv_offsetc           
      ób   ‡ ‡‡— dt           dt           dt           dt           dt          f
ˆˆ ˆfd„}|S )z•
    This function adds the correct offsets to the `q_idx` and `kv_idx` as the torch API can only accept lengths,
    not start and end indices.
    r,   r-   r.   r/   r   c                 ó,   •—  ‰| ||‰z   |‰z   ¦  «        S r   r$   )r,   r-   r.   r/   rk   ri   rj   s       €€€r   rF   z0add_offsets_to_mask_function.<locals>.inner_maskÇ   s#   ø€ Øˆ}˜Y¨°%¸(Ñ2BÀFÈYÑDVÑWÔWÐWr    rG   )ri   rj   rk   rF   s   ``` r   Úadd_offsets_to_mask_functionrn   Á   se   øøø€ ðX�cð X­Sð X½ð XÅcð XÍdð Xð Xð Xð Xð Xð Xð Xð Xð Ðr    Úattention_maskÚ	kv_lengthc                 óŽ   — | }| �@||z   | j         d         z
  x}dk    r't          j        j                             | d|f¦  «        }|S )zh
    From the 2D attention mask, prepare the correct padding mask to use by potentially padding it.
    Néÿÿÿÿr   )Úshaper'   ÚnnÚ
functionalÚpad)ro   rp   rk   Úlocal_padding_maskÚpadding_lengths        r   Úprepare_padding_maskry   Í   sY   € ð (ÐØÐ!à'¨)Ñ3°nÔ6JÈ2Ô6NÑNÐNˆNÐRSÒSÐSÝ!&¤Ô!4×!8Ò!8¸È!È^ÐI\Ñ!]Ô!]ÐØÐr    c                 ój   — ||z   | j         d         z
  x}dk    rt          j        | d|fd¬¦  «        } | S )zÏ
    Pads the `block_sequence_ids` in case the total length is less than `kv_length`.
    Usually that happens with `StaticCache` generation or generating without cache.
    Pads to the right with `-1`.
    rr   r   )rv   Úvalue)rs   ÚFrv   )rP   ro   rp   rk   rx   s        r   Úmaybe_pad_block_sequence_idsr}   Ù   sL   € ð $ iÑ/Ð2DÔ2JÈ2Ô2NÑNÐNˆÐRSÒSÐSÝœUÐ#5¸A¸~Ð;NÐVXÐYÑYÔYÐØÐr    Útensorc                 óV   — |                       ¦   «         |                      ¦   «         k    S )ziSimilar to `tensor.all()`, but uses an implementation with `tensor.sum()`, which is actually much faster.)ÚsumÚnumel)r~   s    r   Úfast_allr‚   æ   s   € à�:Š:‰<Œ<˜6Ÿ<š<™>œ>Ò)Ð)r    Úq_lengthÚlocal_attention_sizec                 óv  — | �;| j         d         |k    r*t          j        || j        ¬¦  «        |z   }| dd…|f         } t	          | ¦  «        rdS |�||k    rdS |dk    s||k    r| �t          | ¦  «        rdS |dk    r;| �7t          | dd…d|…f         ¦  «        rt          | dd…|d…f          ¦  «        rdS dS )a×  
    Detects whether the causal mask can be ignored in case PyTorch's SDPA is used, rather relying on SDPA's `is_causal` argument.

    In case no token is masked in the 2D `padding_mask` argument, if `query_length == 1` or
    `key_value_length == query_length`, we rather rely on SDPA `is_causal` argument to use causal/non-causal masks,
    allowing to dispatch to the flash attention kernel (that can otherwise not be used if a custom `attn_mask` is
    passed).
    Nrr   ©r*   Fr   Tr   )rs   r'   Úaranger*   r   r‚   )ra   rƒ   rp   rj   rk   r„   Úmask_indicess          r   Ú_ignore_causal_mask_sdpar‰   ë   s  € ð  Ð LÔ$6°rÔ$:¸YÒ$FÐ$FÝ”| I°lÔ6IÐJÑJÔJÈYÑVˆØ# A A A | OÔ4ˆõ �,ÑÔð ØˆuàÐ'¨IÐ9MÒ,MÐ,MØˆuð
 	�AŠˆ˜ hÒ.Ð.°\Ð5IÍXÐVbÑMcÔMcÐ5IØˆtð �1‚}€}ØÐ¥¨,°q°q°q¸)¸8¸)°|Ô*DÑ!EÔ!EÐÍ(ÐT`ÐabÐabÐabÐdlÐdmÐdmÐamÔTnÐSnÑJoÔJoÐàˆtàˆ5r    c                 óh   — t          | ¦  «        rdS |�||k    rdS | €dS |                      ¦   «         S )zÃ
    XPU-specific logic for determining if we can skip bidirectional mask creation.

    For XPU devices, we have special handling:
    - Skip if no padding and no local attention constraint
    FNT)r   r4   ©ra   rp   r„   s      r   Ú _can_skip_bidirectional_mask_xpurŒ     sP   € õ �,ÑÔð Øˆuð Ð'¨IÐ9MÒ,MÐ,MØˆuàÐàˆtð ×ÒÑÔÐr    c                 ó”   — t           rt          | ||¦  «        S t          | ¦  «        s | �|                      ¦   «         r
|�||k     rdS dS )a²  
    Detects whether the bidirectional mask can be ignored in case PyTorch's SDPA is used.

    In case no token is masked in the 2D `padding_mask` argument and no local attention constraint applies
    (i.e. `local_attention_size` is None or `kv_length < local_attention_size`), we skip mask creation,
    allowing to dispatch to the flash attention kernel (that can otherwise not be used if a custom `attn_mask` is
    passed).
    NTF)Ú_is_torch_xpu_availablerŒ   r   r4   r‹   s      r   Ú_ignore_bidirectional_mask_sdpar�   4  sf   € õ ð _õ 0°¸iÐI]Ñ^Ô^Ð^õ
 �|Ñ$Ô$ðàÐ! \×%5Ò%5Ñ%7Ô%7Ð!à!Ð)¨YÐ9MÒ-MÐ-Màˆtàˆ5r    c                 óF   — g d¢}|D ]}t          j        | |d¬¦  «        } Œ| S )aJ  
    Used to vmap our mask_functions over the all 4 dimensions (b_idx, h_idx, q_idx, kv_idx) of the inputs.
    Using vmap here allows us to keep the performance of vectorized ops, while having a single set of primitive
    functions between attention interfaces (i.e. between flex and sdpa/eager, FA2 being a bit different).
    ))NNNr   )NNr   N)Nr   NN)r   NNNr   )Úin_dimsÚout_dims)r'   Úvmap)ri   Ú
dimensionsÚdimss      r   Ú_vmap_expansion_sdpar–   S  s?   € ð nÐmÐm€JØð Lð LˆÝœ
 =¸$ÈÐKÑKÔKˆˆØÐr    Úbatch_indicesÚhead_indicesÚ	q_indicesÚ
kv_indicesc                 ó~   — | dd…dddf         } |ddd…ddf         }|dddd…df         }|ddddd…f         }| |||fS )aÌ  
    Used to broadcast our mask_functions over the all 4 dimensions (b_idx, h_idx, q_idx, kv_idx) of the inputs.
    Allows the usage of any index-based mask function without relying on vmap.

    NOTE: This is limited to index based functions only and is not guaranteed to work otherwise.

    Reference:
        - https://github.com/huggingface/optimum-onnx/blob/c123e8f4fab61b54a8e0e31ce74462bcacca576e/optimum/exporters/onnx/model_patcher.py#L362-L365
    Nr$   )r—   r˜   r™   rš   s       r   Ú_non_vmap_expansion_sdparœ   `  so   € ð " ! ! ! T¨4°Ð"5Ô6€MØ  a a a¨¨tÐ 3Ô4€LØ˜$  a a a¨Ð-Ô.€IØ˜D $¨¨a¨a¨aÐ/Ô0€JØ˜,¨	°:Ð=Ð=r    FÚcpuÚ
batch_sizeÚ
local_sizeÚallow_is_causal_skipÚallow_is_bidirectional_skipÚallow_torch_fixÚuse_vmapr*   c                 óÔ  — t          |||¦  «        }|rt          ||||||¦  «        rdS |	rt          |||¦  «        rdS |�t          |t	          |¦  «        ¦  «        }t          j        | |¬¦  «        }t          j        d|¬¦  «        }t          j        ||¬¦  «        |z   }t          j        ||¬¦  «        |z   }|s. |t          ||||¦  «        Ž }|                     | d||¦  «        }nXt          rBt          ¦   «         5   t          |¦  «        ||||¦  «        }ddd¦  «         n# 1 swxY w Y   nt          d¦  «        ‚t          s|
r|t          j        | dd¬¦  «        z  }|S )uy  
    Create a 4D boolean mask of shape `(batch_size, 1, query_length, kv_length)` where a value of True indicates that
    the element should take part in the attention computation, and False that it should not.
    This function can only be used with torch>=2.5, as the context manager is otherwise not available.

    Args:
        batch_size (`int`):
            The batch size of the input sequence.
        q_length (`int`):
            The size that the query states will have during the attention computation.
        kv_length (`int`):
            The size that the key and value states will have during the attention computation.
        kv_offset (`int`, optional):
            An optional offset to indicate at which first position the key and values states will refer to.
        q_offset (`int`, optional):
            An optional offset to indicate at which first position the query states will refer to.
        mask_function (`Callable`):
            The mask factory function describing the mask pattern.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length)
        local_size (`int`, optional):
            The size of the local attention, if we do not use full attention. This is used only if `allow_is_causal_skip=True`
            to try to skip mask creation if possible.
        allow_is_causal_skip (`bool`, optional):
            Whether to allow to return `None` for the mask under conditions where we can use the `is_causal` argument in
            `torch.sdpa` instead. Default to `True`.
        allow_is_bidirectional_skip (`bool`, optional):
            Whether to allow to return `None` for the mask under conditions where we do not have to add any bias,
            i.e. full attention without any padding. Default to `False`.
        allow_torch_fix (`bool`, optional):
            Whether to update the mask in case a query is not attending to any tokens, to solve a bug in torch's older
            versions. We need an arg to skip it when using eager. By default `True`.
        use_vmap (`bool`, optional):
            Whether to use `vmap` during the mask construction or not. Allows powerful custom patterns that may not be
            index-based (for the cost of speed performance). By default `False`.
        device (`torch.device` or `str`, optional):
            An optional device to create the mask on.


    ## Creating a simple causal mask:

    To create the following causal mask:

        0 â–  â¬š â¬š â¬š â¬š
        1 â–  â–  â¬š â¬š â¬š
        2 â–  â–  â–  â¬š â¬š
        3 â–  â–  â–  â–  â¬š
        4 â–  â–  â–  â–  â– 

    You can do

    ```python
    >>> sdpa_mask(batch_size=1, q_length=5, kv_length=5)
    >>> tensor([[[[ True, False, False, False, False],
                  [ True,  True, False, False, False],
                  [ True,  True,  True, False, False],
                  [ True,  True,  True,  True, False],
                  [ True,  True,  True,  True,  True]]]])
    ```

    ## Creating a sliding window mask:

    To create the following sliding window mask (`sliding_window=3`):

        0 â–  â¬š â¬š â¬š â¬š
        1 â–  â–  â¬š â¬š â¬š
        2 â–  â–  â–  â¬š â¬š
        3 â¬š â–  â–  â–  â¬š
        4 â¬š â¬š â–  â–  â– 

    You can do

    ```python
    >>> sdpa_mask(batch_size=1, q_length=5, kv_length=5, mask_function=sliding_window_causal_mask_function(3))
    >>> tensor([[[[ True, False, False, False, False],
                  [ True,  True, False, False, False],
                  [ True,  True,  True, False, False],
                  [False,  True,  True,  True, False],
                  [False, False,  True,  True,  True]]]])
    ```

    ## Creating a chunked attention mask

    To create the following chunked attention mask (`chunk_size=3`):

        0 â–  â¬š â¬š â¬š â¬š
        1 â–  â–  â¬š â¬š â¬š
        2 â–  â–  â–  â¬š â¬š
        3 â¬š â¬š â¬š â–  â¬š
        4 â¬š â¬š â¬š â–  â– 

    You can do

    ```python
    >>> sdpa_mask(batch_size=1, q_length=5, kv_length=5, mask_function=chunked_causal_mask_function(3, torch.zeros(1, dtype=int)))
    >>> tensor([[[[ True, False, False, False, False],
                [ True,  True, False, False, False],
                [ True,  True,  True, False, False],
                [False, False, False,  True, False],
                [False, False, False,  True,  True]]]])
    ```

    Nr†   r   rr   zœThe vmap functionality for mask creation is only supported from torch>=2.6. Please update your torch version or use `use_vmap=False` with index-based masks.T)ÚdimÚkeepdim)ry   r‰   r�   r6   rd   r'   r‡   rœ   ÚexpandÚ#_is_torch_greater_or_equal_than_2_6r   r–   Ú
ValueErrorÚ#_is_torch_greater_or_equal_than_2_5r4   )rž   rƒ   rp   rj   rk   ri   ro   rŸ   r    r¡   r¢   r£   r*   Úkwargsra   Úbatch_arangeÚhead_arangeÚq_arangeÚ	kv_aranges                      r   Ú	sdpa_maskr°   s  s  € õp (¨¸	À9ÑMÔM€Lð
 ð Õ 8Ø�h 	¨8°YÀ
ñ!ô !ð ð ˆtØ"ð Õ'FÀ|ÐU^Ð`jÑ'kÔ'kð Øˆtð ÐÝ! -Õ1FÀ|Ñ1TÔ1TÑUÔUˆå”< 
°6Ð:Ñ:Ô:€LÝ”,˜q¨Ð0Ñ0Ô0€KÝŒ|˜H¨VÐ4Ñ4Ô4°xÑ?€HÝ”˜Y¨vÐ6Ñ6Ô6¸ÑB€Ið ð 
à&˜Õ(@ÀÈ{Ð\dÐfoÑ(pÔ(pÐqˆà'×.Ò.¨z¸2¸xÈÑSÔSˆˆõ 
-ð 
õ %Ñ&Ô&ð 	qð 	qØ@Õ1°-Ñ@Ô@ÀÈ{Ð\dÐfoÑpÔpˆNð	qð 	qð 	qñ 	qô 	qð 	qð 	qð 	qð 	qð 	qð 	qøøøð 	qð 	qð 	qð 	qøõ
 ð_ñ
ô 
ð 	
õ /ð [°?ð [Ø'­%¬)°^°OÈÐUYÐ*ZÑ*ZÔ*ZÑZˆàÐs   ÄD)Ä)D-Ä0D-r%   c                 ó&  — |                      dd¦  «        }|                      dd¦  «        }t          d| ||||||d|d|	|
dœ|¤Ž}|�It          j        |¦  «        j        }t          j        |t          j        d|j        |¬¦  «        |¦  «        }|S )	a$  
    Create a 4D float mask of shape `(batch_size, 1, query_length, kv_length)` where a value of 0 indicates that
    the element should take part in the attention computation, and -inf (minimum value for the given `dtype`) that
    it should not.

    Args:
        batch_size (`int`):
            The batch size of the input sequence.
        q_length (`int`):
            The size that the query states will have during the attention computation.
        kv_length (`int`):
            The size that the key and value states will have during the attention computation.
        q_offset (`int`, optional):
            An optional offset to indicate at which first position the query states will refer to.
        kv_offset (`int`, optional):
            An optional offset to indicate at which first position the key and values states will refer to.
        mask_function (`Callable`):
            The mask factory function describing the mask pattern.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length)
        dtype (`torch.dtype`, optional):
            The dtype to use for the mask. By default, `torch.float32`.
        allow_is_bidirectional_skip (`bool`, optional):
            Whether to allow to return `None` for the mask under conditions where we do not have to add any bias,
            i.e. full attention without any padding. Default to `False`.
        use_vmap (`bool`, optional):
            Whether to use `vmap` during the mask construction or not. Allows powerful custom patterns that may not be
            index-based (for the cost of speed performance). By default `False`.
        device (`torch.device` or `str`, optional):
            An optional device to create the mask on.
    r    Nr¢   F)rž   rƒ   rp   rj   rk   ri   ro   r    r¡   r¢   r£   r*   g        ©r*   r%   r$   )Úpopr°   r'   ÚfinfoÚminÚwherer~   r*   )rž   rƒ   rp   rj   rk   ri   ro   r%   r¡   r£   r*   r«   Ú_r1   Ú	min_dtypes                  r   Ú
eager_maskr¹     sº   € ð\ 	�
Š
Ð)¨4Ñ0Ô0€AØ�
Š
Ð$ dÑ+Ô+€AÝð ØØØØØØ#Ø%Ø"Ø$?ØØØðð ð ðð €Dð  ÐÝ”K Ñ&Ô&Ô*ˆ	åŒ{˜4¥¤¨c¸$¼+ÈUÐ!SÑ!SÔ!SÐU^Ñ_Ô_ˆØ€Kr    c                 óv   — |�6|dd…| d…f         }|j         d         |k    r|                     ¦   «         rd}|S )a„  
    Create the attention mask necessary to use FA2. Since FA2 is un-padded by definition, here we simply return
    `None` if the mask is fully causal, or we return the 2D mask which will then be used to extract the seq_lens.
    We just slice it in case of sliding window.

    Args:
        batch_size (`int`):
            The batch size of the input sequence.
        q_length (`int`):
            The size that the query states will have during the attention computation.
        kv_length (`int`):
            The size that the key and value states will have during the attention computation.
        q_offset (`int`, optional):
            An optional offset to indicate at which first position the query states will refer to.
        kv_offset (`int`, optional):
            An optional offset to indicate at which first position the key and values states will refer to.
        mask_function (`Callable`):
            The mask factory function describing the mask pattern.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length)
    Nr   )rs   r4   )rž   rƒ   rp   rj   rk   ri   ro   r«   s           r   Úflash_attention_maskr»   f  sS   € ð> Ð!à'¨¨¨¨I¨:¨;¨;¨Ô7ˆð Ô Ô" iÒ/Ð/°N×4FÒ4FÑ4HÔ4HÐ/Ø!ˆNàÐr    c           	      ó¨  — |�£t          |¦  «        s”|j        d         t          z  dz   t          z  }	|	|j        d         z
  }	t          s/|	dk    r)t          j        j                             |dd|	f¬¦  «        }t          |||¦  «        }
t          |t          |
¦  «        ¦  «        }t          |||¦  «        }t          || d|||t          ¬¦  «        }|S )a´  
    Create a 4D block mask which is a compressed representation of the full 4D block causal mask. BlockMask is essential
    for performant computation of flex attention. See: https://pytorch.org/blog/flexattention/

    Args:
        batch_size (`int`):
            The batch size of the input sequence.
        q_length (`int`):
            The size that the query states will have during the attention computation.
        kv_length (`int`):
            The size that the key and value states will have during the attention computation.
        q_offset (`int`, optional):
            An optional offset to indicate at which first position the query states will refer to.
        kv_offset (`int`, optional):
            An optional offset to indicate at which first position the key and values states will refer to.
        mask_function (`Callable`):
            The mask factory function describing the mask pattern.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length)
        device (`torch.device` or `str`, optional):
            An optional device to create the mask on.
    Nr   r   )r{   rv   )Úmask_modÚBÚHÚQ_LENÚKV_LENr*   Ú_compile)r‚   rs   Úflex_default_block_sizer¨   r'   rt   ru   rv   ry   r6   rd   rn   r   )rž   rƒ   rp   rj   rk   ri   ro   r*   r«   Úpad_lenra   Ú
block_masks               r   Úflex_attention_maskrÆ   ‘  sð   € ðD Ð!­(°>Ñ*BÔ*BÐ!ð #Ô(¨Ô+Õ/FÑFÈ!ÑKÕOfÑfˆØ˜NÔ0°Ô3Ñ3ˆÝ2ð 	`°wÀ²{°{Ý"œXÔ0×4Ò4°^È1ÐSTÐV]ÐR^Ð4Ñ_Ô_ˆNå+¨N¸IÀyÑQÔQˆÝ! -Õ1FÀ|Ñ1TÔ1TÑUÔUˆõ 1°ÀÈ)ÑTÔT€Mõ #ØØ
Ø
ØØØÝ4ðñ ô €Jð Ðr    c                   ó    — e Zd ZeeeeeedœZdS )ÚAttentionMaskInterface)ÚsdpaÚeagerÚflash_attention_2Úflash_attention_3Úflash_attention_4Úflex_attentionN)Ú__name__Ú
__module__Ú__qualname__r°   r¹   r»   rÆ   Ú_global_mappingr$   r    r   rÈ   rÈ   Î  s.   € € € € € ð ØØ1Ø1Ø1Ø-ðð €O€O€Or    rÈ   ÚALL_MASK_ATTENTION_FUNCTIONSÚposition_idsc                 óî   — | dd…dd…f         dz
  }t          j        | |d¬¦  «        }|dk                         d¦  «        }t          |¦  «        s$|dd…df         dk                         ¦   «         rdS |S )a>  
    Find the indices of the sequence to which each new query token in the sequence belongs when using packed
    tensor format (i.e. several sequences packed in the same batch dimension).

    Args:
        position_ids (`torch.Tensor`)
            A 2D tensor of shape (batch_size, query_length) indicating the positions of each token in the sequences.

    Returns:
        A 2D tensor where each similar integer indicates that the tokens belong to the same sequence. For example, if we
        pack 3 sequences of 2, 3 and 1 tokens respectively along a single batch dim, this will return [[0, 0, 1, 1, 1, 2]].

        If the there is only one sequence in each batch item (and we don't compile), then we return `None` indicating
        no packed sequences. This is the same as [[0, 0, 0, 0, 0, 0]] for the example above.
    Nr   rr   )Úprependr¥   r   )r'   ÚdiffÚcumsumr   r4   )rÔ   Úfirst_dummy_valueÚposition_diffre   s       r   Úfind_packed_sequence_indicesrÛ   ß  s•   € ð* % Q Q Q¨¨¨ UÔ+¨aÑ/ÐÝ”J˜|Ð5FÈBÐOÑOÔO€MØ)¨QÒ.×6Ò6°rÑ:Ô:Ðõ Ð*Ñ+Ô+ð Ð1EÀaÀaÀaÈÀeÔ1LÐPQÒ1Q×0VÒ0VÑ0XÔ0Xð ØˆtàÐr    ÚconfigÚinputs_embedsÚpast_key_valuesÚ	layer_idxÚencoder_hidden_statesc                 óô  — t          |t          j        t          f¦  «        r!t	          |j        ¦  «        dk    r	d|dddddfS | j        t          j        vrdS |�1|j	        dk    r&| 
                    |j        t          j        ¬¦  «        }|j        d         }|�e|                     |¦  «        }t          |t          j        ¦  «        r| 
                    |j        ¦  «        n|}|                     ||¦  «        \  }	}
n'd}|€|�|j        d         n|}	d}
n|j        d	         d}
}	d}|�G|€E|€C|j        d         }||j        d         k    r|                     |d	¦  «        }t!          |¦  «        }d
||||	||
fS )aµ	  
    Perform some common pre-processing of the mask arguments we get from the modeling code. Mostly determine the
    key-value length and offsets, and if we should early exit or not.

    Args:
        config (`PreTrainedConfig`):
            The model config.
        inputs_embeds (`torch.Tensor`):
            The input embeddings of shape (batch_size, query_length, hidden_dim). This is used only to infer the
            batch size, query length and dtype.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length).
            It can also be an already prepared 4D mask, in which case it is returned as-is.
        past_key_values (`Cache`, optional):
            The past key values, if we use a cache.
        position_ids (`torch.Tensor`, optional)
            A 2D tensor of shape (batch_size, query_length) indicating the positions of each token in the sequences.
        layer_idx (`int`, optional):
            If `past_key_values` is not None, this is the layer index of the cache from which to get the key-value
            length and offset. Indeed, for hybrid caches, different layers may return different lengths.
        encoder_hidden_states (`torch.Tensor`, optional):
            The input embeddings of shape (batch_size, kv_length, hidden_dim). If provided, it is used instead of
            `inputs_embeds` to infer the kv length.

    Returns:
        early_exit (`bool`):
            Whether we should early exit mask creation, and return the mask as-is.
        attention_mask (`torch.Tensor` or `BlockMask` or `None`):
            The attention mask to either return immediately, or to use in downstream mask creation.
        packed_sequence_mask (`torch.Tensor`, optional):
            In case we detected packed sequence format, this is a tensor where each similar integer indicates that
            the tokens belong to the same sequence.
        q_length (`int`):
            The size that the query states will have during the attention computation.
        kv_length (`int`):
            The size that the key and value states will have during the attention computation.
        q_offset (`int`, optional):
            An optional offset to indicate at which first position the query states will refer to.
        kv_offset (`int`):
            An offset to indicate at which first position the key and values states will refer to.
    é   TN)TNNNNNNé   r²   r   r   rr   F)Ú
isinstancer'   ÚTensorr   Úlenrs   Ú_attn_implementationrÓ   rÒ   Úndimr)   r*   r(   Úget_query_offsetÚget_mask_sizesr§   rÛ   )rÜ   rÝ   ro   rÞ   rÔ   rß   rà   rƒ   rj   rp   rk   re   rž   s                r   Ú_preprocess_mask_argumentsrë   ÿ  s¸  € õf �.¥5¤<µÐ";Ñ<Ô<ð BÅÀ^ÔEYÑAZÔAZÐ^_ÒA_ÐA_Ø�^ T¨4°°t¸TÐAÐAð Ô"Õ*FÔ*VÐVÐVØ7Ð7ð Ð! nÔ&9¸QÒ&>Ð&>Ø'×*Ò*°-Ô2FÍeÌjÐ*ÑYÔYˆàÔ" 1Ô%€HàÐ"Ø"×3Ò3°IÑ>Ô>ˆõ 9CÀ8ÍUÌ\Ñ8ZÔ8ZÐh�8—;’;˜}Ô3Ñ4Ô4Ð4Ð`hˆØ.×=Ò=¸hÈ	ÑRÔRÑˆ	�9�9ð ˆàÐ!à:OÐ:[Ð-Ô3°AÔ6Ð6ÐaiˆIØˆIˆIð $2Ô#7¸Ô#;¸Q�yˆIð  ÐØÐ NÐ$:¸Ð?VØ"Ô(¨Ô+ˆ
à˜Ô+¨AÔ.Ò.Ð.Ø'×.Ò.¨z¸2Ñ>Ô>ˆLÝ;¸LÑIÔIÐà�.Ð"6¸À)ÈXÐW`Ð`Ð`r    Úor_mask_functionÚand_mask_functionc	                 ó$  — t          | dd¦  «        st          | |||||¬¦  «        S |€6t          |d¦  «        r$d|j        v r|j                             d¦  «        }nd}t          | |||||¦  «        \  }	}}
}}}}|	r|S |j        d         |j        |j        }}}t          }t          | j                 }d}t          |dd¦  «        o|d	k     }|�*t          st          d
¦  «        ‚t          ||¦  «        }d}d}|�*t          st          d
¦  «        ‚t          ||¦  «        }d}d}|
�t          |t!          |
¦  «        ¦  «        }d}|�1t#          ||||¦  «        }t          |t%          |¦  «        ¦  «        }d} ||||||||||| ||¬¦  «        }|S )aÒ	  
    Create a standard causal mask based on the attention implementation used (stored in the config). If `past_key_values`
    has an hybrid cache structure, this function will return the mask corresponding to one of the "full_attention" layers (to align
    to what is needed in the `modeling_xxx.py` files).

    Args:
        config (`PreTrainedConfig`):
            The model config.
        inputs_embeds (`torch.Tensor`):
            The input embeddings of shape (batch_size, query_length, hidden_dim). This is used only to infer the
            batch size, query length and dtype.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length).
            It can also be an already prepared 4D mask, in which case it is returned as-is.
        cache_position (`torch.Tensor`):
            Deprecated and unused.
        past_key_values (`Cache`, optional):
            The past key values, if we use a cache.
        position_ids (`torch.Tensor`, optional)
            A 2D tensor of shape (batch_size, query_length) indicating the positions of each token in the sequences.
        or_mask_function (`Callable`, optional):
            An optional mask function to combine with the causal mask function (by doing the union of both). This is
            useful to easily overlay another mask on top of the causal one, for example for image tokens handling.
        and_mask_function (`Callable`, optional):
            An optional mask function to combine with the causal mask function (by doing the intersection of both). This is
            useful to easily overlay another mask on top of the causal one, for example for image tokens handling.
        block_sequence_ids (`torch.Tensor`, *optional*):
            A tensor of same shape as input IDs indicating to which block or group each token belongs to. Tokens from
            the same block will keep a bidirectional mask within the block, attending causally to the past. Index `-1`
            can be used for blocks that have to keep complete causality within itself.
        layer_idx (`int`, *optional*):
            The cache layer to size the mask against. By default, the first "full_attention" layer is used, which is
            correct whenever all layers of a given mask type have seen the same tokens. Pass it explicitly for caches
            where layers of the same type hold different lengths (e.g. per-depth MTP streams).
    Ú	is_causalT©rÞ   rì   rí   NÚ
is_slidingFr   Úis_compileabler   úLUsing `or_mask_function` or `and_mask_function` arguments require torch>=2.6)rž   rƒ   rp   rj   rk   ri   ro   r    r%   rÜ   r£   r*   )ÚgetattrÚcreate_bidirectional_maskÚhasattrrñ   Úindexrë   rs   r%   r*   r?   rÓ   rç   r¨   r©   r<   r6   rh   r}   rU   )rÜ   rÝ   ro   rÞ   rÔ   rì   rí   rP   rß   Ú
early_exitre   rƒ   rp   rj   rk   rž   r%   r*   Úmask_factory_functionÚmask_interfacer£   r    Úcausal_masks                          r   Úcreate_causal_maskrü   g  sA  € õ` �6˜;¨Ñ-Ô-ð 
Ý(ØØØØ+Ø-Ø/ð
ñ 
ô 
ð 	
ð ÐÝ�? LÑ1Ô1ð 	°e¸Ô?YÐ6YÐ6YØ'Ô2×8Ò8¸Ñ?Ô?ˆIˆIàˆIõ 	# 6¨=¸.È/Ð[gÐirÑsÔsñ _€J�Ð 4°hÀ	È8ÐU^ð ð ØÐà -Ô 3°AÔ 6¸Ô8KÈ]ÔMa�v�€JÝ0ÐÝ1°&Ô2MÔN€Nð
 €Hõ !(¨Ð9IÈ5Ñ QÔ QÐ cÐV^ÐbcÒVcÐdÐð
 Ð#Ý2ð 	mÝÐkÑlÔlÐlÝ (Ð)>Ð@PÑ QÔ QÐØ$ÐØˆØÐ$Ý2ð 	mÝÐkÑlÔlÐlÝ )Ð*?ÐARÑ SÔ SÐØ$ÐØˆð Ð'Ý )Ð*?ÕA^Ð_sÑAtÔAtÑ uÔ uÐØ$ÐØÐ%Ý9Ð:LÈnÐ^gÐirÑsÔsÐÝ (Ð)>Õ@QÐRdÑ@eÔ@eÑ fÔ fÐØ$Ðð !�.ØØØØØØ+Ø%Ø1ØØØØðñ ô €Kð Ðr    c                 ó  — t          |d¦  «        r$d|j        v r|j                             d¦  «        }nd}t          | |||d||¦  «        \  }	}}
}}}}|	r|S |�|n|}|j        d         |j        }}|j        }t          }t          | j	                 }d}d}|�*t          st          d¦  «        ‚t          ||¦  «        }d}d}|�*t          st          d¦  «        ‚t          ||¦  «        }d}d} ||||||||d||| ||¬¦  «        }|S )a;  
    Create a standard bidirectional mask based on the attention implementation used (stored in the config).

    Args:
        config (`PreTrainedConfig`):
            The model config.
        inputs_embeds (`torch.Tensor`):
            The input embeddings of shape (batch_size, query_length, hidden_dim). This is only used to infer metadata
            such as the batch size, query length, dtype, and device.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, kv_length).
            It can also be an already prepared 4D mask of shape (batch_size, 1, query_length, kv_length),
            in which case it is returned as-is.
        encoder_hidden_states (`torch.Tensor`, optional):
            The input embeddings of shape (batch_size, kv_length, hidden_dim). If provided, it is used instead of
            `inputs_embeds` to infer the batch size, kv length and dtype.
        past_key_values (`Cache`, optional):
            The past key values, if we use a cache.
        or_mask_function (`Callable`, optional):
            An optional mask function to combine with the base mask function (by doing the union of both). This is
            useful to easily overlay another mask on top, for example for image tokens handling.
        and_mask_function (`Callable`, optional):
            An optional mask function to combine with the base mask function (by doing the intersection of both). This is
            useful to easily overlay another mask on top, for example for image tokens handling.
    rñ   Fr   NTró   )rž   rƒ   rp   rj   rk   ri   ro   r    r¡   r%   rÜ   r£   r*   )rö   rñ   r÷   rë   rs   r%   r*   rA   rÓ   rç   r¨   r©   r<   r6   )rÜ   rÝ   ro   rà   rÞ   rì   rí   r«   rß   rø   r·   rƒ   rp   rj   rk   Úembedsrž   r%   r*   rù   rú   r¡   r£   s                          r   rõ   rõ   æ  s‘  € õH ˆ Ñ-Ô-ð °%¸?Ô;UÐ2UÐ2UØ#Ô.×4Ò4°UÑ;Ô;ˆ	ˆ	àˆ	õ OiØ�˜~¨ÀÀiÐQfñOô OÑK€J�  8¨Y¸À)ð ð ØÐà&;Ð&GÐ"Ð"È]€FØœ Qœ¨¬�€Jð Ô!€FÝ7ÐÝ1°&Ô2MÔN€Nð #'Ðð €Hð
 Ð#Ý2ð 	mÝÐkÑlÔlÐlÝ (Ð)>Ð@PÑ QÔ QÐØ&+Ð#ØˆØÐ$Ý2ð 	mÝÐkÑlÔlÐlÝ )Ð*?ÐARÑ SÔ SÐØ&+Ð#Øˆð $�^ØØØØØØ+Ø%à"Ø$?ØØØØðñ ô €Nð  Ðr    c	                 óz  — t          | dd¦  «        st          | |||||¬¦  «        S |€6t          |d¦  «        r$d|j        v r|j                             d¦  «        }nd}t          | |||||¦  «        \  }	}}
}}}}|	r|S t          | dd¦  «        }|€t          d¦  «        ‚|j        d         |j        |j	        }}}t          |¦  «        }t          | j                 }d	}t          |d
d	¦  «        o|dk     }|�*t          st          d¦  «        ‚t          ||¦  «        }d	}d}|�*t          st          d¦  «        ‚t          ||¦  «        }d	}d}|
�t          |t!          |
¦  «        ¦  «        }d	}|�1t#          ||||¦  «        }t          |t%          |¦  «        ¦  «        }d	} |||||||||||| ||¬¦  «        }|S )aB
  
    Create a sliding window causal mask based on the attention implementation used (stored in the config). This type
    of attention pattern was mostly democratized by Mistral. If `past_key_values` has an hybrid cache structure, this
    function will return the mask corresponding to one of the "sliding_attention" layers (to align to what is needed in the
    `modeling_xxx.py` files).

    Args:
        config (`PreTrainedConfig`):
            The model config.
        inputs_embeds (`torch.Tensor`):
            The input embeddings of shape (batch_size, query_length, hidden_dim). This is used only to infer the
            batch size, query length and dtype.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length).
            It can also be an already prepared 4D mask, in which case it is returned as-is.
        cache_position (`torch.Tensor`):
            Deprecated and unused.
        past_key_values (`Cache`, optional):
            The past key values, if we use a cache.
        position_ids (`torch.Tensor`, optional)
            A 2D tensor of shape (batch_size, query_length) indicating the positions of each token in the sequences.
        or_mask_function (`Callable`, optional):
            An optional mask function to combine with the sliding causal mask function (by doing the union of both). This is
            useful to easily overlay another mask on top of the sliding causal one, for example for image tokens handling.
        and_mask_function (`Callable`, optional):
            An optional mask function to combine with the sliding causal mask function (by doing the intersection of both). This is
            useful to easily overlay another mask on top of the sliding causal one, for example for image tokens handling.
        block_sequence_ids (`torch.Tensor`, *optional*):
            A tensor of same shape as input IDs indicating to which block or group each token belongs to. Tokens from
            the same block will keep a bidirectional mask within the block, attending causally to the past. Index `-1`
            can be used for blocks that have to keep complete causality within itself.
        layer_idx (`int`, *optional*):
            The cache layer to size the mask against. By default, the first "full_attention" layer is used, which is
            correct whenever all layers of a given mask type have seen the same tokens. Pass it explicitly for caches
            where layers of the same type hold different lengths (e.g. per-depth MTP streams).
    rï   Trð   Nrñ   r   rB   úJCould not find a `sliding_window` argument in the config, or it is not setFrò   r   ró   ©rž   rƒ   rp   rj   rk   ri   ro   r    rŸ   r%   rÜ   r£   r*   )rô   Ú(create_bidirectional_sliding_window_maskrö   rñ   r÷   rë   r©   rs   r%   r*   rX   rÓ   rç   r¨   r<   r6   rh   r}   rU   )rÜ   rÝ   ro   rÞ   rÔ   rì   rí   rP   rß   rø   re   rƒ   rp   rj   rk   rB   rž   r%   r*   rù   rú   r£   r    rû   s                           r   Ú!create_sliding_window_causal_maskr  J  ss  € õb �6˜;¨Ñ-Ô-ð 
Ý7ØØØØ+Ø-Ø/ð
ñ 
ô 
ð 	
ð ÐÝ�? LÑ1Ô1ð 	°d¸oÔ>XÐ6XÐ6XØ'Ô2×8Ò8¸Ñ>Ô>ˆIˆIàˆIõ 	# 6¨=¸.È/Ð[gÐirÑsÔsñ _€J�Ð 4°hÀ	È8ÐU^ð ð ØÐå˜VÐ%5°tÑ<Ô<€NØÐÝÐeÑfÔfÐfà -Ô 3°AÔ 6¸Ô8KÈ]ÔMa�v�€JÝ?ÀÑOÔOÐÝ1°&Ô2MÔN€Nð
 €Hõ !(¨Ð9IÈ5Ñ QÔ QÐ cÐV^ÐbcÒVcÐdÐð
 Ð#Ý2ð 	mÝÐkÑlÔlÐlÝ (Ð)>Ð@PÑ QÔ QÐØ$ÐØˆØÐ$Ý2ð 	mÝÐkÑlÔlÐlÝ )Ð*?ÐARÑ SÔ SÐØ$ÐØˆð Ð'Ý )Ð*?ÕA^Ð_sÑAtÔAtÑ uÔ uÐØ$ÐØÐ%Ý9Ð:LÈnÐ^gÐirÑsÔsÐÝ (Ð)>Õ@QÐRdÑ@eÔ@eÑ fÔ fÐØ$Ðð !�.ØØØØØØ+Ø%Ø1Ø!ØØØØðñ ô €Kð Ðr    c                 óf  — t          |d¦  «        r$d|j        v r|j                             d¦  «        }nd}t          | |||d||¦  «        \  }	}}
}}}}|	r|S t	          | dd¦  «        }|€t          d¦  «        ‚|�|n|}|j        d         |j        |j        }}}t          |¦  «        }t          | j                 }d}d}|�*t          st          d¦  «        ‚t          ||¦  «        }d}d}|�*t          st          d¦  «        ‚t          ||¦  «        }d}d} ||||||||d|||| ||¬	¦  «        }|S )
aJ  
    Create a standard bidirectional sliding window mask based on the attention implementation used (stored in the config).

    Args:
        config (`PreTrainedConfig`):
            The model config.
        inputs_embeds (`torch.Tensor`):
            The input embeddings of shape (batch_size, query_length, hidden_dim). This is only used to infer metadata
            such as the batch size, query length, dtype, and device.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, kv_length).
            It can also be an already prepared 4D mask of shape (batch_size, 1, query_length, kv_length),
            in which case it is returned as-is.
        encoder_hidden_states (`torch.Tensor`, optional):
            The input embeddings of shape (batch_size, kv_length, hidden_dim). If provided, it is used instead of
            `inputs_embeds` to infer the batch size, kv length and dtype.
        past_key_values (`Cache`, optional):
            The past key values, if we use a cache.
        or_mask_function (`Callable`, optional):
            An optional mask function to combine with the base mask function (by doing the union of both). This is
            useful to easily overlay another mask on top, for example for image tokens handling.
        and_mask_function (`Callable`, optional):
            An optional mask function to combine with the base mask function (by doing the intersection of both). This is
            useful to easily overlay another mask on top, for example for image tokens handling.
    rñ   Tr   NrB   r   Fró   )rž   rƒ   rp   rj   rk   ri   ro   r    r¡   rŸ   r%   rÜ   r£   r*   )rö   rñ   r÷   rë   rô   r©   rs   r%   r*   r^   rÓ   rç   r¨   r<   r6   )rÜ   rÝ   ro   rà   rÞ   rì   rí   r«   rß   rø   r·   rƒ   rp   rj   rk   rB   rþ   rž   r%   r*   rù   rú   r£   r¡   s                           r   r  r  Î  s·  € õH ˆ Ñ-Ô-ð °$¸/Ô:TÐ2TÐ2TØ#Ô.×4Ò4°TÑ:Ô:ˆ	ˆ	àˆ	õ OiØ�˜~¨ÀÀiÐQfñOô OÑK€J�  8¨Y¸À)ð ð ØÐå˜VÐ%5°tÑ<Ô<€NØÐÝÐeÑfÔfÐfà&;Ð&GÐ"Ð"È]€FØ &¤¨Q¤°´¸v¼}�v�€JÝFÀ~ÑVÔVÐÝ1°&Ô2MÔN€Nà€HØ"&ÐàÐ#Ý2ð 	mÝÐkÑlÔlÐlÝ (Ð)>Ð@PÑ QÔ QÐØ&+Ð#ØˆØÐ$Ý2ð 	mÝÐkÑlÔlÐlÝ )Ð*?ÐARÑ SÔ SÐØ&+Ð#Øˆà#�^ØØØØØØ+Ø%Ø"Ø$?Ø!ØØØØðñ ô €Nð  Ðr    c                 óÖ  — |€6t          |d¦  «        r$d|j        v r|j                             d¦  «        }nd}t          | |||||¦  «        \  }}}	}
}}}|r|S t	          | dd¦  «        }|€t          d¦  «        ‚t          | ¦  «        r||z   |k    rt          d¦  «        ‚|j        d         |j        |j	        }}}|�A| 
                    d¬	¦  «        t          j        |¦  «        k                         d¬	¦  «        }nt          j        ||t          ¬
¦  «        }t!          ||¦  «        }t"          | j                 }d}t	          |dd¦  «        o|
dk     }|�*t&          st          d¦  «        ‚t)          ||¦  «        }d}d}|�*t&          st          d¦  «        ‚t+          ||¦  «        }d}d}|	�t+          |t-          |	¦  «        ¦  «        }d} |||
||||||||| ||¬¦  «        }|S )aÇ  
    Create a chunked attention causal mask based on the attention implementation used (stored in the config). This type
    of attention pattern was mostly democratized by Llama4. If `past_key_values` has an hybrid cache structure, this
    function will return the mask corresponding to one of the "chunked_attention" layers (to align to what is needed in the
    `modeling_xxx.py` files).

    Args:
        config (`PreTrainedConfig`):
            The model config.
        inputs_embeds (`torch.Tensor`):
            The input embeddings of shape (batch_size, query_length, hidden_dim). This is used only to infer the
            batch size, query length and dtype.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length).
            It can also be an already prepared 4D mask, in which case it is returned as-is.
        cache_position (`torch.Tensor`):
            Deprecated and unused.
        past_key_values (`Cache`, optional):
            The past key values, if we use a cache.
        position_ids (`torch.Tensor`, optional)
            A 2D tensor of shape (batch_size, query_length) indicating the positions of each token in the sequences.
        or_mask_function (`Callable`, optional):
            An optional mask function to combine with the chunked causal mask function (by doing the union of both). This is
            useful to easily overlay another mask on top of the chunked causal one, for example for image tokens handling.
        and_mask_function (`Callable`, optional):
            An optional mask function to combine with the chunked causal mask function (by doing the intersection of both). This is
            useful to easily overlay another mask on top of the chunked causal one, for example for image tokens handling.
        layer_idx (`int`, *optional*):
            The cache layer to size the mask against. By default, the first "full_attention" layer is used, which is
            correct whenever all layers of a given mask type have seen the same tokens. Pass it explicitly for caches
            where layers of the same type hold different lengths (e.g. per-depth MTP streams).
    Nrñ   Tr   Úattention_chunk_sizezQCould not find an `attention_chunk_size` argument in the config, or it is not setzÝFlash attention cannot handle chunked attention, and the key-value length is larger than the chunk size so the chunked pattern cannot be respected. You should use another `attn_implementation` when instantiating the modelrr   )r¥   r²   Frò   r   ró   r  )rö   rñ   r÷   rë   rô   r©   r
   rs   r%   r*   rØ   r'   Ú
zeros_liker€   ÚzerosrH   r`   rÓ   rç   r¨   r<   r6   rh   )rÜ   rÝ   ro   rÞ   rÔ   rì   rí   rß   rø   re   rƒ   rp   rj   rk   rK   rž   r%   r*   Úleft_padding_tokensrù   rú   r£   r    rû   s                           r   Úcreate_chunked_causal_maskr
  *  s�  € ðT Ðå�? LÑ1Ô1ð 	°d¸oÔ>XÐ6XÐ6XØ'Ô2×8Ò8¸Ñ>Ô>ˆIˆIàˆIõ 	# 6¨=¸.È/Ð[gÐirÑsÔsñ _€J�Ð 4°hÀ	È8ÐU^ð ð ØÐå˜Ð!7¸Ñ>Ô>€JØÐÝÐlÑmÔmÐmõ $ FÑ+Ô+ð 
°	¸IÑ0EÈ
Ò0RÐ0RÝð}ñ
ô 
ð 	
ð
 !.Ô 3°AÔ 6¸Ô8KÈ]ÔMa�v�€Jð Ð!à-×4Ò4¸Ð4Ñ<Ô<ÅÔ@PÐQ_Ñ@`Ô@`Ò`×eÒeÐjlÐeÑmÔmÐÐå#œk¨*¸VÍ3ÐOÑOÔOÐÝ8¸ÐEXÑYÔYÐÝ1°&Ô2MÔN€Nð
 €Hõ !(¨Ð9IÈ5Ñ QÔ QÐ cÐV^ÐbcÒVcÐdÐð
 Ð#Ý2ð 	mÝÐkÑlÔlÐlÝ (Ð)>Ð@PÑ QÔ QÐØ$ÐØˆØÐ$Ý2ð 	mÝÐkÑlÔlÐlÝ )Ð*?ÐARÑ SÔ SÐØ$ÐØˆð Ð'Ý )Ð*?ÕA^Ð_sÑAtÔAtÑ uÔ uÐØ$Ðð !�.ØØØØØØ+Ø%Ø1ØØØØØðñ ô €Kð Ðr    c                 óú   — |�|j         dk    rdS |�|                     ¦   «         rdS t          |¦  «        st          j        |dk    ¦  «        rdS |dd…|j        d          d…f                              ¦   «         S )uÕ  Return the 2D padding mask for mamba / linear-attention layers, sized to the local sequence.

    Returns ``None`` (so the consumer skips masking entirely) when any of:
    - the input mask is missing or is already a custom 4D attention mask (no 2D padding signal);
    - the recurrent state already covers past tokens (cached forwards);
    - the mask is all-ones (un-padded batch â€” the masking multiply would be a no-op), skipped
      only outside trace/compile so the graph specialisation stays stable.

    Otherwise we trim the mask to the trailing ``inputs_embeds.shape[1]`` positions so it aligns
    with the current forward's local sequence and the consumer can multiply directly without
    further slicing.
    Nrã   r   )rè   Úhas_previous_stater   r'   r4   rs   Ú
contiguous)rÜ   rÝ   ro   rÞ   r«   s        r   Úcreate_recurrent_attention_maskr  §  s”   € ð& Ð Ô!4¸Ò!9Ð!9ØˆtØÐ" ×'IÒ'IÑ'KÔ'KÐ"ØˆtÝ�nÑ%Ô%ð ­%¬)°NÀaÒ4GÑ*HÔ*Hð Øˆtà˜!˜!˜!˜mÔ1°!Ô4Ð4Ð6Ð6Ð6Ô7×BÒBÑDÔDÐDr    )Úfull_attentionÚlinear_attention)Úsliding_attentionr  )r  r  Úchunked_attentionÚcompressed_sparse_attentionÚheavily_compressed_attentionÚminimax_m3_sparseÚdeepseek_sparse_attentionr  ÚconvÚhybridÚhybrid_slidingc           	      ó
  — |                       ¦   «         }	|	|||||||dœ}
t          |	d¦  «        r�t          |	j        ¦  «        }t	          d„ |D ¦   «         ¦  «        r|S i }|D ]Y}t
          |         }t          |t          ¦  «        r*|                     ¦   «         D ]\  }}||vr |di |
¤Ž||<   ŒŒN |di |
¤Ž||<   ŒZ|S t          |	dd¦  «        �t          di |
¤ŽS t          |	dd¦  «        �t          di |
¤ŽS t          di |
¤ŽS )aº  
    This function mimics how we create the masks in the `modeling_xxx.py` files, and is used in places like `generate`
    in order to easily create the masks in advance, when we compile the forwards with Static caches.

    Args:
        config (`PreTrainedConfig`):
            The model config.
        inputs_embeds (`torch.Tensor`):
            The input embeddings of shape (batch_size, query_length, hidden_dim). This is used only to infer the
            batch size, query length and dtype.
        attention_mask (`torch.Tensor`, optional):
            The 2D attention mask corresponding to padded tokens of shape (batch_size, number_of_seen_tokens+q_length).
            It can also be an already prepared 4D mask, in which case it is returned as-is.
        past_key_values (`Cache`, optional):
            The past key values, if we use a cache.
        position_ids (`torch.Tensor`, optional)
            A 2D tensor of shape (batch_size, query_length) indicating the positions of each token in the sequences.
        or_mask_function (`Callable`, optional):
            An optional mask function to combine with the other mask function (by doing the union of both). This is
            useful to easily overlay another mask on top of the causal one, for example for image tokens handling.
        and_mask_function (`Callable`, optional):
            An optional mask function to combine with the other mask function (by doing the intersection of both). This is
            useful to easily overlay another mask on top of the causal one, for example for image tokens handling.
        block_sequence_ids (`torch.Tensor`, *optional*):
            A tensor of same shape as input IDs indicating to which block or group each token belongs to. Tokens from
            the same block will keep a bidirectional mask within the block, attending causally to the past. Index `-1`
            can be used for blocks that have to keep complete causality within itself.
    )rÜ   rÝ   ro   rÞ   rÔ   rì   rí   rP   Úlayer_typesc              3   ó(   K  — | ]}|t           vV — Œd S r   )Ú&LAYER_PATTERN_TO_MASK_FUNCTION_MAPPING)r   Ú
layer_types     r   r   z,create_masks_for_generate.<locals>.<genexpr>  s(   è è € ÐiÐiÈJˆzÕ!GÐGÐiÐiÐiÐiÐiÐir    rB   Nr  r$   )Úget_text_configrö   Úsetr  Úanyr  rä   ÚdictÚitemsrô   r  r
  rü   )rÜ   rÝ   ro   rÞ   rÔ   rì   rí   rP   r«   Úeffective_configÚmask_kwargsÚlayer_patternsÚcausal_masksÚlayer_patternri   Úactual_patternÚactual_functions                    r   Úcreate_masks_for_generater+  Ó  s“  € ðP ×-Ò-Ñ/Ô/Ðð #Ø&Ø(Ø*Ø$Ø,Ø.Ø0ð	ð 	€Kõ Ð Ñ/Ô/ð 9ÝÐ-Ô9Ñ:Ô:ˆåÐiÐiÐZhÐiÑiÔiÑiÔið 	"Ø!Ð!ØˆØ+ð 		Kð 		KˆMÝBÀ=ÔQˆMå˜-­Ñ.Ô.ð KØ7D×7JÒ7JÑ7LÔ7Lð Vð VÑ3�N Oà%¨\Ð9Ð9Ø7F°Ð7UÐ7UÈÐ7UÐ7U˜ ^Ñ4øðVð
 /<¨mÐ.JÐ.J¸kÐ.JÐ.J�˜]Ñ+Ð+ØÐå	Ð!Ð#3°TÑ	:Ô	:Ð	FÝ0Ð?Ð?°;Ð?Ð?Ð?å	Ð!Ð#9¸4Ñ	@Ô	@Ð	LÝ)Ð8Ð8¨KÐ8Ð8Ð8åÐ,Ð, Ð,Ð,Ð,r    r   )NNNNN)NNNN)RÚcollections.abcr   r'   Útorch.nn.functionalrt   ru   r|   Úcache_utilsr   Úconfiguration_utilsr   Úutilsr   r   Úutils.genericr	   r
   Úutils.import_utilsr   r   r   Ú!torch.nn.attention.flex_attentionr   rÃ   r   r   rå   rª   r¨   rŽ   Ú,torch._dynamo._trace_wrapped_higher_order_opr   Ú
get_loggerrÏ   Úloggerr6   r<   rH   r(   r?   rA   rJ   rO   rU   rX   r\   r^   r`   rd   rh   rn   ry   r}   Ú
BoolTensorr‚   r‰   rŒ   r�   r–   rœ   r*   Ústrr°   Úfloat32r%   r¹   r»   rÆ   rÈ   rÓ   Ú__annotations__rÛ   Útuplerë   rü   rõ   r  r  r
  r  r  r+  r$   r    r   ú<module>r<     s¾  ðð %Ð $Ð $Ð $Ð $Ð $Ð $à €€€Ø Ð Ð Ð Ð Ð Ð Ð Ð à Ð Ð Ð Ð Ð Ø 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2Ø IÐ IÐ IÐ IÐ IÐ IÐ IÐ Iðð ð ð ð ð ð ð ð ð ð  ÐÑ!Ô!ð ØgÐgÐgÐgÐgÐgØNÐNÐNÐNÐNÐNÐNÐNÐNð ”€Ià&?Ð&?ÀÐRVÐ&WÑ&WÔ&WÐ #Ø&?Ð&?ÀÐRVÐ&WÑ&WÔ&WÐ #Ø0Ð0Ñ2Ô2Ð à&ð UØTÐTÐTÐTÐTÐTð 
ˆÔ	˜HÑ	%Ô	%€ð˜xð ¨Hð ð ð ð ð˜hð ¨8ð ð ð ð ð Cð °3ð ¸sð ÈCð ÐTXð ð ð ð ð¨3ð ¸#ð Àcð ÐSVð Ð[_ð ð ð ð ð	¨3ð 	°8ð 	ð 	ð 	ð 	ð	 ð 	°5´<ð 	ÀHð 	ð 	ð 	ð 	ð¨%¬,ð ¸8ð ð ð ð ð$S¸ð SÀð Sð Sð Sð Sð
¸ð 
Àð 
ð 
ð 
ð 
ðh¸sð hÀxð hð hð hð hðV¨Sð VÀÄð VÐQYð Vð Vð Vð Vð¨¬ð ¸ð ð ð ð ð¸¼ð Èð ð ð ð ð	°ð 	ÀCð 	ÐTWð 	Ð\dð 	ð 	ð 	ð 	ð	¨¬¸Ñ)<ð 	Èð 	ÐY\ð 	ÐafÔamÐptÑatð 	ð 	ð 	ð 	ð
Øœð
Ø6;´lÀTÑ6Ið
ØVYð
Øfið
à
„\ð
ð 
ð 
ð 
ð*�UÔ%ð *¨%Ô*:ð *ð *ð *ð *ð (,ð+ð +Ø”, Ñ%ð+àð+ð ð+ð ð	+ð
 ð+ð  ™*ð+ð 
ð+ð +ð +ð +ð\Ø”, Ñ%ðàðð  ™*ðð 
ð	ð ð ð ð< (,ðð Ø”, Ñ%ðàðð  ™*ðð 
ð	ð ð ð ð>
¨ð 
°Xð 
ð 
ð 
ð 
ð>Ø”<ð>Ø/4¬|ð>ØHMÌð>ØbgÔbnð>ð >ð >ð >ð. ØØ2Ø*.Ø!Ø!%Ø(-Ø ØØ!&ðið iØðiàðið ðið ð	ið
 ðið ðið ”L 4Ñ'ðið �d‘
ðið ðið "&ðið ðið ðið ŒL˜3Ñðið „\�DÑðið ið ið ið` ØØ2Ø*.ØœØ(-ØØ!&ðDð DØðDàðDð ðDð ð	Dð
 ðDð ðDð ”L 4Ñ'ðDð Œ;ðDð "&ðDð ðDð ŒL˜3ÑðDð „\ðDð Dð Dð DðV ØØ2Ø*.ð(ð (Øð(àð(ð ð(ð ð	(ð
 ð(ð ð(ð ”L 4Ñ'ð(ð (ð (ð (ð^ ØØ2Ø*.Ø!&ð:ð :Øð:àð:ð ð:ð ð	:ð
 ð:ð ð:ð ”L 4Ñ'ð:ð ŒL˜3Ñð:ð ð:ð :ð :ð :ðz
ð 
ð 
ð 
ð 
Ð-ñ 
ô 
ð 
ð 8NÐ7MÑ7OÔ7OÐ Ð4Ð OÐ OÑ Oð ¨u¬|ð  ÀÄÈtÑ@Sð  ð  ð  ð  ðN 26ðeað eaØðeaà”<ðeað ”L 9Ñ,¨tÑ3ðeað ˜T‘\ð	eað
 ”, Ñ%ðeað �T‰zðeað !œ<¨$Ñ.ðeað ˆ4�” 	Ñ)¨DÑ0°#°sÐ:Ô;ðeað eað eað eaðZ )-Ø(,Ø)-Ø.2Ø ð|ð |Øð|à”<ð|ð ”L 4Ñ'ð|ð ˜T‘\ð	|ð
 ”, Ñ%ð|ð  ‘oð|ð   $‘ð|ð œ tÑ+ð|ð �T‰zð|ð „\�IÑ Ñ$ð|ð |ð |ð |ðF 26Ø$(Ø(,Ø)-ðað aØðaà”<ðað ”L 4Ñ'ðað !œ<¨$Ñ.ð	að
 ˜T‘\ðað  ‘oðað   $‘ðað „\�IÑ Ñ$ðað að að aðR )-Ø(,Ø)-Ø.2Ø ðAð AØðAà”<ðAð ”L 4Ñ'ðAð ˜T‘\ð	Að
 ”, Ñ%ðAð  ‘oðAð   $‘ðAð œ tÑ+ðAð �T‰zðAð „\�IÑ Ñ$ðAð Að Að AðP 26Ø$(Ø(,Ø)-ðYð YØðYà”<ðYð ”L 4Ñ'ðYð !œ<¨$Ñ.ð	Yð
 ˜T‘\ðYð  ‘oðYð   $‘ðYð „\�IÑ Ñ$ðYð Yð Yð YðB )-Ø(,Ø)-Ø ðzð zØðzà”<ðzð ”L 4Ñ'ðzð ˜T‘\ð	zð
 ”, Ñ%ðzð  ‘oðzð   $‘ðzð �T‰zðzð „\�IÑ Ñ$ðzð zð zð zðB %)ð	Eð EØðEà”<ðEð ”L 4Ñ'ðEð ˜T‘\ð	Eð „\�DÑðEð Eð Eð Eð< )Ø:Ø3Ø#DØ$EØ+Ø!3Ø7Ø+Ø!3ÐIhÐiÐiØ,>ÐTsÐtÐtð*ð *Ð &ð( )-Ø(,Ø)-Ø.2ðN-ð N-ØðN-à”<ðN-ð ”L 4Ñ'ðN-ð ˜T‘\ð	N-ð
 ”, Ñ%ðN-ð  ‘oðN-ð   $‘ðN-ð œ tÑ+ðN-ð N-ð N-ð N-ð N-ð N-r    