§
    ‚ŠtjAz ã            
       ó~  — d dl mZmZ d dlmZ d dlZddlmZ ddlm	Z	m
Z
mZmZmZmZ  e	¦   «         rd dlmZ  edd	¬
¦  «        Z ej        e¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ d e¦  «        Z  G d!„ d"e ¦  «        Z! G d#„ d$e!e¦  «        Z" G d%„ d&e!e¦  «        Z# G d'„ d(e!e¦  «        Z$ G d)„ d*e!e¦  «        Z%eeee!e!e!e"e#ed+œ	Z&eeee!e!e!e$e%ed+œ	Z' G d,„ d-¦  «        Z(d.ed/e)e*e+         e,f         fd0„Z- G d1„ d2e(¦  «        Z. G d3„ d4e(¦  «        Z/ G d5„ d6e(¦  «        Z0 G d7„ d8e(¦  «        Z1e/Z2 G d9„ d:e.¦  «        Z3dS );é    )ÚABCÚabstractmethod)ÚIterableNé   )ÚPreTrainedConfig)Úis_hqq_availableÚis_optimum_quanto_availableÚis_quanto_greaterÚis_torch_greater_or_equalÚis_torchdynamo_compilingÚlogging)Ú	Quantizerz2.7T©Ú
accept_devc            	       ó‚  ‡ — e Zd ZU dZdZdZdZedz  ed<   ˆ fd„Z	d„ Z
d„ Zed	ej        d
ej        ddfd„¦   «         Zed	ej        d
ej        deej        ej        f         fd„¦   «         Zededeeef         fd„¦   «         Zedefd„¦   «         Zedefd„¦   «         Zd„ Zd„ Zdd„Zdej        ddfd„Zdefd„Zˆ xZS )ÚCacheLayerMixinz0Base, abstract class for a single layer's cache.FTNÚ_layer_typec                 ó¶   •—  t          ¦   «         j        di |¤Ž | j        �7t          | t          ¦  «        r| t
          | j        <   d S | t          | j        <   d S d S )N© )ÚsuperÚ__init_subclass__r   Ú
issubclassÚStaticLayerÚSTATIC_LAYER_TYPE_MAPPINGÚDYNAMIC_LAYER_TYPE_MAPPING)ÚclsÚkwargsÚ	__class__s     €úV/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/cache_utils.pyr   z!CacheLayerMixin.__init_subclass__#   si   ø€ Ø!�‰ŒÔ!Ð+Ð+ FÐ+Ð+Ð+ØŒ?Ð&Ý˜#�{Ñ+Ô+ð BØ=@Õ)¨#¬/Ñ:Ð:Ð:à>AÕ*¨3¬?Ñ;Ð;Ð;ð	 'Ð&ó    c                 ó0   — d | _         d | _        d| _        d S ©NF)ÚkeysÚvaluesÚis_initialized)Úselfr   s     r   Ú__init__zCacheLayerMixin.__init__+   s   € Ø)-ˆŒ	Ø+/ˆŒØ#ˆÔÐÐr    c                 ó   — | j         j        › S ©N©r   Ú__name__©r&   s    r   Ú__repr__zCacheLayerMixin.__repr__0   ó   € Ø”.Ô)Ð+Ð+r    Ú
key_statesÚvalue_statesÚreturnc                 ó   — d S r)   r   ©r&   r/   r0   s      r   Úlazy_initializationz#CacheLayerMixin.lazy_initialization3   s   € ØadÐadr    c                 ó   — d S r)   r   ©r&   r/   r0   Úargsr   s        r   ÚupdatezCacheLayerMixin.update6   s	   € ð -0¨Cr    Úquery_lengthc                 ó   — d S r)   r   )r&   r9   s     r   Úget_mask_sizeszCacheLayerMixin.get_mask_sizes;   s   € ØDGÀCr    c                 ó   — d S r)   r   r,   s    r   Úget_seq_lengthzCacheLayerMixin.get_seq_length>   ó   € Ø%( Sr    c                 ó   — dS )a  
        Returns the maximum sequence length the layer can hold. A value of `-1` means no maximum, or an undefined
        maximum, for example a dynamic attention layer that grows indefinitely or a linear attention layer that has no
        sequence length dimension.
        Nr   r,   s    r   Úget_max_lengthzCacheLayerMixin.get_max_lengthA   s	   € ð 	ˆr    c                 óœ   — | j         rD| j                             dd¬¦  «        | _        | j                             dd¬¦  «        | _        dS dS ©z(Offload this layer's data to CPU device.ÚcpuT©Únon_blockingN)r%   r#   Útor$   r,   s    r   ÚoffloadzCacheLayerMixin.offloadJ   sP   € àÔð 	CØœ	Ÿš U¸˜Ñ>Ô>ˆDŒIØœ+Ÿ.š.¨¸T˜.ÑBÔBˆDŒKˆKˆKð	Cð 	Cr    c                 óÞ   — | j         rc| j        j        | j        k    rP| j                             | j        d¬¦  «        | _        | j                             | j        d¬¦  «        | _        dS dS dS ©zcIn case of layer offloading, this allows to move the data back to the layer's device ahead of time.TrD   N)r%   r#   ÚdevicerF   r$   r,   s    r   ÚprefetchzCacheLayerMixin.prefetchP   sk   € àÔð 	I 4¤9Ô#3°t´{Ò#BÐ#BØœ	Ÿš T¤[¸t˜ÑDÔDˆDŒIØœ+Ÿ.š.¨¬À4˜.ÑHÔHˆDŒKˆKˆKð	Ið 	IÐ#BÐ#Br    c                 ó  — | j         r2| j                             ¦   «          | j                             ¦   «          t	          | d¦  «        r>t          | j        t          ¦  «        r	d| _        dS | j                             ¦   «          dS dS )ú4Resets the cache values while preserving the objectsÚcumulative_lengthr   N)r%   r#   Úzero_r$   ÚhasattrÚ
isinstancerN   Úintr,   s    r   ÚresetzCacheLayerMixin.resetV   sŽ   € àÔð 	 ØŒI�OŠOÑÔÐØŒK×ÒÑÔÐå�4Ð,Ñ-Ô-ð 	/å˜$Ô0µ#Ñ6Ô6ð /Ø)*�Ô&Ð&Ð&àÔ&×,Ò,Ñ.Ô.Ð.Ð.Ð.ð	/ð 	/r    Úbeam_idxc                 ó.  — |                       ¦   «         dk    r|| j                             d|                     | j        j        ¦  «        ¦  «        | _        | j                             d|                     | j        j        ¦  «        ¦  «        | _        dS dS )z,Reorders this layer's cache for beam search.r   N)r=   r#   Úindex_selectrF   rJ   r$   ©r&   rT   s     r   Úreorder_cachezCacheLayerMixin.reorder_cachec   sy   € à×ÒÑ Ô  1Ò$Ð$Øœ	×.Ò.¨q°(·+²+¸d¼iÔ>NÑ2OÔ2OÑPÔPˆDŒIØœ+×2Ò2°1°h·k²kÀ$Ä+ÔBTÑ6UÔ6UÑVÔVˆDŒKˆKˆKð %Ð$r    c                 ó^   — t                                d¦  «         |                      ¦   «         S ©Nzm`get_max_cache_shape` is deprecated, and will be removed in version 5.16. Please use `get_max_length` instead)ÚloggerÚwarningr@   r,   s    r   Úget_max_cache_shapez#CacheLayerMixin.get_max_cache_shapei   s/   € Ý�ŠØ{ñ	
ô 	
ð 	
ð ×"Ò"Ñ$Ô$Ð$r    ©r1   N)r+   Ú
__module__Ú__qualname__Ú__doc__Úis_compileableÚsupports_early_initr   ÚstrÚ__annotations__r   r'   r-   r   ÚtorchÚTensorr4   Útupler8   rR   r;   r=   r@   rG   rK   rS   Ú
LongTensorrX   r]   Ú__classcell__©r   s   @r   r   r      sà  ø€ € € € € € Ø:Ð:à€NØÐð #€K��t‘Ð"Ð"Ñ"ðBð Bð Bð Bð Bð$ð $ð $ð
,ð ,ð ,ð Ød¨e¬lÐdÈ%Ì,ÐdÐ[_ÐdÐdÐdñ „^Ødàð0Øœ,ð0Ø6;´lð0à	ˆuŒ|˜Uœ\Ð)Ô	*ð0ð 0ð 0ñ „^ð0ð ØG¨3ÐG°5¸¸c¸´?ÐGÐGÐGñ „^ØGàØ( Ð(Ð(Ð(ñ „^Ø(àð ð ð ð ñ „^ððCð Cð CðIð Ið Ið/ð /ð /ð /ðW eÔ&6ð W¸4ð Wð Wð Wð Wð% Sð %ð %ð %ð %ð %ð %ð %ð %r    r   c                   óü   — e Zd ZdZdZdej        dej        ddfd„Zdej        dej        deej        ej        f         fd„Z	d	e
dee
e
f         fd
„Zde
fd„Zde
fd„Zde
ddfd„Zde
ddfd„Zdej        ddfd„ZdS )ÚDynamicLayerzà
    A cache layer that grows dynamically as more tokens are generated. This is the default for generative models.
    It stores the key and value states as tensors of shape `[batch_size, num_heads, seq_len, head_dim]`.
    Fr/   r0   r1   Nc                 óÞ   — |j         |j        c| _         | _        t          j        g | j         | j        ¬¦  «        | _        t          j        g | j         | j        ¬¦  «        | _        d| _        d S ©N©ÚdtyperJ   T)rq   rJ   rf   Útensorr#   r$   r%   r3   s      r   r4   z DynamicLayer.lazy_initializationx   s^   € Ø",Ô"2°JÔ4EÐˆŒ
�D”KÝ”L ¨4¬:¸d¼kÐJÑJÔJˆŒ	Ý”l 2¨T¬ZÀÄÐLÑLÔLˆŒØ"ˆÔÐÐr    c                 óà   — | j         s|                      ||¦  «         t          j        | j        |gd¬¦  «        | _        t          j        | j        |gd¬¦  «        | _        | j        | j        fS )ái  
        Update the key and value caches in-place, and return the necessary keys and value states.

        Args:
            key_states (`torch.Tensor`): The new key states to cache.
            value_states (`torch.Tensor`): The new value states to cache.

        Returns:
            tuple[`torch.Tensor`, `torch.Tensor`]: The key and value states.
        éþÿÿÿ©Údim)r%   r4   rf   Úcatr#   r$   r6   s        r   r8   zDynamicLayer.update~   sn   € ð Ô"ð 	?Ø×$Ò$ Z°Ñ>Ô>Ð>å”I˜tœy¨*Ð5¸2Ð>Ñ>Ô>ˆŒ	Ý”i ¤¨lÐ ;ÀÐDÑDÔDˆŒØŒy˜$œ+Ð%Ð%r    r9   c                 ó<   — d}|                       ¦   «         |z   }||fS )zDReturn the length and offset of the cache, used to generate the maskr   ©r=   ©r&   r9   Ú	kv_offsetÚ	kv_lengths       r   r;   zDynamicLayer.get_mask_sizes“   s(   € àˆ	Ø×'Ò'Ñ)Ô)¨LÑ8ˆ	Ø˜)Ð#Ð#r    c                 ór   — | j         r| j                             ¦   «         dk    rdS | j        j        d         S )ú1Returns the sequence length of the cached states.r   ru   )r%   r#   ÚnumelÚshaper,   s    r   r=   zDynamicLayer.get_seq_length™   s7   € àÔ"ð 	 d¤i§o¢oÑ&7Ô&7¸1Ò&<Ð&<Ø�1ØŒyŒ˜rÔ"Ð"r    c                 ó   — dS )zeReturns the maximum sequence length of the cache object. DynamicLayer does not have a maximum length.éÿÿÿÿr   r,   s    r   r@   zDynamicLayer.get_max_lengthŸ   ó   € àˆrr    Ú
max_lengthc                 óò   — |dk    r$|                       ¦   «         t          |¦  «        z
  }|                       ¦   «         |k    rdS | j        dd|…dd…f         | _        | j        dd|…dd…f         | _        dS )z 
        Crop the past key values up to a new `max_length` in terms of tokens. `max_length` can also be negative
        to remove `max_length` tokens.
        r   N.)r=   Úabsr#   r$   ©r&   r…   s     r   ÚcropzDynamicLayer.crop£   sƒ   € ð
 ˜Š?ˆ?Ø×,Ò,Ñ.Ô.µ°Z±´Ñ@ˆJà×ÒÑ Ô  JÒ.Ð.ØˆFà”I˜c ; J ;°°°Ð1Ô2ˆŒ	Ø”k # {¨
 {°A°A°AÐ"5Ô6ˆŒˆˆr    Úrepeatsc                 ó¾   — |                       ¦   «         dk    rD| j                             |d¬¦  «        | _        | j                             |d¬¦  «        | _        dS dS )z8Repeat the cache `repeats` times in the batch dimension.r   rv   N)r=   r#   Úrepeat_interleaver$   ©r&   rŠ   s     r   Úbatch_repeat_interleavez$DynamicLayer.batch_repeat_interleave±   s]   € à×ÒÑ Ô  1Ò$Ð$Øœ	×3Ò3°GÀÐ3ÑCÔCˆDŒIØœ+×7Ò7¸ÀQÐ7ÑGÔGˆDŒKˆKˆKð %Ð$r    Úindicesc                 óŠ   — |                       ¦   «         dk    r*| j        |df         | _        | j        |df         | _        dS dS )z<Only keep the `indices` in the batch dimension of the cache.r   .N)r=   r#   r$   ©r&   r�   s     r   Úbatch_select_indicesz!DynamicLayer.batch_select_indices·   sI   € à×ÒÑ Ô  1Ò$Ð$Øœ	 '¨3 ,Ô/ˆDŒIØœ+ g¨s lÔ3ˆDŒKˆKˆKð %Ð$r    )r+   r_   r`   ra   Ú
is_slidingrf   rg   r4   rh   r8   rR   r;   r=   r@   r‰   rŽ   r’   r   r    r   rm   rm   p   s_  € € € € € ðð ð
 €Jð#¨e¬lð #È%Ì,ð #Ð[_ð #ð #ð #ð #ð&Øœ,ð&Ø6;´lð&à	ˆuŒ|˜Uœ\Ð)Ô	*ð&ð &ð &ð &ð*$¨3ð $°5¸¸c¸´?ð $ð $ð $ð $ð# ð #ð #ð #ð #ð ð ð ð ð ð7˜sð 7 tð 7ð 7ð 7ð 7ðH¨sð H°tð Hð Hð Hð Hð4¨E¬Lð 4¸Tð 4ð 4ð 4ð 4ð 4ð 4r    rm   c                   óð   ‡ — e Zd ZdZdZdefˆ fd„Zdej        dej        ddfˆ fd	„Z	dej        dej        de
ej        ej        f         fd
„Zdede
eef         fd„Zdefd„Zdefd„Zdeddfˆ fd„Zˆ xZS )ÚDynamicSlidingWindowLayerzì
    A cache layer that grows dynamically as more tokens are generated, up until the sliding window size.
    It stores the key and value states as tensors of shape `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.
    TÚsliding_windowc                 ó¸   •— t          ¦   «                              ¦   «          || _        d| _        t	          j        | j        t          j        ¬¦  «        | _        d S ©Nr   ©rq   )r   r'   r–   rN   rf   rr   ÚlongÚ_sliding_window_tensor)r&   r–   r   r   s      €r   r'   z"DynamicSlidingWindowLayer.__init__Æ   sK   ø€ Ý‰Œ×ÒÑÔÐØ,ˆÔØ!"ˆÔÝ&+¤l°4Ô3FÍeÌjÐ&YÑ&YÔ&YˆÔ#Ð#Ð#r    r/   r0   r1   Nc                 ó”   •— t          ¦   «                              ||¦  «         | j                             | j        ¦  «        | _        d S r)   )r   r4   r›   rF   rJ   )r&   r/   r0   r   s      €r   r4   z-DynamicSlidingWindowLayer.lazy_initializationÌ   s>   ø€ Ý‰Œ×#Ò# J°Ñ=Ô=Ð=Ø&*Ô&A×&DÒ&DÀTÄ[Ñ&QÔ&QˆÔ#Ð#Ð#r    c                 óv  — | j         s|                      ||¦  «         | xj        |j        d         z  c_        t	          j        | j        |gd¬¦  «        }t	          j        | j        |gd¬¦  «        }|dd…dd…| j         dz   d…dd…f         | _        |dd…dd…| j         dz   d…dd…f         | _        ||fS )rt   ru   rv   Nr   )	r%   r4   rN   r�   rf   rx   r#   r$   r–   )r&   r/   r0   r7   r   Úfull_key_statesÚfull_value_statess          r   r8   z DynamicSlidingWindowLayer.updateÐ   sì   € ð Ô"ð 	?Ø×$Ò$ Z°Ñ>Ô>Ð>àÐÔ *Ô"2°2Ô"6Ñ6ÐÔõ  œ) T¤Y°
Ð$;ÀÐDÑDÔDˆÝ!œI t¤{°LÐ&AÀrÐJÑJÔJÐà# A A A q q q¨4Ô+>Ð*>ÀÑ*BÐ*DÐ*DÀaÀaÀaÐ$GÔHˆŒ	Ø'¨¨¨¨1¨1¨1¨tÔ/BÐ.BÀQÑ.FÐ.HÐ.HÈ!È!È!Ð(KÔLˆŒð Ð 1Ð1Ð1r    r9   c                 óž   — | j         | j        k    }t          | j         | j        z
  dz   d¦  «        }|r| j        dz
  |z   }n
| j         |z   }||fS ©úNReturn the length and offset of the cache, used to generate the attention maskr   r   )rN   r–   Úmax)r&   r9   Úis_fullr|   r}   s        r   r;   z(DynamicSlidingWindowLayer.get_mask_sizesí   se   € àÔ(¨DÔ,?Ò?ˆå˜Ô.°Ô1DÑDÀqÑHÈ!ÑLÔLˆ	Øð 	>ØÔ+¨aÑ/°,Ñ>ˆIˆIàÔ.°Ñ=ˆIà˜)Ð#Ð#r    c                 ó   — | j         S ©r   ©rN   r,   s    r   r=   z(DynamicSlidingWindowLayer.get_seq_lengthù   ó   € àÔ%Ð%r    c                 ó   — | j         S ©z+Return the maximum cache shape of the cache©r–   r,   s    r   r@   z(DynamicSlidingWindowLayer.get_max_lengthý   s   € àÔ"Ð"r    r…   c                 óÐ   •— |                       ¦   «         | j        k    rt          d¦  «        ‚t          ¦   «                              |¦  «         | j        j        d         | _        dS )z 
        Crop the past key values up to a new `max_length` in terms of tokens. `max_length` can also be
        negative to remove `max_length` tokens.
        z�Cannot `crop` a `DynamicSlidingWindowLayer` after it has seen more tokens than itssliding window (otherwise some states are lost)ru   N)r=   r–   Ú
ValueErrorr   r‰   r#   r�   rN   )r&   r…   r   s     €r   r‰   zDynamicSlidingWindowLayer.crop  sf   ø€ ð
 ×ÒÑ Ô  DÔ$7Ò7Ð7ÝðBñô ð õ 	‰Œ�Š�ZÑ Ô Ð Ø!%¤¤°Ô!4ˆÔÐÐr    )r+   r_   r`   ra   r“   rR   r'   rf   rg   r4   rh   r8   r;   r=   r@   r‰   rj   rk   s   @r   r•   r•   ¾   sg  ø€ € € € € ðð ð
 €JðZ sð Zð Zð Zð Zð Zð ZðR¨e¬lð RÈ%Ì,ð RÐ[_ð Rð Rð Rð Rð Rð Rð2Øœ,ð2Ø6;´lð2à	ˆuŒ|˜Uœ\Ð)Ô	*ð2ð 2ð 2ð 2ð:
$¨3ð 
$°5¸¸c¸´?ð 
$ð 
$ð 
$ð 
$ð& ð &ð &ð &ð &ð# ð #ð #ð #ð #ð5˜sð 5 tð 5ð 5ð 5ð 5ð 5ð 5ð 5ð 5ð 5ð 5r    r•   c                   óä   ‡ — e Zd ZdZˆ fd„Zdej        ddfd„Zdej        dej        fd„Zˆ fd„Z	ˆ fd	„Z
dˆ fd
„Zdej        ddfˆ fd„Zdeddfˆ fd„Zdeddfˆ fd„Zdej        ddfˆ fd„Zˆ xZS )ÚDynamicIndexedLayera{  
    A cache layer that extends `DynamicLayer` with an extra indexer key cache for Dynamic Sparse Attention (DSA)
    models (e.g. GLM MoE DSA, DeepSeek V32).

    The main K/V cache stores tensors of shape `[batch_size, num_heads, seq_len, head_dim]` (inherited).
    The indexer key cache stores a tensor of shape `[batch_size, seq_len, index_head_dim]` (3D, single-head).
    c                 ód   •— t          ¦   «                              ¦   «          d | _        d| _        d S r"   )r   r'   Úindexer_keysÚis_indexer_initialized)r&   r   r   s     €r   r'   zDynamicIndexedLayer.__init__  s/   ø€ Ý‰Œ×ÒÑÔÐØ15ˆÔØ,1ˆÔ#Ð#Ð#r    Úindexer_key_statesr1   Nc                 ó’   — |j         |j        c| _        | _        t	          j        g | j        | j        ¬¦  «        | _        d| _        d S ro   )rq   rJ   Úindexer_dtypeÚindexer_devicerf   rr   r±   r²   ©r&   r³   s     r   Úlazy_initialization_indexerz/DynamicIndexedLayer.lazy_initialization_indexer  sH   € Ø2DÔ2JÐL^ÔLeÐ/ˆÔ˜DÔ/Ý!œL¨°4Ô3EÈdÔNaÐbÑbÔbˆÔØ&*ˆÔ#Ð#Ð#r    c                 óŒ   — | j         s|                      |¦  «         t          j        | j        |gd¬¦  «        | _        | j        S )a`  
        Update the indexer key cache by concatenation, and return the full indexer keys.

        Args:
            indexer_key_states (`torch.Tensor`): New indexer keys, shape `[batch_size, seq_len, index_head_dim]`.

        Returns:
            `torch.Tensor`: The full cached indexer keys, shape `[batch_size, total_len, index_head_dim]`.
        r   rv   )r²   r¸   rf   rx   r±   r·   s     r   Úupdate_indexerz"DynamicIndexedLayer.update_indexer"  sO   € ð Ô*ð 	AØ×,Ò,Ð-?Ñ@Ô@Ð@Ý!œI tÔ'8Ð:LÐ&MÐSTÐUÑUÔUˆÔØÔ Ð r    c                 óœ   •— t          ¦   «                              ¦   «          | j        r#| j                             dd¬¦  «        | _        d S d S )NrC   TrD   )r   rG   r²   r±   rF   ©r&   r   s    €r   rG   zDynamicIndexedLayer.offload1  sS   ø€ Ý‰Œ�ŠÑÔÐØÔ&ð 	OØ $Ô 1× 4Ò 4°UÈÐ 4Ñ NÔ NˆDÔÐÐð	Oð 	Or    c                 óÔ   •— t          ¦   «                              ¦   «          | j        r=| j        j        | j        k    r*| j                             | j        d¬¦  «        | _        d S d S d S )NTrD   )r   rK   r²   r±   rJ   rF   r¼   s    €r   rK   zDynamicIndexedLayer.prefetch6  so   ø€ Ý‰Œ×ÒÑÔÐØÔ&ð 	U¨4Ô+<Ô+CÀtÄ{Ò+RÐ+RØ $Ô 1× 4Ò 4°T´[ÈtÐ 4Ñ TÔ TˆDÔÐÐð	Uð 	UÐ+RÐ+Rr    c                 óŒ   •— t          ¦   «                              ¦   «          | j        r| j                             ¦   «          d S d S r)   )r   rS   r²   r±   rO   r¼   s    €r   rS   zDynamicIndexedLayer.reset;  sD   ø€ Ý‰Œ�Š‰ŒˆØÔ&ð 	&ØÔ×#Ò#Ñ%Ô%Ð%Ð%Ð%ð	&ð 	&r    rT   c                 ó  •— t          ¦   «                              |¦  «         | j        r\| j                             ¦   «         dk    rA| j                             d|                     | j        j        ¦  «        ¦  «        | _        d S d S d S ©Nr   )r   rX   r²   r±   r€   rV   rF   rJ   )r&   rT   r   s     €r   rX   z!DynamicIndexedLayer.reorder_cache@  s…   ø€ Ý‰Œ×Ò˜hÑ'Ô'Ð'ØÔ&ð 	i¨4Ô+<×+BÒ+BÑ+DÔ+DÀqÒ+HÐ+HØ $Ô 1× >Ò >¸qÀ(Ç+Â+ÈdÔN_ÔNfÑBgÔBgÑ hÔ hˆDÔÐÐð	ið 	iÐ+HÐ+Hr    r…   c                 óP  •— t          ¦   «                              |¦  «         | j        r| j                             ¦   «         dk    rd S |dk    r|n!| j        j        d         t          |¦  «        z
  }| j        j        d         |k    r| j        d d …d |…d d …f         | _        d S d S )Nr   r   )r   r‰   r²   r±   r€   r�   r‡   )r&   r…   Ú	effectiver   s      €r   r‰   zDynamicIndexedLayer.cropE  s´   ø€ Ý‰Œ�Š�ZÑ Ô Ð ØÔ*ð 	¨dÔ.?×.EÒ.EÑ.GÔ.GÈ1Ò.LÐ.LØˆFØ",°¢/ /�J�J°tÔ7HÔ7NÈqÔ7QÕTWÐXbÑTcÔTcÑ7cˆ	ØÔÔ" 1Ô%¨	Ò1Ð1Ø $Ô 1°!°!°!°Z°i°ZÀÀÀÐ2BÔ CˆDÔÐÐð 2Ð1r    rŠ   c                 óÜ   •— t          ¦   «                              |¦  «         | j        r@| j                             ¦   «         dk    r%| j                             |d¬¦  «        | _        d S d S d S )Nr   rv   )r   rŽ   r²   r±   r€   rŒ   )r&   rŠ   r   s     €r   rŽ   z+DynamicIndexedLayer.batch_repeat_interleaveM  sw   ø€ Ý‰Œ×'Ò'¨Ñ0Ô0Ð0ØÔ&ð 	T¨4Ô+<×+BÒ+BÑ+DÔ+DÀqÒ+HÐ+HØ $Ô 1× CÒ CÀGÐQRÐ CÑ SÔ SˆDÔÐÐð	Tð 	TÐ+HÐ+Hr    r�   c                 óÂ   •— t          ¦   «                              |¦  «         | j        r3| j                             ¦   «         dk    r| j        |df         | _        d S d S d S )Nr   .)r   r’   r²   r±   r€   )r&   r�   r   s     €r   r’   z(DynamicIndexedLayer.batch_select_indicesR  sl   ø€ Ý‰Œ×$Ò$ WÑ-Ô-Ð-ØÔ&ð 	@¨4Ô+<×+BÒ+BÑ+DÔ+DÀqÒ+HÐ+HØ $Ô 1°'¸3°,Ô ?ˆDÔÐÐð	@ð 	@Ð+HÐ+Hr    r^   )r+   r_   r`   ra   r'   rf   rg   r¸   rº   rG   rK   rS   ri   rX   rR   r‰   rŽ   r’   rj   rk   s   @r   r¯   r¯     s¿  ø€ € € € € ðð ð2ð 2ð 2ð 2ð 2ð
+¸e¼lð +Ètð +ð +ð +ð +ð
!°´ð !À%Ä,ð !ð !ð !ð !ðOð Oð Oð Oð Oð
Uð Uð Uð Uð Uð
&ð &ð &ð &ð &ð &ð
i eÔ&6ð i¸4ð ið ið ið ið ið ið
D˜sð D tð Dð Dð Dð Dð Dð DðT¨sð T°tð Tð Tð Tð Tð Tð Tð
@¨E¬Lð @¸Tð @ð @ð @ð @ð @ð @ð @ð @ð @ð @r    r¯   c                   óÜ   ‡ — e Zd ZdZdZdZdefˆ fd„Zdej	        dej	        dd	fd
„Z
dej	        dej	        deej	        ej	        f         fd„Zdedeeef         fd„Zdefd„Zdefd„Zˆ xZS )r   aŠ  
    A static cache layer that stores the key and value states as static tensors of shape `[batch_size, num_heads, max_cache_len), head_dim]`.
    It lazily allocates its full backing tensors, and then mutates them in-place. Built for `torch.compile` support.

    Args:
        max_cache_len (`int`):
            Maximum number of tokens that can be stored, used for tensor preallocation.
    TFÚmax_cache_lenc                 ó–   •— t          ¦   «                              ¦   «          || _        t          j        dt
          ¬¦  «        | _        d S r˜   )r   r'   rÆ   rf   rr   rR   rN   ©r&   rÆ   r   r   s      €r   r'   zStaticLayer.__init__e  s>   ø€ Ý‰Œ×ÒÑÔÐØ*ˆÔå!&¤¨aµsÐ!;Ñ!;Ô!;ˆÔÐÐr    r/   r0   r1   Nc                 óú  — |j         |j        c| _         | _        |j        dd…         \  | _        | _        |j        d         | _        |j        d         | _        t          j        | j        | j        | j	        | j        f| j         | j        ¬¦  «        | _
        t          j        | j        | j        | j	        | j        f| j         | j        ¬¦  «        | _        | j                             | j        ¦  «        | _        t          ¦   «         slt          j                             | j
        ¦  «         t          j                             | j        ¦  «         t          j                             | j        ¦  «         d| _        dS )a”  
        Lazy initialization of the keys and values tensors. This allows to get all properties (dtype, device,
        num_heads in case of TP etc...) at runtime directly, which is extremely practical as it avoids moving
        devices, dtypes etc later on for each `update` (which could break the static dynamo addresses as well).

        If this is unwanted, one can call `early_initialization(...)` on the Cache directly, which will call this
        function ahead-of-time (this is required for `torch.export` for example). It is also required whenever the
        prefill itself ends up in a compiled region (with chunked prefill for instance).
        Né   rƒ   rp   T)rq   rJ   r�   Ú
batch_sizeÚ	num_headsÚ
v_head_dimÚ
k_head_dimrf   ÚzerosrÆ   r#   r$   rN   rF   r   Ú_dynamoÚmark_static_addressr%   r3   s      r   r4   zStaticLayer.lazy_initializationk  sG  € ð #-Ô"2°JÔ4EÐˆŒ
�D”KØ*4Ô*:¸2¸A¸2Ô*>Ñ'ˆŒ˜œØ&Ô,¨RÔ0ˆŒØ$Ô*¨2Ô.ˆŒå”KØŒ_˜dœn¨dÔ.@À$Ä/ÐRØ”*Ø”;ð
ñ 
ô 
ˆŒ	õ
 ”kØŒ_˜dœn¨dÔ.@À$Ä/ÐRØ”*Ø”;ð
ñ 
ô 
ˆŒð
 "&Ô!7×!:Ò!:¸4¼;Ñ!GÔ!GˆÔõ (Ñ)Ô)ð 	FÝŒM×-Ò-¨d¬iÑ8Ô8Ð8ÝŒM×-Ò-¨d¬kÑ:Ô:Ð:ÝŒM×-Ò-¨dÔ.DÑEÔEÐEà"ˆÔÐÐr    c                 óÄ  — | j         s|                      ||¦  «         |j        d         }t          j        || j        ¬¦  «        | j        z   }| j                             |¦  «         	 | j         	                    d||¦  «         | j
         	                    d||¦  «         n2# t          $ r% || j        dd…dd…|f<   || j
        dd…dd…|f<   Y nw xY w| j        | j
        fS )rt   ru   ©rJ   rÊ   N)r%   r4   r�   rf   ÚarangerJ   rN   Úadd_r#   Úindex_copy_r$   ÚNotImplementedError)r&   r/   r0   r7   r   r}   Úcache_positions          r   r8   zStaticLayer.update‘  s  € ð Ô"ð 	?Ø×$Ò$ Z°Ñ>Ô>Ð>ð Ô$ RÔ(ˆ	Ýœ i¸¼ÐDÑDÔDÀtÔG]Ñ]ˆàÔ×#Ò# IÑ.Ô.Ð.ð	=ØŒI×!Ò! ! ^°ZÑ@Ô@Ð@ØŒK×#Ò# A ~°|ÑDÔDÐDÐDøÝ"ð 	=ð 	=ð 	=à.8ˆDŒI�a�a�a˜˜˜˜NÐ*Ñ+Ø0<ˆDŒK˜˜˜˜1˜1˜1˜nÐ,Ñ-Ð-Ð-ð	=øøøð
 Œy˜$œ+Ð%Ð%s   Á)8B" Â",CÃCr9   c                 ó   — d}| j         }||fS )r¢   r   ©rÆ   r{   s       r   r;   zStaticLayer.get_mask_sizes³  s   € àˆ	ØÔ&ˆ	Ø˜)Ð#Ð#r    c                 ó"   — | j         r| j        ndS )r   r   )r%   rN   r,   s    r   r=   zStaticLayer.get_seq_length¹  s   € à)-Ô)<ÐCˆtÔ%Ð%À!ÐCr    c                 ó   — | j         S rª   rÚ   r,   s    r   r@   zStaticLayer.get_max_length½  s   € àÔ!Ð!r    )r+   r_   r`   ra   rb   r“   rR   r'   rf   rg   r4   rh   r8   r;   r=   r@   rj   rk   s   @r   r   r   X  s-  ø€ € € € € ðð ð €NØ€Jð< cð <ð <ð <ð <ð <ð <ð$#¨e¬lð $#È%Ì,ð $#Ð[_ð $#ð $#ð $#ð $#ðL &Øœ,ð &Ø6;´lð &à	ˆuŒ|˜Uœ\Ð)Ô	*ð &ð  &ð  &ð  &ðD$¨3ð $°5¸¸c¸´?ð $ð $ð $ð $ðD ð Dð Dð Dð Dð" ð "ð "ð "ð "ð "ð "ð "ð "r    r   c                   ó²   ‡ — e Zd ZdZdZdedefˆ fd„Zdej        dej        de	ej        ej        f         fd	„Z
d
ede	eef         fd„Zdefd„Zˆ fd„Zˆ xZS )ÚStaticSlidingWindowLayeraî  
    A static cache layer that stores the key and value states as static tensors of shape
    `[batch_size, num_heads, min(max_cache_len, sliding_window), head_dim]`. It lazily allocates its full backing
    tensors, and then mutates them in-place. Built for `torch.compile` support.

    Args:
        max_cache_len (`int`):
            Maximum number of tokens that can be stored, used for tensor preallocation.
        sliding_window (`int`):
            The size of the sliding window.
    TrÆ   r–   c                 óz   •— t          ||¦  «        }t          ¦   «                              |¬¦  «         d| _        d S )NrÚ   r   )Úminr   r'   Úcumulative_length_int)r&   rÆ   r–   r   Úeffective_max_cache_lenr   s        €r   r'   z!StaticSlidingWindowLayer.__init__Ñ  s=   ø€ Ý"% n°mÑ"DÔ"DÐÝ‰Œ×ÒÐ'>ÐÑ?Ô?Ð?à%&ˆÔ"Ð"Ð"r    r/   r0   r1   c                 ó  — | j         s|                      ||¦  «         |j        d         }| j        }|| j        k    }| xj        |z  c_        |�r%|j        d         dk    r´| j                             dd¬¦  «        }| j                             dd¬¦  «        }	t          j	        dgt          | j        ¬¦  «        }
||dd…dd…|
f<   ||	dd…dd…|
f<   | j                             |¦  «         | j                             |	¦  «         | j        | j        fS t          j        | j        dd…dd…dd…dd…f         |fd¬¦  «        }t          j        | j        dd…dd…dd…dd…f         |fd¬¦  «        }�n0||z   | j        k    rk|dk    r|}|}�nt          j        | j        dd…dd…d|…dd…f         |fd¬¦  «        }t          j        | j        dd…dd…d|…dd…f         |fd¬¦  «        }n·t          j        || j        ¬	¦  «        | j        z   }	 | j                             d
||¦  «         | j                             d
||¦  «         n2# t"          $ r% || j        dd…dd…|f<   || j        dd…dd…|f<   Y nw xY w| j                             |¦  «         | j        | j        fS | j                             |dd…dd…| j         d…dd…f         ¦  «         | j                             |dd…dd…| j         d…dd…f         ¦  «         ||fS )rt   ru   r   rƒ   )Údimsrp   Nrv   r   rÓ   rÊ   )r%   r4   r�   rá   rÆ   r#   Úrollr$   rf   rr   rR   rJ   Úcopy_rx   rÔ   rN   rÖ   r×   rÕ   )r&   r/   r0   r7   r   r}   Úcurrent_lengthr¤   Únew_keysÚ
new_valuesÚindexrž   rŸ   rØ   s                 r   r8   zStaticSlidingWindowLayer.update×  sº  € ð Ô"ð 	?Ø×$Ò$ Z°Ñ>Ô>Ð>àÔ$ RÔ(ˆ	ØÔ3ˆØ  DÔ$6Ò6ˆàÐ"Ô" iÑ/Ð"Ô"àñ 0	*ð Ô Ô# qÒ(Ð(àœ9Ÿ>š>¨"°2˜>Ñ6Ô6�Ø!œ[×-Ò-¨b°rÐ-Ñ:Ô:�
õ œ b Tµ¸T¼[ÐIÑIÔI�Ø(2�˜˜˜˜A˜A˜A˜u˜Ñ%Ø*6�
˜1˜1˜1˜a˜a˜a ˜;Ñ'ð ”	—’ Ñ)Ô)Ð)Ø”×!Ò! *Ñ-Ô-Ð-ð ”y $¤+Ð-Ð-õ #(¤)¨T¬Y°q°q°q¸!¸!¸!¸Q¸R¸RÀÀÀ°{Ô-CÀZÐ,PÐVXÐ"YÑ"YÔ"Y�Ý$)¤I¨t¬{¸1¸1¸1¸a¸a¸aÀÀÀÀQÀQÀQ¸;Ô/GÈÐ.VÐ\^Ð$_Ñ$_Ô$_Ð!Ñ!à˜iÑ'¨$Ô*<Ò<Ð<à Ò"Ð"Ø",�Ø$0Ð!Ñ!å"'¤)¨T¬Y°q°q°q¸!¸!¸!¸_¸n¸_ÈaÈaÈaÐ7OÔ-PÐR\Ð,]ÐceÐ"fÑ"fÔ"f�Ý$)¤I¨t¬{¸1¸1¸1¸a¸a¸aÀÀ.ÀÐRSÐRSÐRSÐ;SÔ/TÐVbÐ.cÐikÐ$lÑ$lÔ$lÐ!Ð!õ #œ\¨)¸D¼KÐHÑHÔHÈ4ÔKaÑaˆNðAØ”	×%Ò% a¨¸ÑDÔDÐDØ”×'Ò'¨¨>¸<ÑHÔHÐHÐHøÝ&ð Að Að AØ2<�”	˜!˜!˜!˜Q˜Q˜Q Ð.Ñ/Ø4@�”˜A˜A˜A˜q˜q˜q .Ð0Ñ1Ð1Ð1ðAøøøð Ô"×'Ò'¨	Ñ2Ô2Ð2ð ”9˜dœkÐ)Ð)ð 	Œ	�Š˜¨¨¨¨1¨1¨1¨tÔ/AÐ.AÐ.CÐ.CÀQÀQÀQÐ(FÔGÑHÔHÐHØŒ×ÒÐ+¨A¨A¨A¨q¨q¨q°4Ô3EÐ2EÐ2GÐ2GÈÈÈÐ,JÔKÑLÔLÐLàÐ 1Ð1Ð1s   È8I É,I:É9I:r9   c                 óº   — | j         }| j        | j         k    }t          | j        |z
  dz   d¦  «        }|r	||z   dz
  }n| j        |z   |k    r| j        |z   }n|}||fS r¡   )rÆ   rá   r£   )r&   r9   r–   r¤   r|   r}   s         r   r;   z'StaticSlidingWindowLayer.get_mask_sizes&  sƒ   € àÔ+ˆØÔ,°Ô0BÒBˆå˜Ô2°^ÑCÀaÑGÈÑKÔKˆ	àð 	'Ø&¨Ñ5¸Ñ9ˆIˆIàÔ'¨,Ñ6¸ÒGÐGØÔ2°\ÑAˆIˆIð 'ˆIà˜)Ð#Ð#r    c                 ó   — | j         S r¦   )rá   r,   s    r   r=   z'StaticSlidingWindowLayer.get_seq_length8  s   € àÔ)Ð)r    c                 óV   •— t          ¦   «                              ¦   «          d| _        d S rÀ   )r   rS   rá   r¼   s    €r   rS   zStaticSlidingWindowLayer.reset<  s"   ø€ Ý‰Œ�Š‰ŒˆØ%&ˆÔ"Ð"Ð"r    )r+   r_   r`   ra   r“   rR   r'   rf   rg   rh   r8   r;   r=   rS   rj   rk   s   @r   rÞ   rÞ   Â  s   ø€ € € € € ð
ð 
ð €Jð' cð '¸3ð 'ð 'ð 'ð 'ð 'ð 'ðM2Øœ,ðM2Ø6;´lðM2à	ˆuŒ|˜Uœ\Ð)Ô	*ðM2ð M2ð M2ð M2ð^$¨3ð $°5¸¸c¸´?ð $ð $ð $ð $ð$* ð *ð *ð *ð *ð'ð 'ð 'ð 'ð 'ð 'ð 'ð 'ð 'r    rÞ   c                   ór   ‡ — e Zd ZdZdefˆ fd„Zdej        ddfd„Zdej        dej        fd„Z	d
ˆ fd	„Z
ˆ xZS )ÚStaticIndexedLayera  
    A `StaticLayer` with an additional statically-allocated indexer key cache for Dynamic Sparse
    Attention (DSA) models (e.g. GLM MoE DSA, DeepSeek V32). This is the static, `torch.compile`-friendly
    counterpart of `DynamicIndexedLayer`: the indexer key buffer is preallocated once and mutated in-place.

    The main K/V cache is inherited from `StaticLayer` (`[batch_size, num_heads, max_cache_len, head_dim]`).
    The indexer key cache stores a tensor of shape `[batch_size, max_cache_len, index_head_dim]` (3D, single-head).
    rÆ   c                 ó¨   •— t          ¦   «                              |¬¦  «         d | _        d| _        t	          j        dt          ¬¦  «        | _        d S )NrÚ   Fr   r™   )r   r'   r±   r²   rf   rr   rR   Úindexer_cumulative_lengthrÈ   s      €r   r'   zStaticIndexedLayer.__init__K  sM   ø€ Ý‰Œ×Ò }ÐÑ5Ô5Ð5Ø15ˆÔØ,1ˆÔ#õ */¬°a½sÐ)CÑ)CÔ)CˆÔ&Ð&Ð&r    r³   r1   Nc                 ó¬  — |j         |j        c| _        | _        |j        \  }}}t          j        || j        |f| j        | j        ¬¦  «        | _        | j	         
                    | j        ¦  «        | _	        t          ¦   «         sHt
          j                             | j        ¦  «         t
          j                             | j	        ¦  «         d| _        d S ro   )rq   rJ   rµ   r¶   r�   rf   rÏ   rÆ   r±   rñ   rF   r   rÐ   rÑ   r²   )r&   r³   rË   Ú_Úindex_head_dims        r   r¸   z.StaticIndexedLayer.lazy_initialization_indexerS  sÉ   € Ø2DÔ2JÐL^ÔLeÐ/ˆÔ˜DÔ/Ø(:Ô(@Ñ%ˆ
�A�~Ý!œKØ˜Ô+¨^Ð<ØÔ$ØÔ&ð
ñ 
ô 
ˆÔð
 *.Ô)G×)JÒ)JÈ4ÔK^Ñ)_Ô)_ˆÔ&å'Ñ)Ô)ð 	NÝŒM×-Ò-¨dÔ.?Ñ@Ô@Ð@ÝŒM×-Ò-¨dÔ.LÑMÔMÐMØ&*ˆÔ#Ð#Ð#r    c                 óT  — | j         s|                      |¦  «         |j        d         }t          j        || j        ¬¦  «        | j        z   }| j                             |¦  «         	 | j         	                    d||¦  «         n# t          $ r || j        dd…|f<   Y nw xY w| j        S )a.  
        Update the indexer key cache in-place at the current positions, and return the full static buffer.

        Args:
            indexer_key_states (`torch.Tensor`): New indexer keys, shape `[batch_size, seq_len, index_head_dim]`.

        Returns:
            `torch.Tensor`: The full static indexer key cache, shape `[batch_size, max_cache_len, index_head_dim]`.
                Unfilled positions are masked out downstream by the indexer's attention mask, exactly as the
                main `StaticLayer` returns its full preallocated K/V.
        r   rÓ   N)r²   r¸   r�   rf   rÔ   r¶   rñ   rÕ   r±   rÖ   r×   )r&   r³   Úseq_lenrØ   s       r   rº   z!StaticIndexedLayer.update_indexerb  sÓ   € ð Ô*ð 	AØ×,Ò,Ð-?Ñ@Ô@Ð@à$Ô*¨1Ô-ˆÝœ g°dÔ6IÐJÑJÔJÈTÔMkÑkˆàÔ&×+Ò+¨GÑ4Ô4Ð4ð	FØÔ×)Ò)¨!¨^Ð=OÑPÔPÐPÐPøÝ"ð 	Fð 	Fð 	Fà3EˆDÔ˜a˜a˜a Ð/Ñ0Ð0Ð0ð	Føøøð Ô Ð s   Á(B ÂB ÂB c                 ó¾   •— t          ¦   «                              ¦   «          | j        r4| j                             ¦   «          | j                             ¦   «          d S d S r)   )r   rS   r²   r±   rO   rñ   r¼   s    €r   rS   zStaticIndexedLayer.reset}  sY   ø€ Ý‰Œ�Š‰ŒˆØÔ&ð 	3ØÔ×#Ò#Ñ%Ô%Ð%ØÔ*×0Ò0Ñ2Ô2Ð2Ð2Ð2ð	3ð 	3r    r^   )r+   r_   r`   ra   rR   r'   rf   rg   r¸   rº   rS   rj   rk   s   @r   rï   rï   A  s½   ø€ € € € € ðð ðD cð Dð Dð Dð Dð Dð Dð+¸e¼lð +Ètð +ð +ð +ð +ð!°´ð !À%Ä,ð !ð !ð !ð !ð63ð 3ð 3ð 3ð 3ð 3ð 3ð 3ð 3ð 3r    rï   c                   óÈ   ‡ — e Zd ZdZ	 	 	 	 	 ddededed	ed
ef
ˆ fd„Zdej        dej        deej        ej        f         fd„Z	e
d„ ¦   «         Ze
d„ ¦   «         Zdefd„Zˆ xZS )ÚQuantizedLayera  
    A quantized layer similar to what is described in the [KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache paper](https://huggingface.co/papers/2402.02750).
    It allows the model to generate longer sequence length without allocating too much memory for the key and value caches by
    applying quantization.

    The cache has two types of storage, one for original precision and one for the quantized cache. A `residual length`
    is set as a maximum capacity for the original precision cache. When the length goes beyond maximum capacity, the original
    precision cache is discarded and moved into the quantized cache. The quantization is done per-channel with a set `q_group_size`
    for both Keys and Values, in contrast to what was described in the paper.
    é   r   é@   é€   ÚnbitsÚaxis_keyÚ
axis_valueÚq_group_sizeÚresidual_lengthc                 óœ   •— t          ¦   «                              ¦   «          || _        || _        || _        || _        || _        d| _        d S rÀ   )r   r'   rý   rþ   rÿ   r   r  rN   ©r&   rý   rþ   rÿ   r   r  r   s         €r   r'   zQuantizedLayer.__init__�  sN   ø€ õ 	‰Œ×ÒÑÔÐØˆŒ
Ø ˆŒØ$ˆŒØ(ˆÔØ.ˆÔØ!"ˆÔÐÐr    r/   r0   r1   c                 ó’  — | xj         |j        d         z  c_         | j        s€|                      ||¦  «         |                      |                     ¦   «         | j        ¬¦  «        | _        |                      |                     ¦   «         | j        ¬¦  «        | _	        ||fS |  
                    | j        ¦  «        }|  
                    | j	        ¦  «        }t          j        || j        |gd¬¦  «        }t          j        || j        |gd¬¦  «        }| j                             ¦   «         dk    rÑ| j        j        d         dz   | j        k    r³|                      |                     ¦   «         | j        ¬¦  «        | _        |                      |                     ¦   «         | j        ¬¦  «        | _	        t          j        g |j        |j        ¬¦  «        | _        t          j        g |j        |j        ¬¦  «        | _        nDt          j        | j        |gd¬¦  «        | _        t          j        | j        |gd¬¦  «        | _        ||fS )rt   ru   )Úaxisrv   rú   r   rp   )rN   r�   r%   r4   Ú	_quantizeÚ
contiguousrþ   Ú_quantized_keysrÿ   Ú_quantized_valuesÚ_dequantizerf   rx   r#   r$   rw   r  rr   rq   rJ   )	r&   r/   r0   r7   r   Údequant_keysÚdequant_valuesÚkeys_to_returnÚvalues_to_returns	            r   r8   zQuantizedLayer.update   s  € ð 	ÐÔ *Ô"2°2Ô"6Ñ6ÐÔð Ô"ð 	,Ø×$Ò$ Z°Ñ>Ô>Ð>Ø#'§>¢>°*×2GÒ2GÑ2IÔ2IÐPTÔP] >Ñ#^Ô#^ˆDÔ Ø%)§^¢^°L×4KÒ4KÑ4MÔ4MÐTXÔTc ^Ñ%dÔ%dˆDÔ"Ø˜|Ð+Ð+à×'Ò'¨Ô(<Ñ=Ô=ˆØ×)Ò)¨$Ô*@ÑAÔAˆÝœ L°$´)¸ZÐ#HÈbÐQÑQÔQˆÝ œ9 n°d´kÀ<Ð%PÐVXÐYÑYÔYÐØŒ9�=Š=‰?Œ?˜aÒÐ D¤I¤O°BÔ$7¸!Ñ$;¸tÔ?SÒ$SÐ$SØ#'§>¢>°.×2KÒ2KÑ2MÔ2MÐTXÔTa >Ñ#bÔ#bˆDÔ Ø%)§^¢^Ð4D×4OÒ4OÑ4QÔ4QÐX\ÔXg ^Ñ%hÔ%hˆDÔ"Ýœ R¨zÔ/?È
ÔHYÐZÑZÔZˆDŒIÝœ, r°Ô1AÈ*ÔJ[Ð\Ñ\Ô\ˆDŒKˆKåœ	 4¤9¨jÐ"9¸rÐBÑBÔBˆDŒIÝœ) T¤[°,Ð$?ÀRÐHÑHÔHˆDŒKàÐ/Ð/Ð/r    c                 ó   — d S r)   r   )r&   rr   r  s      r   r  zQuantizedLayer._quantizeÅ  s   € Ø'* sr    c                 ó   — d S r)   r   )r&   Úq_tensors     r   r
  zQuantizedLayer._dequantizeÈ  r>   r    c                 ó   — | j         S r¦   r§   r,   s    r   r=   zQuantizedLayer.get_seq_lengthË  r¨   r    ©rú   r   r   rû   rü   )r+   r_   r`   ra   rR   r'   rf   rg   rh   r8   r   r  r
  r=   rj   rk   s   @r   rù   rù   „  s  ø€ € € € € ð	ð 	ð ØØØØ"ð#ð #àð#ð ð#ð ð	#ð
 ð#ð ð#ð #ð #ð #ð #ð #ð #0Øœ,ð#0Ø6;´lð#0à	ˆuŒ|˜Uœ\Ð)Ô	*ð#0ð #0ð #0ð #0ðJ Ø*Ð*ñ „^Ø*àØ(Ð(ñ „^Ø(ð& ð &ð &ð &ð &ð &ð &ð &ð &r    rù   c                   óL   ‡ — e Zd Z	 	 	 	 	 ddedededed	ef
ˆ fd
„Zd„ Zd„ Zˆ xZS )ÚQuantoQuantizedLayerrú   r   rû   rü   rý   rþ   rÿ   r   r  c                 óê  •— t          ¦   «                              |||||¬¦  «         t          ¦   «         st          d¦  «        ‚t	          dd¬¦  «        rddlm}m}m} nt          d¦  «        ‚| j	        d	vrt          d
| j	        › �¦  «        ‚| j        dvrt          d| j        › �¦  «        ‚| j        dvrt          d| j        › �¦  «        ‚| j	        dk    r|n|| _         |¦   «         | _        d S )N©rý   rþ   rÿ   r   r  zžYou need to install optimum-quanto in order to use KV cache quantization with optimum-quanto backend. Please install it via  with `pip install optimum-quanto`z0.2.5Tr   r   )ÚMaxOptimizerÚqint2Úqint4ziYou need optimum-quanto package version to be greater or equal than 0.2.5 to use `QuantoQuantizedLayer`. )rÊ   rú   zA`nbits` for `quanto` backend has to be one of [`2`, `4`] but got )r   rƒ   zE`axis_key` for `quanto` backend has to be one of [`0`, `-1`] but got zG`axis_value` for `quanto` backend has to be one of [`0`, `-1`] but got rú   )r   r'   r	   ÚImportErrorr
   Úoptimum.quantor  r  r  rý   r­   rþ   rÿ   ÚqtypeÚ	optimizer)
r&   rý   rþ   rÿ   r   r  r  r  r  r   s
            €r   r'   zQuantoQuantizedLayer.__init__Ñ  sR  ø€ õ 	‰Œ×ÒØØØ!Ø%Ø+ð 	ñ 	
ô 	
ð 	
õ +Ñ,Ô,ð 
	ÝðTñô ð õ ˜w°4Ð8Ñ8Ô8ð 	ØAÐAÐAÐAÐAÐAÐAÐAÐAÐAÐAåØ{ñô ð ð Œ:˜VÐ#Ð#ÝÐmÐaeÔakÐmÐmÑnÔnÐnàŒ= Ð'Ð'ÝÐtÐeiÔerÐtÐtÑuÔuÐuàŒ? 'Ð)Ð)ÝØkÐZ^ÔZiÐkÐkñô ð ð #œj¨Ašo˜o�U�U°5ˆŒ
Ø%˜™œˆŒˆˆr    c                 ó�   — ddl m} |                      || j        || j        ¦  «        \  }} ||| j        |||| j        ¦  «        }|S )Nr   )Úquantize_weight)r  r   r  r  r   )r&   rr   r  r   ÚscaleÚ	zeropointÚqtensors          r   r  zQuantoQuantizedLayer._quantizeü  sX   € Ø2Ð2Ð2Ð2Ð2Ð2àŸ>š>¨&°$´*¸dÀDÔDUÑVÔVÑˆˆyØ!�/ &¨$¬*°d¸EÀ9ÈdÔN_Ñ`Ô`ˆØˆr    c                 ó*   — |                      ¦   «         S r)   )Ú
dequantize)r&   r#  s     r   r
  z QuantoQuantizedLayer._dequantize  s   € Ø×!Ò!Ñ#Ô#Ð#r    r  ©r+   r_   r`   rR   r'   r  r
  rj   rk   s   @r   r  r  Ð  s¢   ø€ € € € € ð ØØØØ"ð)(ð )(àð)(ð ð)(ð ð	)(ð
 ð)(ð ð)(ð )(ð )(ð )(ð )(ð )(ðVð ð ð$ð $ð $ð $ð $ð $ð $r    r  c                   óL   ‡ — e Zd Z	 	 	 	 	 ddedededed	ef
ˆ fd
„Zd„ Zd„ Zˆ xZS )ÚHQQQuantizedLayerrú   r   rû   rü   rý   rþ   rÿ   r   r  c                 óf  •— t          ¦   «                              |||||¬¦  «         t          ¦   «         st          d¦  «        ‚| j        dvrt          d| j        › �¦  «        ‚| j        dvrt          d| j        › �¦  «        ‚| j        dvrt          d| j        › �¦  «        ‚t          | _	        d S )Nr  zYou need to install `HQQ` in order to use KV cache quantization with HQQ backend. Please install it via  with `pip install hqq`)r   rÊ   é   rú   é   zM`nbits` for `HQQ` backend has to be one of [`1`, `2`, `3`, `4`, `8`] but got )r   r   zA`axis_key` for `HQQ` backend has to be one of [`0`, `1`] but got zC`axis_value` for `HQQ` backend has to be one of [`0`, `1`] but got )
r   r'   r   r  rý   r­   rþ   rÿ   ÚHQQQuantizerÚ	quantizerr  s         €r   r'   zHQQQuantizedLayer.__init__  sê   ø€ õ 	‰Œ×ÒØØØ!Ø%Ø+ð 	ñ 	
ô 	
ð 	
õ  Ñ!Ô!ð 	Ýð@ñô ð ð
 Œ:˜_Ð,Ð,ÝØlÐ`dÔ`jÐlÐlñô ð ð Œ= Ð&Ð&ÝÐpÐaeÔanÐpÐpÑqÔqÐqàŒ? &Ð(Ð(ÝÐtÐcgÔcrÐtÐtÑuÔuÐuå%ˆŒˆˆr    c                 ó„  — | j                              ||| j        j        | j        j        | j        | j        ¬¦  «        \  }}| j        j        |d<   | j                              ||| j        j        ¬¦  «         |d                              |j        ¦  «        |d<   |d                              |j        ¦  «        |d<   ||fS )N)r  rJ   Úcompute_dtyperý   Ú
group_sizer/  )ÚmetarJ   r!  Úzero)	r-  Úquantizer#   rJ   rq   rý   r   ÚcudarF   )r&   rr   r  r#  r1  s        r   r  zHQQQuantizedLayer._quantize+  s¶   € Øœ×/Ò/ØØØ”9Ô#Øœ)œ/Ø”*ØÔ(ð 0ñ 
ô 
‰ˆ�ð !%¤	¤ˆˆ_ÑØŒ×Ò˜G¨$°t´yÔ7GÐÑHÔHÐHØ˜Wœ×(Ò(¨¬Ñ8Ô8ˆˆW‰Ø˜F”|—’ w¤~Ñ6Ô6ˆˆV‰Ø˜ˆ}Ðr    c                 óF   — |\  }}| j                              ||¦  «        }|S r)   )r-  r%  )r&   r#  Úquant_tensorr1  rr   s        r   r
  zHQQQuantizedLayer._dequantize:  s(   € Ø$Ñˆ�dØ”×*Ò*¨<¸Ñ>Ô>ˆØˆr    r  r&  rk   s   @r   r(  r(    s¢   ø€ € € € € ð ØØØØ"ð!&ð !&àð!&ð ð!&ð ð	!&ð
 ð!&ð ð!&ð !&ð !&ð !&ð !&ð !&ðFð ð ðð ð ð ð ð ð r    r(  c            
       ó:  — e Zd ZdZdZdZddefd„Zd„ Ze		 	 	 dd
e
j        dz  de
j        dz  deddfd„¦   «         Ze	dd
e
j        dede
j        fd„¦   «         Ze	dde
j        dede
j        fd„¦   «         Zd„ Zd„ Zdd„Zde
j        fd„Zd„ Zdefd„Zdefd„ZdS )ÚLinearAttentionCacheLayerMixinzABase, abstract class for a linear attention single layer's cache.TFr   Únumber_of_statesc                 óT  — || _         t                               t          |¦  «        ¦  «        | _        t                               t          |¦  «        ¦  «        | _        t                               t          |¦  «        d¦  «        | _        t                               t          |¦  «        d¦  «        | _        t                               t          |¦  «        d¦  «        | _        t                               t          |¦  «        ¦  «        | _	        d | _
        d | _        d| _        d S r"   )r9  ÚdictÚfromkeysÚrangeÚconv_statesÚrecurrent_statesÚis_conv_states_initializedÚis_recurrent_states_initializedÚhas_previous_stateÚconv_kernel_sizerJ   rq   Úrecord_past©r&   r9  r   s      r   r'   z'LinearAttentionCacheLayerMixin.__init__H  sÜ   € Ø 0ˆÔå;?¿=º=ÍÐO_ÑI`ÔI`Ñ;aÔ;aˆÔÝ@DÇÂÍeÐTdÑNeÔNeÑ@fÔ@fˆÔÝ*.¯-ª-½Ð>NÑ8OÔ8OÐQVÑ*WÔ*WˆÔ'Ý/3¯}ª}½UÐCSÑ=TÔ=TÐV[Ñ/\Ô/\ˆÔ,Ý"&§-¢-µÐ6FÑ0GÔ0GÈÑ"OÔ"OˆÔÝ $§¢­eÐ4DÑ.EÔ.EÑ FÔ FˆÔØˆŒØˆŒ
Ø ˆÔÐÐr    c                 ó   — | j         j        › S r)   r*   r,   s    r   r-   z'LinearAttentionCacheLayerMixin.__repr__U  r.   r    Nr   r>  r?  Ú	state_idxr1   c                 ó   — d S r)   r   )r&   r>  r?  rG  s       r   r4   z2LinearAttentionCacheLayerMixin.lazy_initializationX  s	   € ð ˆsr    c                 ó   — d S r)   r   )r&   r>  rG  s      r   Úupdate_conv_statez0LinearAttentionCacheLayerMixin.update_conv_state`  s   € Ø`cÐ`cr    c                 ó   — d S r)   r   )r&   r?  rG  s      r   Úupdate_recurrent_statez5LinearAttentionCacheLayerMixin.update_recurrent_statec  s   € ØjmÐjmr    c                 ó  — t          | j        ¦  «        D ]p}| j        |         r*| j        |                              dd¬¦  «        | j        |<   | j        |         r*| j        |                              dd¬¦  «        | j        |<   ŒqdS rB   )r=  r9  r@  r>  rF   rA  r?  ©r&   Úis     r   rG   z&LinearAttentionCacheLayerMixin.offloadf  sŸ   € å�tÔ,Ñ-Ô-ð 	að 	aˆAØÔ.¨qÔ1ð WØ&*Ô&6°qÔ&9×&<Ò&<¸UÐQUÐ&<Ñ&VÔ&V�Ô  Ñ#ØÔ3°AÔ6ð aØ+/Ô+@ÀÔ+C×+FÒ+FÀuÐ[_Ð+FÑ+`Ô+`�Ô% aÑ(øð		að 	ar    c                 ó�  — t          | j        ¦  «        D ]°}| j        |         rJ| j        |         j        | j        k    r/| j        |                              | j        d¬¦  «        | j        |<   | j        |         rJ| j        |         j        | j        k    r/| j        |                              | j        d¬¦  «        | j        |<   Œ±dS rI   )r=  r9  r@  r>  rJ   rF   rA  r?  rN  s     r   rK   z'LinearAttentionCacheLayerMixin.prefetchn  sÖ   € å�tÔ,Ñ-Ô-ð 	gð 	gˆAØÔ.¨qÔ1ð ]°dÔ6FÀqÔ6IÔ6PÐTXÔT_Ò6_Ð6_Ø&*Ô&6°qÔ&9×&<Ò&<¸T¼[ÐW[Ð&<Ñ&\Ô&\�Ô  Ñ#ØÔ3°AÔ6ð g¸4Ô;PÐQRÔ;SÔ;ZÐ^bÔ^iÒ;iÐ;iØ+/Ô+@ÀÔ+C×+FÒ+FÀtÄ{ÐaeÐ+FÑ+fÔ+f�Ô% aÑ(øð		gð 	gr    c                 óø   — t          | j        ¦  «        D ]d}| j        |         r| j        |                              ¦   «          | j        |         r| j        |                              ¦   «          d| j        |<   ŒedS )rM   FN)r=  r9  r@  r>  rO   rA  r?  rB  rN  s     r   rS   z$LinearAttentionCacheLayerMixin.resetv  sŠ   € å�tÔ,Ñ-Ô-ð 	/ð 	/ˆAØÔ.¨qÔ1ð ,ØÔ  Ô#×)Ò)Ñ+Ô+Ð+ØÔ3°AÔ6ð 1ØÔ% aÔ(×.Ò.Ñ0Ô0Ð0Ø).ˆDÔ# AÑ&Ð&ð	/ð 	/r    rT   c                 ól  — t          | j        ¦  «        D ]ž}| j        |         rA| j        |                              d|                     | j        ¦  «        ¦  «        | j        |<   | j        |         rA| j        |                              d|                     | j        ¦  «        ¦  «        | j        |<   ŒŸdS )úDReorders the cache for beam search, given the selected beam indices.r   N)	r=  r9  r@  r>  rV   rF   rJ   rA  r?  )r&   rT   rO  s      r   rX   z,LinearAttentionCacheLayerMixin.reorder_cache  s»   € å�tÔ,Ñ-Ô-ð 	nð 	nˆAØÔ.¨qÔ1ð dØ&*Ô&6°qÔ&9×&FÒ&FÀqÈ(Ï+Ê+ÐVZÔVaÑJbÔJbÑ&cÔ&c�Ô  Ñ#àÔ3°AÔ6ð nØ+/Ô+@ÀÔ+C×+PÒ+PÐQRÐT\×T_ÒT_Ð`dÔ`kÑTlÔTlÑ+mÔ+m�Ô% aÑ(øð	nð 	nr    c                 ó   — d| _         dS )a  
        Calling this function will activate past state recording, meaning that a call to `update_conv_states` will
        wait for a call to `crop` before restricting the size of the `conv_states` to `conv_kernel_size`, to be able
        to retrieve previous full states.
        TN)rD  r,   s    r   Úactivate_past_recordingz6LinearAttentionCacheLayerMixin.activate_past_recordingˆ  s   € ð  ˆÔÐÐr    Útokens_to_removec                 ój  — | j         st          d¦  «        ‚|dk    rt          d¦  «        ‚t          | j        ¦  «        D ]r}t	          |¦  «        }|dk    r,| j        |         d| j        |          d …f         | j        |<   ŒC| j        |         d| | j        |         z
  | …f         | j        |<   Œsd S )Nz“`crop` was called, but the current layer does not track past states. Call `activate_past_recording` before `crop` to be able to rollback the cache.r   zkLinear attention layers can only be cropped by passing a negative int, to specify how many tokens to remove.)rD  ÚRuntimeErrorr=  r9  r‡   r>  rC  )r&   rV  rO  s      r   r‰   z#LinearAttentionCacheLayerMixin.crop�  sû   € ØÔð 	Ýð;ñô ð ð ˜aÒÐÝØ}ñô ð õ �tÔ,Ñ-Ô-ð 		ð 		ˆAÝ"Ð#3Ñ4Ô4Ðð   1Ò$Ð$Ø&*Ô&6°qÔ&9¸#ÀÔ@UÐVWÔ@XÐ?XÐ?ZÐ?ZÐ:ZÔ&[�Ô  Ñ#Ð#à&*Ô&6°qÔ&9ØÐ*Ð*¨TÔ-BÀ1Ô-EÑEÐIYÐHYÐYÐYô'�Ô  Ñ#Ð#ð		ð 		r    c                 ó   — dS )Nrƒ   r   r,   s    r   r@   z-LinearAttentionCacheLayerMixin.get_max_length¥  r„   r    ©r   )NNr   ©r   r^   )r+   r_   r`   ra   rb   rc   rR   r'   r-   r   rf   rg   r4   rJ  rL  rG   rK   rS   ri   rX   rU  r‰   r@   r   r    r   r8  r8  @  s¹  € € € € € ØKÐKð €NàÐð!ð !¨ð !ð !ð !ð !ð,ð ,ð ,ð ð ,0Ø04Øð	ð à”\ DÑ(ðð  œ,¨Ñ-ðð ð	ð
 
ðð ð ñ „^ðð ØcÐc¨U¬\ÐcÀcÐcÐRWÔR^ÐcÐcÐcñ „^ØcàØmÐm°u´|ÐmÐPSÐmÐ\aÔ\hÐmÐmÐmñ „^Ømðað að aðgð gð gð/ð /ð /ð /ðn eÔ&6ð nð nð nð nð ð  ð  ð Sð ð ð ð ð* ð ð ð ð ð ð r    r8  c                   óº   — e Zd Z	 	 	 	 ddej        dz  dej        dz  dededz  ddf
d„Z	 ddej        dededz  dej        fd	„Zddej        dedej        fd
„ZdS )ÚLinearAttentionLayerNr   r>  r?  rG  rC  r1   c                 óL  — |�¿| j         €|j        |j         c| _        | _         |€|j        d         n|}|| j        |<   t	          j        g |j        d d…         ¢|‘R |j        |j         ¬¦  «        | j        |<   t          ¦   «         s1| j        s*t          j	         
                    | j        |         ¦  «         d| j        |<   |�`t	          j        |¦  «        | j        |<   t          ¦   «         s*t          j	         
                    | j        |         ¦  «         d| j        |<   d S d S )Nrƒ   rp   T)rJ   rq   r�   rC  rf   rÏ   r>  r   rD  rÐ   rÑ   r@  Ú
zeros_liker?  rA  )r&   r>  r?  rG  rC  s        r   r4   z(LinearAttentionLayer.lazy_initialization«  sF  € ð Ð"ØŒ{Ð"Ø*5Ô*;¸[Ô=OÐ'�”
˜DœKð 9IÐ8P˜{Ô0°Ô4Ð4ÐVfÐØ/?ˆDÔ! )Ñ,å*/¬+Ø;�+Ô# C R CÔ(Ð;Ð*:Ð;Ð;Ø!Ô'Ø"Ô)ð+ñ +ô +ˆDÔ˜YÑ'õ ,Ñ-Ô-ð O°dÔ6Fð OÝ”×1Ò1°$Ô2BÀ9Ô2MÑNÔNÐNØ9=ˆDÔ+¨IÑ6àÐ'å/4Ô/?Ð@PÑ/QÔ/QˆDÔ! )Ñ,å+Ñ-Ô-ð TÝ”×1Ò1°$Ô2GÈ	Ô2RÑSÔSÐSØ>BˆDÔ0°Ñ;Ð;Ð;ð (Ð'r    c                 ó(  — | j         |         s|                      |||¬¦  «         | j        |         st|}d| j        |<   | j        s`|j        d         | j        |         k     rD| j        |         |j        d         z
  }t          j        j         	                    ||dfd¬¦  «        }n#t          j
        | j        |         |gd¬¦  «        }| j        s7| j        |                              |d| j        |          d…f         ¦  «         n
|| j        |<   |S )	a  
        Update the linear attention cache in-place, and return the necessary conv states.

        Args:
            conv_states (`torch.Tensor`): The new conv states to cache.

        Returns:
            `torch.Tensor`: The updated conv states.
        )r>  rG  rC  Trƒ   r   )Úvaluerv   .N)r@  r4   rB  rD  r�   rC  rf   ÚnnÚ
functionalÚpadrx   r>  ræ   )r&   r>  rG  rC  r   Úfull_conv_statesÚpadding_lengths          r   rJ  z&LinearAttentionLayer.update_conv_stateÌ  sI  € ð Ô.¨yÔ9ð 	vØ×$Ò$°È	ÐdtÐ$ÑuÔuÐuð Ô& yÔ1ð 
	]Ø*ÐØ15ˆDÔ# IÑ.àÔ#ð kÐ(8Ô(>¸rÔ(BÀTÔEZÐ[dÔEeÒ(eÐ(eØ!%Ô!6°yÔ!AÐDTÔDZÐ[]ÔD^Ñ!^�Ý#(¤8Ô#6×#:Ò#:Ð;KÈnÐ^_ÐM`ÐhiÐ#:Ñ#jÔ#jÐ øõ  %œy¨$Ô*:¸9Ô*EÀ{Ð)SÐY[Ð\Ñ\Ô\Ðð Ôð 	;àÔ˜YÔ'×-Ò-Ð.>¸sÀTÔEZÐ[dÔEeÐDeÐDgÐDgÐ?gÔ.hÑiÔiÐiÐið +;ˆDÔ˜YÑ'ð  Ðr    c                 ó¤   — | j         |         s|                      ||¬¦  «         | j        |                              |¦  «         | j        |         S )zý
        Update the linear attention cache in-place, and return the necessary ssm states.

        Args:
            smm_states (`torch.Tensor`): The new ssm states to cache.

        Returns:
            `torch.Tensor`: The updated ssm states.
        )r?  rG  )rA  r4   r?  ræ   )r&   r?  rG  r   s       r   rL  z+LinearAttentionLayer.update_recurrent_stateô  s[   € ð Ô3°IÔ>ð 	]Ø×$Ò$Ð6FÐR[Ð$Ñ\Ô\Ð\àÔ˜iÔ(×.Ò.Ð/?Ñ@Ô@Ð@ØÔ$ YÔ/Ð/r    )NNr   N)r   Nr[  )	r+   r_   r`   rf   rg   rR   r4   rJ  rL  r   r    r   r]  r]  ª  s  € € € € € ð ,0Ø04ØØ'+ðCð Cà”\ DÑ(ðCð  œ,¨Ñ-ðCð ð	Cð
  ™*ðCð 
ðCð Cð Cð CðD ]að& ð & Ø œ<ð& Ø47ð& ØORÐUYÉzð& à	Œð& ð & ð & ð & ðP0ð 0°u´|ð 0ÐPSð 0ÐfkÔfrð 0ð 0ð 0ð 0ð 0ð 0r    r]  c                   ób   — e Zd ZdZddefd„Zdd„Zd„ Zd	„ Zdd
„Z	de
j        fd„Zdeddfd„ZdS )Ú$LinearAttentionAndFullAttentionLayerFr   r9  c                 ór   — t                                | ¦  «         t                               | |¬¦  «         d S ©N©r9  )rm   r'   r]  rE  s      r   r'   z-LinearAttentionAndFullAttentionLayer.__init__	  s6   € Ý×Ò˜dÑ#Ô#Ð#Ý×%Ò% dÐ=MÐ%ÑNÔNÐNÐNÐNr    r1   Nc                 óê   — t          |¦  «        dk    r%t          |¦  «        dk    rt          j        | g|¢R Ž  t          |¦  «        dk    r%t          |¦  «        dv rt          j        | fi |¤Ž d S d S d S ©NrÊ   r   )r   rÊ   r*  )Úlenrm   r4   r]  ©r&   r7   r   s      r   r4   z8LinearAttentionAndFullAttentionLayer.lazy_initialization  s‡   € åˆt‰9Œ9˜Š>ˆ>�c &™kœk¨QÒ.Ð.ÝÔ,¨TÐ9°DÐ9Ð9Ð9Ð9õ ˆt‰9Œ9˜Š>ˆ>�c &™kœk¨YÐ6Ð6Ý Ô4°TÐDÐD¸VÐDÐDÐDÐDÐDð ˆ>Ð6Ð6r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )rm   rG   r]  r,   s    r   rG   z,LinearAttentionAndFullAttentionLayer.offload  s0   € Ý×Ò˜TÑ"Ô"Ð"Ý×$Ò$ TÑ*Ô*Ð*Ð*Ð*r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )rm   rK   r]  r,   s    r   rK   z-LinearAttentionAndFullAttentionLayer.prefetch  s0   € Ý×Ò˜dÑ#Ô#Ð#Ý×%Ò% dÑ+Ô+Ð+Ð+Ð+r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )r]  rS   rm   r,   s    r   rS   z*LinearAttentionAndFullAttentionLayer.reset  s0   € Ý×"Ò" 4Ñ(Ô(Ð(Ý×Ò˜4Ñ Ô Ð Ð Ð r    rT   c                 ór   — t                                | |¦  «         t                               | |¦  «         dS ©rS  N)r]  rX   rm   rW   s     r   rX   z2LinearAttentionAndFullAttentionLayer.reorder_cache"  s4   € å×*Ò*¨4°Ñ:Ô:Ð:Ý×"Ò" 4¨Ñ2Ô2Ð2Ð2Ð2r    r…   c                 ór   — t                                | |¦  «         t                               | |¦  «         d S r)   )r]  r‰   rm   rˆ   s     r   r‰   z)LinearAttentionAndFullAttentionLayer.crop'  s4   € Ý×!Ò! $¨
Ñ3Ô3Ð3Ý×Ò˜$ 
Ñ+Ô+Ð+Ð+Ð+r    rZ  r^   )r+   r_   r`   rb   rR   r'   r4   rG   rK   rS   rf   ri   rX   r‰   r   r    r   ri  ri    sÉ   € € € € € à€NðOð O¨ð Oð Oð Oð OðEð Eð Eð Eð+ð +ð +ð,ð ,ð ,ð!ð !ð !ð !ð3 eÔ&6ð 3ð 3ð 3ð 3ð
,˜sð , tð ,ð ,ð ,ð ,ð ,ð ,r    ri  c                   óZ   — e Zd ZdZddedefd„Zdd„Zdd	„Zd
ej	        fd„Z
deddfd„ZdS )Ú-LinearAttentionAndSlidingWindowAttentionLayerFr   r–   r9  c                 óv   — t                                | |¬¦  «         t                               | |¬¦  «         d S )Nr«   rl  )r•   r'   r]  )r&   r–   r9  r   s       r   r'   z6LinearAttentionAndSlidingWindowAttentionLayer.__init__0  s;   € Ý!×*Ò*¨4ÀÐ*ÑOÔOÐOÝ×%Ò% dÐ=MÐ%ÑNÔNÐNÐNÐNr    r1   Nc                 óê   — t          |¦  «        dk    r%t          |¦  «        dk    rt          j        | g|¢R Ž  t          |¦  «        dk    r%t          |¦  «        dv rt          j        | fi |¤Ž d S d S d S rn  )ro  r•   r4   r]  rp  s      r   r4   zALinearAttentionAndSlidingWindowAttentionLayer.lazy_initialization4  s‡   € åˆt‰9Œ9˜Š>ˆ>�c &™kœk¨QÒ.Ð.Ý%Ô9¸$ÐFÀÐFÐFÐFÐFõ ˆt‰9Œ9˜Š>ˆ>�c &™kœk¨YÐ6Ð6Ý Ô4°TÐDÐD¸VÐDÐDÐDÐDÐDð ˆ>Ð6Ð6r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )r]  rS   r•   r,   s    r   rS   z3LinearAttentionAndSlidingWindowAttentionLayer.reset=  s0   € Ý×"Ò" 4Ñ(Ô(Ð(Ý!×'Ò'¨Ñ-Ô-Ð-Ð-Ð-r    rT   c                 ór   — t                                | |¦  «         t                               | |¦  «         dS ru  )r]  rX   r•   rW   s     r   rX   z;LinearAttentionAndSlidingWindowAttentionLayer.reorder_cacheA  s4   € å×*Ò*¨4°Ñ:Ô:Ð:Ý!×/Ò/°°hÑ?Ô?Ð?Ð?Ð?r    r…   c                 ór   — t                                | |¦  «         t                               | |¦  «         d S r)   )r]  r‰   r•   rˆ   s     r   r‰   z2LinearAttentionAndSlidingWindowAttentionLayer.cropF  s4   € Ý×!Ò! $¨
Ñ3Ô3Ð3Ý!×&Ò& t¨ZÑ8Ô8Ð8Ð8Ð8r    rZ  r^   )r+   r_   r`   rb   rR   r'   r4   rS   rf   ri   rX   r‰   r   r    r   rx  rx  ,  s¸   € € € € € à€NðOð O sð O¸cð Oð Oð Oð OðEð Eð Eð Eð.ð .ð .ð .ð@ eÔ&6ð @ð @ð @ð @ð
9˜sð 9 tð 9ð 9ð 9ð 9ð 9ð 9r    rx  c                   óR   — e Zd Zddedefd„Zdd„Zd„ Zd	„ Zdd
„Zde	j
        fd„ZdS )Ú*LinearAttentionAndStaticFullAttentionLayerr   rÆ   r9  c                 ót   — t                                | |¦  «         t                               | |¬¦  «         d S rk  )r   r'   r]  )r&   rÆ   r9  r   s       r   r'   z3LinearAttentionAndStaticFullAttentionLayer.__init__L  s8   € Ý×Ò˜T =Ñ1Ô1Ð1Ý×%Ò% dÐ=MÐ%ÑNÔNÐNÐNÐNr    r1   Nc                 óê   — t          |¦  «        dk    r%t          |¦  «        dk    rt          j        | g|¢R Ž  t          |¦  «        dk    r%t          |¦  «        dv rt          j        | fi |¤Ž d S d S d S rn  )ro  r   r4   r]  rp  s      r   r4   z>LinearAttentionAndStaticFullAttentionLayer.lazy_initializationP  s‡   € åˆt‰9Œ9˜Š>ˆ>�c &™kœk¨QÒ.Ð.ÝÔ+¨DÐ8°4Ð8Ð8Ð8Ð8õ ˆt‰9Œ9˜Š>ˆ>�c &™kœk¨YÐ6Ð6Ý Ô4°TÐDÐD¸VÐDÐDÐDÐDÐDð ˆ>Ð6Ð6r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )r   rG   r]  r,   s    r   rG   z2LinearAttentionAndStaticFullAttentionLayer.offloadY  s0   € Ý×Ò˜DÑ!Ô!Ð!Ý×$Ò$ TÑ*Ô*Ð*Ð*Ð*r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )r   rK   r]  r,   s    r   rK   z3LinearAttentionAndStaticFullAttentionLayer.prefetch]  s0   € Ý×Ò˜TÑ"Ô"Ð"Ý×%Ò% dÑ+Ô+Ð+Ð+Ð+r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )r]  rS   r   r,   s    r   rS   z0LinearAttentionAndStaticFullAttentionLayer.reseta  s0   € Ý×"Ò" 4Ñ(Ô(Ð(Ý×Ò˜$ÑÔÐÐÐr    rT   c                 ór   — t                                | |¦  «         t                               | |¦  «         dS ru  )r]  rX   r   rW   s     r   rX   z8LinearAttentionAndStaticFullAttentionLayer.reorder_cachee  s4   € å×*Ò*¨4°Ñ:Ô:Ð:Ý×!Ò! $¨Ñ1Ô1Ð1Ð1Ð1r    rZ  r^   )r+   r_   r`   rR   r'   r4   rG   rK   rS   rf   ri   rX   r   r    r   r  r  K  sª   € € € € € ðOð O cð O¸Sð Oð Oð Oð OðEð Eð Eð Eð+ð +ð +ð,ð ,ð ,ð ð  ð  ð  ð2 eÔ&6ð 2ð 2ð 2ð 2ð 2ð 2r    r  c                   óJ   — e Zd Zddededefd„Zdd„Zdd	„Zd
ej        fd„Z	dS )Ú3LinearAttentionAndStaticSlidingWindowAttentionLayerr   rÆ   r–   r9  c                 óx   — t                                | ||¬¦  «         t                               | |¬¦  «         d S )N)rÆ   r–   rl  )rÞ   r'   r]  )r&   rÆ   r–   r9  r   s        r   r'   z<LinearAttentionAndStaticSlidingWindowAttentionLayer.__init__l  s>   € Ý ×)Ò)¨$¸mÐ\jÐ)ÑkÔkÐkÝ×%Ò% dÐ=MÐ%ÑNÔNÐNÐNÐNr    r1   Nc                 óê   — t          |¦  «        dk    r%t          |¦  «        dk    rt          j        | g|¢R Ž  t          |¦  «        dk    r%t          |¦  «        dv rt          j        | fi |¤Ž d S d S d S rn  )ro  rÞ   r4   r]  rp  s      r   r4   zGLinearAttentionAndStaticSlidingWindowAttentionLayer.lazy_initializationp  s‡   € åˆt‰9Œ9˜Š>ˆ>�c &™kœk¨QÒ.Ð.Ý$Ô8¸ÐEÀÐEÐEÐEÐEõ ˆt‰9Œ9˜Š>ˆ>�c &™kœk¨YÐ6Ð6Ý Ô4°TÐDÐD¸VÐDÐDÐDÐDÐDð ˆ>Ð6Ð6r    c                 ón   — t                                | ¦  «         t                               | ¦  «         d S r)   )r]  rS   rÞ   r,   s    r   rS   z9LinearAttentionAndStaticSlidingWindowAttentionLayer.resety  s0   € Ý×"Ò" 4Ñ(Ô(Ð(Ý ×&Ò& tÑ,Ô,Ð,Ð,Ð,r    rT   c                 ór   — t                                | |¦  «         t                               | |¦  «         dS ru  )r]  rX   rÞ   rW   s     r   rX   zALinearAttentionAndStaticSlidingWindowAttentionLayer.reorder_cache}  s4   € å×*Ò*¨4°Ñ:Ô:Ð:Ý ×.Ò.¨t°XÑ>Ô>Ð>Ð>Ð>r    rZ  r^   )
r+   r_   r`   rR   r'   r4   rS   rf   ri   rX   r   r    r   r‡  r‡  k  s•   € € € € € ðOð O cð O¸3ð OÐRUð Oð Oð Oð OðEð Eð Eð Eð-ð -ð -ð -ð? eÔ&6ð ?ð ?ð ?ð ?ð ?ð ?r    r‡  )	Úfull_attentionÚsliding_attentionÚchunked_attentionÚconvÚmoeÚlinear_attentionÚhybridÚhybrid_slidingÚdeepseek_sparse_attentionc            
       ó€  — e Zd ZdZ	 	 	 	 d:deeez           dz  deeez           dz  dedefd	„Z	d
„ Z
d„ Zd;dedefd„Zd;dedefd„Zdej        dej        dedeej        ej        f         fd„Z	 d<dej        dededej        fd„Z	 d<dej        dededej        fd„Zdej        dedej        fd„Zdedeee         z  deee         z  dej        d ej        f
d!„Zd<dedefd"„Zd=dedz  defd#„Zd>dedz  dedz  defd$„Zd%ededeeef         fd&„Zd<dedefd'„Zd(„ Zd)ej        fd*„Z d+efd,„Z!d-efd.„Z"d/ej        fd0„Z#d1„ Z$e%defd2„¦   «         Z&e%defd3„¦   «         Z'e%defd4„¦   «         Z(e%dee         fd5„¦   «         Z)e%dee         fd6„¦   «         Z*d<dedefd7„Z+e%defd8„¦   «         Z,e%defd9„¦   «         Z-dS )?ÚCachea³  
    A `Cache` is mostly a list of `CacheLayerMixin` objects, one per model layer. It serves as a container for
    the Cache of each layer.

    Args:
        layers (`Optional`, *optional*):
            A list of pre-created `CacheLayerMixin` or `LinearAttentionCacheLayerMixin`. If omitted (`None`), then `layer_class_to_replicate`
            will be used.
        layer_class_to_replicate (`type[CacheLayerMixin | LinearAttentionCacheLayerMixin]`, *optional*):
            Only used if `layers` is omitted (`None`), in which case it will be used as the base class for each layer,
            and the layers will be added lazily as soon as `update` is called with a `layer_idx` greater than the current
            list of layers.
        offloading (`bool`, *optional*, defaults to `False`):
            Whether to perform offloading of the layers to `cpu`, to save GPU memory.
        offload_only_non_sliding (`bool`, *optional*, defaults to `True`):
            If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
            usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).
    NFTÚlayersÚlayer_class_to_replicateÚ
offloadingÚoffload_only_non_slidingc                 ó  — |�|�t          d¦  «        ‚|€|€t          d¦  «        ‚|�|ng | _        || _        || _        | j        rF|| _        t
          rt          j        ¦   «         nt          j                             ¦   «         | _	        d S d S )Na  You can construct a Cache either from a list `layers` of all the predefined `CacheLayer`, or from a `layer_class_to_replicate`, in which case the Cache will append a new layer corresponding to `layer_class_to_replicate` for each new call to `update` with an idx not already in the Cache.z_You should provide exactly one of `layers` or `layer_class_to_replicate` to initialize a Cache.)
r­   r—  r˜  r™  Úonly_non_slidingÚ#_is_torch_greater_or_equal_than_2_7rf   ÚStreamr4  Úprefetch_stream)r&   r—  r˜  r™  rš  s        r   r'   zCache.__init__º  s·   € ð ÐÐ":Ð"FÝðqñô ð ð
 ˆ>Ð6Ð>ÝØqñô ð ð !'Ð 2�f�f¸ˆŒØ(@ˆÔ%Ø$ˆŒØŒ?ð 	rØ$<ˆDÔ!Ý5XÐ#q¥5¤<¡>¤> >Õ^cÔ^h×^oÒ^oÑ^qÔ^qˆDÔ Ð Ð ð	rð 	rr    c                 ó0   — | j         j        › d| j        › d�S )Nz(layers=ú))r   r+   r—  r,   s    r   r-   zCache.__repr__Ò  s    € Ø”.Ô)ÐAÐA°4´;ÐAÐAÐAÐAr    c                 ó*   — t          | j        ¦  «        S )zN
        This value corresponds to the number of layers in the model.
        )ro  r—  r,   s    r   Ú__len__zCache.__len__Õ  s   € õ �4”;ÑÔÐr    Ú	layer_idxrœ  c                 ó¶  ‡— ˆfd„t          | j        | j        ¦  «        D ¦   «         }	 |||d…                              d¦  «        z   }n%# t          $ r |                     d¦  «        }Y nw xY wt
          r| j        n#t          j         	                    | j        ¦  «        5  | j
        |                              ¦   «          ddd¦  «         dS # 1 swxY w Y   dS )aG  
        Prefetch the next offloaded layer on its device, starting at `layer_idx` and circling back to the beginning
        if needed. Linear-attention layers are never offloaded and are skipped, as are sliding layers when
        `only_non_sliding`. Note that we use a non-default stream for this, to avoid blocking.
        c                 ó&   •— g | ]\  }}| o‰o| ‘ŒS r   r   )Ú.0Ú	is_linearr“   rœ  s      €r   ú
<listcomp>z"Cache.prefetch.<locals>.<listcomp>å  s=   ø€ ð 
ð 
ð 
á%�	˜:ð ˆMÐCÐ#3Ð#B¸
ÐCð
ð 
ð 
r    NT)Úzipr¨  r“   rê   r­   r�  rŸ  rf   r4  Ústreamr—  rK   )r&   r¤  rœ  Úis_offloadeds     ` r   rK   zCache.prefetchÝ  s:  ø€ ð
ð 
ð 
ð 
å),¨T¬^¸T¼_Ñ)MÔ)Mð
ñ 
ô 
ˆð	1à! L°°°Ô$<×$BÒ$BÀ4Ñ$HÔ$HÑHˆIˆIøåð 	1ð 	1ð 	1Ø$×*Ò*¨4Ñ0Ô0ˆIˆIˆIð	1øøøõ &IÐuˆTÔ!Ð!ÍeÌj×N_ÒN_Ð`dÔ`tÑNuÔNuð 	.ð 	.ØŒK˜	Ô"×+Ò+Ñ-Ô-Ð-ð	.ð 	.ð 	.ñ 	.ô 	.ð 	.ð 	.ð 	.ð 	.ð 	.ð 	.ð 	.øøøð 	.ð 	.ð 	.ð 	.ð 	.ð 	.s#   © A
 Á
A,Á+A,Â! CÃCÃCc                 óf   — |r| j         |         s!| j        |                              ¦   «          dS dS )a  
        Offload a given `layer_idx`. If `only_non_sliding` is True, it will offload `layer_idx` only if it is a
        non-sliding layer. Note that we do it on the default stream, so that we ensure all earlier
        computation in the layer's `update` methods are finished.
        N)r“   r—  rG   )r&   r¤  rœ  s      r   rG   zCache.offloadô  sC   € ð !ð 	- T¤_°YÔ%?ð 	-ØŒK˜	Ô"×*Ò*Ñ,Ô,Ð,Ð,Ð,ð	-ð 	-r    r/   r0   r1   c                 ó  — | j         �\t          | j        ¦  «        |k    rD| j                             |                       ¦   «         ¦  «         t          | j        ¦  «        |k    °D| j        rZt
          j                             |j        ¦  «         	                    | j
        ¦  «         |                      |dz   | j        ¦  «          | j        |         j        ||g|¢R i |¤Ž\  }}| j        r|                      || j        ¦  «         ||fS )aá  
        Updates the cache with the new `key_states` and `value_states` for the layer `layer_idx`.

        Parameters:
            key_states (`torch.Tensor`):
                The new key states to cache.
            value_states (`torch.Tensor`):
                The new value states to cache.
            layer_idx (`int`):
                The index of the layer to cache the states for.

        Return:
            A tuple containing the updated key and value states.
        Nr   )r˜  ro  r—  Úappendr™  rf   r4  Údefault_streamrJ   Úwait_streamrŸ  rK   rœ  r8   rG   )r&   r/   r0   r¤  r7   r   r#   r$   s           r   r8   zCache.updateý  s	  € ð$ Ô(Ð4Ý�d”kÑ"Ô" iÒ/Ð/Ø”×"Ò" 4×#@Ò#@Ñ#BÔ#BÑCÔCÐCõ �d”kÑ"Ô" iÒ/Ð/ð Œ?ð 	@åŒJ×%Ò% jÔ&7Ñ8Ô8×DÒDÀTÔEYÑZÔZÐZØ�MŠM˜) a™-¨Ô)>Ñ?Ô?Ð?à4�t”{ 9Ô-Ô4°ZÀÐ_ÐPTÐ_Ð_Ð_ÐX^Ð_Ð_‰ˆˆfàŒ?ð 	;Ø�LŠL˜ DÔ$9Ñ:Ô:Ð:à�Vˆ|Ðr    r   r>  rG  c                 ó˜   — t          | j        |         t          ¦  «        st          d¦  «        ‚ | j        |         j        ||fi |¤Ž}|S )ak  
        Updates the cache with the new `conv_states` for the layer `layer_idx`.

        Parameters:
            conv_states (`torch.Tensor`):
                The new conv states to cache.
            layer_idx (`int`):
                The index of the layer to cache the states for.

        Return:
            `torch.Tensor`: The updated conv states.
        ú?Cannot call `update_conv_state` on a non-LinearAttention layer!)rQ   r—  r8  r­   rJ  )r&   r>  r¤  rG  r   s        r   rJ  zCache.update_conv_state  sX   € õ" ˜$œ+ iÔ0Õ2PÑQÔQð 	`ÝÐ^Ñ_Ô_Ð_Ø>�d”k )Ô,Ô>¸{ÈIÐ`Ð`ÐY_Ð`Ð`ˆØÐr    r?  c                 ó˜   — t          | j        |         t          ¦  «        st          d¦  «        ‚ | j        |         j        ||fi |¤Ž}|S )am  
        Updates the cache with the new `recurrent_states` for the layer `layer_idx`.

        Parameters:
            smm_states (`torch.Tensor`):
                The new ssm states to cache.
            layer_idx (`int`):
                The index of the layer to cache the states for.

        Return:
            `torch.Tensor`: The updated ssm states.
        r³  )rQ   r—  r8  r­   rL  )r&   r?  r¤  rG  r   s        r   rL  zCache.update_recurrent_state5  s[   € õ" ˜$œ+ iÔ0Õ2PÑQÔQð 	`ÝÐ^Ñ_Ô_Ð_ØH˜4œ; yÔ1ÔHÐIYÐ[dÐoÐoÐhnÐoÐoÐØÐr    r³   c           	      óÞ   — t          | j        |         d¦  «        s3t          d|› dt          | j        |         ¦  «        j        › d�¦  «        ‚| j        |                              |¦  «        S )a©  
        Updates the indexer key cache for layer `layer_idx`.

        Parameters:
            indexer_key_states (`torch.Tensor`):
                The new indexer key states to cache, shape `[batch_size, seq_len, index_head_dim]`.
            layer_idx (`int`):
                The index of the layer to cache the states for.

        Return:
            `torch.Tensor`: The updated indexer key states (full cache).
        rº   z&Cannot call `update_indexer` on layer z which is a zY; it has no indexer key cache (expected a `DynamicIndexedLayer` or `StaticIndexedLayer`).)rP   r—  r­   Útyper+   rº   )r&   r³   r¤  s      r   rº   zCache.update_indexerK  s‰   € õ �t”{ 9Ô-Ð/?Ñ@Ô@ð 	ÝðO¸ð Oð OÝ˜œ IÔ.Ñ/Ô/Ô8ðOð Oð Oñô ð ð
 Œ{˜9Ô%×4Ò4Ð5GÑHÔHÐHr    rË   rÌ   Úhead_dimrq   rJ   c                 óÈ  — t          |t          ¦  «        r|gt          | ¦  «        z  }t          |t          ¦  «        r|gt          | ¦  «        z  }t          |¦  «        t          | j        ¦  «        k    r5t	          dt          |¦  «        › dt          | j        ¦  «        › d�¦  «        ‚t          |¦  «        t          | j        ¦  «        k    r5t	          dt          |¦  «        › dt          | j        ¦  «        › d�¦  «        ‚t          | j        ||¦  «        D ]F\  }}}|j        r|j        rŒt          j	        ||d|f||¬¦  «        }	| 
                    |	|	¦  «         ŒGdS )zÐ
        Initialize all the layers in advance (it's otherwise lazily initialized on the first `update` call).
        This is useful for our `export` recipes, as `export` needs everything in advance.
        z,`num_head` was provided as a list of length z, but the Cache currently has z layersz,`head_dim` was provided as a list of length r   rp   N)rQ   rR   ro  r—  r­   rª  rc   r%   rf   rÏ   r4   )
r&   rË   rÌ   r·  rq   rJ   ÚlayerÚlayer_num_headsÚlayer_head_dimÚfake_kv_tensors
             r   Úearly_initializationzCache.early_initialization`  s§  € õ �i¥Ñ%Ô%ð 	0Ø"˜¥c¨$¡i¤iÑ/ˆIÝ�h¥Ñ$Ô$ð 	.Ø �z¥C¨¡I¤IÑ-ˆHåˆy‰>Œ>�S ¤Ñ-Ô-Ò-Ð-Ýð G½sÀ9¹~¼~ð  Gð  GÕmpÐquÔq|Ñm}Ôm}ð  Gð  Gð  Gñô ð õ ˆx‰=Œ=�C ¤Ñ,Ô,Ò,Ð,Ýð G½sÀ9¹~¼~ð  Gð  GÕmpÐquÔq|Ñm}Ôm}ð  Gð  Gð  Gñô ð õ 7:¸$¼+ÀyÐRZÑ6[Ô6[ð 	Fð 	FÑ2ˆE�? NØÔ,ð °Ô0Dð Øõ #œ[¨*°oÀqÈ.Ð)YÐafÐouÐvÑvÔvˆNà×%Ò% n°nÑEÔEÐEÐEð	Fð 	Fr    c                 ó�  ‡ — |t          ‰ j        ¦  «        k    rdS t          ‰ j        |         t          ¦  «        sm|dk    rt	          d|› d�¦  «        ‚	 t          ˆ fd„t          t          ‰ ¦  «        ¦  «        D ¦   «         ¦  «        }n# t          $ r t	          d¦  «        ‚w xY w‰ j        |                              ¦   «         S )z=Returns the sequence length of the cache for the given layer.r   z+You called `get_seq_length` on layer index úR, but this layer is a LinearAttention layer, which does not track sequence length.c              3   ó\   •K  — | ]&}t          ‰j        |         t          ¦  «        ¯"|V — Œ'd S r)   ©rQ   r—  r   ©r§  Úidxr&   s     €r   ú	<genexpr>z'Cache.get_seq_length.<locals>.<genexpr>”  ó;   øè è € Ð rÐ r¨ÅJÈtÌ{Ð[^ÔO_ÕapÑDqÔDqÐ r Ð rÐ rÐ rÐ rÐ rÐ rr    z{`get_seq_length` can only be called on Attention layers, and the current Cache seem to only contain LinearAttention layers.)	ro  r—  rQ   r   r­   Únextr=  ÚStopIterationr=   ©r&   r¤  s   ` r   r=   zCache.get_seq_length…  sí   ø€ à�˜DœKÑ(Ô(Ò(Ð(Ø�1õ ˜$œ+ iÔ0µ/ÑBÔBð 	à˜AŠ~ˆ~Ý ð6À)ð 6ð 6ð 6ñô ð ðå Ð rÐ rÐ rÐ rµµc¸$±i´iÑ0@Ô0@Ð rÑ rÔ rÑrÔr�	�	øÝ ð ð ð Ý ð.ñô ð ðøøøð Œ{˜9Ô%×4Ò4Ñ6Ô6Ð6s   Á5B ÂB&c                 ó¸   — |�|t          | j        ¦  «        k    rdS |€t          d„ | j        D ¦   «         ¦  «        S | j        |                              ¦   «         S )aª  
        Returns the maximum length of the cache. If `layer_idx` is not provided (default), this returns the maximum
        accross all layers. Otherwise, return the maximum supported value for the given layer.
        A value of `-1` means no maximum, or undefined maximum, e.g. for dynamic attention layers that can grow indefinitely,
        or linear attention layer that do not have a sequence length dimension.
        Nrƒ   c              3   ó>   K  — | ]}|                      ¦   «         V — Œd S r)   )r@   ©r§  r¹  s     r   rÄ  z'Cache.get_max_length.<locals>.<genexpr>©  s.   è è € ÐGÐG°%�u×+Ò+Ñ-Ô-ÐGÐGÐGÐGÐGÐGr    )ro  r—  r£   r@   rÈ  s     r   r@   zCache.get_max_length�  sa   € ð Ð  Yµ#°d´kÑ2BÔ2BÒ%BÐ%BØ�2àÐÝÐGÐG¸4¼;ÐGÑGÔGÑGÔGÐGà”;˜yÔ)×8Ò8Ñ:Ô:Ð:r    c                 óî  ‡ — |�|t          ‰ j        ¦  «        k    rdS |€Y	 t          ˆ fd„t          t          ‰ ¦  «        dz
  dd¦  «        D ¦   «         ¦  «        }nP# t          $ r t          d¦  «        ‚w xY wt          ‰ j        |         t          ¦  «        st          d|› d�¦  «        ‚|€1t          ‰ j        |         j	         
                    ¦   «         ¦  «        S ‰ j        |         j	        |         S )	zYReturns whether the LinearAttention layer at index `layer_idx` has previous state or not.NFc              3   ó\   •K  — | ]&}t          ‰j        |         t          ¦  «        ¯"|V — Œ'd S r)   )rQ   r—  r8  rÂ  s     €r   rÄ  z+Cache.has_previous_state.<locals>.<genexpr>µ  sO   øè è € ð !ð !àÝ! $¤+¨cÔ"2Õ4RÑSÔSð!Øð!ð !ð !ð !ð !ð !r    r   rƒ   z`has_previous_state` can only be called on LinearAttention layers, and the current Cache seem to only contain Attention layers.z/You called `has_previous_state` on layer index zJ, but this layer is an Attention layer, which does not support calling it.)ro  r—  rÆ  r=  rÇ  r­   rQ   r8  ÚallrB  r$   )r&   r¤  rG  s   `  r   rB  zCache.has_previous_state­  s5  ø€ àÐ  Yµ#°d´kÑ2BÔ2BÒ%BÐ%BØ�5ð Ðð
Ý ð !ð !ð !ð !å$¥S¨¡Y¤Y°¡]°B¸Ñ;Ô;ð!ñ !ô !ñ ô �	�	øõ
 !ð ð ð Ý ð5ñô ð ðøøøõ
 ˜DœK¨	Ô2Õ4RÑSÔSð 	Ýð/À)ð /ð /ð /ñô ð ð ÐÝ�t”{ 9Ô-Ô@×GÒGÑIÔIÑJÔJÐJØŒ{˜9Ô%Ô8¸ÔCÐCs   ¡:A ÁA6r9   c                 ó–  ‡ — |t          ‰ j        ¦  «        k    r|dfS t          ‰ j        |         t          ¦  «        sm|dk    rt	          d|› d�¦  «        ‚	 t          ˆ fd„t          t          ‰ ¦  «        ¦  «        D ¦   «         ¦  «        }n# t          $ r t	          d¦  «        ‚w xY w‰ j        |                              |¦  «        S )a  
        Return a tuple (kv_length, kv_offset) corresponding to the length and offset that will be returned for
        the given layer at `layer_idx`.
        The masks are then prepared according to the given lengths (kv_length, kv_offset) and patterns for each layer.
        r   z+You called `get_mask_sizes` on layer index r¿  c              3   ó\   •K  — | ]&}t          ‰j        |         t          ¦  «        ¯"|V — Œ'd S r)   rÁ  rÂ  s     €r   rÄ  z'Cache.get_mask_sizes.<locals>.<genexpr>à  rÅ  r    z{`get_mask_sizes` can only be called on Attention layers, and the current Cache seem to only contain LinearAttention layers.)	ro  r—  rQ   r   r­   rÆ  r=  rÇ  r;   ©r&   r9   r¤  s   `  r   r;   zCache.get_mask_sizesË  sö   ø€ ð �˜DœKÑ(Ô(Ò(Ð(Ø �?Ð"õ ˜$œ+ iÔ0µ/ÑBÔBð 	à˜AŠ~ˆ~Ý ð6À)ð 6ð 6ð 6ñô ð ðå Ð rÐ rÐ rÐ rµµc¸$±i´iÑ0@Ô0@Ð rÑ rÔ rÑrÔr�	�	øÝ ð ð ð Ý ð.ñô ð ðøøøð Œ{˜9Ô%×4Ò4°\ÑBÔBÐBs   Á5B ÂB(c                 ó.   — |                       |¬¦  «        S )z³Returns the current offset of the query for the given `layer_idx`. It's always equal to the cache length, i.e.
        `get_seq_length(layer_idx)`, except for MTP layers.
        )r¤  rz   rÈ  s     r   Úget_query_offsetzCache.get_query_offseté  s   € ð
 ×"Ò"¨YÐ"Ñ7Ô7Ð7r    c                 óŒ   — t          t          | j        ¦  «        ¦  «        D ]!}| j        |                              ¦   «          Œ"dS )z$Recursively reset all layers tensorsN)r=  ro  r—  rS   rÈ  s     r   rS   zCache.resetð  sI   € å�s 4¤;Ñ/Ô/Ñ0Ô0ð 	+ð 	+ˆIØŒK˜	Ô"×(Ò(Ñ*Ô*Ð*Ð*ð	+ð 	+r    rT   c                 óŽ   — t          t          | j        ¦  «        ¦  «        D ]"}| j        |                              |¦  «         Œ#dS )z!Reorder the cache for beam searchN)r=  ro  r—  rX   )r&   rT   r¤  s      r   rX   zCache.reorder_cacheõ  sK   € å�s 4¤;Ñ/Ô/Ñ0Ô0ð 	;ð 	;ˆIØŒK˜	Ô"×0Ò0°Ñ:Ô:Ð:Ð:ð	;ð 	;r    r…   c                 óŽ   — t          t          | j        ¦  «        ¦  «        D ]"}| j        |                              |¦  «         Œ#dS )z"Crop the cache to the given lengthN)r=  ro  r—  r‰   )r&   r…   r¤  s      r   r‰   z
Cache.cropú  sK   € å�s 4¤;Ñ/Ô/Ñ0Ô0ð 	4ð 	4ˆIØŒK˜	Ô"×'Ò'¨
Ñ3Ô3Ð3Ð3ð	4ð 	4r    rŠ   c                 óŽ   — t          t          | j        ¦  «        ¦  «        D ]"}| j        |                              |¦  «         Œ#dS )zRepeat and interleave the cacheN)r=  ro  r—  rŽ   )r&   rŠ   r¤  s      r   rŽ   zCache.batch_repeat_interleaveÿ  sO   € å�s 4¤;Ñ/Ô/Ñ0Ô0ð 	Dð 	DˆIØŒK˜	Ô"×:Ò:¸7ÑCÔCÐCÐCð	Dð 	Dr    r�   c                 óŽ   — t          t          | j        ¦  «        ¦  «        D ]"}| j        |                              |¦  «         Œ#dS )zSelect indices from the cacheN)r=  ro  r—  r’   )r&   r�   r¤  s      r   r’   zCache.batch_select_indices  sO   € å�s 4¤;Ñ/Ô/Ñ0Ô0ð 	Að 	AˆIØŒK˜	Ô"×7Ò7¸Ñ@Ô@Ð@Ð@ð	Að 	Ar    c                 óÂ   — t          t          | j        ¦  «        ¦  «        D ]<}t          | j        |         d¦  «        r| j        |                              ¦   «          Œ=dS )a  
        Calling this function will activate past state recording, meaning that cache with fixed size such as a linear cache will
        wait for a call to `crop` before restricting the size of its cached states, in order to be able to retrieve previous full states.
        rU  N)r=  ro  r—  rP   rU  rÈ  s     r   rU  zCache.activate_past_recording	  sh   € õ
 �s 4¤;Ñ/Ô/Ñ0Ô0ð 	Að 	AˆIÝ�t”{ 9Ô-Ð/HÑIÔIð AØ”˜IÔ&×>Ò>Ñ@Ô@Ð@øð	Að 	Ar    c                 ó    — d„ | j         D ¦   «         }|sdS t          t          |¦  «        ¦  «        dk    rt          d|› �¦  «        ‚|d         S )z¡Return the batch size of the cache, or ``-1`` if no layer has been initialized yet
        (e.g. an all-linear-attention cache queried before the first forward).c                 ó<   — g | ]}t          |d ¦  «        ¯|j        ‘ŒS )rË   )rP   rË   rË  s     r   r©  z$Cache.batch_size.<locals>.<listcomp>  s*   € Ð\Ð\Ð\ u½wÀuÈlÑ?[Ô?[Ð\�%Ô"Ð\Ð\Ð\r    rƒ   r   z0The batch size is not consistent across layers: r   )r—  ro  Úsetr­   )r&   r$   s     r   rË   zCache.batch_size  sb   € ð ]Ð\°´Ð\Ñ\Ô\ˆØð 	Ø�2Ý�s�6‰{Œ{ÑÔ˜aÒÐÝÐXÐPVÐXÐXÑYÔYÐYØ�aŒyÐr    c                 ór   — t          | j        ¦  «        dk    rdS t          d„ | j        D ¦   «         ¦  «        S )z&Return whether the cache is compilabler   Fc              3   ó$   K  — | ]}|j         V — Œd S r)   )rb   rË  s     r   rÄ  z'Cache.is_compileable.<locals>.<genexpr>%  s%   è è € ÐAÐA¨E�5Ô'ÐAÐAÐAÐAÐAÐAr    )ro  r—  rÎ  r,   s    r   rb   zCache.is_compileable  s=   € õ ˆtŒ{ÑÔ˜qÒ Ð Ø�5ÝÐAÐA°T´[ÐAÑAÔAÑAÔAÐAr    c                 ó|   — d„ | j         D ¦   «         }t          |¦  «        dk    ot          d„ |D ¦   «         ¦  «        S )z,Return whether the cache data is initializedc                 ó    — g | ]}|j         ¯	|‘ŒS r   )rc   rË  s     r   r©  z(Cache.is_initialized.<locals>.<listcomp>*  s    € ÐNÐNÐN˜E°EÔ4MÐN�%ÐNÐNÐNr    r   c              3   ó$   K  — | ]}|j         V — Œd S r)   )r%   rË  s     r   rÄ  z'Cache.is_initialized.<locals>.<genexpr>+  s%   è è € Ð&PÐ&PÀ uÔ';Ð&PÐ&PÐ&PÐ&PÐ&PÐ&Pr    )r—  ro  rÎ  )r&   r—  s     r   r%   zCache.is_initialized'  sF   € ð OÐN T¤[ÐNÑNÔNˆÝ�6‰{Œ{˜QŠÐP¥3Ð&PÐ&PÈÐ&PÑ&PÔ&PÑ#PÔ#PÐPr    c                 ó$   — d„ | j         D ¦   «         S )z9Return whether the layers of the cache are sliding windowc                 ó0   — g | ]}t          |d d¦  «        ‘ŒS )r“   F)ÚgetattrrË  s     r   r©  z$Cache.is_sliding.<locals>.<listcomp>0  s$   € ÐMÐMÐM¸•˜˜|¨UÑ3Ô3ÐMÐMÐMr    ©r—  r,   s    r   r“   zCache.is_sliding-  s   € ð NÐMÀÄÐMÑMÔMÐMr    c                 ó$   — d„ | j         D ¦   «         S )z¼Return whether the layers of the cache are linear attention (Mamba/SSM) layers. Note that layers containing
        both linear and full attention states will return False by this functionc                 ód   — g | ]-}t          |t          ¦  «        ot          |t          ¦  «         ‘Œ.S r   )rQ   r8  ri  rË  s     r   r©  z#Cache.is_linear.<locals>.<listcomp>6  sL   € ð 
ð 
ð 
ð õ �uÕ<Ñ=Ô=ð LÝ˜uÕ&JÑKÔKÐKð
ð 
ð 
r    rå  r,   s    r   r¨  zCache.is_linear2  s'   € ð
ð 
ð œð
ñ 
ô 
ð 	
r    c                 ó`   — t                                d¦  «         |                      |¦  «        S rZ   ©r[   Úwarning_oncer@   rÈ  s     r   r]   zCache.get_max_cache_shape<  ó3   € Ý×ÒØ{ñ	
ô 	
ð 	
ð ×"Ò" 9Ñ-Ô-Ð-r    c                 ó^   — t                                d¦  «         |                      ¦   «         S )Nzi`max_cache_len` is deprecated, and will be removed in version 5.16. Please use `get_max_length()` insteadré  r,   s    r   rÆ   zCache.max_cache_lenB  s1   € å×ÒØwñ	
ô 	
ð 	
ð ×"Ò"Ñ$Ô$Ð$r    c                 óD   — t                                d¦  «         | j        S )Nzp`max_batch_size` is deprecated, and will be removed in version 5.16. Please use the simpler `batch_size` instead)r[   rê  rË   r,   s    r   Úmax_batch_sizezCache.max_batch_sizeI  s'   € å×ÒØ~ñ	
ô 	
ð 	
ð ŒÐr    )NNFT)Tr[  r)   )NN).r+   r_   r`   ra   Úlistr   r8  r¶  Úboolr'   r-   r£  rR   rK   rG   rf   rg   rh   r8   rJ  rL  rº   rq   rJ   r½  r=   r@   rB  r;   rÓ  rS   ri   rX   r‰   rŽ   r’   rU  ÚpropertyrË   rb   r%   r“   r¨  r]   rÆ   rî  r   r    r   r–  r–  ¦  sI  € € € € € ðð ð* QUØbfØ Ø)-ðrð rà�_Ð'EÑEÔFÈÑMðrð #' Ð9WÑ'WÔ"XÐ[_Ñ"_ðrð ð	rð
 #'ðrð rð rð rð0Bð Bð Bð ð  ð  ð.ð . #ð .¸ð .ð .ð .ð .ð.-ð - ð -¸ð -ð -ð -ð -ð Øœ,ð Ø6;´lð ØORð à	ˆuŒ|˜Uœ\Ð)Ô	*ð ð  ð  ð  ðF KLðð Ø œ<ðØ47ðØDGðà	Œðð ð ð ð. PQð ð  Ø %¤ð Ø9<ð ØILð à	Œð ð  ð  ð  ð,I°´ð IÈ#ð IÐRWÔR^ð Ið Ið Ið Ið*#Fàð#Fð ˜˜cœ‘?ð#Fð ˜˜Sœ	‘/ð	#Fð
 Œ{ð#Fð ”ð#Fð #Fð #Fð #FðJ7ð 7¨ð 7°Cð 7ð 7ð 7ð 7ð0;ð ;¨¨d©
ð ;¸cð ;ð ;ð ;ð ;ð Dð D¨C°$©Jð DÈ#ÐPTÉ*ð DÐ`dð Dð Dð Dð Dð<C¨3ð C¸3ð CÀ5ÈÈcÈÄ?ð Cð Cð Cð Cð<8ð 8¨#ð 8°cð 8ð 8ð 8ð 8ð+ð +ð +ð
; eÔ&6ð ;ð ;ð ;ð ;ð
4˜sð 4ð 4ð 4ð 4ð
D¨sð Dð Dð Dð Dð
A¨E¬Lð Að Að Að Að
Að Að Að ð
˜Cð 
ð 
ð 
ñ „Xð
ð ðB ð Bð Bð Bñ „XðBð ðQ ð Qð Qð Qñ „XðQð
 ðN˜D œJð Nð Nð Nñ „XðNð ð
˜4 œ:ð 
ð 
ð 
ñ „Xð
ð.ð .¨Sð .¸ð .ð .ð .ð .ð ð%˜sð %ð %ð %ñ „Xð%ð ð ð ð ð ñ „Xðð ð r    r–  Úconfigr1   c                 ó6  — t          | dd¦  «        }|€~t          | dd¦  «        �d„ t          | j        ¦  «        D ¦   «         }nNt          | dd¦  «        �d„ t          | j        ¦  «        D ¦   «         }nd„ t          | j        ¦  «        D ¦   «         }t          | dd¦  «        }|�|d	k    r|d| j         …         }i }d
|v sd|v r
| j        |d<   d|v r
| j        |d<   d|v sd|v r| |d<   t          d„ |D ¦   «         ¦  «        rt          | dd¦  «        |d<   ||fS )z™
    From a `config`, extract the layer types if not present already, as well as the kwargs needed to initialize
    the corresponding layer caches.
    Úlayer_typesNr–   c                 ó   — g | ]}d ‘ŒS )r�  r   ©r§  ró   s     r   r©  z.get_layer_types_and_kwargs.<locals>.<listcomp>Z  ó   € ÐXÐXÐX°1Ð.ÐXÐXÐXr    Úattention_chunk_sizec                 ó   — g | ]}d ‘ŒS )rŽ  r   rö  s     r   r©  z.get_layer_types_and_kwargs.<locals>.<listcomp>\  r÷  r    c                 ó   — g | ]}d ‘ŒS )rŒ  r   rö  s     r   r©  z.get_layer_types_and_kwargs.<locals>.<listcomp>^  s   € ÐUÐUÐU°Ð+ÐUÐUÐUr    Únum_kv_shared_layersr   r�  r“  rŽ  Úheavily_compressed_attentionÚcompressed_sparse_attentionrò  c              3   ó   K  — | ]}|d v V — Œ	dS ))r�  r‘  r’  r“  Nr   )r§  Ú
layer_types     r   rÄ  z-get_layer_types_and_kwargs.<locals>.<genexpr>o  s)   è è € Ð
pÐ
pÐV`ˆ:ÐQÐQÐ
pÐ
pÐ
pÐ
pÐ
pÐ
pr    Únumber_of_conv_statesr   r9  )rä  r=  Únum_hidden_layersrû  r–   rø  Úany)rò  rô  rû  Úlayer_kwargss       r   Úget_layer_types_and_kwargsr  Q  s…  € õ
 ˜& -°Ñ6Ô6€KàÐÝ�6Ð+¨TÑ2Ô2Ð>ØXÐX½¸fÔ>VÑ8WÔ8WÐXÑXÔXˆKˆKÝ�VÐ3°TÑ:Ô:ÐFØXÐX½¸fÔ>VÑ8WÔ8WÐXÑXÔXˆKˆKàUÐUµU¸6Ô;SÑ5TÔ5TÐUÑUÔUˆKõ # 6Ð+AÀ4ÑHÔHÐØÐ'Ð,@À1Ò,DÐ,DØ!Ð"@ VÔ%@Ð$@Ð"@ÔAˆð €LØ˜kÐ)Ð)Ð-=ÀÐ-LÐ-LØ)/Ô)>ˆÐ%Ñ&Ø˜kÐ)Ð)Ø)/Ô)DˆÐ%Ñ&à%¨Ð4Ð4Ð8UÐYdÐ8dÐ8dØ!'ˆ�XÑå
Ð
pÐ
pÐdoÐ
pÑ
pÔ
pÑpÔpð WÝ+2°6Ð;RÐTUÑ+VÔ+VˆÐ'Ñ(à˜Ð$Ð$r    c            	       ó|   ‡ — e Zd ZdZ	 	 	 	 ddeeej        dz  df                  dz  dedz  de	de	fˆ fd	„Z
d
„ Zˆ xZS )ÚDynamicCachea*
  
    A cache that grows dynamically as more tokens are generated. This is the default for generative models.
    It stores the key and value states as a list of `CacheLayer`, one for each layer. The expected shape for each tensor
    in the `CacheLayer`s is `[batch_size, num_heads, seq_len, head_dim]`.
    If a config is passed, it will additionally check for sliding or hybrid cache structure, greatly reducing the
    memory requirement of the cached tensors to `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        ddp_cache_data (`Iterable[tuple[torch.Tensor, torch.Tensor]]`, *optional*):
            It was originally added for compatibility with `torch.distributed` (DDP). In a nutshell, it is
            `map(gather_map, zip(*caches))`, i.e. each item in the iterable contains the key and value states
            for a layer gathered across replicas by torch.distributed (shape=[global batch size, num_heads, seq_len, head_dim]).
            Note: it needs to be the 1st arg as well to work correctly
        config (`PreTrainedConfig`, *optional*):
            The config of the model for which this Cache will be used. If passed, it will be used to check for sliding
            or hybrid layer structure, greatly reducing the memory requirement of the cached tensors to
            `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.
        offloading (`bool`, *optional*, defaults to `False`):
            Whether to perform offloading of the layers to `cpu`, to save GPU memory.
        offload_only_non_sliding (`bool`, *optional*, defaults to `False`):
            If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
            usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).

    Example:

    ```python
    >>> from transformers import AutoTokenizer, AutoModelForCausalLM, DynamicCache

    >>> model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2-0.5B-Instruct")
    >>> tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen2-0.5B-Instruct")

    >>> inputs = tokenizer(text="My name is Qwen2", return_tensors="pt")

    >>> # Prepare a cache class and pass it to model's forward
    >>> past_key_values = DynamicCache(config=model.config)
    >>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
    >>> outputs.past_key_values # access cache filled with key/values from generation
    ```
    NFÚddp_cache_data.rò  r™  rš  c                 óÆ  •‡— g }|�6|                      d¬¦  «        }t          |¦  «        \  }Šˆfd„|D ¦   «         }|�Àt          |¦  «        D ]°\  }}	|€~t          |	¦  «        dk    r|	d         nd }
|
�>|
d                              ¦   «         }|                     t          |¬¦  «        ¦  «         n!|                     t          ¦   «         ¦  «         ||                              |	d         |	d         ¦  «        \  }}Œ±t          |¦  «        dk    r+t          ¦   «          
                    t          ||¬	¦  «         d S t          ¦   «          
                    |||¬
¦  «         d S )NT©Údecoderc                 ó4   •— g | ]}t          |         d i ‰¤Ž‘ŒS ©r   )r   ©r§  rÿ  r  s     €r   r©  z)DynamicCache.__init__.<locals>.<listcomp>­  s.   ø€ ÐkÐkÐkÐQ[Õ0°Ô<ÐLÐL¸|ÐLÐLÐkÐkÐkr    r*  rÊ   r   r«   r   )r˜  r™  rš  ©r—  r™  rš  )Úget_text_configr  Ú	enumeratero  Úitemr¯  r•   rm   r8   r   r'   )r&   r  rò  r™  rš  r—  Údecoder_configrô  r¤  Úkv_and_optional_slidingÚsliding_window_tensorr–   ró   r  r   s                @€r   r'   zDynamicCache.__init__   s˜  øø€ ð ˆàÐØ#×3Ò3¸DÐ3ÑAÔAˆNÝ(BÀ>Ñ(RÔ(RÑ%ˆK˜àkÐkÐkÐkÐ_jÐkÑkÔkˆFð Ð%å6?ÀÑ6OÔ6Oð hð hÑ2�	Ð2à�>õ KNÐNeÑJfÔJfÐjkÒJkÐJkÐ,CÀAÔ,FÐ,FÐquÐ)à,Ð8à)>¸qÔ)A×)FÒ)FÑ)HÔ)H˜ØŸšÕ&?È~Ð&^Ñ&^Ô&^Ñ_Ô_Ð_Ð_àŸš¥l¡n¤nÑ5Ô5Ð5à˜iÔ(×/Ò/Ð0GÈÔ0JÐLcÐdeÔLfÑgÔg‘��1�1õ ˆv‰;Œ;˜!ÒÐÝ‰GŒG×ÒÝ)5Ø%Ø)Að ñ ô ð ð ð õ ‰GŒG×Ò F°zÐ\tÐÑuÔuÐuÐuÐur    c              #   ó^   K  — | j         D ]"}|j        |j        t          |dd ¦  «        fV — Œ#d S )Nr›   )r—  r#   r$   rä  )r&   r¹  s     r   Ú__iter__zDynamicCache.__iter__Ì  sM   è è € Ø”[ð 	[ð 	[ˆEØ”*˜eœl­G°EÐ;SÐUYÑ,ZÔ,ZÐZÐZÐZÐZÐZð	[ð 	[r    )NNFF)r+   r_   r`   ra   r   rh   rf   rg   r   rð  r'   r  rj   rk   s   @r   r  r  u  sÅ   ø€ € € € € ð(ð (ðX LPØ*.Ø Ø).ð*vð *và   u¤|°dÑ':¸CÐ'?Ô!@ÔAÀDÑHð*vð ! 4Ñ'ð*vð ð	*vð
 #'ð*vð *vð *vð *vð *vð *vðX[ð [ð [ð [ð [ð [ð [r    r  c            	       ó:   ‡ — e Zd ZdZ	 	 d	dedededefˆ fd„Zˆ xZS )
ÚStaticCachea¸  
    Static Cache class to be used with `torch.compile(model)` and `torch.export()`. It will check the `config`
    for potential hybrid cache structure, and initialize each layer accordingly.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        config (`PreTrainedConfig`):
            The config of the model for which this Cache will be used. It will be used to check for sliding
            or hybrid layer structure, and initialize each layer accordingly.
        max_cache_len (`int`):
            The maximum number of tokens that this Cache should hold.
        offloading (`bool`, *optional*, defaults to `False`):
            Whether to perform offloading of the layers to `cpu`, to save GPU memory.
        offload_only_non_sliding (`bool`, *optional*, defaults to `True`):
            If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
            usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).

    Example:

    ```python
    >>> from transformers import AutoTokenizer, AutoModelForCausalLM, StaticCache

    >>> model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-2-7b-chat-hf")
    >>> tokenizer = AutoTokenizer.from_pretrained("meta-llama/Llama-2-7b-chat-hf")

    >>> inputs = tokenizer(text="My name is Llama", return_tensors="pt")

    >>> # Prepare a cache class and pass it to model's forward
    >>> # Leave empty space for 10 new tokens, which can be used when calling forward iteratively 10 times to generate
    >>> max_generated_length = inputs.input_ids.shape[1] + 10
    >>> past_key_values = StaticCache(config=model.config, max_cache_len=max_generated_length)
    >>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
    >>> outputs.past_key_values # access cache filled with key/values from generation
    StaticCache()
    ```
    FTrò  rÆ   r™  rš  c                 óÄ   •‡— t          |                     d¬¦  «        ¦  «        \  }Š|‰d<   ˆfd„|D ¦   «         }t          ¦   «                              |||¬¦  «         d S )NTr	  rÆ   c                 ó4   •— g | ]}t          |         d i ‰¤Ž‘ŒS r  )r   r  s     €r   r©  z(StaticCache.__init__.<locals>.<listcomp>  s-   ø€ ÐfÐfÐfÈJÕ+¨JÔ7ÐGÐG¸,ÐGÐGÐfÐfÐfr    r  )r  r  r   r'   )
r&   rò  rÆ   r™  rš  r   rô  r—  r  r   s
           @€r   r'   zStaticCache.__init__ù  st   øø€ õ %?¸v×?UÒ?UÐ^bÐ?UÑ?cÔ?cÑ$dÔ$dÑ!ˆ�\Ø(5ˆ�_Ñ%àfÐfÐfÐfÐZeÐfÑfÔfˆÝ‰Œ×Ò °:ÐXpÐÑqÔqÐqÐqÐqr    )FT)	r+   r_   r`   ra   r   rR   rð  r'   rj   rk   s   @r   r  r  Ñ  sŠ   ø€ € € € € ð$ð $ðV !Ø)-ðrð rà ðrð ðrð ð	rð
 #'ðrð rð rð rð rð rð rð rð rð rr    r  c                   óL   ‡ — e Zd ZdZ	 	 	 	 	 ddededed	ed
ededefˆ fd„Zˆ xZS )ÚQuantizedCacheaš  
    A quantizer cache similar to what is described in the
    [KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache paper](https://huggingface.co/papers/2402.02750).
    It allows the model to generate longer sequence length without allocating too much memory for keys and values
    by applying quantization.
    The cache has two types of storage, one for original precision and one for the
    quantized cache. A `residual length` is set as a maximum capacity for the original precision cache. When the
    length goes beyond maximum capacity, the original precision cache is discarded and moved into the quantized cache.
    The quantization is done per-channel with a set `q_group_size` for both keys and values, in contrast to what was
    described in the paper.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        backend (`str`):
            The quantization backend to use. One of `("quanto", "hqq").
        config (`PreTrainedConfig`):
            The config of the model for which this Cache will be used.
        nbits (`int`, *optional*, defaults to 4):
            The number of bits for quantization.
        axis_key (`int`, *optional*, defaults to 0):
            The axis on which to quantize the keys.
        axis_value (`int`, *optional*, defaults to 0):
            The axis on which to quantize the values.
        q_group_size (`int`, *optional*, defaults to 64):
            Quantization is done per-channel according to a set `q_group_size` for both keys and values.
        residual_length (`int`, *optional*, defaults to 128):
            Maximum capacity for the original precision cache
    rú   r   rû   rü   Úbackendrò  rý   rþ   rÿ   r   r  c                 óÀ  •‡‡‡‡‡‡— |dk    rt           Šn!|dk    rt          Šnt          d|› d�¦  «        ‚|                     d¬¦  «        }t	          |¦  «        \  }}	t          |¦  «        dhz
  }
t          |
¦  «        dk    rt          d	|
› �¦  «        ‚ˆˆˆˆˆˆfd
„t          |j        ¦  «        D ¦   «         }t          ¦   «          
                    |¬¦  «         d S )NÚquantoÚhqqzUnknown quantization backend `ú`Tr	  rŒ  r   z{`QuantizedCache` is only supported for models with only full attention layers. We found the following invalid layer types: c           	      ó.   •— g | ]} ‰‰‰‰‰‰¦  «        ‘ŒS r   r   )r§  ró   rþ   rÿ   Úlayer_classrý   r   r  s     €€€€€€r   r©  z+QuantizedCache.__init__.<locals>.<listcomp>@  s;   ø€ ð 
ð 
ð 
àð ˆK˜˜x¨°\À?ÑSÔSð
ð 
ð 
r    rå  )r  r(  r­   r  r  rÜ  ro  r=  r  r   r'   )r&   r  rò  rý   rþ   rÿ   r   r  rô  ró   Úinvalid_layer_typesr—  r#  r   s      `````    @€r   r'   zQuantizedCache.__init__'  s.  øøøøøøø€ ð �hÒÐÝ.ˆKˆKØ˜ÒÐÝ+ˆKˆKåÐH¸gÐHÐHÐHÑIÔIÐIà×'Ò'°Ð'Ñ5Ô5ˆÝ3°FÑ;Ô;‰ˆ�QÝ! +Ñ.Ô.Ð2BÐ1CÑCÐÝÐ"Ñ#Ô# aÒ'Ð'Ýð0Ø-ð0ð 0ñô ð ð
ð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
å˜6Ô3Ñ4Ô4ð
ñ 
ô 
ˆõ 	‰Œ×Ò ÐÑ'Ô'Ð'Ð'Ð'r    r  )	r+   r_   r`   ra   rd   r   rR   r'   rj   rk   s   @r   r  r    s¢   ø€ € € € € ðð ðD ØØØØ"ð(ð (àð(ð !ð(ð ð	(ð
 ð(ð ð(ð ð(ð ð(ð (ð (ð (ð (ð (ð (ð (ð (ð (r    r  c                   ó   — e Zd ZdZdd„Zd„ Zdefd„Zd„ Zdd	e	de	fd
„Z
dd	e	dz  de	fd„Zd„ Zdej        fd„Zdefd„Zde	fd„Zde	fd„Zdej        fd„Zde	d	e	dee	e	f         fd„Zed„ ¦   «         Zedefd„¦   «         Zd„ Zdd	e	de	fd„ZdS ) ÚEncoderDecoderCachea­  
    Base, abstract class for all encoder-decoder caches. Can be used to hold combinations of self-attention and
    cross-attention caches.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        caches (`Iterable`):
            Usually an iterable of length 2, containing 2 `Cache` objects, the first one for self-attention, the
            second one for cross-attention. Can optionally also be an iterable of length 1, containing a
            `tuple[tuple[torch.Tensor]]` (usually used for compatibility with torch dp and ddp).

    Example:

    ```python
    >>> from transformers import AutoProcessor, AutoModelForCausalLM, DynamicCache, EncoderDecoderCache

    >>> model = AutoModelForCausalLM.from_pretrained("openai/whisper-small")
    >>> processor = AutoProcessor.from_pretrained("openai/whisper-small")

    >>> inputs = processor(audio=YOUR-AUDIO, return_tensors="pt")

    >>> # Prepare cache classes for encoder and decoder and pass it to model's forward
    >>> self_attention_cache = DynamicCache(config=self.config)
    >>> cross_attention_cache = DynamicCache(config=self.config)
    >>> past_key_values = EncoderDecoderCache(self_attention_cache, cross_attention_cache)
    >>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
    >>> outputs.past_key_values # access cache filled with key/values from generation
    EncoderDecoderCache()
    ```
    r1   Nc           	      óN  — t          |¦  «        dk    rõg g }}|d         D ]¿}t          |¦  «        dk    r;|                     |d d…         ¦  «         |                     |dd …         ¦  «         ŒPt          |¦  «        dk    r;|                     |d d…         ¦  «         |                     |dd …         ¦  «         Œžt          dt          |¦  «        ›d|›�¦  «        ‚t          |¦  «        | _        t          |¦  «        | _        n¾t          |¦  «        dk    rŒt          |d         t          ¦  «        rt          |d         t          ¦  «        s;t          d	t          |d         ¦  «        ›d
t          |d         ¦  «        ›�¦  «        ‚|d         | _        |d         | _        nt          dt          |¦  «        › �¦  «        ‚i | _
        t          t          | j        ¦  «        ¦  «        D ]5}t          | j                             |¦  «        dk    ¦  «        | j
        |<   Œ6d S )Nr   r   é   r*  rú   rÊ   z$Expected len(combined_cache_data) = z% to be 4 or 6.
combined_cache_data = z;One of the two arguments is not a Cache: type(caches[0]) = z, type(caches[1]) = zExpected 1 or 2 arguments, got )ro  r¯  r­   r  Úself_attention_cacheÚcross_attention_cacherQ   r–  Ú	TypeErrorr¶  Ú
is_updatedr=  rð  r=   )r&   ÚcachesÚself_attention_cache_dataÚcross_attention_cache_dataÚcombined_cache_datar¤  s         r   r'   zEncoderDecoderCache.__init__h  sB  € åˆv‰;Œ;˜!ÒÐØDFÈÐ'AÐ%Ø'-¨a¤yð 	xð 	xÐ#ÝÐ*Ñ+Ô+¨qÒ0Ð0Ø-×4Ò4Ð5HÈÈ!ÈÔ5LÑMÔMÐMØ.×5Ò5Ð6IÈ!È"È"Ô6MÑNÔNÐNÐNåÐ,Ñ-Ô-°Ò2Ð2Ø-×4Ò4Ð5HÈÈ!ÈÔ5LÑMÔMÐMØ.×5Ò5Ð6IÈ!È"È"Ô6MÑNÔNÐNÐNå$Ð%vµÐ5HÑ1IÔ1IÐ%vÐ%vÐ^qÐ%vÐ%vÑwÔwÐwÝ(4Ð5NÑ(OÔ(OˆDÔ%Ý)5Ð6PÑ)QÔ)QˆDÔ&Ð&å�‰[Œ[˜AÒÐÝ˜f Qœi­Ñ/Ô/ð xµzÀ&ÈÄ)ÍUÑ7SÔ7Sð xÝÐ vÍDÐQWÐXYÔQZÉOÌOÐ vÐ vÕbfÐgmÐnoÔgpÑbqÔbqÐ vÐ vÑwÔwÐwØ(.¨q¬	ˆDÔ%Ø)/°¬ˆDÔ&Ð&õ ÐL½sÀ6¹{¼{ÐLÐLÑMÔMÐMàˆŒÝ�s 4Ô#=Ñ>Ô>Ñ?Ô?ð 	hð 	hˆIÝ)-¨dÔ.H×.WÒ.WÐXaÑ.bÔ.bÐefÒ.fÑ)gÔ)gˆDŒO˜IÑ&Ð&ð	hð 	hr    c              #   óX   K  — t          | j        | j        ¦  «        D ]\  }}||z   V — ŒdS )zuReturns tuples of style (self_attn_k, self_attn_v, self_attn_sliding, cross_attn_k, cross_attn_v, cross_attn_sliding)N)rª  r)  r*  )r&   Úself_attention_layerÚcross_attention_layers      r   r  zEncoderDecoderCache.__iter__†  sL   è è € å;>¸tÔ?XÐZ^ÔZtÑ;uÔ;uð 	?ð 	?Ñ7Ð Ð"7Ø&Ð)>Ñ>Ð>Ð>Ð>Ð>ð	?ð 	?r    c                 ó@   — | j         j        › d| j        › d| j        › d�S )Nz(self_attention_cache=z, cross_attention_cache=r¡  )r   r+   r)  r*  r,   s    r   r-   zEncoderDecoderCache.__repr__‹  s=   € àŒ~Ô&ð -ð -¸dÔ>Wð -ð -ØÔ)ð-ð -ð -ð	
r    c                 ó*   — t          | j        ¦  «        S )z®
        Support for backwards-compatible `past_key_values` length, e.g. `len(past_key_values)`. This value corresponds
        to the number of layers in the model.
        )ro  r)  r,   s    r   r£  zEncoderDecoderCache.__len__‘  s   € õ
 �4Ô,Ñ-Ô-Ð-r    r   r¤  c                 ó6   — | j                              |¦  «        S )zYReturns the sequence length of the cached states. A layer index can be optionally passed.)r)  r=   rÈ  s     r   r=   z"EncoderDecoderCache.get_seq_length˜  ó   € àÔ(×7Ò7¸	ÑBÔBÐBr    c                 ó6   — | j                              |¦  «        S )zKReturns the maximum sequence length (i.e. max capacity) of the cache object)r)  r@   rÈ  s     r   r@   z"EncoderDecoderCache.get_max_lengthœ  r7  r    c                 ó’   — | j                              ¦   «          | j                             ¦   «          | j        D ]}d| j        |<   Œd S r"   )r)  rS   r*  r,  rÈ  s     r   rS   zEncoderDecoderCache.reset   sV   € ØÔ!×'Ò'Ñ)Ô)Ð)ØÔ"×(Ò(Ñ*Ô*Ð*Øœð 	/ð 	/ˆIØ).ˆDŒO˜IÑ&Ð&ð	/ð 	/r    rT   c                 ón   — | j                              |¦  «         | j                             |¦  «         dS ru  )r)  rX   r*  rW   s     r   rX   z!EncoderDecoderCache.reorder_cache¦  s6   € àÔ!×/Ò/°Ñ9Ô9Ð9ØÔ"×0Ò0°Ñ:Ô:Ð:Ð:Ð:r    Úmethodc           	      óü   — t          | j        t          ¦  «        rt          | j        t          ¦  «        sGt	          d|› d| j                             ¦   «         › d| j                             ¦   «         › d�¦  «        ‚d S )Nr!  z)` is only defined for dynamic cache, got z" for the self attention cache and z for the cross attention cache.)rQ   r)  r  r*  r+  Ú__str__)r&   r;  s     r   Úcheck_dynamic_cachez'EncoderDecoderCache.check_dynamic_cache«  sŸ   € å�tÔ0µ,Ñ?Ô?ð	å˜4Ô5µ|ÑDÔDð	õ ðm�Fð mð mÀTÔE^×EfÒEfÑEhÔEhð mð mØ'+Ô'A×'IÒ'IÑ'KÔ'Kðmð mð mñô ð ð		ð 	r    Úmaximum_lengthc                 óx   — |                       | j        j        ¦  «         | j                             |¦  «         dS )zó
        Crop the past key values up to a new `maximum_length` in terms of tokens. `maximum_length` can also be
        negative to remove `maximum_length` tokens. This is used in assisted decoding and contrastive search (on the Hub).
        N)r>  r‰   r+   r)  )r&   r?  s     r   r‰   zEncoderDecoderCache.crop¶  s:   € ð
 	× Ò  ¤Ô!3Ñ4Ô4Ð4ØÔ!×&Ò& ~Ñ6Ô6Ð6Ð6Ð6r    rŠ   c                 ó¬   — |                       | j        j        ¦  «         | j                             |¦  «         | j                             |¦  «         dS )zaRepeat the cache `repeats` times in the batch dimension. Used in contrastive search (on the Hub).N)r>  rŽ   r+   r)  r*  r�   s     r   rŽ   z+EncoderDecoderCache.batch_repeat_interleave¾  sP   € à× Ò  Ô!=Ô!FÑGÔGÐGØÔ!×9Ò9¸'ÑBÔBÐBØÔ"×:Ò:¸7ÑCÔCÐCÐCÐCr    r�   c                 ó¬   — |                       | j        j        ¦  «         | j                             |¦  «         | j                             |¦  «         dS )zeOnly keep the `indices` in the batch dimension of the cache. Used in contrastive search (on the Hub).N)r>  r’   r+   r)  r*  r‘   s     r   r’   z(EncoderDecoderCache.batch_select_indicesÄ  sP   € à× Ò  Ô!:Ô!CÑDÔDÐDØÔ!×6Ò6°wÑ?Ô?Ð?ØÔ"×7Ò7¸Ñ@Ô@Ð@Ð@Ð@r    r9   c                 ó8   — | j                              ||¦  «        S r)   )r)  r;   rÑ  s      r   r;   z"EncoderDecoderCache.get_mask_sizesÊ  s   € ØÔ(×7Ò7¸ÀiÑPÔPÐPr    c                 ó   — | j         j        S r)   )r)  r“   r,   s    r   r“   zEncoderDecoderCache.is_slidingÍ  s   € àÔ(Ô3Ð3r    c                 ó   — | j         j        S r)   )r)  rb   r,   s    r   rb   z"EncoderDecoderCache.is_compileableÑ  s   € àÔ(Ô7Ð7r    c                 ó8   — | j                              ¦   «          d S r)   )r)  rU  r,   s    r   rU  z+EncoderDecoderCache.activate_past_recordingÕ  s   € ØÔ!×9Ò9Ñ;Ô;Ð;Ð;Ð;r    c                 ó`   — t                                d¦  «         |                      |¦  «        S rZ   ré  rÈ  s     r   r]   z'EncoderDecoderCache.get_max_cache_shapeØ  rë  r    r^   r[  r)   )r+   r_   r`   ra   r'   r  rd   r-   r£  rR   r=   r@   rS   rf   ri   rX   r>  r‰   rŽ   rg   r’   rh   r;   rñ  r“   rð  rb   rU  r]   r   r    r   r&  r&  G  s%  € € € € € ðð ð@hð hð hð hð<?ð ?ð ?ð

˜#ð 
ð 
ð 
ð 
ð.ð .ð .ðCð C¨ð C°Cð Cð Cð Cð CðCð C¨¨d©
ð C¸cð Cð Cð Cð Cð/ð /ð /ð; eÔ&6ð ;ð ;ð ;ð ;ð
¨#ð ð ð ð ð7 3ð 7ð 7ð 7ð 7ðD¨sð Dð Dð Dð DðA¨E¬Lð Að Að Að AðQ¨3ð Q¸3ð QÀ5ÈÈcÈÄ?ð Qð Qð Qð Qð ð4ð 4ñ „Xð4ð ð8 ð 8ð 8ð 8ñ „Xð8ð<ð <ð <ð.ð .¨Sð .¸ð .ð .ð .ð .ð .ð .r    r&  c                   óN   ‡ — e Zd Zddefˆ fd„Zdededeeef         fˆ fd„Zˆ xZS )ÚMtpCacher   r¤  c                 óV   •— |dz   }t          ¦   «                              |¦  «        |z   S ©Nr   )r   rÓ  )r&   r¤  Ú
mtp_offsetr   s      €r   rÓ  zMtpCache.get_query_offsetä  s)   ø€ à ‘]ˆ
Ý‰wŒw×'Ò'¨	Ñ2Ô2°ZÑ?Ð?r    r9   r1   c                 óf   •— |dz   }t          ¦   «                              ||¦  «        \  }}|||z   fS rK  )r   r;   )r&   r9   r¤  rL  r}   r|   r   s         €r   r;   zMtpCache.get_mask_sizesé  s:   ø€ Ø ‘]ˆ
Ý$™wœw×5Ò5°lÀIÑNÔNÑˆ	�9Ø˜) jÑ0Ð0Ð0r    r[  )r+   r_   r`   rR   rÓ  rh   r;   rj   rk   s   @r   rI  rI  ã  sŒ   ø€ € € € € ð@ð @¨#ð @ð @ð @ð @ð @ð @ð
1¨3ð 1¸3ð 1À5ÈÈcÈÄ?ð 1ð 1ð 1ð 1ð 1ð 1ð 1ð 1ð 1ð 1r    rI  )4Úabcr   r   Úcollections.abcr   rf   Úconfiguration_utilsr   Úutilsr   r	   r
   r   r   r   Úhqq.core.quantizer   r,  r�  Ú
get_loggerr+   r[   r   rm   r•   r¯   r   rÞ   rï   rù   r  r(  r8  r]  ri  rx  r  r‡  r   r   r–  rh   rï  rd   r;  r  r  r  r  r&  ÚSlidingWindowCacherI  r   r    r   ú<module>rU     s¿  ðØ #Ð #Ð #Ð #Ð #Ð #Ð #Ð #Ø $Ð $Ð $Ð $Ð $Ð $à €€€à 1Ð 1Ð 1Ð 1Ð 1Ð 1ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ÐÑÔð <Ø;Ð;Ð;Ð;Ð;Ð;à&?Ð&?ÀÐRVÐ&WÑ&WÔ&WÐ #ð 
ˆÔ	˜HÑ	%Ô	%€ðS%ð S%ð S%ð S%ð S%�cñ S%ô S%ð S%ðlK4ð K4ð K4ð K4ð K4�?ñ K4ô K4ð K4ð\N5ð N5ð N5ð N5ð N5 ñ N5ô N5ð N5ðbF@ð F@ð F@ð F@ð F@˜,ñ F@ô F@ð F@ðRg"ð g"ð g"ð g"ð g"�/ñ g"ô g"ð g"ðT|'ð |'ð |'ð |'ð |'˜{ñ |'ô |'ð |'ð~@3ð @3ð @3ð @3ð @3˜ñ @3ô @3ð @3ðFI&ð I&ð I&ð I&ð I&�\ñ I&ô I&ð I&ðX4$ð 4$ð 4$ð 4$ð 4$˜>ñ 4$ô 4$ð 4$ðn6ð 6ð 6ð 6ð 6˜ñ 6ô 6ð 6ðrgð gð gð gð g Sñ gô gð gðTX0ð X0ð X0ð X0ð X0Ð9ñ X0ô X0ð X0ðv$,ð $,ð $,ð $,ð $,Ð+?Àñ $,ô $,ð $,ðN9ð 9ð 9ð 9ð 9Ð4HÐJcñ 9ô 9ð 9ð>2ð 2ð 2ð 2ð 2Ð1EÀ{ñ 2ô 2ð 2ð@?ð ?ð ?ð ?ð ?Ð:NÐPhñ ?ô ?ð ?ð4 #à2Ø2ð !ØØ,à2ØCà!4ðð Ð ð$ "à1Ø1à ØØ,à8ØIà!3ðð Ð ð"hð hð hð hð hñ hô hð hðV!%Ð'7ð !%¸EÀ$ÀsÄ)ÈTÀ/Ô<Rð !%ð !%ð !%ð !%ðHY[ð Y[ð Y[ð Y[ð Y[�5ñ Y[ô Y[ð Y[ðx4rð 4rð 4rð 4rð 4r�%ñ 4rô 4rð 4rðn<(ð <(ð <(ð <(ð <(�Uñ <(ô <(ð <(ð~U.ð U.ð U.ð U.ð U.˜%ñ U.ô U.ð U.ðr !Ð ð	1ð 	1ð 	1ð 	1ð 	1ˆ|ñ 	1ô 	1ð 	1ð 	1ð 	1r    