Ë
    îÍ:j¼ ã                   óÀ  — d dl mZmZ d dlmZ d dlmZmZ d dlZddl	m
Z
 ddlmZmZmZmZmZ  e«       rd dlmZ  ed	d
¬«      Z ej*                  e«      Z G d„ de«      Z G d„ de«      Z G d„ de«      Z G d„ de«      Z G d„ de«      Z G d„ de«      Z G d„ de«      Z G d„ de«      Z G d„ d«      Z  G d„ de «      Z! G d „ d!e «      Z" G d"„ d#e «      Z# G d$„ d%e «      Z$ G d&„ d'e«      Z% G d(„ d)e«      Z& G d*„ d+e!«      Z' G d,„ d-e"«      Z( G d.„ d/e"«      Z) G d0„ d1e"«      Z* G d2„ d3e"«      Z+ G d4„ d5e"«      Z, G d6„ d7e#«      Z- G d8„ d9e#«      Z. G d:„ d;e «      Z/y)<é    )ÚABCÚabstractmethod)ÚIterable)ÚAnyÚOptionalNé   )ÚPretrainedConfig)Úis_hqq_availableÚis_quanto_greaterÚis_torch_greater_or_equalÚis_torchdynamo_compilingÚlogging)Ú	Quantizerz2.7T©Ú
accept_devc                   óv  — e Zd ZdZdZd„ Zd„ Zedej                  fd„«       Z
e	 ddej                  dej                  d	eeeef      d
eej                  ej                  f   fd„«       Zedej                  d
eeef   fd„«       Zed
efd„«       Zed
efd„«       Zd„ Zd„ Zdd„Zdej0                  d
dfd„Zy)ÚCacheLayerMixinz0Base, abstract class for a single layer's cache.Fc                 ó.   — d | _         d | _        d| _        y ©NF)ÚkeysÚvaluesÚis_initialized©Úselfs    úm/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/transformers/cache_utils.pyÚ__init__zCacheLayerMixin.__init__   s   € Ø,0ˆŒ	Ø.2ˆŒØ#ˆÕó    c                 ó0   — | j                   j                  › S ©N)Ú	__class__Ú__name__r   s    r   Ú__repr__zCacheLayerMixin.__repr__$   s   € Ø—.‘.×)Ñ)Ð*Ð+r   Ú
key_statesc                  ó   — y r   © ©r   r#   s     r   Úlazy_initializationz#CacheLayerMixin.lazy_initialization'   s   € Ø=@r   NÚvalue_statesÚcache_kwargsÚreturnc                  ó   — y r   r%   ©r   r#   r(   r)   s       r   ÚupdatezCacheLayerMixin.update*   s   € ð -0r   Úcache_positionc                  ó   — y r   r%   )r   r.   s     r   Úget_mask_sizeszCacheLayerMixin.get_mask_sizes/   s   € ØORr   c                  ó   — y r   r%   r   s    r   Úget_seq_lengthzCacheLayerMixin.get_seq_length2   ó   € Ø%(r   c                  ó   — y r   r%   r   s    r   Úget_max_cache_shapez#CacheLayerMixin.get_max_cache_shape5   s   € Ø*-r   c                 ó¦   — | j                   rE| j                  j                  dd¬«      | _        | j                  j                  dd¬«      | _        yy)z(Offload this layer's data to CPU device.ÚcpuT©Únon_blockingN)r   r   Útor   r   s    r   ÚoffloadzCacheLayerMixin.offload8   s@   € à×ÒØŸ	™	Ÿ™ U¸˜Ó>ˆDŒIØŸ+™+Ÿ.™.¨¸T˜.ÓBˆD�Kð r   c                 ó  — | j                   r}| j                  j                  | j                  k7  rY| j                  j                  | j                  d¬«      | _        | j                  j                  | j                  d¬«      | _        yyy)zcIn case of layer offloading, this allows to move the data back to the layer's device ahead of time.Tr8   N)r   r   Údevicer:   r   r   s    r   ÚprefetchzCacheLayerMixin.prefetch>   sa   € à×Ò 4§9¡9×#3Ñ#3°t·{±{Ò#BØŸ	™	Ÿ™ T§[¡[¸t˜ÓDˆDŒIØŸ+™+Ÿ.™.¨¯©À4˜.ÓHˆD�Kð $CÐr   c                 ó¬   — | j                   r4| j                  j                  «        | j                  j                  «        t	        | d«      rd| _        yy)z4Resets the cache values while preserving the objectsÚcumulative_lengthr   N)r   r   Úzero_r   Úhasattrr@   r   s    r   ÚresetzCacheLayerMixin.resetD   sA   € à×ÒØ�I‰I�O‰OÔØ�K‰K×ÑÔä�4Ð,Ô-Ø%&ˆDÕ"ð .r   Úbeam_idxc                 ó<  — | j                  «       dkD  r‰| j                  j                  d|j                  | j                  j                  «      «      | _        | j
                  j                  d|j                  | j
                  j                  «      «      | _        yy)z,Reorders this layer's cache for beam search.r   N)r2   r   Úindex_selectr:   r=   r   ©r   rD   s     r   Úreorder_cachezCacheLayerMixin.reorder_cacheM   sn   € à×ÑÓ  1Ò$ØŸ	™	×.Ñ.¨q°(·+±+¸d¿i¹i×>NÑ>NÓ2OÓPˆDŒIØŸ+™+×2Ñ2°1°h·k±kÀ$Ç+Á+×BTÑBTÓ6UÓVˆD�Kð %r   r   ©r*   N)r!   Ú
__module__Ú__qualname__Ú__doc__Úis_compileabler   r"   r   ÚtorchÚTensorr'   r   ÚdictÚstrr   Útupler-   Úintr0   r2   r5   r;   r>   rC   Ú
LongTensorrH   r%   r   r   r   r      s  „ Ù:à€Nò$ò
,ð Ø@¨e¯l©lÒ@ó Ø@ààmqñ0ØŸ,™,ð0Ø6;·l±lð0ØRZÐ[_Ð`cÐehÐ`hÑ[iÑRjð0à	ˆu�|‰|˜UŸ\™\Ð)Ñ	*ò0ó ð0ð ØR¨U¯\©\ÐR¸eÀCÈÀH¹oÒRó ØRàØ( Ò(ó Ø(àØ- SÒ-ó Ø-òCòIó'ðW e×&6Ñ&6ð W¸4ô Wr   r   c                   óD  — e Zd ZdZdZdej                  fd„Z	 ddej                  dej                  dee	e
ef      deej                  ej                  f   fd	„Zd
ej                  deeef   fd„Zdefd„Zdefd„Zdeddfd„Zdeddfd„Zdej                  ddfd„Zy)ÚDynamicLayerzà
    A cache layer that grows dynamically as more tokens are generated. This is the default for generative models.
    It stores the key and value states as tensors of shape `[batch_size, num_heads, seq_len, head_dim]`.
    Fr#   c                 ó  — |j                   |j                  c| _         | _        t        j                  g | j                   | j                  ¬«      | _        t        j                  g | j                   | j                  ¬«      | _        d| _        y )N©Údtyper=   T)rY   r=   rN   Útensorr   r   r   r&   s     r   r'   z DynamicLayer.lazy_initialization\   s^   € Ø",×"2Ñ"2°J×4EÑ4EÐˆŒ
�D”KÜ—L‘L ¨4¯:©:¸d¿k¹kÔJˆŒ	Ü—l‘l 2¨T¯Z©ZÀÇÁÔLˆŒØ"ˆÕr   Nr(   r)   r*   c                 ó  — | j                   s| j                  |«       t        j                  | j                  |gd¬«      | _        t        j                  | j
                  |gd¬«      | _        | j                  | j
                  fS )áÆ  
        Update the key and value caches in-place, and return the necessary keys and value states.

        Args:
            key_states (`torch.Tensor`): The new key states to cache.
            value_states (`torch.Tensor`): The new value states to cache.
            cache_kwargs (`dict[str, Any]`, *optional*): Additional arguments for the cache.

        Returns:
            tuple[`torch.Tensor`, `torch.Tensor`]: The key and value states.
        éþÿÿÿ©Údim)r   r'   rN   Úcatr   r   r,   s       r   r-   zDynamicLayer.updateb   sd   € ð$ ×"Ò"Ø×$Ñ$ ZÔ0ä—I‘I˜tŸy™y¨*Ð5¸2Ô>ˆŒ	Ü—i‘i §¡¨lÐ ;ÀÔDˆŒØ�y‰y˜$Ÿ+™+Ð%Ð%r   r.   c                 óR   — d}|j                   d   }| j                  «       |z   }||fS )zDReturn the length and offset of the cache, used to generate the maskr   )Úshaper2   )r   r.   Ú	kv_offsetÚquery_lengthÚ	kv_lengths        r   r0   zDynamicLayer.get_mask_sizes{   s5   € àˆ	Ø%×+Ñ+¨AÑ.ˆØ×'Ñ'Ó)¨LÑ8ˆ	Ø˜)Ð#Ð#r   c                 óˆ   — | j                   r| j                  j                  «       dk(  ry| j                  j                  d   S )ú1Returns the sequence length of the cached states.r   r]   )r   r   Únumelrb   r   s    r   r2   zDynamicLayer.get_seq_length‚   s3   € à×"Ò" d§i¡i§o¡oÓ&7¸1Ò&<ØØ�y‰y�‰˜rÑ"Ð"r   c                  ó   — y)zeReturns the maximum sequence length of the cache object. DynamicLayer does not have a maximum length.éÿÿÿÿr%   r   s    r   r5   z DynamicLayer.get_max_cache_shapeˆ   s   € àr   Ú
max_lengthc                 óÚ   — |dk  r| j                  «       t        |«      z
  }| j                  «       |k  ry| j                  dd|…dd…f   | _        | j                  dd|…dd…f   | _        y)z 
        Crop the past key values up to a new `max_length` in terms of tokens. `max_length` can also be negative
        to remove `max_length` tokens.
        r   N.)r2   Úabsr   r   )r   rk   s     r   ÚcropzDynamicLayer.cropŒ   sl   € ð
 ˜Š>Ø×,Ñ,Ó.´°Z³Ñ@ˆJà×ÑÓ  JÒ.Øà—I‘I˜c ; J ;²Ð1Ñ2ˆŒ	Ø—k‘k # {¨
 {²AÐ"5Ñ6ˆ�r   Úrepeatsc                 ó´   — | j                  «       dkD  rE| j                  j                  |d¬«      | _        | j                  j                  |d¬«      | _        yy)z8Repeat the cache `repeats` times in the batch dimension.r   r^   N)r2   r   Úrepeat_interleaver   ©r   ro   s     r   Úbatch_repeat_interleavez$DynamicLayer.batch_repeat_interleaveš   sN   € à×ÑÓ  1Ò$ØŸ	™	×3Ñ3°GÀÐ3ÓCˆDŒIØŸ+™+×7Ñ7¸ÀQÐ7ÓGˆD�Kð %r   Úindicesc                 ó„   — | j                  «       dkD  r-| j                  |df   | _        | j                  |df   | _        yy)z<Only keep the `indices` in the batch dimension of the cache.r   .N)r2   r   r   ©r   rt   s     r   Úbatch_select_indicesz!DynamicLayer.batch_select_indices    s@   € à×ÑÓ  1Ò$ØŸ	™	 '¨3 ,Ñ/ˆDŒIØŸ+™+ g¨s lÑ3ˆD�Kð %r   r   )r!   rJ   rK   rL   Ú
is_slidingrN   rO   r'   r   rP   rQ   r   rR   r-   rS   r0   r2   r5   rn   rs   rw   r%   r   r   rV   rV   T   sì   „ ñð
 €Jð#¨e¯l©ló #ð 26ñ	&à—L‘Lð&ð —l‘lð&ð ˜t C¨ H™~Ñ.ð	&ð
 
ˆu�|‰|˜UŸ\™\Ð)Ñ	*ó&ð2$¨U¯\©\ð $¸eÀCÈÀH¹oó $ð# ó #ð Só ð7˜sð 7 tó 7ðH¨sð H°tó Hð4¨E¯L©Lð 4¸Tô 4r   rV   c                   ó  ‡ — e Zd ZdZdZdefˆ fd„Z	 ddej                  dej                  de	e
eef      d	eej                  ej                  f   fd
„Zdej                  d	eeef   fd„Zd	efd„Zd	efd„Zded	dfˆ fd„Zˆ xZS )ÚDynamicSlidingWindowLayerzì
    A cache layer that grows dynamically as more tokens are generated, up until the sliding window size.
    It stores the key and value states as tensors of shape `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.
    TÚsliding_windowc                 ó>   •— t         ‰| �  «        || _        d| _        y ©Nr   )Úsuperr   r{   r@   )r   r{   r    s     €r   r   z"DynamicSlidingWindowLayer.__init__¯   s   ø€ Ü‰ÑÔØ,ˆÔØ!"ˆÕr   Nr#   r(   r)   r*   c                 ó¤  — | j                   s| j                  |«       | xj                  |j                  d   z  c_        t	        j
                  | j                  |gd¬«      }t	        j
                  | j                  |gd¬«      }|dd…dd…| j                   dz   d…dd…f   | _        |dd…dd…| j                   dz   d…dd…f   | _        ||fS )r\   r]   r^   Nr   )	r   r'   r@   rb   rN   r`   r   r   r{   )r   r#   r(   r)   Úfull_key_statesÚfull_value_statess         r   r-   z DynamicSlidingWindowLayer.update´   sÆ   € ð$ ×"Ò"Ø×$Ñ$ ZÔ0à×Ò *×"2Ñ"2°2Ñ"6Ñ6Õô  Ÿ)™) T§Y¡Y°
Ð$;ÀÔDˆÜ!ŸI™I t§{¡{°LÐ&AÀrÔJÐà#¢A¢q¨4×+>Ñ+>Ð*>ÀÑ*BÑ*DÂaÐ$GÑHˆŒ	Ø'ªª1¨t×/BÑ/BÐ.BÀQÑ.FÑ.HÊ!Ð(KÑLˆŒð Ð 1Ð1Ð1r   r.   c                 óô   — |j                   d   }| j                  | j                  k\  }t        | j                  | j                  z
  dz   d«      }|r| j                  dz
  |z   }||fS | j                  |z   }||fS ©úNReturn the length and offset of the cache, used to generate the attention maskr   r   )rb   r@   r{   Úmax)r   r.   rd   Úis_fullrc   re   s         r   r0   z(DynamicSlidingWindowLayer.get_mask_sizesÕ   sŒ   € à%×+Ñ+¨AÑ.ˆØ×(Ñ(¨D×,?Ñ,?Ñ?ˆä˜×.Ñ.°×1DÑ1DÑDÀqÑHÈ!ÓLˆ	ÙØ×+Ñ+¨aÑ/°,Ñ>ˆIð ˜)Ð#Ð#ð ×.Ñ.°Ñ=ˆIà˜)Ð#Ð#r   c                 ó   — | j                   S ©rg   ©r@   r   s    r   r2   z(DynamicSlidingWindowLayer.get_seq_lengthâ   ó   € à×%Ñ%Ð%r   c                 ó   — | j                   S ©z+Return the maximum cache shape of the cache©r{   r   s    r   r5   z-DynamicSlidingWindowLayer.get_max_cache_shapeæ   s   € à×"Ñ"Ð"r   rk   c                 ó°   •— | j                  «       | j                  k\  rt        d«      ‚t        ‰| �  |«       | j
                  j                  d   | _        y)z 
        Crop the past key values up to a new `max_length` in terms of tokens. `max_length` can also be
        negative to remove `max_length` tokens.
        z�Cannot `crop` a `DynamicSlidingWindowLayer` after it has seen more tokens than itssliding window (otherwise some states are lost)r]   N)r2   r{   Ú
ValueErrorr~   rn   r   rb   r@   )r   rk   r    s     €r   rn   zDynamicSlidingWindowLayer.cropê   sR   ø€ ð
 ×ÑÓ  D×$7Ñ$7Ò7ÜðBóð ô 	‰‰�ZÔ Ø!%§¡§¡°Ñ!4ˆÕr   r   )r!   rJ   rK   rL   rx   rS   r   rN   rO   r   rP   rQ   r   rR   r-   r0   r2   r5   rn   Ú__classcell__©r    s   @r   rz   rz   §   sÂ   ø„ ñð
 €Jð# sõ #ð 26ñ	2à—L‘Lð2ð —l‘lð2ð ˜t C¨ H™~Ñ.ð	2ð
 
ˆu�|‰|˜UŸ\™\Ð)Ñ	*ó2ðB$¨U¯\©\ð $¸eÀCÈÀH¹oó $ð& ó &ð# Só #ð5˜sð 5 t÷ 5ñ 5r   rz   c                   ó  ‡ — e Zd ZdZdZdZdefˆ fd„Zdej                  fd„Z
	 ddej                  dej                  d	eeeef      d
eej                  ej                  f   fd„Zdej                  d
eeef   fd„Zd
efd„Zd
efd„Zˆ xZS )ÚStaticLayeraŠ  
    A static cache layer that stores the key and value states as static tensors of shape `[batch_size, num_heads, max_cache_len), head_dim]`.
    It lazily allocates its full backing tensors, and then mutates them in-place. Built for `torch.compile` support.

    Args:
        max_cache_len (`int`):
            Maximum number of tokens that can be stored, used for tensor preallocation.
    TFÚmax_cache_lenc                 ó0   •— t         ‰| �  «        || _        y r   )r~   r   r”   )r   r”   r    s     €r   r   zStaticLayer.__init__  s   ø€ Ü‰ÑÔØ*ˆÕr   r#   c                 óÄ  — |j                   \  | _        | _        }| _        |j                  |j
                  c| _        | _        t        j                  | j                  | j                  | j                  | j                  f| j                  | j
                  ¬«      | _	        t        j                  | j                  | j                  | j                  | j                  f| j                  | j
                  ¬«      | _
        t        «       sRt        j                  j                  | j                  «       t        j                  j                  | j                  «       d| _        y)a6  
        Lazy initialization of the keys and values tensors. This allows to get all properties (dtype, device,
        num_heads in case of TP etc...) at runtime directly, which is extremely practical as it avoids moving
        devices, dtypes etc later on for each `update` (which could break the static dynamo addresses as well).

        If this is unwanted, one can call `early_initialization(...)` on the Cache directly, which will call this
        function ahead-of-time (this is required for `torch.export` for example). Note that for `compile`, as we
        internally don't compile the prefill, this is guaranteed to have been called already when compiling.
        If compiling the prefill as well, e.g. calling `model.compile(...)` before `generate` with a static cache,
        it is still supported in general, but without guarantees depending on the compilation options (e.g. cuda graphs,
        i.e. `mode="reduce-overhead"` is known to fail). But it will in general work correctly, and prefill should
        not be compiled anyway for performances!
        rX   TN)rb   Úmax_batch_sizeÚ	num_headsÚhead_dimrY   r=   rN   Úzerosr”   r   r   r   Ú_dynamoÚmark_static_addressr   )r   r#   Ú_s      r   r'   zStaticLayer.lazy_initialization	  s÷   € ð AK×@PÑ@PÑ=ˆÔ˜Tœ^¨Q°´Ø",×"2Ñ"2°J×4EÑ4EÐˆŒ
�D”Kä—K‘KØ× Ñ  $§.¡.°$×2DÑ2DÀdÇmÁmÐTØ—*‘*Ø—;‘;ô
ˆŒ	ô
 —k‘kØ× Ñ  $§.¡.°$×2DÑ2DÀdÇmÁmÐTØ—*‘*Ø—;‘;ô
ˆŒô (Ô)Ü�M‰M×-Ñ-¨d¯i©iÔ8Ü�M‰M×-Ñ-¨d¯k©kÔ:à"ˆÕr   r(   r)   r*   c                 óæ  — | j                   s| j                  |«       |�|j                  d«      nd}|�|n-t        j                  |j
                  d   | j                  ¬«      }	 | j                  j                  d||«       | j                  j                  d||«       | j                  | j                  fS # t        $ r/ || j                  dd…dd…|f<   || j                  dd…dd…|f<   Y ŒOw xY w)r\   Nr.   r]   ©r=   é   )r   r'   ÚgetrN   Úarangerb   r=   r   Úindex_copy_r   ÚNotImplementedError)r   r#   r(   r)   r.   s        r   r-   zStaticLayer.update.  sæ   € ð$ ×"Ò"Ø×$Ñ$ ZÔ0ð @LÐ?W˜×)Ñ)Ð*:Ô;Ð]aˆà,Ð8‰N¼e¿l¹lÈ:×K[ÑK[Ð\^ÑK_Ðhl×hsÑhsÔ>tð 	ð
	=Ø�I‰I×!Ñ! ! ^°ZÔ@Ø�K‰K×#Ñ# A ~°|ÔDð
 �y‰y˜$Ÿ+™+Ð%Ð%øô	 #ò 	=à.8ˆD�I‰I’aš˜NÐ*Ñ+Ø0<ˆD�K‰Kšš1˜nÐ,Ó-ð	=ús   Á&:B8 Â85C0Ã/C0r.   c                 ó&   — d}| j                   }||fS )r„   r   ©r”   )r   r.   rc   re   s       r   r0   zStaticLayer.get_mask_sizesT  s   € àˆ	Ø×&Ñ&ˆ	Ø˜)Ð#Ð#r   c                 óx   — | j                   r-| j                  d   j                  d¬«      j                  «       S dS )rg   )r   r   rj   r^   r   )r   r   ÚanyÚsumr   s    r   r2   zStaticLayer.get_seq_lengthZ  s6   € ð 7;×6IÒ6I�—	‘	˜$‘×#Ñ#¨Ð#Ó+×0Ñ0Ó2ÐPÈqÐPr   c                 ó   — | j                   S rŒ   r¦   r   s    r   r5   zStaticLayer.get_max_cache_shape`  s   € à×!Ñ!Ð!r   r   )r!   rJ   rK   rL   rM   rx   rS   r   rN   rO   r'   r   rP   rQ   r   rR   r-   r0   r2   r5   r�   r‘   s   @r   r“   r“   ø   sÂ   ø„ ñð €NØ€Jð+ cõ +ð##¨e¯l©ló ##ðR 26ñ	$&à—L‘Lð$&ð —l‘lð$&ð ˜t C¨ H™~Ñ.ð	$&ð
 
ˆu�|‰|˜UŸ\™\Ð)Ñ	*ó$&ðL$¨U¯\©\ð $¸eÀCÈÀH¹oó $ðQ ó Qð" S÷ "r   r“   c                   óð   ‡ — e Zd ZdZdZdedefˆ fd„Z	 ddej                  dej                  de	e
eef      d	eej                  ej                  f   fd
„Zdej                  d	eeef   fd„Zd	efd„Zˆ xZS )ÚStaticSlidingWindowLayeraî  
    A static cache layer that stores the key and value states as static tensors of shape
    `[batch_size, num_heads, min(max_cache_len, sliding_window), head_dim]`. It lazily allocates its full backing
    tensors, and then mutates them in-place. Built for `torch.compile` support.

    Args:
        max_cache_len (`int`):
            Maximum number of tokens that can be stored, used for tensor preallocation.
        sliding_window (`int`):
            The size of the sliding window.
    Tr”   r{   c                 óL   •— t        ||«      }t        ‰| �	  |¬«       d| _        y )Nr¦   r   )Úminr~   r   r@   )r   r”   r{   Úeffective_max_cache_lenr    s       €r   r   z!StaticSlidingWindowLayer.__init__t  s)   ø€ Ü"% n°mÓ"DÐÜ‰ÑÐ'>ÐÔ?Ø!"ˆÕr   r#   r(   r)   r*   c                 óÚ  — | j                   s| j                  |«       |�|j                  d«      nd}|�|n-t        j                  |j
                  d   | j                  ¬«      }| j                  }|| j                  k\  }| xj                  |j
                  d   z  c_        |�r>|j
                  d   dk(  rÇ| j                  j                  dd¬«      }| j                  j                  dd¬«      }t        j                  dgt        | j                  ¬«      }	||dd…dd…|	f<   ||dd…dd…|	f<   | j                  j                  |«       | j                  j                  |«       | j                  | j                  fS t        j                  | j                  dd…dd…dd…dd…f   |fd¬	«      }
t        j                  | j                  dd…dd…dd…dd…f   |fd¬	«      }ná||j
                  d
   z   | j                  kD  ro|dk(  r|}
|}n¸t        j                  | j                  dd…dd…d|…dd…f   |fd¬	«      }
t        j                  | j                  dd…dd…d|…dd…f   |fd¬	«      }nS	 | j                  j!                  d
||«       | j                  j!                  d
||«       | j                  | j                  fS | j                  j                  |
dd…dd…| j                   d…dd…f   «       | j                  j                  |dd…dd…| j                   d…dd…f   «       |
|fS # t"        $ r/ || j                  dd…dd…|f<   || j                  dd…dd…|f<   Y Œ½w xY w)r\   Nr.   r]   rŸ   r   rj   )ÚdimsrX   r^   r    r   )r   r'   r¡   rN   r¢   rb   r=   r@   r”   r   Úrollr   rZ   rS   Úcopy_r`   r£   r¤   )r   r#   r(   r)   r.   r@   r†   Únew_keysÚ
new_valuesÚindexr€   r�   s               r   r-   zStaticSlidingWindowLayer.updatey  s)  € ð$ ×"Ò"Ø×$Ñ$ ZÔ0ð @LÐ?W˜×)Ñ)Ð*:Ô;Ð]aˆà,Ð8‰N¼e¿l¹lÈ:×K[ÑK[Ð\^ÑK_Ðhl×hsÑhsÔ>tð 	ð !×2Ñ2ÐØ# t×'9Ñ'9Ñ9ˆà×Ò *×"2Ñ"2°2Ñ"6Ñ6Õâð ×Ñ Ñ# qÒ(àŸ9™9Ÿ>™>¨"°2˜>Ó6�Ø!Ÿ[™[×-Ñ-¨b°rÐ-Ó:�
ô Ÿ™ b T´¸T¿[¹[ÔI�Ø(2�ššA˜u˜Ñ%Ø*6�
š1ša ˜;Ñ'ð —	‘	—‘ Ô)Ø—‘×!Ñ! *Ô-à—y‘y $§+¡+Ð-Ð-ô #(§)¡)¨T¯Y©Y²qº!¸Q¹RÂ°{Ñ-CÀZÐ,PÐVXÔ"Y�Ü$)§I¡I¨t¯{©{º1ºaÀÁÂQ¸;Ñ/GÈÐ.VÐ\^Ô$_Ñ!à ×!1Ñ!1°!Ñ!4Ñ4°t×7IÑ7IÒIà  AÒ%Ø",�Ø$0Ñ!ä"'§)¡)¨T¯Y©Y²qº!Ð=OÐ>OÐ=OÒQRÐ7RÑ-SÐU_Ð,`ÐfhÔ"i�Ü$)§I¡I¨t¯{©{º1ºaÐASÐBSÐASÒUVÐ;VÑ/WÐYeÐ.fÐlnÔ$oÑ!ðAØ—	‘	×%Ñ% a¨¸ÔDØ—‘×'Ñ'¨¨>¸<ÔHð —9‘9˜dŸk™kÐ)Ð)ð 	�	‰	�‰˜ªª1¨t×/AÑ/AÐ.AÑ.CÂQÐ(FÑGÔHØ�‰×ÑÐ+ªAªq°4×3EÑ3EÐ2EÑ2GÊÐ,JÑKÔLàÐ 1Ð1Ð1øô 'ò AØ2<�—	‘	š!šQ Ð.Ñ/Ø4@�—‘šAšq .Ð0Ó1ðAús   É2:L2 Ì25M*Í)M*r.   c                 ó  — |j                   d   }| j                  }| j                  | j                  k\  }t        | j                  |z
  dz   d«      }|r||z   dz
  }||fS | j                  |z   |kD  r| j                  |z   }||fS |}||fS rƒ   )rb   r”   r@   r…   )r   r.   rd   r{   r†   rc   re   s          r   r0   z'StaticSlidingWindowLayer.get_mask_sizesÊ  s²   € à%×+Ñ+¨AÑ.ˆØ×+Ñ+ˆØ×(Ñ(¨D×,>Ñ,>Ñ>ˆä˜×.Ñ.°Ñ?À!ÑCÀQÓGˆ	áØ&¨Ñ5¸Ñ9ˆIð ˜)Ð#Ð#ð ×#Ñ# lÑ2°^ÒCØ×.Ñ.°Ñ=ˆIð
 ˜)Ð#Ð#ð 'ˆIà˜)Ð#Ð#r   c                 ó   — | j                   S rˆ   r‰   r   s    r   r2   z'StaticSlidingWindowLayer.get_seq_lengthÝ  rŠ   r   r   )r!   rJ   rK   rL   rx   rS   r   rN   rO   r   rP   rQ   r   rR   r-   r0   r2   r�   r‘   s   @r   r¬   r¬   e  sª   ø„ ñ
ð €Jð# cð #¸3õ #ð 26ñ	O2à—L‘LðO2ð —l‘lðO2ð ˜t C¨ H™~Ñ.ð	O2ð
 
ˆu�|‰|˜UŸ\™\Ð)Ñ	*óO2ðb$¨U¯\©\ð $¸eÀCÈÀH¹oó $ð&& ÷ &r   r¬   c                   óö   ‡ — e Zd ZdZ	 	 	 	 	 ddededededef
ˆ fd„Z	 ddej                  d	ej                  d
ee	e
ef      deej                  ej                  f   fd„Zed„ «       Zed„ «       Zdefd„Zˆ xZS )ÚQuantizedLayera  
    A quantized layer similar to what is described in the [KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache paper](https://huggingface.co/papers/2402.02750).
    It allows the model to generate longer sequence length without allocating too much memory for the key and value caches by
    applying quantization.

    The cache has two types of storage, one for original precision and one for the quantized cache. A `residual length`
    is set as a maximum capacity for the original precision cache. When the length goes beyond maximum capacity, the original
    precision cache is discarded and moved into the quantized cache. The quantization is done per-channel with a set `q_group_size`
    for both Keys and Values, in contrast to what was described in the paper.
    ÚnbitsÚaxis_keyÚ
axis_valueÚq_group_sizeÚresidual_lengthc                 óv   •— t         ‰| �  «        || _        || _        || _        || _        || _        d| _        y r}   )r~   r   r»   r¼   r½   r¾   r¿   r@   ©r   r»   r¼   r½   r¾   r¿   r    s         €r   r   zQuantizedLayer.__init__î  s=   ø€ ô 	‰ÑÔØˆŒ
Ø ˆŒØ$ˆŒØ(ˆÔØ.ˆÔØ!"ˆÕr   r#   r(   r)   r*   c                 ó  — | xj                   |j                  d   z  c_         | j                  su| j                  |«       | j	                  |j                  «       | j                  ¬«      | _        | j	                  |j                  «       | j                  ¬«      | _	        ||fS | j                  | j                  «      }| j                  | j                  «      }t        j                  || j                  |gd¬«      }t        j                  || j                  |gd¬«      }| j                  j                  «       dk(  rï| j                  j                  d   dz   | j                   k\  rÆ| j	                  |j                  «       | j                  ¬«      | _        | j	                  |j                  «       | j                  ¬«      | _	        t        j"                  g |j$                  |j&                  ¬«      | _        t        j"                  g |j$                  |j&                  ¬«      | _        ||fS t        j                  | j                  |gd¬«      | _        t        j                  | j                  |gd¬«      | _        ||fS )r\   r]   )Úaxisr^   é   r   rX   )r@   rb   r   r'   Ú	_quantizeÚ
contiguousr¼   Ú_quantized_keysr½   Ú_quantized_valuesÚ_dequantizerN   r`   r   r   r_   r¿   rZ   rY   r=   )r   r#   r(   r)   Údequant_keysÚdequant_valuesÚkeys_to_returnÚvalues_to_returns           r   r-   zQuantizedLayer.updateþ  s   € ð" 	×Ò *×"2Ñ"2°2Ñ"6Ñ6Õð ×"Ò"Ø×$Ñ$ ZÔ0Ø#'§>¡>°*×2GÑ2GÓ2IÐPT×P]ÑP] >Ó#^ˆDÔ Ø%)§^¡^°L×4KÑ4KÓ4MÐTX×TcÑTc ^Ó%dˆDÔ"Ø˜|Ð+Ð+à×'Ñ'¨×(<Ñ(<Ó=ˆØ×)Ñ)¨$×*@Ñ*@ÓAˆÜŸ™ L°$·)±)¸ZÐ#HÈbÔQˆÜ Ÿ9™9 n°d·k±kÀ<Ð%PÐVXÔYÐØ�9‰9�=‰=‹?˜aÒ D§I¡I§O¡O°BÑ$7¸!Ñ$;¸t×?SÑ?SÒ$SØ#'§>¡>°.×2KÑ2KÓ2MÐTX×TaÑTa >Ó#bˆDÔ Ø%)§^¡^Ð4D×4OÑ4OÓ4QÐX\×XgÑXg ^Ó%hˆDÔ"ÜŸ™ R¨z×/?Ñ/?È
×HYÑHYÔZˆDŒIÜŸ,™, r°×1AÑ1AÈ*×J[ÑJ[Ô\ˆDŒKð
 Ð/Ð/Ð/ô Ÿ	™	 4§9¡9¨jÐ"9¸rÔBˆDŒIÜŸ)™) T§[¡[°,Ð$?ÀRÔHˆDŒKàÐ/Ð/Ð/r   c                  ó   — y r   r%   )r   rZ   rÃ   s      r   rÅ   zQuantizedLayer._quantize'  s   € Ø'*r   c                  ó   — y r   r%   )r   Úq_tensors     r   rÉ   zQuantizedLayer._dequantize*  r3   r   c                 ó   — | j                   S rˆ   r‰   r   s    r   r2   zQuantizedLayer.get_seq_length-  rŠ   r   ©rÄ   r   r   é@   é€   r   )r!   rJ   rK   rL   rS   r   rN   rO   r   rP   rQ   r   rR   r-   r   rÅ   rÉ   r2   r�   r‘   s   @r   rº   rº   â  sÐ   ø„ ñ	ð ØØØØ"ñ#àð#ð ð#ð ð	#ð
 ð#ð õ#ð( 26ñ	'0à—L‘Lð'0ð —l‘lð'0ð ˜t C¨ H™~Ñ.ð	'0ð
 
ˆu�|‰|˜UŸ\™\Ð)Ñ	*ó'0ðR Ù*ó Ø*àÙ(ó Ø(ð& ÷ &r   rº   c                   óL   ‡ — e Zd Z	 	 	 	 	 d	dededededef
ˆ fd„Zd„ Zd„ Zˆ xZS )
ÚQuantoQuantizedLayerr»   r¼   r½   r¾   r¿   c                 ó   •— t         ‰	| �  |||||¬«       t        dd¬«      rddlm}m}m} nt        d«      ‚| j                  dvrt        d	| j                  › �«      ‚| j                  d
vrt        d| j                  › �«      ‚| j                  d
vrt        d| j                  › �«      ‚| j                  dk(  r|n|| _         |«       | _        y )N©r»   r¼   r½   r¾   r¿   z0.2.5Tr   r   )ÚMaxOptimizerÚqint2Úqint4ziYou need optimum-quanto package version to be greater or equal than 0.2.5 to use `QuantoQuantizedCache`. )r    rÄ   zA`nbits` for `quanto` backend has to be one of [`2`, `4`] but got )r   rj   zE`axis_key` for `quanto` backend has to be one of [`0`, `-1`] but got zG`axis_value` for `quanto` backend has to be one of [`0`, `-1`] but got rÄ   )r~   r   r   Úoptimum.quantorÙ   rÚ   rÛ   ÚImportErrorr»   r�   r¼   r½   ÚqtypeÚ	optimizer)
r   r»   r¼   r½   r¾   r¿   rÙ   rÚ   rÛ   r    s
            €r   r   zQuantoQuantizedLayer.__init__3  sá   ø€ ô 	‰ÑØØØ!Ø%Ø+ð 	ô 	
ô ˜W°Õ6ßAÒAäØ{óð ð �:‰:˜VÑ#ÜÐ`Ðae×akÑakÐ`lÐmÓnÐnà�=‰= Ñ'ÜÐdÐei×erÑerÐdsÐtÓuÐuà�?‰? 'Ñ)ÜØYÐZ^×ZiÑZiÐYjÐkóð ð #Ÿj™j¨Ašo‘U°5ˆŒ
Ù%›ˆ�r   c                 óª   — ddl m} | j                  || j                  || j                  «      \  }} ||| j                  |||| j                  «      }|S )Nr   )Úquantize_weight)rÜ   rá   rß   rÞ   r¾   )r   rZ   rÃ   rá   ÚscaleÚ	zeropointÚqtensors          r   rÅ   zQuantoQuantizedLayer._quantizeY  sK   € Ý2àŸ>™>¨&°$·*±*¸dÀD×DUÑDUÓVÑˆˆyÙ! &¨$¯*©*°d¸EÀ9Èd×N_ÑN_Ó`ˆØˆr   c                 ó"   — |j                  «       S r   )Ú
dequantize)r   rä   s     r   rÉ   z QuantoQuantizedLayer._dequantize`  s   € Ø×!Ñ!Ó#Ð#r   rÒ   ©r!   rJ   rK   rS   r   rÅ   rÉ   r�   r‘   s   @r   rÖ   rÖ   2  sT   ø„ ð ØØØØ"ñ$(àð$(ð ð$(ð ð	$(ð
 ð$(ð õ$(òLö$r   rÖ   c                   óL   ‡ — e Zd Z	 	 	 	 	 d	dededededef
ˆ fd„Zd„ Zd„ Zˆ xZS )
ÚHQQQuantizedLayerr»   r¼   r½   r¾   r¿   c                 óR  •— t         ‰| �  |||||¬«       t        «       st        d«      ‚| j                  dvrt        d| j                  › �«      ‚| j                  dvrt        d| j                  › �«      ‚| j                  dvrt        d| j                  › �«      ‚t        | _	        y )NrØ   z4You need to install `hqq` to use `HQQQuantizedLayer`)r   r    é   rÄ   é   zM`nbits` for `HQQ` backend has to be one of [`1`, `2`, `3`, `4`, `8`] but got )r   r   zA`axis_key` for `HQQ` backend has to be one of [`0`, `1`] but got zC`axis_value` for `HQQ` backend has to be one of [`0`, `1`] but got )
r~   r   r
   rÝ   r»   r�   r¼   r½   ÚHQQQuantizerÚ	quantizerrÁ   s         €r   r   zHQQQuantizedLayer.__init__e  s¼   ø€ ô 	‰ÑØØØ!Ø%Ø+ð 	ô 	
ô  Ô!ÜÐTÓUÐUà�:‰:˜_Ñ,ÜØ_Ð`d×`jÑ`jÐ_kÐlóð ð �=‰= Ñ&ÜÐ`Ðae×anÑanÐ`oÐpÓqÐqà�?‰? &Ñ(ÜÐbÐcg×crÑcrÐbsÐtÓuÐuä%ˆ�r   c                 óä  — | j                   j                  ||| j                  j                  | j                  j                  | j
                  | j                  ¬«      \  }}| j                  j                  |d<   | j                   j                  ||| j                  j                  ¬«       |d   j                  |j                  «      |d<   |d   j                  |j                  «      |d<   ||fS )N)rÃ   r=   Úcompute_dtyper»   Ú
group_sizerð   )Úmetar=   râ   Úzero)	rî   Úquantizer   r=   rY   r»   r¾   Úcudar:   )r   rZ   rÃ   rä   rò   s        r   rÅ   zHQQQuantizedLayer._quantize…  sÄ   € ØŸ™×/Ñ/ØØØ—9‘9×#Ñ#ØŸ)™)Ÿ/™/Ø—*‘*Ø×(Ñ(ð 0ó 
‰ˆ�ð !%§	¡	§¡ˆˆ_ÑØ�‰×Ñ˜G¨$°t·y±y×7GÑ7GÐÔHØ˜W™×(Ñ(¨¯©Ó8ˆˆW‰Ø˜F‘|—‘ w§~¡~Ó6ˆˆV‰Ø˜ˆ}Ðr   c                 óH   — |\  }}| j                   j                  ||«      }|S r   )rî   ræ   )r   rä   Úquant_tensorrò   rZ   s        r   rÉ   zHQQQuantizedLayer._dequantize”  s'   € Ø$Ñˆ�dØ—‘×*Ñ*¨<¸Ó>ˆØˆr   rÒ   rç   r‘   s   @r   ré   ré   d  sT   ø„ ð ØØØØ"ñ&àð&ð ð&ð ð	&ð
 ð&ð õ&ò@ör   ré   c                   ó¸  — e Zd ZdZ	 	 	 	 d-deee      deee      dedefd„Z	d„ Z
d.d	ed
efd„Zd.d	ed
efd„Z	 d/dej                  dej                  d	edeeeef      deej                  ej                  f   f
d„Zdedededej*                  dej,                  f
d„Zd0d	edefd„Zdej                  d	edeeef   fd„Zd0d	edefd„Zd„ Zdej8                  fd„Zdefd „Zd!efd"„Zd#ej                  fd$„Z e!defd%„«       Z"e!defd&„«       Z#e!defd'„«       Z$e!defd(„«       Z%e!dee   fd)„«       Z&d	edeej                  ej                  f   fd*„Z'd+„ Z(d,„ Z)y)1ÚCachean  
    A `Cache` is mostly a list of `CacheLayerMixin` objects, one per model layer. It serves as a container for
    the Cache of each layer.

    Args:
        layers (`Optional`, *optional*):
            A list of pre-created `CacheLayerMixin`. If omitted (`None`), then `layer_class_to_replicate` will
            be used.
        layer_class_to_replicate (`type[CacheLayerMixin]`, *optional*):
            Only used if `layers` is omitted (`None`), in which case it will be used as the base class for each layer,
            and the layers will be added lazily as soon as `update` is called with a `layer_idx` greater than the current
            list of layers.
        offloading (`bool`, *optional*, defaults to `False`):
            Whether to perform offloading of the layers to `cpu`, to save GPU memory.
        offload_only_non_sliding (`bool`, *optional*, defaults to `True`):
            If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
            usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).
    NÚlayersÚlayer_class_to_replicateÚ
offloadingÚoffload_only_non_slidingc                 ó  — |�|�t        d«      ‚|€|€t        d«      ‚|�|ng | _        || _        || _        | j                  rE|| _        t
        rt        j                  «       nt        j                  j                  «       | _	        y y )Na  You can construct a Cache either from a list `layers` of all the predefined `CacheLayer`, or from a `layer_class_to_replicate`, in which case the Cache will append a new layer corresponding to `layer_class_to_replicate` for each new call to `update` with an idx not already in the Cache.z_You should provide exactly one of `layers` or `layer_class_to_replicate` to initialize a Cache.)
r�   rú   rû   rü   Úonly_non_slidingÚ#_is_torch_greater_or_equal_than_2_7rN   ÚStreamrõ   Úprefetch_stream)r   rú   rû   rü   rý   s        r   r   zCache.__init__®  s•   € ð ÐÐ":Ð"FÜðqóð ð
 ˆ>Ð6Ð>ÜØqóð ð !'Ð 2‘f¸ˆŒØ(@ˆÔ%Ø$ˆŒØ�?Š?Ø$<ˆDÔ!Ý5X¤5§<¡<¤>Ô^c×^hÑ^h×^oÑ^oÓ^qˆDÕ ð r   c                 óN   — | j                   j                  › d| j                  › d�S )Nz(layers=ú))r    r!   rú   r   s    r   r"   zCache.__repr__Æ  s$   € Ø—.‘.×)Ñ)Ð*¨(°4·;±;°-¸qÐAÐAr   Ú	layer_idxrÿ   c                 ó´  — |r#	 || j                   |d j                  d«      z   }n|t        | j                  «      k  r|nd}t
        r| j                  n(t        j                  j                  | j                  «      5  | j                  |   j                  «        ddd«       y# t        $ r | j                   j                  d«      }Y Œˆw xY w# 1 sw Y   yxY w)a<  
        Prefetch a given layer on its device. If `only_non_sliding` is True, it will try to prefetch only the layers
        which are non-sliding. If the `layer_idx` is outside the range, this will circle back to the first layers.
        Note that we use a non-default stream for this, to avoid blocking.
        NFr   )rx   r¶   r�   Úlenrú   r   r  rN   rõ   Ústreamr>   ©r   r  rÿ   s      r   r>   zCache.prefetchÉ  s¿   € ñ ð9Ø%¨¯©¸	¸
Ð(C×(IÑ(IÈ%Ó(PÑP‘	ð
 &/´°T·[±[Ó1AÒ%A™	ÀqˆIõ &IˆT×!Ò!ÌeÏjÉj×N_ÑN_Ð`d×`tÑ`tÓNuñ 	.Ø�K‰K˜	Ñ"×+Ñ+Ô-÷	.ð 	.øô ò 9Ø ŸO™O×1Ñ1°%Ó8’	ð9ú÷	.ð 	.ús   „!B$ Á=CÂ$$CÃ
CÃCc                 ób   — |r| j                   |   s| j                  |   j                  «        yy)a  
        Offload a given `layer_idx`. If `only_non_sliding` is True, it will offload `layer_idx` only if it is a
        non-sliding layer. Note that we do it on the default stream, so that we ensure all earlier
        computation in the layer's `update` methods are finished.
        N)rx   rú   r;   r	  s      r   r;   zCache.offloadÝ  s-   € ñ ! T§_¡_°YÒ%?Ø�K‰K˜	Ñ"×*Ñ*Õ,ð &@r   r#   r(   r)   r*   c                 óF  — | j                   �Zt        | j                  «      |k  rB| j                  j                  | j                  «       «       t        | j                  «      |k  rŒB| j                  rat
        j                  j                  |j                  «      j                  | j                  «       | j                  |dz   | j                  «       | j                  |   j                  |||«      \  }}| j                  r| j                  || j                  «       ||fS )a·  
        Updates the cache with the new `key_states` and `value_states` for the layer `layer_idx`.

        Parameters:
            key_states (`torch.Tensor`):
                The new key states to cache.
            value_states (`torch.Tensor`):
                The new value states to cache.
            layer_idx (`int`):
                The index of the layer to cache the states for.
            cache_kwargs (`dict[str, Any]`, *optional*):
                Additional arguments for the cache subclass. These are specific to each subclass and allow new types of
                cache to be created.

        Return:
            A tuple containing the updated key and value states.
        r   )rû   r  rú   Úappendrü   rN   rõ   Údefault_streamr=   Úwait_streamr  r>   rÿ   r-   r;   )r   r#   r(   r  r)   r   r   s          r   r-   zCache.updateæ  sß   € ð2 ×(Ñ(Ð4Ü�d—k‘kÓ" iÒ/Ø—‘×"Ñ" 4×#@Ñ#@Ó#BÔCô �d—k‘kÓ" iÓ/ð �?Š?ä�J‰J×%Ñ% j×&7Ñ&7Ó8×DÑDÀT×EYÑEYÔZØ�M‰M˜) a™-¨×)>Ñ)>Ô?à—{‘{ 9Ñ-×4Ñ4°ZÀÈ|Ó\‰ˆˆfà�?Š?Ø�L‰L˜ D×$9Ñ$9Ô:à�Vˆ|Ðr   Ú
batch_sizer˜   r™   rY   r=   c                 ó€   — t        j                  ||d|f||¬«      }| j                  D ]  }|j                  |«       Œ y)zÐ
        Initialize all the layers in advance (it's otherwise lazily initialized on the first `update` call).
        This is useful for our `export` recipes, as `export` needs everything in advance.
        r   rX   N)rN   rš   rú   r'   )r   r  r˜   r™   rY   r=   Úfake_keys_tensorÚlayers           r   Úearly_initializationzCache.early_initialization  sD   € ô !Ÿ;™;¨
°I¸qÀ(Ð'KÐSXÐagÔhÐà—[‘[ò 	8ˆEØ×%Ñ%Ð&6Õ7ñ	8r   c                 ón   — |t        | j                  «      k\  ry| j                  |   j                  «       S )z=Returns the sequence length of the cache for the given layer.r   )r  rú   r2   ©r   r  s     r   r2   zCache.get_seq_length  s.   € àœ˜DŸK™KÓ(Ò(ØØ�{‰{˜9Ñ%×4Ñ4Ó6Ð6r   r.   c                 ó�   — |t        | j                  «      k\  r|j                  d   dfS | j                  |   j                  |«      S )a  
        Return a tuple (kv_length, kv_offset) corresponding to the length and offset that will be returned for
        the given layer at `layer_idx`.
        The masks are then prepared according to the given lengths (kv_length, kv_offset) and patterns for each layer.
        r   )r  rú   rb   r0   ©r   r.   r  s      r   r0   zCache.get_mask_sizes$  sE   € ð œ˜DŸK™KÓ(Ò(Ø!×'Ñ'¨Ñ*¨AÐ-Ð-Ø�{‰{˜9Ñ%×4Ñ4°^ÓDÐDr   c                 ón   — |t        | j                  «      k\  ry| j                  |   j                  «       S )zaReturns maximum sequence length of the cache object. Dynamic caches do not have a maximum length.rj   )r  rú   r5   r  s     r   r5   zCache.get_max_cache_shape0  s0   € ð œ˜DŸK™KÓ(Ò(ØØ�{‰{˜9Ñ%×9Ñ9Ó;Ð;r   c                 ó„   — t        t        | j                  «      «      D ]  }| j                  |   j                  «        Œ! y)z$Recursively reset all layers tensorsN)Úranger  rú   rC   r  s     r   rC   zCache.reset8  s4   € äœs 4§;¡;Ó/Ó0ò 	+ˆIØ�K‰K˜	Ñ"×(Ñ(Õ*ñ	+r   rD   c                 ó†   — t        t        | j                  «      «      D ]   }| j                  |   j                  |«       Œ" y)z!Reorder the cache for beam searchN)r  r  rú   rH   )r   rD   r  s      r   rH   zCache.reorder_cache=  s6   € äœs 4§;¡;Ó/Ó0ò 	;ˆIØ�K‰K˜	Ñ"×0Ñ0°Õ:ñ	;r   rk   c                 ó†   — t        t        | j                  «      «      D ]   }| j                  |   j                  |«       Œ" y)z"Crop the cache to the given lengthN)r  r  rú   rn   )r   rk   r  s      r   rn   z
Cache.cropB  s6   € äœs 4§;¡;Ó/Ó0ò 	4ˆIØ�K‰K˜	Ñ"×'Ñ'¨
Õ3ñ	4r   ro   c                 ó†   — t        t        | j                  «      «      D ]   }| j                  |   j                  |«       Œ" y)zRepeat and interleave the cacheN)r  r  rú   rs   )r   ro   r  s      r   rs   zCache.batch_repeat_interleaveG  s8   € äœs 4§;¡;Ó/Ó0ò 	DˆIØ�K‰K˜	Ñ"×:Ñ:¸7ÕCñ	Dr   rt   c                 ó†   — t        t        | j                  «      «      D ]   }| j                  |   j                  |«       Œ" y)zSelect indices from the cacheN)r  r  rú   rw   )r   rt   r  s      r   rw   zCache.batch_select_indicesL  s8   € äœs 4§;¡;Ó/Ó0ò 	AˆIØ�K‰K˜	Ñ"×7Ñ7¸Õ@ñ	Ar   c                 ó¦   — | j                   D �cg c]  }|j                  ‘Œ }}t        t        |«      «      dkD  rt	        d|› �«      ‚|d   S c c}w )z*Return the maximum batch size of the cacher   z0Max batch size is not consistent across layers: r   )rú   r—   r  Úsetr�   ©r   r  r   s      r   r—   zCache.max_batch_sizeQ  sV   € ð 59·K±KÖ@¨5�%×&Ó&Ð@ˆÐ@ÜŒs�6‹{Ó˜aÒÜÐOÐPVÈxÐXÓYÐYØ�a‰yÐùò As   �Ac                 óh   — | j                   D �cg c]  }|j                  ‘Œ }}t        |«      S c c}w )z,Return the maximum cache length of the cache)rú   r”   r…   r!  s      r   r”   zCache.max_cache_lenY  s1   € ð 48·;±;Ö?¨%�%×%Ó%Ð?ˆÐ?Ü�6‹{Ðùò @s   �/c                 ól   — t        | j                  «      dk(  ryt        d„ | j                  D «       «      S )z'Return whether the cache is compileabler   Fc              3   ó4   K  — | ]  }|j                   –— Œ y ­wr   )rM   ©Ú.0r  s     r   ú	<genexpr>z'Cache.is_compileable.<locals>.<genexpr>e  s   è ø€ ÒA¨E�5×'Õ'ÑAùó   ‚©r  rú   Úallr   s    r   rM   zCache.is_compileable_  s-   € ô ˆt�{‰{Ó˜qÒ ØÜÑA°T·[±[ÔAÓAÐAr   c                 ón   — t        | j                  «      dkD  xr t        d„ | j                  D «       «      S )z,Return whether the cache data is initializedr   c              3   ó4   K  — | ]  }|j                   –— Œ y ­wr   )r   r%  s     r   r'  z'Cache.is_initialized.<locals>.<genexpr>j  s   è ø€ Ò+ZÀU¨E×,@Õ,@Ñ+Zùr(  r)  r   s    r   r   zCache.is_initializedg  s,   € ô �4—;‘;Ó !Ñ#ÒZ¬Ñ+ZÈdÏkÉkÔ+ZÓ(ZÐZr   c                 óV   — | j                   D �cg c]  }t        |dd«      ‘Œ c}S c c}w )z9Return whether the layers of the cache are sliding windowrx   F)rú   Úgetattr)r   r  s     r   rx   zCache.is_slidingl  s'   € ð BFÇÁÖM¸”˜˜|¨UÕ3ÒMÐMùÒMs   �&c                 óÞ   — |t        | j                  «      k  r2| j                  |   j                  | j                  |   j                  fS t	        dt        | j                  «      › d|› �«      ‚©z˜
        Support for backwards-compatible `past_key_values` indexing, e.g. `past_key_values[0][0].shape[2]` to get the
        sequence length.
        zCache only has z. layers, attempted to access layer with index )r  rú   r   r   ÚKeyErrorr  s     r   Ú__getitem__zCache.__getitem__q  sh   € ð
 ”s˜4Ÿ;™;Ó'Ò'Ø—;‘;˜yÑ)×.Ñ.°·±¸IÑ0F×0MÑ0MÐMÐMäØ!¤# d§k¡kÓ"2Ð!3Ð3aÐbkÐalÐmóð r   c              #   ó¦   K  — t        t        | «      «      D ]6  }| j                  |   j                  | j                  |   j                  f–— Œ8 y­w©z˜
        Support for backwards-compatible `past_key_values` iteration, e.g. `for x in past_key_values:` to iterate over
        keys and values
        N)r  r  rú   r   r   r  s     r   Ú__iter__zCache.__iter__}  sK   è ø€ ô
 œs 4›yÓ)ò 	OˆIØ—;‘;˜yÑ)×.Ñ.°·±¸IÑ0F×0MÑ0MÐNÓNñ	Oùs   ‚AAc                 ó,   — t        | j                  «      S )zN
        This value corresponds to the number of layers in the model.
        )r  rú   r   s    r   Ú__len__zCache.__len__…  s   € ô �4—;‘;ÓÐr   )NNFT)Tr   ©r   )*r!   rJ   rK   rL   r   Úlistr   ÚtypeÚboolr   r"   rS   r>   r;   rN   rO   rP   rQ   r   rR   r-   rY   r=   r  r2   r0   r5   rC   rT   rH   rn   rs   rw   Úpropertyr—   r”   rM   r   rx   r2  r5  r7  r%   r   r   rù   rù   š  sx  „ ñð* 37ØDHØ Ø)-ñrà˜˜oÑ.Ñ/ðrð #+¨4°Ñ+@Ñ"Aðrð ð	rð
 #'órò0Bñ. #ð .¸ó .ñ(- ð -¸ó -ð 26ñ'à—L‘Lð'ð —l‘lð'ð ð	'ð
 ˜t C¨ H™~Ñ.ð'ð 
ˆu�|‰|˜UŸ\™\Ð)Ñ	*ó'ðR8Øð8Ø*-ð8Ø9<ð8ØEJÇ[Á[ð8ØZ_×ZfÑZfó8ñ7¨ð 7°Có 7ð
E¨U¯\©\ð 
EÀcð 
EÈeÐTWÐY\ÐT\Éoó 
Eñ<¨Sð <¸ó <ò+ð
; e×&6Ñ&6ó ;ð
4˜só 4ð
D¨só Dð
A¨E¯L©Ló Að
 ð ò ó ðð ð˜sò ó ðð
 ðB ò Bó ðBð ð[ ò [ó ð[ð ðN˜D ™Jò Nó ðNð
 Sð 
¨U°5·<±<ÀÇÁÐ3MÑ-Nó 
òOó r   rù   c            	       ó  ‡ — e Zd ZdZ	 	 	 	 ddeeeej                  ej                  f         dee	   de
de
fˆ fd„Zdeeej                  ej                  f      fd„Zed	eeej                  ej                  f      dd fd
„«       Zˆ xZS )ÚDynamicCachea*
  
    A cache that grows dynamically as more tokens are generated. This is the default for generative models.
    It stores the key and value states as a list of `CacheLayer`, one for each layer. The expected shape for each tensor
    in the `CacheLayer`s is `[batch_size, num_heads, seq_len, head_dim]`.
    If a config is passed, it will additionally check for sliding or hybrid cache structure, greatly reducing the
    memory requirement of the cached tensors to `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        ddp_cache_data (`Iterable[tuple[torch.Tensor, torch.Tensor]]`, *optional*):
            It was originally added for compatibility with `torch.distributed` (DDP). In a nutshell, it is
            `map(gather_map, zip(*caches))`, i.e. each item in the iterable contains the key and value states
            for a layer gathered across replicas by torch.distributed (shape=[global batch size, num_heads, seq_len, head_dim]).
            Note: it needs to be the 1st arg as well to work correctly
        config (`PretrainedConfig`, *optional*):
            The config of the model for which this Cache will be used. If passed, it will be used to check for sliding
            or hybrid layer structure, greatly reducing the memory requirement of the cached tensors to
            `[batch_size, num_heads, min(seq_len, sliding_window), head_dim]`.
        offloading (`bool`, *optional*, defaults to `False`):
            Whether to perform offloading of the layers to `cpu`, to save GPU memory.
        offload_only_non_sliding (`bool`, *optional*, defaults to `False`):
            If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
            usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).

    Example:

    ```python
    >>> from transformers import AutoTokenizer, AutoModelForCausalLM, DynamicCache

    >>> model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2-0.5B-Instruct")
    >>> tokenizer = AutoTokenizer.from_pretrained("Qwen/Qwen2-0.5B-Instruct")

    >>> inputs = tokenizer(text="My name is Qwen2", return_tensors="pt")

    >>> # Prepare a cache class and pass it to model's forward
    >>> past_key_values = DynamicCache(config=model.config)
    >>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
    >>> outputs.past_key_values # access cache filled with key/values from generation
    ```
    Úddp_cache_dataÚconfigrü   rý   c                 óš  •— g }|�¿|j                  d¬«      }t        |dd «      xs t        |dd «      }t        |dd «      }|€&t        |j                  «      D �	cg c]  }	|�dnd‘Œ
 }}	t	        |d«      r|d |j
                    }|D ];  }
|
d	v r|j                  t        |¬
«      «       Œ#|j                  t        «       «       Œ= |�It        |«      D ];  \  }\  }}|€|j                  t        «       «       ||   j                  ||«      \  }	}	Œ= t        |«      dk(  rt        ‰| �5  t        ||¬«       y t        ‰| �5  |||¬«       y c c}	w )NT©Údecoderr{   Úattention_chunk_sizeÚlayer_typesÚsliding_attentionÚfull_attentionÚnum_kv_shared_layers)rF  Úchunked_attentionr�   r   )rû   rü   rý   ©rú   rü   rý   )Úget_text_configr.  r  Únum_hidden_layersrB   rH  r  rz   rV   Ú	enumerater-   r  r~   r   )r   r?  r@  rü   rý   rú   Údecoder_configr{   rE  r�   Ú
layer_typer  r#   r(   r    s                 €r   r   zDynamicCache.__init__¹  s�  ø€ ð ˆàÐØ#×3Ñ3¸DÐ3ÓAˆNÜ$ ^Ð5EÀtÓLò ÔPWØÐ 6¸óQˆNô " .°-ÀÓFˆKØÐ"ô # >×#CÑ#CÓDöàð ,:Ð+EÑ'ÐK[Ñ[ð�ð ô
 �~Ð'=Ô>Ø)Ð*P¨^×-PÑ-PÐ,PÐQ�à)ò 2�
ð Ð!KÑKØ—M‘MÔ";È>Ô"ZÕ[à—M‘M¤,£.Õ1ð2ð Ð%ä9BÀ>Ó9Rò JÑ5�	Ñ5˜J¨à�>Ø—M‘M¤,£.Ô1à˜iÑ(×/Ñ/°
¸LÓI‘�‘1ðJô ˆv‹;˜!ÒÜ‰GÑÜ)5Ø%Ø)Að õ ô ‰GÑ F°zÐ\tÐÕuùòEs   ÁEr*   c                 ód   — d}| j                   D ]  }||j                  |j                  ffz  }Œ  |S )zŒ
        Converts the `Cache` instance into the its equivalent in the legacy cache format. Used for
        backward compatibility.
        r%   )rú   r   r   )r   Úlegacy_cacher  s      r   Úto_legacy_cachezDynamicCache.to_legacy_cacheí  s<   € ð
 ˆØ—[‘[ò 	:ˆEØ˜eŸj™j¨%¯,©,Ð7Ð9Ñ9‰Lð	:àÐr   Úpast_key_valuesc                 ó®   —  | «       }|€t         j                  d«       |�4t        t        |«      «      D ]  }||   \  }}|j	                  |||«       Œ |S )z‚
        Converts a cache in the legacy cache format into an equivalent `Cache`. Used for
        backward compatibility.
        ú9past_key_values should not be None in from_legacy_cache())ÚloggerÚwarning_oncer  r  r-   )ÚclsrS  Úcacher  r#   r(   s         r   Úfrom_legacy_cachezDynamicCache.from_legacy_cache÷  sg   € ñ “ˆØÐ"Ü×ÑÐ [Ô\ØÐ&Ü"¤3 Ó#7Ó8ò B�	Ø+:¸9Ñ+EÑ(�
˜LØ—‘˜Z¨°yÕAðBð ˆr   )NNFF)r!   rJ   rK   rL   r   r   rR   rN   rO   r	   r;  r   rR  ÚclassmethodrZ  r�   r‘   s   @r   r>  r>  Ž  sÉ   ø„ ñ(ðX QUØ-1Ø Ø).ñ2và  ¨%°·±¸e¿l¹lÐ0JÑ*KÑ!LÑMð2vð Ð)Ñ*ð2vð ð	2vð
 #'õ2vðh  u¨U¯\©\¸5¿<¹<Ð-GÑ'HÑ!Ió ð ð°°e¸E¿L¹LÈ%Ï,É,Ð<VÑ6WÑ0Xð Ð]kò ó ôr   r>  c            	       ó:   ‡ — e Zd ZdZ	 	 ddedededefˆ fd„Zˆ xZS )ÚStaticCachea¸  
    Static Cache class to be used with `torch.compile(model)` and `torch.export()`. It will check the `config`
    for potential hybrid cache structure, and initialize each layer accordingly.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        config (`PretrainedConfig`):
            The config of the model for which this Cache will be used. It will be used to check for sliding
            or hybrid layer structure, and initialize each layer accordingly.
        max_cache_len (`int`):
            The maximum number of tokens that this Cache should hold.
        offloading (`bool`, *optional*, defaults to `False`):
            Whether to perform offloading of the layers to `cpu`, to save GPU memory.
        offload_only_non_sliding (`bool`, *optional*, defaults to `True`):
            If `offloading` is `True`, this further decides if only the non-sliding layers will be offloaded (because
            usually the sliding layers are small in size, so there is no need to offload them, and skipping it is faster).

    Example:

    ```python
    >>> from transformers import AutoTokenizer, AutoModelForCausalLM, StaticCache

    >>> model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-2-7b-chat-hf")
    >>> tokenizer = AutoTokenizer.from_pretrained("meta-llama/Llama-2-7b-chat-hf")

    >>> inputs = tokenizer(text="My name is Llama", return_tensors="pt")

    >>> # Prepare a cache class and pass it to model's forward
    >>> # Leave empty space for 10 new tokens, which can be used when calling forward iteratively 10 times to generate
    >>> max_generated_length = inputs.input_ids.shape[1] + 10
    >>> past_key_values = StaticCache(config=model.config, max_cache_len=max_generated_length)
    >>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
    >>> outputs.past_key_values # access cache filled with key/values from generation
    StaticCache()
    ```
    r@  r”   rü   rý   c                 ó†  •— |j                  d¬«      }t        |dd «      }|€‚t        |dd «      �#t        |j                  «      D �cg c]  }d‘Œ }}nRt        |dd «      �#t        |j                  «      D �cg c]  }d‘Œ }}n"t        |j                  «      D �cg c]  }d‘Œ }}t	        |d	«      r|d |j
                    }g }|D ]Y  }	|	dk(  rt        ||j                  ¬
«      }
n)|	dk(  rt        ||j                  ¬
«      }
nt        |¬«      }
|j                  |
«       Œ[ t        ‰| �1  |||¬«       y c c}w c c}w c c}w )NTrB  rE  r{   rF  rD  rI  rG  rH  )r”   r{   r¦   rJ  )rK  r.  r  rL  rB   rH  r¬   r{   rD  r“   r  r~   r   )r   r@  r”   rü   rý   ÚkwargsrE  r�   rú   rO  r  r    s              €r   r   zStaticCache.__init__/  sU  ø€ ð ×'Ñ'°Ð'Ó5ˆÜ˜f m°TÓ:ˆàÐÜ�vÐ/°Ó6ÐBÜ<AÀ&×BZÑBZÓ<[Ö\°qÒ2Ð\�Ñ\Ü˜Ð!7¸Ó>ÐJÜ<AÀ&×BZÑBZÓ<[Ö\°qÒ2Ð\�Ñ\ä9>¸v×?WÑ?WÓ9XÖY°AÒ/ÐY�ÐYä�6Ð1Ô2Ø%Ð&D¨×)DÑ)DÐ(DÐEˆKàˆØ%ò 	!ˆJØÐ0Ò0Ü0¸}Ð]c×]rÑ]rÔs‘ØÐ2Ò2ô 1Ø"/À×@[Ñ@[ô‘ô $°-Ô@�Ø�M‰M˜%Õ ð	!ô 	‰Ñ °:ÐXpÐÕqùò/ ]ùâ\ùâYs   Á	D4Á7	D9Â	D>)FT)	r!   rJ   rK   rL   r	   rS   r;  r   r�   r‘   s   @r   r]  r]    sG   ø„ ñ$ðV !Ø)-ñ$rà ð$rð ð$rð ð	$rð
 #'÷$rñ $rr   r]  c                   óL   ‡ — e Zd ZdZ	 	 	 	 	 d
dededededededefˆ fd	„Zˆ xZS )ÚQuantizedCacheaš  
    A quantizer cache similar to what is described in the
    [KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache paper](https://huggingface.co/papers/2402.02750).
    It allows the model to generate longer sequence length without allocating too much memory for keys and values
    by applying quantization.
    The cache has two types of storage, one for original precision and one for the
    quantized cache. A `residual length` is set as a maximum capacity for the original precision cache. When the
    length goes beyond maximum capacity, the original precision cache is discarded and moved into the quantized cache.
    The quantization is done per-channel with a set `q_group_size` for both keys and values, in contrast to what was
    described in the paper.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        backend (`str`):
            The quantization backend to use. One of `("quanto", "hqq").
        config (`PretrainedConfig`):
            The config of the model for which this Cache will be used.
        nbits (`int`, *optional*, defaults to 4):
            The number of bits for quantization.
        axis_key (`int`, *optional*, defaults to 0):
            The axis on which to quantize the keys.
        axis_value (`int`, *optional*, defaults to 0):
            The axis on which to quantize the values.
        q_group_size (`int`, *optional*, defaults to 64):
            Quantization is done per-channel according to a set `q_group_size` for both keys and values.
        residual_length (`int`, *optional*, defaults to 128):
            Maximum capacity for the original precision cache
    Úbackendr@  r»   r¼   r½   r¾   r¿   c           
      óú   •— |dk(  rt         }n|dk(  rt        }nt        d|› d�«      ‚|j                  d¬«      }t	        |j
                  «      D �	cg c]  }	 ||||||«      ‘Œ }
}	t        ‰| �  |
¬«       y c c}	w )NÚquantoÚhqqzUnknown quantization backend `ú`TrB  )rú   )rÖ   ré   r�   rK  r  rL  r~   r   )r   rb  r@  r»   r¼   r½   r¾   r¿   Úlayer_classr�   rú   r    s              €r   r   zQuantizedCache.__init__u  s•   ø€ ð �hÒÜ.‰KØ˜ÒÜ+‰KäÐ=¸g¸YÀaÐHÓIÐIà×'Ñ'°Ð'Ó5ˆô ˜6×3Ñ3Ó4ö
àñ ˜˜x¨°\À?ÕSð
ˆð 
ô 	‰Ñ ÐÕ'ùò	
s   ÁA8rÒ   )	r!   rJ   rK   rL   rQ   r	   rS   r   r�   r‘   s   @r   ra  ra  V  sh   ø„ ñðD ØØØØ"ñ(àð(ð !ð(ð ð	(ð
 ð(ð ð(ð ð(ð ÷(ñ (r   ra  c                   ó  — e Zd ZdZd#d„Zdefd„Zd„ Zdede	e
j                  e
j                  e
j                  e
j                  f   fd„Zd	„ Zde	e	e
j                        fd
„Zedeee	e
j$                  df         dd fd„«       Zd$dedefd„Zd„ Zde
j,                  fd„Zdefd„Zdefd„Zdededdfd„Zdefd„Zde
j                  fd„Zdefd„Zde
j                  dede	eef   fd „Zed!„ «       Z ede!fd"„«       Z"y)%ÚEncoderDecoderCachea­  
    Base, abstract class for all encoder-decoder caches. Can be used to hold combinations of self-attention and
    cross-attention caches.

    See `Cache` for details on common methods that are implemented by all cache classes.

    Args:
        caches (`Iterable`):
            Usually an iterable of length 2, containing 2 `Cache` objects, the first one for self-attention, the
            second one for cross-attention. Can optionally also be an iterable of length 1, containing a
            `tuple[tuple[torch.Tensor]]` (usually used for compatibility with torch dp and ddp).

    Example:

    ```python
    >>> from transformers import AutoProcessor, AutoModelForCausalLM, DynamicCache, EncoderDecoderCache

    >>> model = AutoModelForCausalLM.from_pretrained("openai/whisper-small")
    >>> processor = AutoProcessor.from_pretrained("openai/whisper-small")

    >>> inputs = processor(audio=YOUR-AUDIO, return_tensors="pt")

    >>> # Prepare cache classes for encoder and decoder and pass it to model's forward
    >>> self_attention_cache = DynamicCache(config=self.config)
    >>> cross_attention_cache = DynamicCache(config=self.config)
    >>> past_key_values = EncoderDecoderCache(self_attention_cache, cross_attention_cache)
    >>> outputs = model(**inputs, past_key_values=past_key_values, use_cache=True)
    >>> outputs.past_key_values # access cache filled with key/values from generation
    EncoderDecoderCache()
    ```
    r*   Nc           	      ó
  — t        |«      dk(  rŽt        «       | _        t        «       | _        t	        |d   «      D ]^  \  }}|d d \  }}| j                  j                  |||«       t        |«      dkD  sŒ:|dd  \  }}| j                  j                  |||«       Œ` n‰t        |«      dk(  rdt        |d   t        «      rt        |d   t        «      s)t        dt        |d   «      ›dt        |d   «      ›�«      ‚|d   | _        |d   | _        nt        dt        |«      › �«      ‚i | _        t        t        | j                  «      «      D ]6  }t        | j                  j                  |«      dkD  «      | j                  |<   Œ8 y )Nr   r   r    z;One of the two arguments is not a Cache: type(caches[0]) = z, type(caches[1]) = zExpected 1 or 2 arguments, got )r  r>  Úself_attention_cacheÚcross_attention_cacherM  r-   Ú
isinstancerù   Ú	TypeErrorr:  r�   Ú
is_updatedr  r;  r2   )r   Úcachesr  Úkey_value_statesr#   r(   s         r   r   zEncoderDecoderCache.__init__¯  s„  € äˆv‹;˜!ÒÜ(4«ˆDÔ%Ü)5«ˆDÔ&ä/8¸À¹Ó/Cò [Ñ+�	Ð+Ø+;¸B¸QÐ+?Ñ(�
˜LØ×)Ñ)×0Ñ0°¸\È9ÔUÜÐ'Ó(¨1Ó,Ø/?ÀÀÐ/CÑ,�J Ø×.Ñ.×5Ñ5°jÀ,ÐPYÕZñ[ô �‹[˜AÒÜ˜f Q™i¬Ô/´zÀ&ÈÁ)ÌUÔ7SÜÐ"^ÌDÐQWÐXYÑQZËOÐK_Ð_tÔbfÐgmÐnoÑgpÓbqÐauÐ vÓwÐwØ(.¨q©	ˆDÔ%Ø)/°©ˆDÕ&ô Ð>¼sÀ6»{¸mÐLÓMÐMàˆŒÜœs 4×#=Ñ#=Ó>Ó?ò 	hˆIÜ)-¨d×.HÑ.H×.WÑ.WÐXaÓ.bÐefÑ.fÓ)gˆD�O‰O˜IÒ&ñ	hr   c                 óh   — | j                   j                  › d| j                  › d| j                  › d�S )Nz(self_attention_cache=z, cross_attention_cache=r  )r    r!   rk  rl  r   s    r   r"   zEncoderDecoderCache.__repr__É  s;   € à�~‰~×&Ñ&Ð'Ð'=¸d×>WÑ>WÐ=XÐXpØ×)Ñ)Ð*¨!ð-ð	
r   c              #   óV  K  — t        t        | «      «      D ]Ž  }| j                  j                  |   j                  | j                  j                  |   j
                  | j                  j                  |   j                  | j                  j                  |   j
                  f–— Œ� y­wr4  )r  r  rk  rú   r   r   rl  r  s     r   r5  zEncoderDecoderCache.__iter__Ï  s’   è ø€ ô
 œs 4›yÓ)ò 	ˆIà×)Ñ)×0Ñ0°Ñ;×@Ñ@Ø×)Ñ)×0Ñ0°Ñ;×BÑBØ×*Ñ*×1Ñ1°)Ñ<×AÑAØ×*Ñ*×1Ñ1°)Ñ<×CÑCð	ó ñ	ùs   ‚B'B)r  c                 óf  — |t        | «      k  rŠ| j                  j                  |   j                  | j                  j                  |   j                  | j
                  j                  |   j                  | j
                  j                  |   j                  fS t        dt        | «      › d|› �«      ‚r0  )r  rk  rú   r   r   rl  r1  r  s     r   r2  zEncoderDecoderCache.__getitem__Ü  s£   € ð
 ”s˜4“yÒ à×)Ñ)×0Ñ0°Ñ;×@Ñ@Ø×)Ñ)×0Ñ0°Ñ;×BÑBØ×*Ñ*×1Ñ1°)Ñ<×AÑAØ×*Ñ*×1Ñ1°)Ñ<×CÑCð	ð ô ˜_¬S°«Y¨KÐ7eÐfoÐepÐqÓrÐrr   c                 ó,   — t        | j                  «      S )z®
        Support for backwards-compatible `past_key_values` length, e.g. `len(past_key_values)`. This value corresponds
        to the number of layers in the model.
        )r  rk  r   s    r   r7  zEncoderDecoderCache.__len__ë  s   € ô
 �4×,Ñ,Ó-Ð-r   c                 ó  — d}t        | j                  «      dkD  rOt        | j                  j	                  «       | j                  j	                  «       «      D ]  \  }}|||z   fz  }Œ |S | j                  j	                  «       }|S )z[Converts the `EncoderDecoderCache` instance into its equivalent in the legacy cache format.r%   r   )r  rl  Úziprk  rR  )r   rQ  Ú	self_attnÚ
cross_attns       r   rR  z#EncoderDecoderCache.to_legacy_cacheò  sŽ   € àˆÜˆt×)Ñ)Ó*¨QÒ.Ü),Ø×)Ñ)×9Ñ9Ó;¸T×=WÑ=W×=gÑ=gÓ=ió*ò :Ñ%�	˜:ð  ¨ZÑ!7Ð 9Ñ9‘ð:ð Ðð  ×4Ñ4×DÑDÓFˆLØÐr   rS  .c                 ó`  —  | t        «       t        «       «      }|€t        j                  d«       |S t        |«      D ]m  \  }}|dd \  }}|j                  j                  |||«       t        |«      dkD  sŒ:|dd \  }}|j                  j                  |||«       d|j                  |<   Œo |S )zUConverts a cache in the legacy cache format into an equivalent `EncoderDecoderCache`.NrU  r    T)	r>  rV  rW  rM  rk  r-   r  rl  ro  )rX  rS  rY  r  rq  r#   r(   s          r   rZ  z%EncoderDecoderCache.from_legacy_cacheþ  sÄ   € ñ
 ”L“N¤L£NÓ3ˆØÐ"Ü×ÑÐ [Ô\ð ˆô 09¸Ó/Iò 7Ñ+�	Ð+Ø+;¸B¸QÐ+?Ñ(�
˜LØ×*Ñ*×1Ñ1°*¸lÈIÔVÜÐ'Ó(¨1Ó,Ø/?ÀÀÐ/CÑ,�J Ø×/Ñ/×6Ñ6°zÀ<ÐQZÔ[Ø26�E×$Ñ$ YÒ/ð7ð ˆr   c                 ó8   — | j                   j                  |«      S )zYReturns the sequence length of the cached states. A layer index can be optionally passed.)rk  r2   r  s     r   r2   z"EncoderDecoderCache.get_seq_length  s   € à×(Ñ(×7Ñ7¸	ÓBÐBr   c                 ó¬   — | j                   j                  «        | j                  j                  «        | j                  D ]  }d| j                  |<   Œ y r   )rk  rC   rl  ro  r  s     r   rC   zEncoderDecoderCache.reset  sG   € Ø×!Ñ!×'Ñ'Ô)Ø×"Ñ"×(Ñ(Ô*ØŸ™ò 	/ˆIØ).ˆD�O‰O˜IÒ&ñ	/r   rD   c                 óp   — | j                   j                  |«       | j                  j                  |«       y)zDReorders the cache for beam search, given the selected beam indices.N)rk  rH   rl  rG   s     r   rH   z!EncoderDecoderCache.reorder_cache  s*   € à×!Ñ!×/Ñ/°Ô9Ø×"Ñ"×0Ñ0°Õ:r   Úmethodc           	      óö   — t        | j                  t        «      rt        | j                  t        «      sEt	        d|› d| j                  j                  «       › d| j                  j                  «       › d�«      ‚y )Nrf  z)` is only defined for dynamic cache, got z" for the self attention cache and z for the cross attention cache.)rm  rk  r>  rl  r�   Ú__str__)r   r~  s     r   Úcheck_dynamic_cachez'EncoderDecoderCache.check_dynamic_cache  sw   € ä�t×0Ñ0´,Ô?Ü˜4×5Ñ5´|ÔDäØ�F�8ÐDÀT×E^ÑE^×EfÑEfÓEhÐDið j'Ø'+×'AÑ'A×'IÑ'IÓ'KÐ&LÐLkðmóð ð Er   Úmaximum_lengthc                 ó„   — | j                  | j                  j                  «       | j                  j                  |«       y)zó
        Crop the past key values up to a new `maximum_length` in terms of tokens. `maximum_length` can also be
        negative to remove `maximum_length` tokens. This is used in assisted decoding and contrastive search (on the Hub).
        N)r�  rn   r!   rk  )r   r‚  s     r   rn   zEncoderDecoderCache.crop*  s0   € ð
 	× Ñ  §¡×!3Ñ!3Ô4Ø×!Ñ!×&Ñ& ~Õ6r   Úfull_batch_sizeÚ
split_sizezlist[EncoderDecoderCache]c                 ó"  — | j                  | j                  j                  «       | j                  j                  ||«      }| j                  j                  ||«      }g }t        ||«      D ]   \  }}|j                  t        ||«      «       Œ" |S )z¨
        Split the current instance into a list of `DynamicCache` by the batch size. This will be used by
        `_split_model_inputs()` in `generation.utils`
        )r�  Úbatch_splitr!   rk  rl  rw  r  ri  )r   r„  r…  rk  rl  Úoutrx  ry  s           r   r‡  zEncoderDecoderCache.batch_split2  s�   € ð
 	× Ñ  ×!1Ñ!1×!:Ñ!:Ô;Ø#×8Ñ8×DÑDÀ_ÐV`ÓaÐØ $× :Ñ :× FÑ FÀÐXbÓ cÐàˆÜ%(Ð)=Ð?TÓ%Uò 	CÑ!ˆI�zØ�J‰JÔ*¨9°jÓAÕBð	Càˆ
r   ro   c                 óº   — | j                  | j                  j                  «       | j                  j                  |«       | j                  j                  |«       y)zaRepeat the cache `repeats` times in the batch dimension. Used in contrastive search (on the Hub).N)r�  rs   r!   rk  rl  rr   s     r   rs   z+EncoderDecoderCache.batch_repeat_interleave@  sD   € à× Ñ  ×!=Ñ!=×!FÑ!FÔGØ×!Ñ!×9Ñ9¸'ÔBØ×"Ñ"×:Ñ:¸7ÕCr   rt   c                 óº   — | j                  | j                  j                  «       | j                  j                  |«       | j                  j                  |«       y)zeOnly keep the `indices` in the batch dimension of the cache. Used in contrastive search (on the Hub).N)r�  rw   r!   rk  rl  rv   s     r   rw   z(EncoderDecoderCache.batch_select_indicesF  sD   € à× Ñ  ×!:Ñ!:×!CÑ!CÔDØ×!Ñ!×6Ñ6°wÔ?Ø×"Ñ"×7Ñ7¸Õ@r   c                 ó6   — | j                   j                  «       S )zKReturns the maximum sequence length (i.e. max capacity) of the cache object)rk  r5   r   s    r   r5   z'EncoderDecoderCache.get_max_cache_shapeL  s   € à×(Ñ(×<Ñ<Ó>Ð>r   r.   c                 ó:   — | j                   j                  ||«      S r   )rk  r0   r  s      r   r0   z"EncoderDecoderCache.get_mask_sizesP  s   € Ø×(Ñ(×7Ñ7¸È	ÓRÐRr   c                 ó.   — | j                   j                  S r   )rk  rx   r   s    r   rx   zEncoderDecoderCache.is_slidingS  s   € à×(Ñ(×3Ñ3Ð3r   c                 ó.   — | j                   j                  S r   )rk  rM   r   s    r   rM   z"EncoderDecoderCache.is_compileableW  s   € à×(Ñ(×7Ñ7Ð7r   rI   r8  )#r!   rJ   rK   rL   r   rQ   r"   r5  rS   rR   rN   rO   r2  r7  rR  r[  r   r   ÚFloatTensorrZ  r2   rC   rT   rH   r�  rn   r‡  rs   rw   r5   r0   r<  rx   r;  rM   r%   r   r   ri  ri  Ž  s›  „ ñó@hð4
˜#ó 
òðs Sð s¨U°5·<±<ÀÇÁÈuÏ|É|Ð]b×]iÑ]iÐ3iÑ-jó sò.ð
  u¨U¯\©\Ñ':Ñ!;ó 
ð ðØ& x°°e×6GÑ6GÈÐ6LÑ0MÑ'NÑOðà	òó ðñ"C¨ð C°Có Cò/ð; e×&6Ñ&6ó ;ð
¨#ó ð7 3ó 7ð¨3ð ¸Cð ÐD_ó ðD¨só DðA¨E¯L©Ló Að? Só ?ðS¨U¯\©\ð SÀcð SÈeÐTWÐY\ÐT\Éoó Sð ñ4ó ð4ð ð8 ò 8ó ñ8r   ri  c                   ó(   ‡ — e Zd Zdedefˆ fd„Zˆ xZS )ÚSlidingWindowLayerr”   r{   c                 óP   •— t         j                  d«       t        ‰| �  ||«       y )NzŽ`SlidingWindowLayer` is deprecated and will be removed in version v4.59 Use `StaticSlidingWindowLayer` instead, which is a better name for it.©rV  rW  r~   r   ©r   r”   r{   r    s      €r   r   zSlidingWindowLayer.__init__`  s(   ø€ Ü×ÑðUô	
ô 	‰Ñ˜¨Õ7r   ©r!   rJ   rK   rS   r   r�   r‘   s   @r   r‘  r‘  _  ó   ø„ ð8 cð 8¸3÷ 8ñ 8r   r‘  c                   ó(   ‡ — e Zd Zdedefˆ fd„Zˆ xZS )ÚChunkedSlidingLayerr”   r{   c                 óP   •— t         j                  d«       t        ‰| �  ||«       y )Nzš`ChunkedSlidingLayer` is deprecated and will be removed in version v4.59 Use `StaticSlidingWindowLayer` instead, which has the exact same functionalities.r“  r”  s      €r   r   zChunkedSlidingLayer.__init__i  s(   ø€ Ü×Ñð`ô	
ô 	‰Ñ˜¨Õ7r   r•  r‘   s   @r   r˜  r˜  h  r–  r   r˜  c                   ó    ‡ — e Zd Zdˆ fd„Zˆ xZS )ÚOffloadedCachec                 óP   •— t         j                  d«       t        ‰| �  d¬«       y )Nzo`OffloadedCache` is deprecated and will be removed in version v4.59 Use `DynamicCache(offloading=True)` insteadT)rü   r“  )r   r    s    €r   r   zOffloadedCache.__init__r  s(   ø€ Ü×Ñð:ô	
ô 	‰Ñ DÐÕ)r   rI   )r!   rJ   rK   r   r�   r‘   s   @r   r›  r›  q  s   ø„ ÷*ñ *r   r›  c                   ó(   ‡ — e Zd Zdedefˆ fd„Zˆ xZS )ÚOffloadedStaticCacher@  r”   c                 óT   •— t         j                  d«       t        ‰| �  ||d¬«       y )Nzy`OffloadedStaticCache` is deprecated and will be removed in version v4.59 Use `StaticCache(..., offloading=True)` insteadT©r@  r”   rü   r“  ©r   r@  r”   Úargsr_  r    s        €r   r   zOffloadedStaticCache.__init__{  s-   ø€ Ü×Ñð>ô	
ô 	‰Ñ °mÐPTÐÕUr   ©r!   rJ   rK   r	   rS   r   r�   r‘   s   @r   rž  rž  z  ó    ø„ ðVÐ/ð VÀ÷ Vñ Vr   rž  c                   ó(   ‡ — e Zd Zdedefˆ fd„Zˆ xZS )ÚSlidingWindowCacher@  r”   c                 óR   •— t         j                  d«       t        ‰| �  ||¬«       y )Nz™`SlidingWindowCache` is deprecated and will be removed in version v4.59 Use `StaticCache(...)` instead which will correctly infer the type of each layer.©r@  r”   r“  r¡  s        €r   r   zSlidingWindowCache.__init__„  ó+   ø€ Ü×Ñð`ô	
ô 	‰Ñ °mÐÕDr   r£  r‘   s   @r   r¦  r¦  ƒ  ó    ø„ ðEÐ/ð EÀ÷ Eñ Er   r¦  c                   ó(   ‡ — e Zd Zdedefˆ fd„Zˆ xZS )ÚHybridCacher@  r”   c                 óR   •— t         j                  d«       t        ‰| �  ||¬«       y )Nz’`HybridCache` is deprecated and will be removed in version v4.59 Use `StaticCache(...)` instead which will correctly infer the type of each layer.r¨  r“  r¡  s        €r   r   zHybridCache.__init__�  r©  r   r£  r‘   s   @r   r¬  r¬  Œ  rª  r   r¬  c                   ó(   ‡ — e Zd Zdedefˆ fd„Zˆ xZS )ÚHybridChunkedCacher@  r”   c                 óR   •— t         j                  d«       t        ‰| �  ||¬«       y )Nz™`HybridChunkedCache` is deprecated and will be removed in version v4.59 Use `StaticCache(...)` instead which will correctly infer the type of each layer.r¨  r“  r¡  s        €r   r   zHybridChunkedCache.__init__–  r©  r   r£  r‘   s   @r   r¯  r¯  •  rª  r   r¯  c                   ó(   ‡ — e Zd Zdedefˆ fd„Zˆ xZS )ÚOffloadedHybridCacher@  r”   c                 óT   •— t         j                  d«       t        ‰| �  ||d¬«       y )Nz©`OffloadedHybridCache` is deprecated and will be removed in version v4.59 Use `StaticCache(..., offload=True)` instead which will correctly infer the type of each layer.Tr   r“  r¡  s        €r   r   zOffloadedHybridCache.__init__Ÿ  s.   ø€ Ü×Ñðnô	
ô 	‰Ñ °mÐPTÐÕUr   r£  r‘   s   @r   r²  r²  ž  r¤  r   r²  c                   óD   ‡ — e Zd Z	 	 	 	 	 ddedededededefˆ fd„Zˆ xZS )	ÚQuantoQuantizedCacher@  r»   r¼   r½   r¾   r¿   c           	      óZ   •— t         j                  d«       t        ‰| �  d||||||«       y )Nz~`QuantoQuantizedCache` is deprecated and will be removed in version v4.59 Use `QuantizedCache(backend='quanto', ...)` instead.rd  r“  ©r   r@  r»   r¼   r½   r¾   r¿   r    s          €r   r   zQuantoQuantizedCache.__init__¨  s5   ø€ ô 	×ÑðCô	
ô 	‰Ñ˜ 6¨5°(¸JÈÐVeÕfr   rÒ   r£  r‘   s   @r   rµ  rµ  §  s`   ø„ ð ØØØØ"ñgà ðgð ðgð ð	gð
 ðgð ðgð ÷gñ gr   rµ  c                   óD   ‡ — e Zd Z	 	 	 	 	 ddedededededefˆ fd„Zˆ xZS )	ÚHQQQuantizedCacher@  r»   r¼   r½   r¾   r¿   c           	      óZ   •— t         j                  d«       t        ‰| �  d||||||«       y )Nzx`HQQQuantizedCache` is deprecated and will be removed in version v4.59 Use `QuantizedCache(backend='hqq', ...)` instead.re  r“  r·  s          €r   r   zHQQQuantizedCache.__init__¹  s5   ø€ ô 	×Ñð@ô	
ô 	‰Ñ˜ ¨¨x¸À\ÐSbÕcr   rÒ   r£  r‘   s   @r   r¹  r¹  ¸  s`   ø„ ð ØØØØ"ñdà ðdð ðdð ð	dð
 ðdð ðdð ÷dñ dr   r¹  c                   ó   — e Zd ZdZdd„Zy)Ú	SinkCachea  
    It is now a `custom_generate` repository on the Hub: https://huggingface.co/transformers-community/sink_cache.
    See [these docs](https://huggingface.co/docs/transformers/generation_strategies#custom-decoding-methods) for
    general `custom_generate`usage.
    Nc                 ó   — t        d«      ‚)Nz©`SinkCache` has been moved as a `custom_generate` repository on the Hub: https://huggingface.co/transformers-community/sink_cache. See the repository for usage examples.)r¤   )r   r_  s     r   r   zSinkCache.__init__Ñ  s   € Ü!ðoó
ð 	
r   rI   )r!   rJ   rK   rL   r   r%   r   r   r¼  r¼  É  s   „ ñô
r   r¼  )0Úabcr   r   Úcollections.abcr   Útypingr   r   rN   Úconfiguration_utilsr	   Úutilsr
   r   r   r   r   Úhqq.core.quantizer   rí   r   Ú
get_loggerr!   rV  r   rV   rz   r“   r¬   rº   rÖ   ré   rù   r>  r]  ra  ri  r‘  r˜  r›  rž  r¦  r¬  r¯  r²  rµ  r¹  r¼  r%   r   r   ú<module>rÅ     sœ  ðß #Ý $ß  ã å 1÷õ ñ ÔÝ;á&?ÀÐRVÔ&WÐ #ð 
ˆ×	Ñ	˜HÓ	%€ô7W�cô 7WôtP4�?ô P4ôfN5 ô N5ôbj"�/ô j"ôZz&˜{ô z&ôzM&�\ô M&ô`/$˜>ô /$ôd3˜ô 3÷lq ñ q ôhv�5ô vôrLr�%ô Lrô^5(�Uô 5(ôpK8˜%ô K8ôb8Ð1ô 8ô8Ð2ô 8ô*�\ô *ôV˜;ô VôE˜ô EôE�+ô EôE˜ô EôV˜;ô Vôg˜>ô gô"d˜ô dô"
�õ 
r   