Ë
    îÍ:jo  ã                   ód  — d dl mZ d dlZddlmZ ddlmZ 	  e«       r	d dlmZ eZ	n e
d«      ‚	 	 	 	 	 	 	 	 dd	ej                  j                   d
ej"                  dej"                  dej"                  deej"                     dedej"                  fd„Zy# e$ rZ ee«      Zd„ Z	Y dZ[ŒydZ[ww xY w)é    )ÚOptionalNé   )ÚPagedAttentionCache)Úis_flash_attn_2_available)Úflash_attn_varlen_funczŽFlash Attention 2 is not installed. Please refer to https://huggingface.co/docs/transformers/perf_infer_gpu_one#flashattention-2 to install itc                  ó&   — t        dt        › �«      ‚)Nz)flash_attn_varlen_func is not available: )Ú	ExceptionÚmsg)ÚargsÚkwargss     úz/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/transformers/integrations/flash_paged.pyÚFLASH_ATTN_VARLEN_FUNCr      s   € ÜÐCÄCÀ5ÐIÓJÐJó    ÚmoduleÚqÚkÚvÚattention_maskÚcacheÚreturnc           	      óÒ  — t        | dd«      sdn| j                  dz
  df}|dk(  rdnd}|�" |j                  ||| j                  fi |¤Ž\  }}t	        |t
        «      r
||   }|	|   }	|
�t        |
d	«      r|
j                  }nt        }d
|v rd
|j                  d
«      ini } ||j                  dd«      j                  d«      j                  «       |j                  «       |j                  «       |j                  t        j                  «      |j                  t        j                  «      j!                  «       ||	f| j"                  d|dœ|¤Ž}t	        |t$        «      r|d   }|dfS )aæ  Perform the forward pass of attention with paged key-value cache.

    This function handles the cache updates and performs the attention computation
    using the flash_attn_varlen_func for efficient processing.

    Args:
        q: (total_q, nheads, headdim), where total_q = total number of query tokens in the batch.
        k: (total_k, nheads_k, headdim), where total_k = total number of key tokens in the batch.  but if there is a block table it can be the full k
        v: (total_k, nheads_k, headdim), where total_k = total number of key tokens in the batch.  but if there is a block table it can be the full v
        cu_seq_lens_q: (batch_size + 1,), dtype torch.int32. The cumulative sequence lengths
           of the sequences in the batch, used to index into q.
        cu_seq_lens_k: (batch_size + 1,), dtype torch.int32. The cumulative sequence lengths
           of the sequences in the batch, used to index into kv.
        max_seqlen_q: int. Maximum query sequence length in the batch.
        max_seqlen_k: int. Maximum key sequence length in the batch.
        dropout_p: float. Dropout probability.
        softmax_scale: float. The scaling of QK^T before applying softmax.
            Default to 1 / sqrt(headdim).
        causal: bool. Whether to apply causal attention mask (e.g., for auto-regressive modeling).
        window_size: (left, right). If not (-1, -1), implements sliding window local attention.
        softcap: float. Anything > 0 activates softcapping attention.
    Úsliding_windowF)éÿÿÿÿr   é   r   Úfull_attentionÚsliding_attentionNr   Ús_auxr   T)Úsoftmax_scaleÚcausalÚwindow_size)Úgetattrr   ÚupdateÚ	layer_idxÚ
isinstanceÚdictÚhasattrr   r   ÚgetÚ	transposeÚsqueezeÚ
contiguousÚtoÚtorchÚint32ÚcloneÚscalingÚtuple)r   r   r   r   r   r   Úcu_seq_lens_qÚcu_seq_lens_kÚmax_seqlen_qÚmax_seqlen_kÚimplementationr   r   Ú
layer_typer   Úcustom_kwargsÚattn_outputs                    r   Úpaged_attention_forwardr9      ss  € ôH &-¨VÐ5EÀuÔ%M‘XÐTZ×TiÑTiÐlmÑTmÐopÐSq€NØ%3°xÒ%?Ñ!ÐEX€Jð ÐØˆu�|‰|˜A˜q &×"2Ñ"2Ñ=°fÑ=‰ˆˆ1ô �-¤Ô&Ø% jÑ1ˆØ# JÑ/ˆàÐ!¤g¨nÐ>VÔ&WØ!/×!FÑ!FÑä!7Ðà6=ÀÑ6G�W˜fŸj™j¨Ó1Ñ2ÈR€Má(Ø	�‰�A�qÓ×!Ñ! !Ó$×/Ñ/Ó1Ø	�‰‹Ø	�‰‹Ø×ÑœŸ™Ó%Ø×ÑœŸ™Ó%×+Ñ+Ó-ØØðð —n‘nØØ"ñð ñ€Kô �+œuÔ%Ø! !‘nˆØ˜ÐÐr   )NNNNNNN)Útypingr   r,   Úgeneration.continuous_batchingr   Úutilsr   Ú
flash_attnr   r   ÚRuntimeErrorr	   ÚeÚreprr
   ÚnnÚModuleÚTensorr9   © r   r   ú<module>rE      sè   ðÝ ã å @Ý -ðKÙ Ô"Ý5à!7Ñáð ]ó
ð 	
ð 	ð" .2Ø!%ØØØØØñFØ�H‰H�O‰OðFà‡|�|ðFð ‡|�|ðFð ‡|�|ð	Fð
 ˜UŸ\™\Ñ*ðFð ðFð ‡\�\ôFøð ò KÙ
ˆq‹'€C÷KûðKús   ˜B ÂB/ÂB*Â*B/