Ë
    èÍ:j"…  ã                   ó0  — d Z ddlmZmZmZ ddlZddlmZm	Z	m
Z
  G d„ de
j                  «      Z G d„ de
j                  «      Z G d	„ d
e
j                  «      Z G d„ de
j                  «      Z G d„ de
j                  «      Z G d„ de
j                  «      Z G d„ de
j$                  «      Z G d„ de
j(                  «      Z G d„ de
j,                  «      Z G d„ de
j0                  «      Z G d„ de
j,                  «      Z G d„ de
j0                  «      Zy)z0Declares specification of the Transformer model.é    )ÚOptionalÚTupleÚUnionN)Úattention_specÚcommon_specÚ
model_specc            /       óL  — e Zd Zddej                  j
                  dej                  j                  ddddddddddddddddfdedede	d	e	d
ej                  dedej                  de	de	de	de	de	de	de
e   de
e   de
e   de	de
ej                     dedede
e   de
e	   de	f.d„Zy)ÚTransformerEncoderSpecTFé   Né'  Ú
num_layersÚ	num_headsÚpre_normÚno_final_normÚ
activationÚnum_source_embeddingsÚembeddings_mergeÚlayernorm_embeddingÚrelative_positionÚrelative_attention_biasÚffn_gluÚrms_normÚmulti_query_attentionÚnum_heads_kvÚhead_dimÚ
rotary_dimÚrotary_interleaveÚrotary_scaling_typeÚrotary_scaling_factorÚrotary_baseÚsliding_windowÚqk_normÚpre_post_layer_normc                 ó"  — |r|�|dk7  rt        d«      ‚d}|| _        t        j                  d«      j	                  |«      | _        || _        t        j                  d«      j	                  |«      | _        t        j                  d«      j	                  |«      | _        t        |«      D �cg c]  }t        j                  «       ‘Œ c}| _        d| _        |	s|
st        «       | _        |r|st        j                   |¬«      | _        |rt        j                   |¬«      | _        |�)t        j                  d«      j	                  |«      | _        t        |«      D �cg c]  }t)        |	|
||||||||||||¬	«      ‘Œ c}| _        yc c}w c c}w )
a'  Initializes a Transformer encoder specification.

        Args:
          num_layers: Number of layers.
          num_heads: Number of attention heads.
          pre_norm: Enable the pre-norm Transformer architecture.
          no_final_norm: Disable the final layer norm in the pre-norm architecture.
          activation: Activation to apply in the feed-forward network.
          num_source_embeddings: Number of source embeddings.
          embeddings_merge: When :obj:`num_source_embeddings` > 1, specify how the
            embeddings are merged.
          layernorm_embedding: Apply layer normalization after the embedding layer.
          relative_position: Use relative position representations in the self-attention
            layers as described in https://arxiv.org/abs/1803.02155.
          relative_attention_bias: Use relative attention bias in the self-attention
            layers as described in the T5 paper https://arxiv.org/abs/1910.10683.
          ffn_glu: Use gated linear units in the FFN layers as described in
            https://arxiv.org/abs/2002.05202.
          rms_norm: Use the root mean square layer normalization.
          multi_query_attention: Use multi-query attention (alias for num_heads_kv=1).
          num_heads_kv: Number of attention heads for the key and value.
          head_dim: Number of dimensions per attention head.
          rotary_dim: Apply rotary embeddings to these first N dimensions. If 0, rotary
            embeddings are applied to all dimensions.
          rotary_interleave: Interleave the head dimensions when rotary embeddings are applied.
            Otherwise the head dimensions are sliced in half.
          rotary_scaling_type: Type of RoPE scaling.
          rotary_scaling_factor: Factor used in the RoPE scaling.
          rotary_base: The base period of the rotary embeddings.
          sliding_window: Max sequence length to retain in KV Cache.
          qk_norm: Apply layer normalization to the query and key projections.
          pre_post_layer_norm: Add post layer norm for each pre norm layer.
        Nr   ú5Enabling multi_query_attention implies num_heads_kv=1Úint16Úint8T©r   Úint32)r   r   r   r   r   r   r!   r   r   r   r   r    r"   r#   )Ú
ValueErrorr   ÚnpÚdtypeÚtyper   r   r   r   Úranger   ÚEmbeddingsSpecÚ
embeddingsÚscale_embeddingsÚPositionEncoderSpecÚposition_encodingsÚLayerNormSpecÚ
layer_normr   r!   ÚTransformerEncoderLayerSpecÚlayer)Úselfr   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r    r!   r"   r#   Ú_s                            úw/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/ctranslate2/specs/transformer_spec.pyÚ__init__zTransformerEncoderSpec.__init__   sz  € ñx !ØÐ'¨L¸AÒ,=Ü ØKóð ð ˆLà%:ˆÔ"ÜŸ™ 'Ó*×/Ñ/°	Ó:ˆŒØ ˆŒÜŸ(™( 6Ó*×/Ñ/°
Ó;ˆŒÜ "§¡¨Ó 0× 5Ñ 5Ð6FÓ GˆÔä27Ð8MÓ2Nö
Ø-.ŒK×&Ñ&Õ(ò
ˆŒð !%ˆÔÙ Ñ)@Ü&9Ó&;ˆDÔ#Ù™MÜ)×7Ñ7ÀÔJˆDŒOÙÜ'2×'@Ñ'@È(Ô'SˆDÔ$ØÐ%Ü"$§(¡(¨7Ó"3×"8Ñ"8¸Ó"HˆDÔô& ˜:Ó&ö#
ð" ô! (Ø"3Ø(?ØØ!Ø)Ø!Ø-Ø%Ø"3Ø$7Ø&;Ø'ØØ$7öò
ˆ�
ùò
ùò
s   Â-FÅ F)Ú__name__Ú
__module__Ú__qualname__r   Ú
ActivationÚRELUÚEmbeddingsMergeÚCONCATÚintÚboolr   r   ÚRotaryScalingTypeÚfloatr;   © ó    r:   r
   r
   
   s‡  „ ð
 Ø#Ø-8×-CÑ-C×-HÑ-HØ%&Ø8C×8SÑ8S×8ZÑ8ZØ$)Ø"'Ø(-ØØØ&+Ø&*Ø"&Ø$(Ø"&ØJNØ'(Ø"Ø(,Ø"'Ø$)ñ1g
àðg
ð ðg
ð ð	g
ð
 ðg
ð  ×*Ñ*ðg
ð  #ðg
ð &×5Ñ5ðg
ð "ðg
ð  ðg
ð "&ðg
ð ðg
ð ðg
ð  $ðg
ð ˜s‘mðg
ð  ˜3‘-ð!g
ð" ˜S‘Mð#g
ð$  ð%g
ð& & n×&FÑ&FÑGð'g
ð(  %ð)g
ð* ð+g
ð, ! ™ð-g
ð. ˜$‘ð/g
ð0 "ô1g
rH   r
   c            M       ó   — e Zd Zdej                  j
                  ddddddddddddddddddddddddddddddddddf$ded	ed
edej                  dedededededededededededededee   dedee	j                     dedededed ed!ed"ed#ed$ee   d%ee   d&ee   d'eej                     d(ee   d)ee   d*ed+ed,ee   d-efLd.„Zed/„ «       Zy)0ÚTransformerDecoderSpecTFéÿÿÿÿr   Nr   r   r   r   r   r   r   Úwith_encoder_attentionr   Úproject_in_outr   r   Úalignment_layerÚalignment_headsr   r   ÚalibiÚalibi_use_positive_positionsÚscale_alibir   r   r   r   r    Ú original_max_position_embeddingsÚmax_position_embeddingsÚparallel_residualÚshared_layer_normr#   r   r   r   r!   Ú
quant_typeÚquant_group_sizeÚ
quant_bitsr"   Úv_normÚ external_pre_post_encoder_layersÚmerged_encoder_attentionc'           	      ó¢  — t        «       | _        |r|st        d«      ‚|rt        d«      ‚|r|�|dk7  rt        d«      ‚d}t        j                  d«      j                  |«      | _        || _        t        j                  d«      j                  |«      | _        t        j                  d«      j                  |«      | _	        t        j                  d«      j                  |«      | _
        t        j                  «       | _        d| _        t        j                   | _        || _        || _        || _        |�)t        j                  d	«      j                  |«      | _        |	s|
s|s|€t-        «       | _        |r|st        j0                  |¬
«      | _        |rt        j0                  |¬
«      | _        t        j6                  «       | _        t;        |«      D �'cg c]M  }'t=        d&i d|“d|	“d|
“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|#“d|$“d|%“d |&“Ž‘ŒO c}'| _        d!| _         t        j                   | _!        |xs ||k7  | j                  d"<   |r2t        j6                  «       | _"        t        j6                  «       | _#        | r.| | j                  d#<   |"| j                  d$<   |!| j                  d%<   yyc c}'w )'a.  Initializes a Transformer decoder specification.

        Args:
          num_layers: Number of layers.
          num_heads: Number of attention heads.
          pre_norm: Enable the pre-norm Transformer architecture.
          activation: Activation to apply in the feed-forward network.
          layernorm_embedding: Apply layer normalization after the embedding layer.
          with_encoder_attention: Enable the encoder attention sublayers.
          no_final_norm: Disable the final layer norm in the pre-norm architecture.
          project_in_out: Add linear transformations after the embedding layer and before
            the final layer.
          relative_position: Use relative position representations in the self-attention
            layers as described in https://arxiv.org/abs/1803.02155.
          relative_attention_bias: Use relative attention bias in the self-attention
            layers as described in the T5 paper https://arxiv.org/abs/1910.10683.
          alignment_layer: Layer index selected for alignment.
          alignment_heads: Number of attention heads selected for alignment.
          ffn_glu: Use gated linear units in the FFN layers as described in
            https://arxiv.org/abs/2002.05202.
          rms_norm: Use the root mean square layer normalization.
          alibi: Use attention with linear biases.
          alibi_use_positive_positions: Use positive positions in the ALiBi definition.
          scale_alibi: Apply the dot product scale factor to ALiBi.
          rotary_dim: Apply rotary embeddings to these first N dimensions. If 0, rotary
            embeddings are applied to all dimensions.
          rotary_interleave: Interleave the head dimensions when rotary embeddings are applied.
            Otherwise the head dimensions are sliced in half.
          rotary_scaling_type: Type of RoPE scaling.
          rotary_scaling_factor: Factor used in the RoPE scaling.
          rotary_base: The base period of the rotary embeddings.
          original_max_position_embeddings: The original max position embeddings
            for Su rope embeddings
          max_position_embeddings: The max position embeddings for Su rope embeddings
          parallel_residual: Use parallel residual connections in each layer block, as used
            by the GPT-J and GPT-NeoX models.
          shared_layer_norm: When using parallel residual, share the input and post
            attention layer norms.
          pre_post_layer_norm: Add post layer norm for each pre norm layer
          multi_query_attention: Use multi-query attention (alias for num_heads_kv=1).
          num_heads_kv: Number of attention heads for the key and value.
          sliding_window: Max sequence length to retain in KV Cache.
          quant_type: quantization type used (like awq... for lower bit quantization)
          quant_group_size: group size of the lower bit quantization
          quant_bits: number of bit of the quantization (ex: 4bit)
          external_pre_post_encoder_layers: if the encoder attention pre and processing
            is done outside the attention.
        z/The GPT-J block expects a pre-norm architecturez-The GPT-J block does not have cross attentionNr   r%   r&   r'   Tr)   r(   rL   r   r   r   r   r   r   r   r   r    rS   rT   rU   rV   r#   r   r   r!   r"   rZ   r[   r\   Fr   Úquantization_typeÚquantization_bitsÚquantization_group_sizerG   )$ÚdictÚ_configr*   r+   r,   r-   r   r   r   rN   rO   r   r/   r0   r1   r   ÚOPTIONALÚscale_outputsrP   rQ   rR   r!   r2   r3   r4   r5   r   Ú
LinearSpecÚ
projectionr.   ÚTransformerDecoderLayerSpecr7   Ústart_from_zero_embeddingÚfinal_logit_softcappingÚ
project_inÚproject_out)(r8   r   r   r   r   r   rL   r   rM   r   r   rN   rO   r   r   rP   rQ   rR   r   r   r   r   r    rS   rT   rU   rV   r#   r   r   r   r!   rW   rX   rY   r"   rZ   r[   r\   r9   s(                                           r:   r;   zTransformerDecoderSpec.__init__v   s  € ôt “vˆŒÙÙÜ Ð!RÓSÐSÙ%Ü Ð!PÓQÐQá ØÐ'¨L¸AÒ,=Ü ØKóð ð ˆLäŸ™ 'Ó*×/Ñ/°	Ó:ˆŒØ ˆŒÜŸ(™( 6Ó*×/Ñ/°
Ó;ˆŒÜ!Ÿx™x¨Ó0×5Ñ5°oÓFˆÔÜ!Ÿx™x¨Ó0×5Ñ5°oÓFˆÔÜ%×4Ñ4Ó6ˆŒØ $ˆÔÜ'×0Ñ0ˆÔØˆŒ
Ø,HˆÔ)Ø&ˆÔØÐ%Ü"$§(¡(¨7Ó"3×"8Ñ"8¸Ó"HˆDÔá!Ù+ÙØÐ"ä&9Ó&;ˆDÔ#Ù™MÜ)×7Ñ7ÀÔJˆDŒOÙÜ'2×'@Ñ'@È(Ô'SˆDÔ$Ü%×0Ñ0Ó2ˆŒô4 ˜:Ó&ö3
ð2 ô1 (ò Ù'=ðá"3ðñ )@ðñ  ð	ñ
 "ðñ &ðñ #4ðñ %8ðñ '<ðñ (ðñ 2Rðñ )@ðñ #4ðñ #4ðñ %8ðñ  *ð!ñ" "ð#ñ$  .ð%ñ&  ð'ñ( ð)ñ* 2Rð+ñ, *Bò-ò
ˆŒ
ð8 */ˆÔ&Ü'1×':Ñ':ˆÔ$Ø0Eò 1
Ø˜IÑ%ð 	�‰Ð,Ñ-ñ Ü)×4Ñ4Ó6ˆDŒOÜ*×5Ñ5Ó7ˆDÔáØ0:ˆD�L‰LÐ,Ñ-Ø0:ˆD�L‰LÐ,Ñ-Ø6FˆD�L‰LÐ2Ò3ð ùòM
s   ÇAKc                 ó   — | j                   S ©N)rb   ©r8   s    r:   ÚconfigzTransformerDecoderSpec.config"  s   € à�|‰|ÐrH   )r<   r=   r>   r   r?   r@   rC   rD   r   r   rE   rF   ÚQuantizationr;   Úpropertyro   rG   rH   r:   rJ   rJ   u   s§  „ ð
 Ø-8×-CÑ-C×-HÑ-HØ$)Ø'+Ø#Ø$Ø"'Ø(-Ø!Ø ØØØØ-2Ø!Ø$(Ø"&ØJNØ'(Ø"Ø01Ø'(Ø"'Ø"'Ø$)Ø&+Ø&*Ø"&Ø(,Ø9=Ø*.Ø$(ØØØ;@Ø).ñOjGàðjGð ðjGð ð	jGð
  ×*Ñ*ðjGð "ðjGð !%ðjGð ðjGð ðjGð  ðjGð "&ðjGð ðjGð ðjGð ðjGð ðjGð  ð!jGð" '+ð#jGð$ ð%jGð& ˜S‘Mð'jGð(  ð)jGð* & n×&FÑ&FÑGð+jGð,  %ð-jGð. ð/jGð0 +.ð1jGð2 "%ð3jGð4  ð5jGð6  ð7jGð8 "ð9jGð:  $ð;jGð< ˜s‘mð=jGð> ˜3‘-ð?jGð@ ! ™ðAjGðB ˜[×5Ñ5Ñ6ðCjGðD # 3™-ðEjGðF ˜S‘MðGjGðH ðIjGðJ ðKjGðL +3°4©.ðMjGðN #'óOjGðX ñó ñrH   rJ   c                   ój   — e Zd Z	 	 	 	 	 	 	 	 	 	 	 	 	 	 d	dee   dedeej                     dededefd„Z	y)
r6   Nr   r   r   r   r    r#   c                 ó¬  — t        j                  d||||||||	|
|||¬«      | _        t        ||¬«      | _        |r™t        j                  |¬«      | _        t        j                  |¬«      | _        t        j                  |¬«      | _	        t        j                  |¬«      | _
        t        | j                  d«       t        | j                  d«       y y )NT)Úself_attentionr   r   r   r   r   r!   r   r   r   r   r    r"   ©Úglur   r(   r5   )r   ÚMultiHeadAttentionSpecrt   ÚFeedForwardSpecÚffnr   r4   Úinput_layer_normÚpost_attention_layer_normÚpre_feedforward_layer_normÚpost_feedforward_layer_normÚdelattr)r8   r   r   r   r   r   r   r!   r   r   r   r   r    r"   r#   s                  r:   r;   z$TransformerEncoderLayerSpec.__init__(  sÏ   € ô" -×CÑCØØ/Ø$;ØØ%ØØ)Ø!Ø/Ø 3Ø"7Ø#Øô
ˆÔô # w¸ÔBˆŒáÜ$/×$=Ñ$=ÀxÔ$PˆDÔ!Ü-8×-FÑ-FØ!ô.ˆDÔ*ô /:×.GÑ.GØ!ô/ˆDÔ+ô 0;×/HÑ/HØ!ô0ˆDÔ,ô �D×'Ñ'¨Ô6Ü�D—H‘H˜lÕ+ð rH   )FFFFNNNNTNr   r   FF)
r<   r=   r>   r   rC   rD   r   rE   rF   r;   rG   rH   r:   r6   r6   '  s~   „ ð  Ø %ØØØØØØ$(Ø"&ØJNØ'(Ø"ØØ$)ñ/,ð ˜S‘Mð/,ð  ð/,ð & n×&FÑ&FÑGð/,ð  %ð/,ð ð/,ð "ô/,rH   r6   c                   ó@   — e Zd Z	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dd„Zy)rg   Nc           	      óÔ  — t        j                  di dd“d|“d|“d|“d|“d|“d|“d	|	“d
|
“d|“d|“d|“d|“d|“d|“d|“d|“Ž| _        |r$|s"t        j                  ||||||du ¬«      | _        t	        ||¬«      | _        |rz|rt        j                  «       | _        n2t        j                  «       | _	        t        j                  «       | _
        t        | j                  d«       t        | j
                  d«       |rÒt        j                  |¬«      | _	        t        j                  |¬«      | _
        |r8|r6t        j                  |¬«      | _        t        j                  |¬«      | _        t        j                  |¬«      | _        t        j                  |¬«      | _        t        | j                  d«       t        | j
                  d«       t         j"                  | _        y )Nrt   Tr   r   r   r   r   r   r   r    rS   rT   r   r   r!   r"   rZ   r\   F)r   r   r   r!   r"   Úhas_normru   r5   r(   rG   )r   rw   rt   Ú	attentionrx   ry   r   r4   rV   rz   r{   r~   Ú*external_post_encoder_attention_layer_normÚ)external_pre_encoder_attention_layer_normr|   r}   r   rc   Úlayer_scalar)r8   rL   r   r   r   r   r   r   r   r   r    rS   rT   rU   rV   r#   r   r   r!   r"   rZ   r[   r\   s                          r:   r;   z$TransformerDecoderLayerSpec.__init__[  s  € ô2 -×CÑCò 
Ùð
á/ð
ñ %<ð
ñ ð	
ñ
 "ð
ñ 0ð
ñ !4ð
ñ #8ð
ñ $ð
ñ .Nð
ñ %<ð
ñ &ð
ñ ð
ñ *ð
ñ ð
ñ  ð!
ñ" &>ð#
ˆÔñ( "Ñ*BÜ+×BÑBØ!Ø)Ø!Ø-ØØ9¸UÐBôˆDŒNô # w¸ÔBˆŒáÙ Ü)4×)BÑ)BÓ)D�Õ&ä(3×(AÑ(AÓ(C�Ô%Ü1<×1JÑ1JÓ1L�Ô.ä�D×'Ñ'¨Ô6Ü�D—H‘H˜lÔ+áä$/×$=Ñ$=ÀxÔ$PˆDÔ!Ü-8×-FÑ-FØ!ô.ˆDÔ*ñ &Ñ*Jä×-Ñ-°xÔ@ð Ô?ô  ×-Ñ-°xÔ@ð Ô>ô
 /:×.GÑ.GØ!ô/ˆDÔ+ô 0;×/HÑ/HØ!ô0ˆDÔ,ô �D×'Ñ'¨Ô6Ü�D—H‘H˜lÔ+ä&×/Ñ/ˆÕrH   )TFFFFNTNr   r   r   r   FFFNNNFFFF©r<   r=   r>   r;   rG   rH   r:   rg   rg   Z  sL   „ ð  $ØØ %ØØØØØ ØØØ)*Ø !ØØØ!ØØØØØØ).Ø!&ô/]0rH   rg   c                   ó   — e Zd Zdd„Zy)rx   c                 óÖ   — t        j                  |¬«      | _        t        j                  «       | _        t        j                  «       | _        |rt        j                  «       | _        y y )Nr(   )r   r4   r5   re   Úlinear_0Úlinear_1Úlinear_0_noact)r8   rv   r   s      r:   r;   zFeedForwardSpec.__init__¼  sM   € Ü%×3Ñ3¸XÔFˆŒÜ#×.Ñ.Ó0ˆŒÜ#×.Ñ.Ó0ˆŒÙÜ"-×"8Ñ"8Ó":ˆDÕð rH   N)FFr†   rG   rH   r:   rx   rx   »  s   „ ô;rH   rx   c                   ó   — e Zd Zd„ Zy)r2   c                 ó.   — t         j                  | _        y rm   )r   rc   Ú	encodingsrn   s    r:   r;   zPositionEncoderSpec.__init__Å  s   € Ü#×,Ñ,ˆ�rH   Nr†   rG   rH   r:   r2   r2   Ä  s   „ ó-rH   r2   c                   ó0   ‡ — e Zd ZdZddee   fˆ fd„Zˆ xZS )ÚTransformerConfigz%Configuration for Transformer models.Úlayer_norm_epsilonc                 ó(   •— t        ‰| �  dd|i|¤Ž y)z·Initializes the configuration for Transformer models.

        Args:
          layer_norm_epsilon: The layer norm epsilon value.
          **kwargs: Additional configuration.
        r‘   NrG   ©Úsuperr;   ©r8   r‘   ÚkwargsÚ	__class__s      €r:   r;   zTransformerConfig.__init__Ì  ó   ø€ ô 	‰ÑÑIÐ,>ÐIÀ&ÓIrH   rm   ©r<   r=   r>   Ú__doc__r   rF   r;   Ú__classcell__©r—   s   @r:   r�   r�   É  s   ø„ Ù/ñJ¨8°E©?÷ Jñ JrH   r�   c                    óT  ‡ — e Zd ZdZdedefˆ fd„Zedddej                  j                  dddej                  j                  dddddfd	eeeeef   f   d
ededededej                  dedededej                  dededededefd„«       Zed„ «       Zed„ «       Zd„ Zd„ Zd„ Zˆ xZS )ÚTransformerSpecz©Describes a Transformer model.

    The specification is invariant to hidden dimensions but requires to
    explicitly set the number of layers and attention heads.
    ÚencoderÚdecoderc                 ó
  •— t        |t        «      st        d«      ‚t        |t        «      st        d«      ‚t        ‰| �  «        || _        || _        | j                  j                  d| j                  j                  «       y)z¢Initializes a Transformer model specification.

        Args:
          encoder: The encoder specification.
          decoder: The decoder specification.
        ú1encoder argument must be a TransformerEncoderSpecú1decoder argument must be a TransformerDecoderSpecr   N)Ú
isinstancer
   Ú	TypeErrorrJ   r”   r;   rŸ   r    rb   Úadd_attributer   )r8   rŸ   r    r—   s      €r:   r;   zTransformerSpec.__init__Ý  sm   ø€ ô ˜'Ô#9Ô:ÜÐOÓPÐPÜ˜'Ô#9Ô:ÜÐOÓPÐPä‰ÑÔØˆŒØˆŒØ�‰×"Ñ"Ø# T§\¡\×%GÑ%Gõ	
rH   FTrK   r   r   r   Úwith_relative_positionr   r   r   rN   rO   r   r   r   r   r   r   r   c                 ó´   — t        |t        t        f«      r|\  }}n||}}t        ||||||	|
||||||¬«      }t	        |||||||||||||¬«      } | ||«      S )a•  Creates a Transformer model specification.

        Args:
          num_layers: Number of encoder and decoder layers, or a 2-tuple if the
            number is different.
          num_heads: Number of attention heads.
          with_relative_position: Use relative position representations in the self-attention
            layers as described in https://arxiv.org/abs/1803.02155.
          pre_norm: Enable the pre-norm Transformer architecture.
          no_final_norm: Disable the final layer norm in the pre-norm architecture.
          activation: Activation to apply in the feed-forward network.
          alignment_layer: Layer index selected for alignment.
          alignment_heads: Number of attention heads selected for alignment.
          num_source_embeddings: Number of source embeddings.
          embeddings_merge: When :obj:`num_source_embeddings` > 1, specify how the
            embeddings are merged.
          layernorm_embedding: Apply layer normalization after the embedding layer.
          relative_attention_bias: Use relative attention bias in the self-attention
            layers as described in the T5 paper https://arxiv.org/abs/1910.10683.
          ffn_glu: Use gated linear units in the FFN layer as described in
            https://arxiv.org/abs/2002.05202.
          rms_norm: Use the root mean square layer normalization.
          multi_query_attention: Use multi-query attention.
        )r   r   r   r   r   r   r   r   r   r   r   )r   r   r   r   r   r   rN   rO   r   r   r   )r¤   ÚlistÚtupler
   rJ   )Úclsr   r   r§   r   r   r   rN   rO   r   r   r   r   r   r   r   Únum_encoder_layersÚnum_decoder_layersrŸ   r    s                       r:   Úfrom_configzTransformerSpec.from_configò  s�   € ôV �j¤4¬ -Ô0Ø5?Ñ2ÐÑ 2à5?ÀÐ 2Ðä(ØØØØ'Ø!Ø"7Ø-Ø 3Ø4Ø$;ØØØ"7ô
ˆô  )ØØØØ'Ø!Ø 3Ø4Ø$;Ø+Ø+ØØØ"7ô
ˆñ  �7˜GÓ$Ð$rH   c                  ó   — y)Nrž   rG   rn   s    r:   ÚnamezTransformerSpec.nameD  s   € à rH   c                  ó   — y)Né   rG   rn   s    r:   ÚrevisionzTransformerSpec.revisionH  ó   € àrH   c                 ó   — t        «       S rm   )r�   rn   s    r:   Úget_default_configz"TransformerSpec.get_default_configL  s   € Ü Ó"Ð"rH   c                 ó‚   — | j                   j                  D �cg c]  }|j                  j                  d   ‘Œ c}S c c}w ©Nr   ©rŸ   r0   ÚweightÚshape)r8   Úspecs     r:   Úget_source_vocabulary_sizez*TransformerSpec.get_source_vocabulary_sizeO  s/   € Ø15·±×1HÑ1HÖI¨�—‘×!Ñ! !Ó$ÒIÐIùÒIs   ™ <c                 ó\   — | j                   j                  j                  j                  d   S r¸   ©r    r0   rº   r»   rn   s    r:   Úget_target_vocabulary_sizez*TransformerSpec.get_target_vocabulary_sizeR  ó#   € Ø�|‰|×&Ñ&×-Ñ-×3Ñ3°AÑ6Ð6rH   )r<   r=   r>   rš   r
   rJ   r;   Úclassmethodr   r?   r@   rA   rB   r   rC   r   rD   r®   rq   r°   r³   r¶   r½   rÀ   r›   rœ   s   @r:   rž   rž   Ö  sn  ø„ ñð
Ø-ð
Ø8Nõ
ð* ð
 (-ØØ#Ø-8×-CÑ-C×-HÑ-HØ!Ø Ø%&Ø8C×8SÑ8S×8ZÑ8ZØ$)Ø(-ØØØ&+ñ!O%à˜#˜u S¨# X™Ð.Ñ/ðO%ð ðO%ð !%ð	O%ð
 ðO%ð ðO%ð  ×*Ñ*ðO%ð ðO%ð ðO%ð  #ðO%ð &×5Ñ5ðO%ð "ðO%ð "&ðO%ð ðO%ð ðO%ð   $ò!O%ó ðO%ðb ñ!ó ð!ð ñó ðò#òJö7rH   rž   c                   ó0   ‡ — e Zd ZdZddee   fˆ fd„Zˆ xZS )ÚTransformerDecoderModelConfigz-Configuration for Transformer decoder models.r‘   c                 ó(   •— t        ‰| �  dd|i|¤Ž y)z¿Initializes the configuration for Transformer decoder models.

        Args:
          layer_norm_epsilon: The layer norm epsilon value.
          **kwargs: Additional configuration.
        r‘   NrG   r“   r•   s      €r:   r;   z&TransformerDecoderModelConfig.__init__Y  r˜   rH   rm   r™   rœ   s   @r:   rÄ   rÄ   V  ó   ø„ Ù7ñJ¨8°E©?÷ Jñ JrH   rÄ   c            B       ó¸  ‡ — e Zd ZdZdefˆ fd„Zedej                  j                  ddddddddddddddd	d	ddddddddddddfd
e
de
dedej                  dedededededededededee
   dedeej                     dedede
de
deded ed!ed"ee
   d#ee
   d$ee
   d%eej                      d&ee
   d'ee
   d(ed)ef@d*„«       Zed+„ «       Zed,„ «       Zd-„ Zd.„ Zˆ xZS )/ÚTransformerDecoderModelSpecz3Describes a Transformer decoder model (e.g. GPT-2).r    c                 óö   •— t        |t        «      st        d«      ‚t        ‰| �  «        || _        | j
                  j                  j                  «       D ]!  \  }}| j                  j                  ||«       Œ# y)z|Initializes a Transformer decoder model specification.

        Args:
          decoder: The decoder specification.
        r£   N)
r¤   rJ   r¥   r”   r;   r    ro   Úitemsrb   r¦   )r8   r    ÚkeyÚvaluer—   s       €r:   r;   z$TransformerDecoderModelSpec.__init__f  sh   ø€ ô ˜'Ô#9Ô:ÜÐOÓPÐPä‰ÑÔØˆŒØŸ,™,×-Ñ-×3Ñ3Ó5ò 	3‰JˆC�Ø�L‰L×&Ñ& s¨EÕ2ñ	3rH   TFNr   r   r   r   r   r   r   r   r   rM   r§   r   r   rP   rQ   rR   r   r   r   r   r    rS   rT   rU   rV   r#   r   r   r   r!   rW   rX   rY   r"   rZ   c!                 óâ   — t        ||fi d|“d|“d|“dd“d|“d|“d|“d	|	“d
|
“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d|“d | “Ž}! | |!«      S )!a!
  Creates a Transformer decoder model specification.

        Args:
          num_layers: Number of decoder layers.
          num_heads: Number of attention heads.
          pre_norm: Enable the pre-norm Transformer architecture.
          activation: Activation to apply in the feed-forward network.
          layernorm_embedding: Apply layer normalization after the embedding layer.
          no_final_norm: Do not apply layer normalization after the last decoder block.
          project_in_out: Add a linear layer after the embedding layer and another one
            before the final output projection.
          with_relative_position: Enable relative position representations modules.
          ffn_glu: Use gated linear units in the FFN layers as described in
            https://arxiv.org/abs/2002.05202.
          rms_norm: Use the root mean square layer normalization.
          alibi: Use attention with linear biases.
          alibi_use_positive_positions: Use positive positions in the ALiBi definition.
          scale_alibi: Apply the dot product scale factor to ALiBi.
          rotary_dim: Apply rotary embeddings to these first N dimensions. If 0, rotary
            embeddings are applied to all dimensions.
          rotary_interleave: Interleave the head dimensions when rotary embeddings are applied.
            Otherwise the head dimensions are sliced in half.
          rotary_scaling_type: Type of RoPE scaling.
          rotary_scaling_factor: Factor used in the RoPE scaling.
          rotary_base: The base period of the rotary embeddings.
          original_max_position_embeddings: The original max position embeddings
            for Su rope embeddings
          max_position_embeddings: The max position embeddings for Su rope embeddings
          parallel_residual: Use parallel residual connections in each layer block, as used
            by the GPT-J and GPT-NeoX models.
          shared_layer_norm: When using parallel residual, share the input and post
            attention layer norms.
          pre_post_layer_norm: add post layer norm for each pre norm layer
          multi_query_attention: Use multi-query attention (alias for num_heads_kv=1).
          num_heads_kv: Number of attention heads for the key and value.
          head_dim: Number of head
          sliding_window: max sequence length to retain KV cache
          quant_type: quantization type used (like awq... for lower bit quantization)
          quant_group_size: group size of the lower bit quantization
          quant_bits: number of bit of the quantization (ex: 4bit)
        r   r   r   rL   Fr   rM   r   r   r   rP   rQ   rR   r   r   r   r   r    rS   rT   rU   rV   r#   r   r   r   r!   rW   rX   rY   r"   rZ   )rJ   )"r«   r   r   r   r   r   r   rM   r§   r   r   rP   rQ   rR   r   r   r   r   r    rS   rT   rU   rV   r#   r   r   r   r!   rW   rX   rY   r"   rZ   r    s"                                     r:   r®   z'TransformerDecoderModelSpec.from_configt  s^  € ôZ )ØØò"
ñ ð"
ñ "ð	"
ñ
 !4ð"
ñ $)ð"
ñ (ð"
ñ *ð"
ñ 5ð"
ñ ð"
ñ ð"
ñ ð"
ñ *Fð"
ñ $ð"
ñ "ð"
ñ  0ð!"
ñ" !4ð#"
ñ$ #8ð%"
ñ& $ð'"
ñ( .Nð)"
ñ* %<ð+"
ñ, 0ð-"
ñ. 0ð/"
ñ0 !4ð1"
ñ2 #8ð3"
ñ4 &ð5"
ñ6 ð7"
ñ8 *ð9"
ñ: "ð;"
ñ< .ð="
ñ> "ð?"
ñ@ ðA"
ñB ðC"
ˆñH �7‹|ÐrH   c                  ó   — y)NrJ   rG   rn   s    r:   r°   z TransformerDecoderModelSpec.nameç  ó   € à'rH   c                  ó   — y)Né   rG   rn   s    r:   r³   z$TransformerDecoderModelSpec.revisionë  r´   rH   c                 ó   — t        «       S rm   )rÄ   rn   s    r:   r¶   z.TransformerDecoderModelSpec.get_default_configï  ó   € Ü,Ó.Ð.rH   c                 ó\   — | j                   j                  j                  j                  d   S r¸   r¿   rn   s    r:   Úget_vocabulary_sizez/TransformerDecoderModelSpec.get_vocabulary_sizeò  rÁ   rH   )r<   r=   r>   rš   rJ   r;   rÂ   r   r?   r@   rC   rD   r   r   rE   rF   rp   r®   rq   r°   r³   r¶   rÕ   r›   rœ   s   @r:   rÈ   rÈ   c  s[  ø„ Ù=ð3Ð 6õ 3ð ð
 Ø-8×-CÑ-C×-HÑ-HØ$)Ø#Ø$Ø',ØØØØ-2Ø!Ø$(Ø"&ØJNØ'(Ø"Ø01Ø'(Ø"'Ø"'Ø$)Ø&+Ø&*Ø"&Ø(,Ø9=Ø*.Ø$(ØØñCpàðpð ðpð ð	pð
  ×*Ñ*ðpð "ðpð ðpð ðpð !%ðpð ðpð ðpð ðpð '+ðpð ðpð ˜S‘Mðpð   ð!pð" & n×&FÑ&FÑGð#pð$  %ð%pð& ð'pð( +.ð)pð* "%ð+pð,  ð-pð.  ð/pð0 "ð1pð2  $ð3pð4 ˜s‘mð5pð6 ˜3‘-ð7pð8 ! ™ð9pð: ˜[×5Ñ5Ñ6ð;pð< # 3™-ð=pð> ˜S‘Mð?pð@ ðApðB òCpó ðpðd ñ(ó ð(ð ñó ðò/ö7rH   rÈ   c                   ó0   ‡ — e Zd ZdZddee   fˆ fd„Zˆ xZS )ÚTransformerEncoderModelConfigz-Configuration for Transformer encoder models.r‘   c                 ó(   •— t        ‰| �  dd|i|¤Ž y)z¿Initializes the configuration for Transformer encoder models.

        Args:
          layer_norm_epsilon: The layer norm epsilon value.
          **kwargs: Additional configuration.
        r‘   NrG   r“   r•   s      €r:   r;   z&TransformerEncoderModelConfig.__init__ù  r˜   rH   rm   r™   rœ   s   @r:   r×   r×   ö  rÆ   rH   r×   c                   óž   ‡ — e Zd ZdZdej
                  j                  fdededej
                  fˆ fd„Z	e
d„ «       Ze
d„ «       Zd	„ Zd
„ Zˆ xZS )ÚTransformerEncoderModelSpecz2Describes a Transformer encoder model (e.g. BERT).FrŸ   Úpooling_layerÚpooling_activationc                 óP  •— t        |t        «      st        d«      ‚t        ‰| �  «        || _        | j                  j                  d| j
                  j                  «       |rCt        j                  «       | _        t        j                  d«      j                  |«      | _        yy)zûInitializes a Transformer encoder model specification.

        Args:
          encoder: The encoder specification.
          pooling_layer: Add the pooling layer.
          pooling_activation: The activation to apply after the pooling layer.
        r¢   r   r'   N)r¤   r
   r¥   r”   r;   rŸ   rb   r¦   r   r   re   Úpooler_denser+   r,   r-   Úpooler_activation)r8   rŸ   rÛ   rÜ   r—   s       €r:   r;   z$TransformerEncoderModelSpec.__init__  s‡   ø€ ô ˜'Ô#9Ô:ÜÐOÓPÐPä‰ÑÔØˆŒØ�‰×"Ñ"Ø# T§\¡\×%GÑ%Gô	
ñ Ü +× 6Ñ 6Ó 8ˆDÔÜ%'§X¡X¨fÓ%5×%:Ñ%:Ð;MÓ%NˆDÕ"ð rH   c                  ó   — y)Nr
   rG   rn   s    r:   r°   z TransformerEncoderModelSpec.name   rÏ   rH   c                  ó   — y)Nr   rG   rn   s    r:   r³   z$TransformerEncoderModelSpec.revision$  r´   rH   c                 ó   — t        «       S rm   )r×   rn   s    r:   r¶   z.TransformerEncoderModelSpec.get_default_config(  rÓ   rH   c                 ób   — | j                   j                  d   j                  j                  d   S r¸   r¹   rn   s    r:   rÕ   z/TransformerEncoderModelSpec.get_vocabulary_size+  s(   € Ø�|‰|×&Ñ& qÑ)×0Ñ0×6Ñ6°qÑ9Ð9rH   )r<   r=   r>   rš   r   r?   ÚTanhr
   rD   r;   rq   r°   r³   r¶   rÕ   r›   rœ   s   @r:   rÚ   rÚ     sw   ø„ Ù<ð
 $Ø5@×5KÑ5K×5PÑ5Pñ	Oà'ðOð ðOð (×2Ñ2õ	Oð4 ñ(ó ð(ð ñó ðò/ö:rH   rÚ   )rš   Útypingr   r   r   Únumpyr+   Úctranslate2.specsr   r   r   Ú	LayerSpecr
   rJ   r6   rg   rx   r2   ÚSequenceToSequenceModelConfigr�   ÚSequenceToSequenceModelSpecrž   ÚLanguageModelConfigrÄ   ÚLanguageModelSpecrÈ   r×   rÚ   rG   rH   r:   ú<module>rí      s  ðÙ 6ç )Ñ )ã ç EÑ Eôh
˜Z×1Ñ1ô h
ôVo˜Z×1Ñ1ô oôd0, *×"6Ñ"6ô 0,ôf^0 *×"6Ñ"6ô ^0ôB;�j×*Ñ*ô ;ô-˜*×.Ñ.ô -ô

J˜
×@Ñ@ô 
Jô}7�j×<Ñ<ô }7ô@
J J×$BÑ$Bô 
JôP7 *×">Ñ">ô P7ôf
J J×$BÑ$Bô 
Jô): *×">Ñ">õ ):rH   