Ë
    îÍ:jUS  ã                   óV  — d Z ddlZddlmZ ddlmZmZmZ ddlZddlm	Z	 ddl
mZ ddlmZ dd	lmZmZ dd
lmZ ddlmZmZmZ ddlmZ  ej2                  e«      Ze G d„ de«      «       Z G d„ de	j:                  «      Z	 d$de	j:                  dej>                  dej>                  dej>                  deej>                     de de fd„Z! G d„ de	j:                  «      Z" G d„ de	j:                  «      Z# G d„ de«      Z$ G d „ d!e	j:                  «      Z% G d"„ d#e	j:                  «      Z&y)%zTPyTorch IdeficsVision model: a copy of CLIPVisionModel using a simpler config objecté    N)Ú	dataclass)ÚCallableÚOptionalÚUnion)Únné   )ÚACT2FN)ÚGradientCheckpointingLayer)ÚBaseModelOutputÚBaseModelOutputWithPooling)ÚALL_ATTENTION_FUNCTIONS)ÚModelOutputÚcan_return_tupleÚloggingé   )ÚIdeficsVisionConfigc                   óÆ   — e Zd ZU dZdZeej                     ed<   dZ	eej                     ed<   dZ
eeej                  df      ed<   dZeeej                  df      ed<   y)ÚIdeficsVisionModelOutputaÝ  
    Base class for vision model's outputs that also contains image embeddings of the pooling of the last hidden states.

    Args:
        image_embeds (`torch.FloatTensor` of shape `(batch_size, output_dim)` *optional* returned when model is initialized with `with_projection=True`):
            The image embeddings obtained by applying the projection layer to the pooler_output.
        last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
            Sequence of hidden-states at the output of the last layer of the model.
        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.

            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
            sequence_length)`.

            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
            heads.
    NÚimage_embedsÚlast_hidden_state.Úhidden_statesÚ
attentions)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   ÚtorchÚFloatTensorÚ__annotations__r   r   Útupler   © ó    úw/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/transformers/models/idefics/vision.pyr   r   '   sr   … ñð* 15€L�(˜5×,Ñ,Ñ-Ó4Ø59Ð�x × 1Ñ 1Ñ2Ó9Ø=A€M�8˜E %×"3Ñ"3°SÐ"8Ñ9Ñ:ÓAØ:>€J�˜˜u×0Ñ0°#Ð5Ñ6Ñ7Ô>r"   r   c                   ó¢   ‡ — e Zd Zdefˆ fd„Zdej                  dededej                  fd„Zddej                  d	e
dej                  fd
„Zˆ xZS )ÚIdeficsVisionEmbeddingsÚconfigc                 óÚ  •— t         ‰| �  «        || _        |j                  | _        |j
                  | _        |j                  | _        t        j                  t        j                  | j                  «      «      | _        t        j                  |j                  | j                  | j                  | j                  d¬«      | _        | j
                  | j                  z  dz  | _        | j                  dz   | _        t        j"                  | j                   | j                  «      | _        | j'                  dt        j(                  | j                   «      j+                  d«      d¬«       y )NF)Úin_channelsÚout_channelsÚkernel_sizeÚstrideÚbiasé   r   Úposition_ids)r   éÿÿÿÿ)Ú
persistent)ÚsuperÚ__init__r&   Úhidden_sizeÚ	embed_dimÚ
image_sizeÚ
patch_sizer   Ú	Parameterr   ÚrandnÚclass_embeddingÚConv2dÚnum_channelsÚpatch_embeddingÚnum_patchesÚnum_positionsÚ	EmbeddingÚposition_embeddingÚregister_bufferÚarangeÚexpand©Úselfr&   Ú	__class__s     €r#   r2   z IdeficsVisionEmbeddings.__init__F   s	  ø€ Ü‰ÑÔØˆŒØ×+Ñ+ˆŒØ ×+Ñ+ˆŒØ ×+Ñ+ˆŒä!Ÿ|™|¬E¯K©K¸¿¹Ó,GÓHˆÔä!Ÿy™yØ×+Ñ+ØŸ™ØŸ™Ø—?‘?Øô 
ˆÔð !ŸO™O¨t¯©Ñ>À1ÑDˆÔØ!×-Ñ-°Ñ1ˆÔÜ"$§,¡,¨t×/AÑ/AÀ4Ç>Á>Ó"RˆÔØ×Ñ˜^¬U¯\©\¸$×:LÑ:LÓ-M×-TÑ-TÐU\Ó-]ÐjoÐÕpr"   Ú
embeddingsÚheightÚwidthÚreturnc                 ó¼  — |j                   d   dz
  }| j                  | j                  «      }|j                   d   dz
  }||k(  r||k(  r|S |dd…df   }|dd…dd…f   }|j                   d   }	|| j                  j                  z  }
|| j                  j                  z  }|
dz   |dz   }}
t        j                  |«      }|j                  dt        |«      t        |«      |	«      }|j                  dddd«      }|j                  t        j                  k(  }|r4t        j                  d«       |j                  t        j                   «      }t"        j$                  j'                  ||
|z  ||z  fd	d
¬«      }|r|j                  t        j                  «      }t        |
«      |j                   d   k7  st        |«      |j                   d   k7  rBt)        dt        |
«      t        |«      f› d|j                   d   |j                   d   f› d�«      ‚|j                  dddd«      j+                  dd|	«      }t        j,                  |j/                  d«      |fd¬«      S )a#  
        This method allows to interpolate the pre-trained position encodings, to be able to use the model on higher
        resolution images.

        Source:
        https://github.com/facebookresearch/dino/blob/de9ee3df6cf39fac952ab558447af1fa1365362a/vision_transformer.py#L174
        r   Nr   r/   gš™™™™™¹?r   r-   zËUpcasting patch_pos_embed to fp32 for interpolation since `upsample_bicubic2d_out_frame` in nn.functional.interpolate is not implemented for 'torch.bfloat16' dtype. This will result in a slight overhead.ÚbicubicF)Úscale_factorÚmodeÚalign_cornerséþÿÿÿzNumber of patches for images (z/) don't match the shape of position embedding (ú)©Údim)Úshaper@   r.   r&   r6   ÚmathÚsqrtÚreshapeÚintÚpermuteÚdtyper   Úbfloat16ÚloggerÚwarning_onceÚtoÚfloatr   Ú
functionalÚinterpolateÚ
ValueErrorÚviewÚcatÚ	unsqueeze)rE   rG   rH   rI   r=   Ú	pos_embedr>   Úclass_pos_embedÚpatch_pos_embedr4   Únum_h_patchesÚnum_w_patchesÚsqrt_num_positionsÚfp32_upcastings                 r#   Úinterpolate_pos_encodingz0IdeficsVisionEmbeddings.interpolate_pos_encoding]   sf  € ð !×&Ñ& qÑ)¨AÑ-ˆØ×+Ñ+¨D×,=Ñ,=Ó>ˆ	Ø!Ÿ™¨Ñ*¨QÑ.ˆØ˜-Ò'¨F°eªOØÐØ#¢A q D™/ˆØ#¢A q¡r EÑ*ˆà×$Ñ$ RÑ(ˆ	Ø $§+¡+×"8Ñ"8Ñ8ˆØ §¡×!7Ñ!7Ñ7ˆð (5°sÑ':¸MÈCÑ<O�}ˆÜ!ŸY™Y }Ó5ÐØ)×1Ñ1°!´SÐ9KÓ5LÌcÐRdÓNeÐgpÓqˆØ)×1Ñ1°!°Q¸¸1Ó=ˆØ(×.Ñ.´%·.±.Ñ@ˆÙÜ×Ñðhôð .×0Ñ0´·±Ó=ˆOÜŸ-™-×3Ñ3ØØ'Ð*<Ñ<¸mÐN`Ñ>`ÐaØØð	 4ó 
ˆñ Ø-×0Ñ0´·±Ó@ˆOÜˆ}Ó ×!6Ñ!6°rÑ!:Ò:¼cÀ-Ó>PÐTc×TiÑTiÐjlÑTmÒ>mÜØ0´°]Ó1CÄSÈÓEWÐ1WÐ0Xð Y0Ø0?×0EÑ0EÀbÑ0IÈ?×K`ÑK`ÐacÑKdÐ0dÐ/eÐefðhóð ð *×1Ñ1°!°Q¸¸1Ó=×BÑBÀ1ÀbÈ)ÓTˆÜ�y‰y˜/×3Ñ3°AÓ6¸ÐHÈaÔPÐPr"   Úpixel_valuesrm   c                 ó`  — |j                   \  }}}}|sJ|| j                  k7  s|| j                  k7  r,t        d|› d|› d| j                  › d| j                  › d�	«      ‚| j                  j                  j
                  }| j                  |j                  |¬«      «      }|j                  d«      j                  dd«      }| j                  j                  |dd«      }	t        j                  |	|gd¬	«      }
|r|
| j                  |
||«      z   }
|
S |
| j                  | j                  «      z   }
|
S )
NzInput image size (Ú*z) doesn't match model (z8). You should try to set `interpolate_pos_encoding=True`)rZ   r-   r   r/   rR   )rT   r5   rb   r<   ÚweightrZ   r^   ÚflattenÚ	transposer9   rC   r   rd   rm   r@   r.   )rE   rn   rm   Ú
batch_sizer;   rH   rI   Útarget_dtypeÚpatch_embedsÚclass_embedsrG   s              r#   ÚforwardzIdeficsVisionEmbeddings.forwardŽ   s8  € Ø2>×2DÑ2DÑ/ˆ
�L &¨%Ù'Ø˜Ÿ™Ò(¨E°T·_±_Ò,DÜ Ø(¨¨°°%°ð 9ØŸ™Ð)¨¨4¯?©?Ð*;Ð;sðuóð ð
 ×+Ñ+×2Ñ2×8Ñ8ˆØ×+Ñ+¨L¯O©OÀ,¨OÓ,OÓPˆà#×+Ñ+¨AÓ.×8Ñ8¸¸AÓ>ˆà×+Ñ+×2Ñ2°:¸qÀ"ÓEˆÜ—Y‘Y ¨lÐ;ÀÔCˆ
ñ $Ø# d×&CÑ&CÀJÐPVÐX]Ó&^Ñ^ˆJð Ðð $ d×&=Ñ&=¸d×>OÑ>OÓ&PÑPˆJàÐr"   ©F)r   r   r   r   r2   r   ÚTensorrX   rm   r   Úboolrx   Ú__classcell__©rF   s   @r#   r%   r%   E   sm   ø„ ðqÐ2õ qð./Q°5·<±<ð /QÈð /QÐUXð /QÐ]b×]iÑ]ió /Qñb E×$5Ñ$5ð ÐQUð Ðbg×bnÑbn÷ r"   r%   ÚmoduleÚqueryÚkeyÚvalueÚattention_maskÚscalingÚdropoutc                 óÀ  — t        j                  ||j                  dd«      «      |z  }|�||z   }t        j                  j                  |dt         j                  ¬«      j                  |j                  «      }t        j                  j                  ||| j                  ¬«      }t        j                  ||«      }	|	j                  dd«      j                  «       }	|	|fS )Nr/   rP   )rS   rZ   )ÚpÚtrainingr   r-   )r   Úmatmulrs   r   r`   ÚsoftmaxÚfloat32r^   rZ   r„   r‡   Ú
contiguous)
r~   r   r€   r�   r‚   rƒ   r„   ÚkwargsÚattn_weightsÚattn_outputs
             r#   Úeager_attention_forwardr�   ©   sº   € ô —<‘<  s§}¡}°R¸Ó'<Ó=ÀÑG€LØÐ!Ø# nÑ4ˆä—=‘=×(Ñ(¨¸2ÄUÇ]Á]Ð(ÓS×VÑVÐW\×WbÑWbÓc€LÜ—=‘=×(Ñ(¨¸È6Ï?É?Ð(Ó[€Lä—,‘,˜|¨UÓ3€KØ×'Ñ'¨¨1Ó-×8Ñ8Ó:€Kà˜Ð$Ð$r"   c                   óÒ   ‡ — e Zd ZdZdefˆ fd„Z	 	 	 d
dej                  deej                     deej                     dee	   de
ej                  eej                     f   f
d	„Zˆ xZS )ÚIdeficsVisionAttentionz=Multi-headed attention from 'Attention Is All You Need' paperr&   c                 ó  •— t         ‰| �  «        || _        |j                  | _        |j
                  | _        | j                  | j                  z  | _        | j                  | j                  z  | j                  k7  r&t        d| j                  › d| j                  › d�«      ‚| j                  dz  | _	        |j                  | _        d| _        t        j                  | j                  | j                  «      | _        t        j                  | j                  | j                  «      | _        t        j                  | j                  | j                  «      | _        t        j                  | j                  | j                  «      | _        y )Nz;embed_dim must be divisible by num_heads (got `embed_dim`: z and `num_heads`: z).g      à¿F)r1   r2   r&   r3   r4   Únum_attention_headsÚ	num_headsÚhead_dimrb   ÚscaleÚattention_dropoutr„   Ú	is_causalr   ÚLinearÚk_projÚv_projÚq_projÚout_projrD   s     €r#   r2   zIdeficsVisionAttention.__init__Ã   s  ø€ Ü‰ÑÔØˆŒØ×+Ñ+ˆŒØ×3Ñ3ˆŒØŸ™¨$¯.©.Ñ8ˆŒØ�=‰=˜4Ÿ>™>Ñ)¨T¯^©^Ò;ÜØMÈdÏnÉnÐM]ð ^Ø—N‘NÐ# 2ð'óð ð —]‘] DÑ(ˆŒ
Ø×/Ñ/ˆŒØˆŒä—i‘i §¡°·±Ó?ˆŒÜ—i‘i §¡°·±Ó?ˆŒÜ—i‘i §¡°·±Ó?ˆŒÜŸ	™	 $§.¡.°$·.±.ÓAˆ�r"   r   r‚   Úcausal_attention_maskÚoutput_attentionsrJ   c           
      ó¤  — |j                   \  }}}| j                  |«      }| j                  |«      }	| j                  |«      }
|j	                  ||| j
                  | j                  «      j                  dd«      }|	j	                  ||| j
                  | j                  «      j                  dd«      }	|
j	                  ||| j
                  | j                  «      j                  dd«      }
| j                  j                  dk7  r|�|�||z   }n|�|}n	|du| _
        t        }| j                  j                  dk7  rt        | j                  j                     } || ||	|
|| j                  | j                  | j                  sdn| j                  ¬«      \  }}|j!                  |||«      j#                  «       }| j%                  |«      }|sd}||fS )z#Input shape: Batch x Time x Channelr   r-   Úflash_attention_2NÚeagerç        )r˜   rƒ   r„   )rT   rœ   rš   r›   rc   r”   r•   rs   r&   Ú_attn_implementationr˜   r�   r   r–   r‡   r„   rW   r‹   r�   )rE   r   r‚   rž   rŸ   rt   Ú
seq_lengthr4   ÚqueriesÚkeysÚvaluesÚattention_interfacerŽ   r�   s                 r#   rx   zIdeficsVisionAttention.forward×   s¬  € ð -:×,?Ñ,?Ñ)ˆ
�J 	à—+‘+˜mÓ,ˆØ�{‰{˜=Ó)ˆØ—‘˜]Ó+ˆà—,‘,˜z¨:°t·~±~ÀtÇ}Á}ÓU×_Ñ_Ð`aÐcdÓeˆØ�y‰y˜ Z°·±ÀÇÁÓO×YÑYÐZ[Ð]^Ó_ˆØ—‘˜Z¨°T·^±^ÀTÇ]Á]ÓS×]Ñ]Ð^_ÐabÓcˆð �;‰;×+Ñ+Ð/BÒBØÐ)Ð.CÐ.OØ!/Ð2GÑ!G‘Ø&Ð2Ø!6‘à2¸$Ð>ˆDŒNä(?ÐØ�;‰;×+Ñ+¨wÒ6Ü"9¸$¿+¹+×:ZÑ:ZÑ"[Ðá$7ØØØØØØ—n‘nØ—J‘JØ#Ÿ}š}‘C°$·,±,ô	%
Ñ!ˆ�\ð "×)Ñ)¨*°jÀ)ÓL×WÑWÓYˆØ—m‘m KÓ0ˆÙ ØˆLØ˜LÐ(Ð(r"   )NNF)r   r   r   r   r   r2   r   rz   r   r{   r    rx   r|   r}   s   @r#   r‘   r‘   À   s†   ø„ ÙGðBÐ2õ Bð. 26Ø8<Ø,1ñ/)à—|‘|ð/)ð ! §¡Ñ.ð/)ð  (¨¯©Ñ5ð	/)ð
 $ D™>ð/)ð 
ˆu�|‰|˜X e§l¡lÑ3Ð3Ñ	4÷/)r"   r‘   c                   óV   ‡ — e Zd Zˆ fd„Zdej
                  dej
                  fd„Zˆ xZS )ÚIdeficsVisionMLPc                 ó  •— t         ‰| �  «        || _        t        |j                     | _        t        j                  |j                  |j                  «      | _
        t        j                  |j                  |j                  «      | _        y ©N)r1   r2   r&   r	   Ú
hidden_actÚactivation_fnr   r™   r3   Úintermediate_sizeÚfc1Úfc2rD   s     €r#   r2   zIdeficsVisionMLP.__init__  sd   ø€ Ü‰ÑÔØˆŒÜ# F×$5Ñ$5Ñ6ˆÔÜ—9‘9˜V×/Ñ/°×1IÑ1IÓJˆŒÜ—9‘9˜V×5Ñ5°v×7IÑ7IÓJˆ�r"   r   rJ   c                 ól   — | j                  |«      }| j                  |«      }| j                  |«      }|S r­   )r±   r¯   r²   )rE   r   s     r#   rx   zIdeficsVisionMLP.forward  s4   € ØŸ™ Ó/ˆØ×*Ñ*¨=Ó9ˆØŸ™ Ó/ˆØÐr"   )r   r   r   r2   r   rz   rx   r|   r}   s   @r#   r«   r«   
  s$   ø„ ôKð U§\¡\ð °e·l±l÷ r"   r«   c                   ó    ‡ — e Zd Zdefˆ fd„Z	 d	dej                  dej                  dej                  dee   de	ej                     f
d„Zˆ xZS )
ÚIdeficsVisionEncoderLayerr&   c                 óD  •— t         ‰| �  «        |j                  | _        t	        |«      | _        t        j                  | j                  |j                  ¬«      | _	        t        |«      | _        t        j                  | j                  |j                  ¬«      | _        y ©N)Úeps)r1   r2   r3   r4   r‘   Ú	self_attnr   Ú	LayerNormÚlayer_norm_epsÚlayer_norm1r«   ÚmlpÚlayer_norm2rD   s     €r#   r2   z"IdeficsVisionEncoderLayer.__init__  sm   ø€ Ü‰ÑÔØ×+Ñ+ˆŒÜ/°Ó7ˆŒÜŸ<™<¨¯©¸F×<QÑ<QÔRˆÔÜ# FÓ+ˆŒÜŸ<™<¨¯©¸F×<QÑ<QÔRˆÕr"   r   r‚   rž   rŸ   rJ   c                 óÎ   — |}| j                  |«      }| j                  ||||¬«      \  }}||z   }|}| j                  |«      }| j                  |«      }||z   }|f}|r||fz  }|S )aI  
        Args:
            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
            attention_mask (`torch.FloatTensor`): attention mask of size
                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
                `(config.encoder_attention_heads,)`.
            output_attentions (`bool`, *optional*):
                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
                returned tensors for more detail.
        )r   r‚   rž   rŸ   )r¼   r¹   r¾   r½   )rE   r   r‚   rž   rŸ   Úresidualr�   Úoutputss           r#   rx   z!IdeficsVisionEncoderLayer.forward#  s’   € ð" !ˆà×(Ñ(¨Ó7ˆØ&*§n¡nØ'Ø)Ø"7Ø/ð	 '5ó '
Ñ#ˆ�|ð ! =Ñ0ˆà ˆØ×(Ñ(¨Ó7ˆØŸ™ Ó/ˆØ  =Ñ0ˆà Ð"ˆáØ˜�Ñ&ˆGàˆr"   ry   )r   r   r   r   r2   r   rz   r   r{   r    r   rx   r|   r}   s   @r#   rµ   rµ     sg   ø„ ðSÐ2õ Sð -2ñ&à—|‘|ð&ð Ÿ™ð&ð  %Ÿ|™|ð	&ð
 $ D™>ð&ð 
ˆu× Ñ Ñ	!÷&r"   rµ   c                   ó®   ‡ — e Zd ZdZdefˆ fd„Ze	 	 	 	 	 ddeej                     deej                     dee
   dee
   dee
   d	eeef   fd
„«       Zˆ xZS )ÚIdeficsVisionEncoderz¿
    Transformer encoder consisting of `config.num_hidden_layers` self attention layers. Each layer is a
    [`IdeficsVisionEncoderLayer`].

    Args:
        config: IdeficsVisionConfig
    r&   c                 óÐ   •— t         ‰| �  «        || _        t        j                  t        |j                  «      D �cg c]  }t        |«      ‘Œ c}«      | _        d| _	        y c c}w )NF)
r1   r2   r&   r   Ú
ModuleListÚrangeÚnum_hidden_layersrµ   ÚlayersÚgradient_checkpointing)rE   r&   Ú_rF   s      €r#   r2   zIdeficsVisionEncoder.__init__V  sQ   ø€ Ü‰ÑÔØˆŒÜ—m‘mÔPUÐV\×VnÑVnÓPoÖ$pÈ1Ô%>¸vÕ%FÒ$pÓqˆŒØ&+ˆÕ#ùò %qs   ½A#r‚   rž   rŸ   Úoutput_hidden_statesÚreturn_dictrJ   c                 ój  — |�|n| j                   j                  }|�|n| j                   j                  }|�|n| j                   j                  }|rdnd}|rdnd}|}	t	        | j
                  «      D ]*  \  }
}|r||	fz   } ||	|||¬«      }|d   }	|sŒ"||d   fz   }Œ, |r||	fz   }t        |	||¬«      S )aÕ  
        Args:
            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
                Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation.
                This is useful if you want more control over how to convert `input_ids` indices into associated vectors
                than the model's internal embedding lookup matrix.
            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            causal_attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
                Causal mask for the text model. Mask values selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            output_attentions (`bool`, *optional*):
                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
                returned tensors for more detail.
            output_hidden_states (`bool`, *optional*):
                Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors
                for more detail.
            return_dict (`bool`, *optional*):
                Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
        Nr!   )rŸ   r   r   )r   r   r   )r&   rŸ   rË   Úuse_return_dictÚ	enumeraterÈ   r   )rE   Úinputs_embedsr‚   rž   rŸ   rË   rÌ   Úencoder_statesÚall_attentionsr   ÚidxÚencoder_layerÚlayer_outputss                r#   rx   zIdeficsVisionEncoder.forward\  s÷   € ðN 2CÐ1NÑ-ÐTX×T_ÑT_×TqÑTqÐà$8Ð$DÑ È$Ï+É+×JjÑJjð 	ð &1Ð%<‘kÀ$Ç+Á+×B]ÑB]ˆá3™¸ˆÙ0™°dˆà%ˆÜ"+¨D¯K©KÓ"8ò 	FÑˆC�Ù#Ø!/°=Ð2BÑ!B�Ù)ØØØ%Ø"3ô	ˆMð *¨!Ñ,ˆMâ Ø!/°=ÀÑ3CÐ2EÑ!E‘ð	Fñ  Ø+¨}Ð.>Ñ>ˆNäØ+¸>ÐVdô
ð 	
r"   )NNNNN)r   r   r   r   r   r2   r   r   r   rz   r{   r   r    r   rx   r|   r}   s   @r#   rÃ   rÃ   M  s¦   ø„ ñð,Ð2õ ,ð ð 26Ø8<Ø,0Ø/3Ø&*ñD
ð ! §¡Ñ.ðD
ð  (¨¯©Ñ5ð	D
ð
 $ D™>ðD
ð ' t™nðD
ð ˜d‘^ðD
ð 
ˆu�oÐ%Ñ	&òD
ó ôD
r"   rÃ   c                   óŒ   ‡ — e Zd Zdefˆ fd„Z	 	 	 	 	 d
deej                     dee   dee   dee   dee   de	e
ef   fd	„Zˆ xZS )ÚIdeficsVisionTransformerr&   c                 ó   •— t         ‰| �  «        || _        |j                  }t	        |«      | _        t        j                  ||j                  ¬«      | _	        t        |«      | _        t        j                  ||j                  ¬«      | _        y r·   )r1   r2   r&   r3   r%   rG   r   rº   r»   Úpre_layrnormrÃ   ÚencoderÚpost_layernorm)rE   r&   r4   rF   s      €r#   r2   z!IdeficsVisionTransformer.__init__¦  sj   ø€ Ü‰ÑÔØˆŒØ×&Ñ&ˆ	ä1°&Ó9ˆŒÜŸL™L¨¸×8MÑ8MÔNˆÔÜ+¨FÓ3ˆŒÜ Ÿl™l¨9¸&×:OÑ:OÔPˆÕr"   rn   rŸ   rË   rm   rÌ   rJ   c                 óÌ  — |�|n| j                   j                  }|�|n| j                   j                  }|�|n| j                   j                  }|€t	        d«      ‚| j                  ||¬«      }| j                  |«      }| j                  ||||¬«      }|d   }|dd…ddd…f   }	| j                  |	«      }	|s
||	f|dd z   S t        ||	|j                  |j                  ¬«      S )z
        Returns:

        Nz You have to specify pixel_values)rm   )rÐ   rŸ   rË   rÌ   r   r   )r   Úpooler_outputr   r   )r&   rŸ   rË   rÎ   rb   rG   rÙ   rÚ   rÛ   r   r   r   )
rE   rn   rŸ   rË   rm   rÌ   r   Úencoder_outputsr   Úpooled_outputs
             r#   rx   z IdeficsVisionTransformer.forward±  s  € ð 2CÐ1NÑ-ÐTX×T_ÑT_×TqÑTqÐà$8Ð$DÑ È$Ï+É+×JjÑJjð 	ð &1Ð%<‘kÀ$Ç+Á+×B]ÑB]ˆàÐÜÐ?Ó@Ð@àŸ™¨ÐOg˜ÓhˆØ×)Ñ)¨-Ó8ˆàŸ,™,Ø'Ø/Ø!5Ø#ð	 'ó 
ˆð ,¨AÑ.ÐØ)ª!¨Q²¨'Ñ2ˆØ×+Ñ+¨MÓ:ˆáØ% }Ð5¸ÈÈÐ8KÑKÐKä)Ø/Ø'Ø)×7Ñ7Ø&×1Ñ1ô	
ð 	
r"   )NNNFN)r   r   r   r   r2   r   r   r   r{   r   r    r   rx   r|   r}   s   @r#   r×   r×   ¥  sˆ   ø„ ðQÐ2õ Qð 59Ø,0Ø/3Ø38Ø&*ñ+
à˜u×0Ñ0Ñ1ð+
ð $ D™>ð+
ð ' t™nð	+
ð
 #+¨4¡.ð+
ð ˜d‘^ð+
ð 
ˆuÐ0Ð0Ñ	1÷+
r"   r×   )r£   )'r   rU   Údataclassesr   Útypingr   r   r   r   r   Úactivationsr	   Úmodeling_layersr
   Úmodeling_outputsr   r   Úmodeling_utilsr   Úutilsr   r   r   Úconfiguration_ideficsr   Ú
get_loggerr   r\   r   ÚModuler%   rz   r_   r�   r‘   r«   rµ   rÃ   r×   r!   r"   r#   ú<module>rê      s3  ðñ [ã Ý !ß ,Ñ ,ã Ý å !Ý 9ß KÝ 5÷ñ õ
 7ð 
ˆ×	Ñ	˜HÓ	%€ð ô?˜{ó ?ó ð?ô:`˜bŸi™iô `ðV ñ%Ø�I‰Ið%à�<‰<ð%ð 
�‰ð%ð �<‰<ð	%ð
 ˜UŸ\™\Ñ*ð%ð ð%ð ó%ô.F)˜RŸY™Yô F)ôT�r—y‘yô ô /Ð :ô /ôfT
˜2Ÿ9™9ô T
ôp7
˜rŸy™yõ 7
r"   