Ë
    îÍ:j/  ã                   óè   — d dl mZmZmZ d dlZddlmZmZm	Z	m
Z
 ddlmZmZmZmZ  e«       r
d dlZddlmZ  e	«       rd dlZ e
j*                  e«      Z e ed¬	«      d
«       G d„ de«      «       Zy)é    )ÚAnyÚUnionÚoverloadNé   )Úadd_end_docstringsÚis_tf_availableÚis_torch_availableÚloggingé   )ÚGenericTensorÚPipelineÚPipelineExceptionÚbuild_pipeline_init_args)Ústable_softmaxT)Úhas_tokenizeraO  
        top_k (`int`, *optional*, defaults to 5):
            The number of predictions to return.
        targets (`str` or `list[str]`, *optional*):
            When passed, the model will limit the scores to the passed targets instead of looking up in the whole
            vocab. If the provided targets are not in the model vocab, they will be tokenized and the first resulting
            token will be used (with a warning, and that might be slower).
        tokenizer_kwargs (`dict`, *optional*):
            Additional dictionary of keyword arguments passed along to the tokenizer.c                   óp  ‡ — e Zd ZdZdZdZdZ	 dedej                  fd„Z
dedej                  fd„Zdefd„Z	 ddeeef   fd	„Zd
„ Zdd„Zd„ Zdd„Zedededeeeef      fd„«       Zedee   dedeeeeef         fd„«       Zdeeee   f   dedeeeeef      eeeeef         f   fˆ fd„Zˆ xZS )ÚFillMaskPipelineFTÚ	input_idsÚreturnc                 ó,  — | j                   dk(  r<t        j                  || j                  j                  k(  «      j                  «       }|S | j                   dk(  r0t        j                  || j                  j                  k(  d¬«      }|S t        d«      ‚)NÚtfÚptF©Úas_tuplezUnsupported framework)	Ú	frameworkr   ÚwhereÚ	tokenizerÚmask_token_idÚnumpyÚtorchÚnonzeroÚ
ValueError)Úselfr   Úmasked_indexs      úu/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/transformers/pipelines/fill_mask.pyÚget_masked_indexz!FillMaskPipeline.get_masked_index\   s€   € Ø�>‰>˜TÒ!ÜŸ8™8 I°·±×1MÑ1MÑ$MÓN×TÑTÓVˆLð
 Ðð	 �^‰^˜tÒ#Ü Ÿ=™=¨°d·n±n×6RÑ6RÑ)RÐ]bÔcˆLð Ðô Ð4Ó5Ð5ó    c                 óà   — | j                  |«      }t        j                  |j                  «      }|dk  r9t	        d| j
                  j                  d| j                  j                  › d�«      ‚y )Nr   ú	fill-maskzNo mask_token (z) found on the input)	r&   ÚnpÚprodÚshaper   ÚmodelÚbase_model_prefixr   Ú
mask_token)r#   r   r$   Únumels       r%   Ú_ensure_exactly_one_mask_tokenz/FillMaskPipeline._ensure_exactly_one_mask_tokene   sg   € Ø×,Ñ,¨YÓ7ˆÜ—‘˜×*Ñ*Ó+ˆØ�1Š9Ü#ØØ—
‘
×,Ñ,Ø! $§.¡.×";Ñ";Ð!<Ð<PÐQóð ð r'   Úmodel_inputsc                 ó˜   — t        |t        «      r|D ]  }| j                  |d   d   «       Œ y |d   D ]  }| j                  |«       Œ y )Nr   r   )Ú
isinstanceÚlistr1   )r#   r2   Úmodel_inputr   s       r%   Úensure_exactly_one_mask_tokenz.FillMaskPipeline.ensure_exactly_one_mask_tokeno   sY   € Ü�l¤DÔ)Ø+ò Q�Ø×3Ñ3°KÀÑ4LÈQÑ4OÕPñQð *¨+Ñ6ò ?�	Ø×3Ñ3°IÕ>ñ?r'   c                 óv   — |€| j                   }|€i } | j                  |fd|i|¤Ž}| j                  |«       |S )NÚreturn_tensors)r   r   r7   )r#   Úinputsr9   Útokenizer_kwargsÚpreprocess_parametersr2   s         r%   Ú
preprocesszFillMaskPipeline.preprocessw   sN   € ð Ð!Ø!Ÿ^™^ˆNØÐ#Ø!Ðà%�t—~‘~ fÑ`¸^Ð`ÐO_Ñ`ˆØ×*Ñ*¨<Ô8ØÐr'   c                 ó:   —  | j                   di |¤Ž}|d   |d<   |S )Nr   © )r-   )r#   r2   Úmodel_outputss      r%   Ú_forwardzFillMaskPipeline._forwardƒ   s*   € Ø"˜Ÿ
™
Ñ2 \Ñ2ˆØ%1°+Ñ%>ˆ�kÑ"ØÐr'   c                 ó   — |�!|j                   d   |k  r|j                   d   }|d   d   }|d   }| j                  dk(  �rt        j                  || j                  j
                  k(  «      j                  «       d d …df   }|j                  «       }|d|d d …f   }t        |d¬«      }|�Pt        j                  t        j                  |d«      |j                  dd«      «      }t        j                  |d«      }t        j                  j                  ||¬«      }	|	j                  j                  «       |	j                  j                  «       }}
nvt!        j"                  || j                  j
                  k(  d	¬
«      j                  d«      }|d|d d …f   }|j%                  d¬«      }|�|d|f   }|j'                  |«      \  }
}g }|
j                   d   dk(  }t)        t+        |
j-                  «       |j-                  «       «      «      D ]è  \  }\  }}g }t+        ||«      D ]¾  \  }}|j                  «       j/                  «       }|�||   j-                  «       }||||   <   |t1        j                  || j                  j2                  k7  «         }| j                  j5                  ||¬«      }||| j                  j5                  |g«      |dœ}|j7                  |«       ŒÀ |j7                  |«       Œê |r|d   S |S )Nr   r   Úlogitsr   éÿÿÿÿ)Úaxisr   )ÚkFr   )Údim.)Úskip_special_tokens)ÚscoreÚtokenÚ	token_strÚsequence)r,   r   r   r   r   r   r   r   Ú	gather_ndÚsqueezeÚreshapeÚexpand_dimsÚmathÚtop_kÚvaluesÚindicesr    r!   ÚsoftmaxÚtopkÚ	enumerateÚzipÚtolistÚcopyr*   Úpad_token_idÚdecodeÚappend)r#   r@   rR   Ú
target_idsr   Úoutputsr$   rC   ÚprobsrV   rS   ÚpredictionsÚresultÚsingle_maskÚiÚ_valuesÚ_predictionsÚrowÚvÚpÚtokensrL   Úpropositions                          r%   ÚpostprocesszFillMaskPipeline.postprocessˆ   sÉ  € àÐ! j×&6Ñ&6°qÑ&9¸EÒ&AØ×$Ñ$ QÑ'ˆEØ! +Ñ.¨qÑ1ˆ	Ø Ñ)ˆà�>‰>˜TÓ!ÜŸ8™8 I°·±×1MÑ1MÑ$MÓN×TÑTÓVÒWXÐZ[ÐW[Ñ\ˆLà—m‘m“oˆGà˜Q ªaÐ/Ñ0ˆFÜ" 6°Ô3ˆEØÐ%ÜŸ™¤R§Z¡Z°°qÓ%9¸:×;MÑ;MÈbÐRSÓ;TÓU�ÜŸ™ u¨aÓ0�ä—7‘7—=‘= ¨%�=Ó0ˆDØ"&§+¡+×"3Ñ"3Ó"5°t·|±|×7IÑ7IÓ7K�K‰Fä Ÿ=™=¨°d·n±n×6RÑ6RÑ)RÐ]bÔc×kÑkÐlnÓoˆLð ˜Q ªaÐ/Ñ0ˆFØ—N‘N r�NÓ*ˆEØÐ%Ø˜c :˜oÑ.�à"'§*¡*¨UÓ"3ÑˆF�KàˆØ—l‘l 1‘o¨Ñ*ˆÜ*3´C¸¿¹»È×I[ÑI[ÓI]Ó4^Ó*_ò 	Ñ&ˆAÑ&�˜ØˆCÜ˜G \Ó2ò (‘��1à"Ÿ™Ó*×/Ñ/Ó1�ØÐ)Ø" 1™×,Ñ,Ó.�Aà*+��| A‘Ñ'à¤§¡¨°4·>±>×3NÑ3NÑ)NÓ OÑP�ð  Ÿ>™>×0Ñ0°È[Ð0ÓY�Ø()°AÀDÇNÁN×DYÑDYÐ[\ÐZ]ÓD^ÐltÑu�Ø—
‘
˜;Õ'ð(ð �M‰M˜#Õð#	ñ$ Ø˜!‘9ÐØˆr'   c           	      óZ  — t        |t        «      r|g}	 | j                  j                  «       }g }|D ]¢  }|j                  |«      }|€|| j                  |ddddd¬«      d   }t        |«      dk(  rt        j                  d|› d�«       ŒX|d   }t        j                  d|› d	| j                  j                  |«      › d
�«       |j                  |«       Œ¤ t        t        |«      «      }t        |«      dk(  rt        d«      ‚t        j                  |«      }|S # t        $ r i }Y Œúw xY w)NFr   T)Úadd_special_tokensÚreturn_attention_maskÚreturn_token_type_idsÚ
max_lengthÚ
truncationr   r   zThe specified target token `zd` does not exist in the model vocabulary. We cannot replace it with anything meaningful, ignoring itz:` does not exist in the model vocabulary. Replacing with `z`.z1At least one target must be provided when passed.)r4   Ústrr   Ú	get_vocabÚ	ExceptionÚgetÚlenÚloggerÚwarningÚconvert_ids_to_tokensr]   r5   Úsetr"   r*   Úarray)r#   ÚtargetsÚvocabr^   ÚtargetÚid_r   s          r%   Úget_target_idszFillMaskPipeline.get_target_ids¿   sX  € Ü�gœsÔ#Ø�iˆGð	Ø—N‘N×,Ñ,Ó.ˆEð ˆ
Øò 	#ˆFØ—)‘)˜FÓ#ˆCØˆ{Ø ŸN™NØØ',Ø*/Ø*/Ø Ø#ð +ó ð ñ�	ô �y“> QÒ&Ü—N‘NØ6°v°hð ?Uð Uôð Ø ‘l�ô
 —‘Ø2°6°(ð ;'Ø'+§~¡~×'KÑ'KÈCÓ'PÐ&QÐQSðUôð ×Ñ˜cÕ"ð5	#ô6 œ#˜j›/Ó*ˆ
Üˆz‹?˜aÒÜÐPÓQÐQÜ—X‘X˜jÓ)ˆ
ØÐøôE ò 	ØŠEð	ús   •D ÄD*Ä)D*c                 óÎ   — i }|�||d<   i }|�| j                  |«      }||d<   |�||d<   | j                  j                  €!t        d| j                  j
                  d«      ‚|i |fS )Nr;   r^   rR   r)   z-The tokenizer does not define a `mask_token`.)r�   r   r   r   r-   r.   )r#   rR   r}   r;   Úpreprocess_paramsÚpostprocess_paramsr^   s          r%   Ú_sanitize_parametersz%FillMaskPipeline._sanitize_parametersè   s‘   € ØÐàÐ'Ø4DÐÐ0Ñ1àÐàÐØ×,Ñ,¨WÓ5ˆJØ/9Ð˜|Ñ,àÐØ*/Ð˜wÑ'à�>‰>×'Ñ'Ð/Ü#Ø˜TŸZ™Z×9Ñ9Ð;jóð ð ! "Ð&8Ð8Ð8r'   r:   Úkwargsc                  ó   — y ©Nr?   ©r#   r:   r†   s      r%   Ú__call__zFillMaskPipeline.__call__ý   s   € ØLOr'   c                  ó   — y rˆ   r?   r‰   s      r%   rŠ   zFillMaskPipeline.__call__   s   € ØX[r'   c                 ón   •— t        ‰| �  |fi |¤Ž}t        |t        «      rt	        |«      dk(  r|d   S |S )a“  
        Fill the masked token in the text(s) given as inputs.

        Args:
            inputs (`str` or `list[str]`):
                One or several texts (or one list of prompts) with masked tokens.
            targets (`str` or `list[str]`, *optional*):
                When passed, the model will limit the scores to the passed targets instead of looking up in the whole
                vocab. If the provided targets are not in the model vocab, they will be tokenized and the first
                resulting token will be used (with a warning, and that might be slower).
            top_k (`int`, *optional*):
                When passed, overrides the number of predictions to return.

        Return:
            A list or a list of list of `dict`: Each result comes as list of dictionaries with the following keys:

            - **sequence** (`str`) -- The corresponding input with the mask token prediction.
            - **score** (`float`) -- The corresponding probability.
            - **token** (`int`) -- The predicted token id (to replace the masked one).
            - **token_str** (`str`) -- The predicted token (to replace the masked one).
        r   r   )ÚsuperrŠ   r4   r5   rw   )r#   r:   r†   r_   Ú	__class__s       €r%   rŠ   zFillMaskPipeline.__call__  s=   ø€ ô0 ‘'Ñ" 6Ñ4¨VÑ4ˆÜ�fœdÔ#¬¨F«°qÒ(8Ø˜1‘:ÐØˆr'   )NN)é   N)NNN)Ú__name__Ú
__module__Ú__qualname__Ú_load_processorÚ_load_image_processorÚ_load_feature_extractorÚ_load_tokenizerr   r*   Úndarrayr&   r1   r7   Údictrs   r=   rA   rl   r�   r…   r   r   r5   rŠ   r   Ú__classcell__)rŽ   s   @r%   r   r      sF  ø„ ð €OØ!ÐØ#ÐØ€Oð2ðh¨-ð ¸B¿J¹Jó ð¸ð È"Ï*É*ó ð?¸-ó ?ð =Añ
à	ˆc�=Ð Ñ	!ó
òó
5òn'óR9ð* ØO˜sÐO¨cÐO°d¸4ÀÀSÀ¹>Ñ6JÒOó ØOàØ[˜t C™yÐ[°CÐ[¸DÀÀdÈ3ÐPSÈ8ÁnÑAUÑ<VÒ[ó Ø[ðØ˜C  c¡˜NÑ+ðØ7:ðà	ˆt�D˜˜c˜‘NÑ# T¨$¨t°C¸°H©~Ñ*>Ñ%?Ð?Ñ	@÷ñ r'   r   )Útypingr   r   r   r   r*   Úutilsr   r   r	   r
   Úbaser   r   r   r   Ú
tensorflowr   Útf_utilsr   r    Ú
get_loggerr�   rx   r   r?   r'   r%   ú<module>r       sx   ðß 'Ñ 'ã ç TÓ Tß VÓ Vñ ÔÛå)ñ ÔÛð 
ˆ×	Ñ	˜HÓ	%€ñ Ù¨4Ô0ðYóô|�xó |óñ|r'   