Ë
    îÍ:j¡  ã                   óˆ  — d dl Z d dlZd dlZd dlmZ d dlmZ d dlmZm	Z	m
Z
mZ d dlZddlmZ ddlmZmZ ddlmZ dd	lmZmZmZ dd
lmZmZ ddlmZmZmZm Z m!Z!m"Z"m#Z#m$Z$m%Z%m&Z&m'Z'm(Z(m)Z) ddl*m+Z+ ddl,m-Z- ddl.m/Z/m0Z0m1Z1m2Z2m3Z3m4Z4m5Z5m6Z6m7Z7  e&«       rd dl8Z8 e(«       rd dl9m:Z;  e)jx                  e=«      Z>dZ? e!de?«       e-d¬«       G d„ de«      «       «       Z@ e"e@j‚                  «      e@_A        e@j‚                  j„                  �8e@j‚                  j„                  j‡                  ddd¬«      e@j‚                  _B        yy)é    N)Údeepcopy)Úpartial)ÚAnyÚCallableÚOptionalÚUnioné   )Úcustom_object_save)ÚBatchFeatureÚget_size_dict)ÚBaseImageProcessorFast)ÚChannelDimensionÚSizeDictÚvalidate_kwargs)ÚUnpackÚVideosKwargs)ÚIMAGE_PROCESSOR_NAMEÚPROCESSOR_NAMEÚVIDEO_PROCESSOR_NAMEÚ
TensorTypeÚadd_start_docstringsÚ	copy_funcÚdownload_urlÚis_offline_modeÚis_remote_urlÚis_torch_availableÚis_torchcodec_availableÚis_torchvision_v2_availableÚlogging)Úcached_file)Úrequires)	Ú
VideoInputÚVideoMetadataÚgroup_videos_by_shapeÚis_valid_videoÚ
load_videoÚmake_batched_metadataÚmake_batched_videosÚreorder_videosÚto_channel_dimension_format)Ú
functionalaÊ  
    Args:
        do_resize (`bool`, *optional*, defaults to `self.do_resize`):
            Whether to resize the video's (height, width) dimensions to the specified `size`. Can be overridden by the
            `do_resize` parameter in the `preprocess` method.
        size (`dict`, *optional*, defaults to `self.size`):
            Size of the output video after resizing. Can be overridden by the `size` parameter in the `preprocess`
            method.
        size_divisor (`int`, *optional*, defaults to `self.size_divisor`):
            The size by which to make sure both the height and width can be divided.
        default_to_square (`bool`, *optional*, defaults to `self.default_to_square`):
            Whether to default to a square video when resizing, if size is an int.
        resample (`PILImageResampling`, *optional*, defaults to `self.resample`):
            Resampling filter to use if resizing the video. Only has an effect if `do_resize` is set to `True`. Can be
            overridden by the `resample` parameter in the `preprocess` method.
        do_center_crop (`bool`, *optional*, defaults to `self.do_center_crop`):
            Whether to center crop the video to the specified `crop_size`. Can be overridden by `do_center_crop` in the
            `preprocess` method.
        crop_size (`dict[str, int]` *optional*, defaults to `self.crop_size`):
            Size of the output video after applying `center_crop`. Can be overridden by `crop_size` in the `preprocess`
            method.
        do_rescale (`bool`, *optional*, defaults to `self.do_rescale`):
            Whether to rescale the video by the specified scale `rescale_factor`. Can be overridden by the
            `do_rescale` parameter in the `preprocess` method.
        rescale_factor (`int` or `float`, *optional*, defaults to `self.rescale_factor`):
            Scale factor to use if rescaling the video. Only has an effect if `do_rescale` is set to `True`. Can be
            overridden by the `rescale_factor` parameter in the `preprocess` method.
        do_normalize (`bool`, *optional*, defaults to `self.do_normalize`):
            Whether to normalize the video. Can be overridden by the `do_normalize` parameter in the `preprocess`
            method. Can be overridden by the `do_normalize` parameter in the `preprocess` method.
        image_mean (`float` or `list[float]`, *optional*, defaults to `self.image_mean`):
            Mean to use if normalizing the video. This is a float or list of floats the length of the number of
            channels in the video. Can be overridden by the `image_mean` parameter in the `preprocess` method. Can be
            overridden by the `image_mean` parameter in the `preprocess` method.
        image_std (`float` or `list[float]`, *optional*, defaults to `self.image_std`):
            Standard deviation to use if normalizing the video. This is a float or list of floats the length of the
            number of channels in the video. Can be overridden by the `image_std` parameter in the `preprocess` method.
            Can be overridden by the `image_std` parameter in the `preprocess` method.
        do_convert_rgb (`bool`, *optional*, defaults to `self.image_std`):
            Whether to convert the video to RGB.
        video_metadata (`VideoMetadata`, *optional*):
            Metadata of the video containing information about total duration, fps and total number of frames.
        do_sample_frames (`int`, *optional*, defaults to `self.do_sample_frames`):
            Whether to sample frames from the video before processing or to process the whole video.
        num_frames (`int`, *optional*, defaults to `self.num_frames`):
            Maximum number of frames to sample when `do_sample_frames=True`.
        fps (`int` or `float`, *optional*, defaults to `self.fps`):
            Target frames to sample per second when `do_sample_frames=True`.
        return_tensors (`str` or `TensorType`, *optional*):
            Returns stacked tensors if set to `pt, otherwise returns a list of tensors.
        data_format (`ChannelDimension` or `str`, *optional*, defaults to `ChannelDimension.FIRST`):
            The channel dimension format for the output video. Can be one of:
            - `"channels_first"` or `ChannelDimension.FIRST`: video in (num_channels, height, width) format.
            - `"channels_last"` or `ChannelDimension.LAST`: video in (height, width, num_channels) format.
            - Unset: Use the channel dimension format of the input video.
        input_data_format (`ChannelDimension` or `str`, *optional*):
            The channel dimension format for the input video. If unset, the channel dimension format is inferred
            from the input video. Can be one of:
            - `"channels_first"` or `ChannelDimension.FIRST`: video in (num_channels, height, width) format.
            - `"channels_last"` or `ChannelDimension.LAST`: video in (height, width, num_channels) format.
            - `"none"` or `ChannelDimension.NONE`: video in (height, width) format.
        device (`torch.device`, *optional*):
            The device to process the videos on. If unset, the device is inferred from the input videos.
        return_metadata (`bool`, *optional*):
            Whether to return video metadata or not.
        z!Constructs a base VideoProcessor.)ÚvisionÚtorchvision)Úbackendsc                   óö  ‡ — e Zd ZdZdZdZdZdZdZdZ	dZ
dZdZdZdZdZdZdZdZdZdZdZeZdgZdee   ddfˆ fd„Zdefd	„Zd
ddefd„Z	 	 d?dede e!   de e"e!e#f      fd„Z$	 	 d?dede"ee%f   de e&   de e'   de(d   f
d„Z)	 	 d?dede e"e*e+f      de e*   de(d   fd„Z, e-e.«      dedee   defd„«       Z/	 d@de(d   de&de&de0de d   de&d e0d!e&d"e#d#e&d$e e"e#e(e#   f      d%e e"e#e(e#   f      d&e e"e*e1f      defd'„Z2e3	 	 	 	 	 dAd(e"e*e4jj                  f   d)e e"e*e4jj                  f      d*e&d+e&d,e e"e*e&f      d-e*fd.„«       Z6dBd/e"e*e4jj                  f   d0e&fd1„Z7e3d(e"e*e4jj                  f   de8e%e*e9f   e%e*e9f   f   fd2„«       Z:e3d3e%e*e9f   fd4„«       Z;de%e*e9f   fd5„Z<de*fd6„Z=d7e"e*e4jj                  f   fd8„Z>d9„ Z?e3d:e"e*e4jj                  f   fd;„«       Z@e3dCd<„«       ZAd@d=e"e*e(e*   e(e(e*      f   fd>„ZBˆ xZCS )DÚBaseVideoProcessorNTgp?FÚpixel_values_videosÚkwargsÚreturnc                 ó  •— t         ‰| �  «        |j                  dd «      | _        |j	                  «       D ]  \  }}	 t        | ||«       Œ |j                  d| j                  «      }|�'t        ||j                  d| j                  «      ¬«      nd | _	        |j                  d| j                  «      }|�t        |d¬	«      nd | _        t        | j                  j                  j!                  «       «      | _        | j"                  D ]E  }|j%                  |«      �t        | |||   «       Œ%t        | |t'        t)        | |d «      «      «       ŒG y # t        $ r%}t        j                  d|› d|› d| › �«       |‚d }~ww xY w)
NÚprocessor_classz
Can't set z with value z for ÚsizeÚdefault_to_square)r6   r7   Ú	crop_size)Ú
param_name)ÚsuperÚ__init__ÚpopÚ_processor_classÚitemsÚsetattrÚAttributeErrorÚloggerÚerrorr6   r   r7   r8   ÚlistÚvalid_kwargsÚ__annotations__ÚkeysÚmodel_valid_processing_keysÚgetr   Úgetattr)Úselfr2   ÚkeyÚvalueÚerrr6   r8   Ú	__class__s          €úx/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/transformers/video_processing_utils.pyr;   zBaseVideoProcessor.__init__®   sj  ø€ Ü‰ÑÔà &§
¡
Ð+<¸dÓ CˆÔð !Ÿ,™,›.ò 	‰JˆC�ðÜ˜˜c 5Õ)ð	ð �z‰z˜& $§)¡)Ó,ˆð Ðô ˜t°v·z±zÐBUÐW[×WmÑWmÓ7nÕoàð 	Œ	ð
 —J‘J˜{¨D¯N©NÓ;ˆ	ØMVÐMbœ y¸[ÕIÐhlˆŒô ,0°×0AÑ0A×0QÑ0Q×0VÑ0VÓ0XÓ+YˆÔ(Ø×3Ñ3ò 	GˆCØ�z‰z˜#‹Ð*Ü˜˜c 6¨#¡;Õ/ä˜˜c¤8¬G°D¸#¸tÓ,DÓ#EÕFñ		Gøô! "ò Ü—‘˜z¨#¨¨l¸5¸'ÀÀtÀfÐMÔNØ�	ûðús   ¾EÅ	F Å E;Å;F c                 ó(   —  | j                   |fi |¤ŽS ©N)Ú
preprocess)rJ   Úvideosr2   s      rO   Ú__call__zBaseVideoProcessor.__call__Í   s   € Øˆt�‰˜vÑ0¨Ñ0Ð0ó    Úvideoztorch.Tensorc                 ó  — t        j                  |«      }|j                  d   dk(  s|dddd…dd…f   dk  j                  «       s|S |dddd…dd…f   dz  }d|dddd…dd…f   z
  dz  |dddd…dd…f   |ddd…dd…dd…f   z  z   }|S )zÏ
        Converts a video to RGB format.

        Args:
            video (`"torch.Tensor"`):
                The video to convert.

        Returns:
            `torch.Tensor`: The converted video.
        éýÿÿÿé   .Néÿ   g     ào@r	   )ÚFÚgrayscale_to_rgbÚshapeÚany)rJ   rV   Úalphas      rO   Úconvert_to_rgbz!BaseVideoProcessor.convert_to_rgbÐ   s¯   € ô ×"Ñ" 5Ó)ˆØ�;‰;�r‰?˜aÒ¨¨c°1²aº¨lÑ(;¸cÑ(A×'FÑ'FÔ'HØˆLð �c˜1ša¢�lÑ# eÑ+ˆØ�U˜3 ¢aª˜?Ñ+Ñ+¨sÑ2°U¸3ÀÂaÊ¸?Ñ5KÈeÐTWÐY[ÐZ[ÐY[Ò]^Ò`aÐTaÑNbÑ5bÑbˆØˆrU   ÚmetadataÚ
num_framesÚfpsc                 óº  — |�|�t        d«      ‚|�|n| j                  }|�|n| j                  }|j                  }|€6|�4|�|j                  €t        d«      ‚t	        ||j                  z  |z  «      }||kD  rt        d|› d|› d�«      ‚|�*t        j                  d|||z  «      j	                  «       }|S t        j                  d|«      j	                  «       }|S )a%  
        Default sampling function which uniformly samples the desired number of frames between 0 and total number of frames.
        If `fps` is passed along with metadata, `fps` frames per second are sampled uniformty. Arguments `num_frames`
        and `fps` are mutually exclusive.

        Args:
            metadata (`VideoMetadata`):
                Metadata of the video containing information about total duration, fps and total number of frames.
            num_frames (`int`, *optional*):
                Maximum number of frames to sample. Defaults to `self.num_frames`.
            fps (`int` or `float`, *optional*):
                Target frames to sample per second. Defaults to `self.fps`.

        Returns:
            np.ndarray:
                Indices to sample video frames.
        zc`num_frames`, `fps`, and `sample_indices_fn` are mutually exclusive arguments, please use only one!zÈAsked to sample `fps` frames per second but no video metadata was provided which is required when sampling with `fps`. Please pass in `VideoMetadata` object or use a fixed `num_frames` per input videoz(Video can't be sampled. The `num_frames=z` exceeds `total_num_frames=z`. r   )Ú
ValueErrorrb   rc   Útotal_num_framesÚintÚtorchÚarange)rJ   ra   rb   rc   r2   rf   Úindicess          rO   Úsample_framesz BaseVideoProcessor.sample_framesé   s  € ð0 ˆ?˜zÐ5ÜØuóð ð $.Ð#9‘Z¸t¿¹ˆ
Ø�_‰c¨$¯(©(ˆØ#×4Ñ4Ðð Ð # /ØÐ 8§<¡<Ð#7Ü ðhóð ô Ð-°·±Ñ<¸sÑBÓCˆJàÐ(Ò(ÜØ:¸:¸,ÐFbÐcsÐbtÐtwÐxóð ð Ð!Ü—l‘l 1Ð&6Ð8HÈ:Ñ8UÓV×ZÑZÓ\ˆGð ˆô —l‘l 1Ð&6Ó7×;Ñ;Ó=ˆGØˆrU   rS   Úvideo_metadataÚdo_sample_framesÚsample_indices_fnc                 óV  — t        |«      }t        ||¬«      }t        |d   «      rW|rUg }g }t        ||«      D ]:  \  }} ||¬«      }	|	|_        |j                  ||	   «       |j                  |«       Œ< |}|}||fS t        |d   «      s�t        |d   t        «      rg| j                  |«      D �
�cg c]:  }
t        j                  |
D �cg c]  }t        j                  |«      ‘Œ c}d¬«      ‘Œ< }}
}|rt        d«      ‚||fS | j                  ||¬«      \  }}||fS c c}w c c}}
w )zB
        Decode input videos and sample frames if needed.
        )rl   r   )ra   ©ÚdimzUSampling frames from a list of images is not supported! Set `do_sample_frames=False`.©rn   )r(   r'   r%   ÚzipÚframes_indicesÚappendÚ
isinstancerC   Úfetch_imagesrh   Ústackr[   Úpil_to_tensorre   Úfetch_videos)rJ   rS   rl   rm   rn   Úsampled_videosÚsampled_metadatarV   ra   rj   ÚimagesÚimages               rO   Ú_decode_and_sample_videosz,BaseVideoProcessor._decode_and_sample_videos  sV  € ô % VÓ,ˆÜ.¨vÀnÔUˆô ˜& ™)Ô$Ñ)9ØˆNØ!ÐÜ#& v¨~Ó#>ò 2‘��xÙ+°XÔ>�Ø*1�Ô'Ø×%Ñ% e¨G¡nÔ5Ø ×'Ñ'¨Õ1ð	2ð
 $ˆFØ-ˆNð �~Ð%Ð%ô    q¡	Ô*Ü˜& ™)¤TÔ*ð #'×"3Ñ"3°FÓ";÷àô —K‘KÀVÖ L¸E¤§¡°Õ!7Ò LÐRSÖTð�ñ ñ $Ü$Øoóð ð �~Ð%Ð%ð *.×):Ñ):¸6ÐUfÐ):Ó)gÑ&�˜à�~Ð%Ð%ùò !Mùós   Â3D%ÃD Ã'D%Ä D%Úinput_data_formatÚdevicec                 ó  — g }|D ]~  }t        |t        j                  «      r>t        |t        j
                  |«      }t        j                  |«      j                  «       }|�|j                  |«      }|j                  |«       Œ€ |S )z:
        Prepare the input videos for processing.
        )rv   ÚnpÚndarrayr*   r   ÚFIRSTrh   Ú
from_numpyÚ
contiguousÚtoru   )rJ   rS   r€   r�   Úprocessed_videosrV   s         rO   Ú_prepare_input_videosz(BaseVideoProcessor._prepare_input_videosF  s€   € ð ÐØò 
	+ˆEä˜%¤§¡Ô,Ü3°EÔ;K×;QÑ;QÐSdÓe�ä×(Ñ(¨Ó/×:Ñ:Ó<�àÐ!ØŸ™ Ó(�à×#Ñ# EÕ*ð
	+ð  ÐrU   c           	      óà  — t        |j                  «       t        | j                  j                  j                  «       «      dgz   ¬«       | j                  j                  D ]  }|j                  |t        | |d «      «       Œ! |j                  d«      }|j                  d«      }|j                  d«      }|j                  d«      }|rt        | j                  fi |¤Žnd }| j                  ||||¬«      \  }}| j                  |||¬«      } | j                  di |¤Ž} | j                  di |¤Ž |j                  d	«       |j                  d
«      }	 | j                  dd|i|¤Ž}
|	r||
d<   |
S )NÚreturn_tensors)Úcaptured_kwargsÚvalid_processor_keysr€   rm   r�   rl   )rl   rm   rn   )rS   r€   r�   Údata_formatÚreturn_metadatarS   © )r   rF   rC   rD   rE   Ú
setdefaultrI   r<   r   rk   r   rŠ   Ú_further_process_kwargsÚ_validate_preprocess_kwargsÚ_preprocess)rJ   rS   r2   Ú
kwarg_namer€   rm   r�   rl   rn   r�   Úpreprocessed_videoss              rO   rR   zBaseVideoProcessor.preprocess]  s„  € ô 	Ø"ŸK™K›MÜ!% d×&7Ñ&7×&GÑ&G×&LÑ&LÓ&NÓ!OÐScÐRdÑ!dõ	
ð ×+Ñ+×;Ñ;ò 	KˆJØ×Ñ˜j¬'°$¸
ÀDÓ*IÕJð	Kð #ŸJ™JÐ':Ó;ÐØ!Ÿ:™:Ð&8Ó9ÐØ—‘˜HÓ%ˆØŸ™Ð$4Ó5ˆáEUœG D×$6Ñ$6ÑA¸&ÒAÐ[_ÐØ!%×!?Ñ!?ØØ)Ø-Ø/ð	 "@ó "
Ñˆ�ð ×+Ñ+°6ÐM^ÐgmÐ+Ónˆà-�×-Ñ-Ñ7°Ñ7ˆØ(ˆ×(Ñ(Ñ2¨6Ò2ð 	�
‰
�=Ô!Ø Ÿ*™*Ð%6Ó7ˆà.˜d×.Ñ.ÑG°fÐGÀÑGÐÙØ4BÐÐ 0Ñ1Ø"Ð"rU   Údo_convert_rgbÚ	do_resizer6   ÚinterpolationzF.InterpolationModeÚdo_center_cropr8   Ú
do_rescaleÚrescale_factorÚdo_normalizeÚ
image_meanÚ	image_stdrŒ   c           	      óà  — t        |«      \  }}i }|j                  «       D ]3  \  }}|r| j                  |«      }|r| j                  |||¬«      }|||<   Œ5 t	        ||«      }t        |«      \  }}i }|j                  «       D ]4  \  }}|r| j                  ||«      }| j                  |||	|
||«      }|||<   Œ6 t	        ||«      }|rt        j                  |d¬«      n|}t        d|i|¬«      S )N)r6   rš   r   rp   r1   )ÚdataÚtensor_type)
r$   r>   r`   Úresizer)   Úcenter_cropÚrescale_and_normalizerh   rx   r   )rJ   rS   r˜   r™   r6   rš   r›   r8   rœ   r�   rž   rŸ   r    rŒ   r2   Úgrouped_videosÚgrouped_videos_indexÚresized_videos_groupedr]   Ústacked_videosÚresized_videosÚprocessed_videos_groupedr‰   s                          rO   r•   zBaseVideoProcessor._preprocessˆ  s1  € ô$ 0EÀVÓ/LÑ,ˆÐ,Ø!#ÐØ%3×%9Ñ%9Ó%;ò 	;Ñ!ˆE�>ÙØ!%×!4Ñ!4°^Ó!D�ÙØ!%§¡¨^À$ÐVc Ó!d�Ø,:Ð" 5Ò)ð	;ô (Ð(>Ð@TÓUˆô 0EÀ^Ó/TÑ,ˆÐ,Ø#%Ð Ø%3×%9Ñ%9Ó%;ò 	=Ñ!ˆE�>ÙØ!%×!1Ñ!1°.À)Ó!L�à!×7Ñ7Ø 
¨N¸LÈ*ÐV_óˆNð /=Ð$ UÒ+ð	=ô *Ð*BÐDXÓYÐÙCQœ5Ÿ;™;Ð'7¸QÕ?ÐWgÐäÐ"7Ð9IÐ!JÐXfÔgÐgrU   Úpretrained_model_name_or_pathÚ	cache_dirÚforce_downloadÚlocal_files_onlyÚtokenÚrevisionc                 ó  — ||d<   ||d<   ||d<   ||d<   |j                  dd«      }|�)t        j                  dt        «       |�t	        d«      ‚|}|�||d	<    | j
                  |fi |¤Ž\  }	} | j                  |	fi |¤ŽS )
a  
        Instantiate a type of [`~video_processing_utils.VideoProcessorBase`] from an video processor.

        Args:
            pretrained_model_name_or_path (`str` or `os.PathLike`):
                This can be either:

                - a string, the *model id* of a pretrained video hosted inside a model repo on
                  huggingface.co.
                - a path to a *directory* containing a video processor file saved using the
                  [`~video_processing_utils.VideoProcessorBase.save_pretrained`] method, e.g.,
                  `./my_model_directory/`.
                - a path or url to a saved video processor JSON *file*, e.g.,
                  `./my_model_directory/video_preprocessor_config.json`.
            cache_dir (`str` or `os.PathLike`, *optional*):
                Path to a directory in which a downloaded pretrained model video processor should be cached if the
                standard cache should not be used.
            force_download (`bool`, *optional*, defaults to `False`):
                Whether or not to force to (re-)download the video processor files and override the cached versions if
                they exist.
            resume_download:
                Deprecated and ignored. All downloads are now resumed by default when possible.
                Will be removed in v5 of Transformers.
            proxies (`dict[str, str]`, *optional*):
                A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
                'http://hostname': 'foo.bar:4012'}.` The proxies are used on each request.
            token (`str` or `bool`, *optional*):
                The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
                the token generated when running `hf auth login` (stored in `~/.huggingface`).
            revision (`str`, *optional*, defaults to `"main"`):
                The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a
                git-based system for storing models and other artifacts on huggingface.co, so `revision` can be any
                identifier allowed by git.


                <Tip>

                To test a pull request you made on the Hub, you can pass `revision="refs/pr/<pr_number>"`.

                </Tip>

            return_unused_kwargs (`bool`, *optional*, defaults to `False`):
                If `False`, then this function returns just the final video processor object. If `True`, then this
                functions returns a `Tuple(video_processor, unused_kwargs)` where *unused_kwargs* is a dictionary
                consisting of the key/value pairs whose keys are not video processor attributes: i.e., the part of
                `kwargs` which has not been used to update `video_processor` and is otherwise ignored.
            subfolder (`str`, *optional*, defaults to `""`):
                In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can
                specify the folder name here.
            kwargs (`dict[str, Any]`, *optional*):
                The values in kwargs of any keys which are video processor attributes will be used to override the
                loaded values. Behavior concerning key/value pairs whose keys are *not* video processor attributes is
                controlled by the `return_unused_kwargs` keyword parameter.

        Returns:
            A video processor of type [`~video_processing_utils.ImagVideoProcessorBase`].

        Examples:

        ```python
        # We can't instantiate directly the base class *VideoProcessorBase* so let's show the examples on a
        # derived class: *LlavaOnevisionVideoProcessor*
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained(
            "llava-hf/llava-onevision-qwen2-0.5b-ov-hf"
        )  # Download video_processing_config from huggingface.co and cache.
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained(
            "./test/saved_model/"
        )  # E.g. video processor (or model) was saved using *save_pretrained('./test/saved_model/')*
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained("./test/saved_model/video_preprocessor_config.json")
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained(
            "llava-hf/llava-onevision-qwen2-0.5b-ov-hf", do_normalize=False, foo=False
        )
        assert video_processor.do_normalize is False
        video_processor, unused_kwargs = LlavaOnevisionVideoProcessor.from_pretrained(
            "llava-hf/llava-onevision-qwen2-0.5b-ov-hf", do_normalize=False, foo=False, return_unused_kwargs=True
        )
        assert video_processor.do_normalize is False
        assert unused_kwargs == {"foo": False}
        ```r®   r¯   r°   r²   Úuse_auth_tokenNúrThe `use_auth_token` argument is deprecated and will be removed in v5 of Transformers. Please use `token` instead.úV`token` and `use_auth_token` are both specified. Please set only the argument `token`.r±   )r<   ÚwarningsÚwarnÚFutureWarningre   Úget_video_processor_dictÚ	from_dict)
Úclsr­   r®   r¯   r°   r±   r²   r2   r´   Úvideo_processor_dicts
             rO   Úfrom_pretrainedz"BaseVideoProcessor.from_pretrained¶  s¿   € ðt (ˆˆ{ÑØ#1ˆÐÑ Ø%5ˆÐ!Ñ"Ø%ˆˆzÑàŸ™Ð$4°dÓ;ˆØÐ%Ü�M‰Mð EÜôð Ð Ü Ølóð ð #ˆEàÐØ#ˆF�7‰Oà'C s×'CÑ'CÐDaÑ'lÐekÑ'lÑ$Ð˜fàˆs�}‰}Ð1Ñ<°VÑ<Ð<rU   Úsave_directoryÚpush_to_hubc           	      ó4  — |j                  dd«      }|�;t        j                  dt        «       |j	                  d«      �t        d«      ‚||d<   t        j                  j                  |«      rt        d|› d�«      ‚t        j                  |d¬	«       |rr|j                  d
d«      }|j                  d|j                  t        j                  j                  «      d   «      } | j                  |fi |¤Ž}| j                  |«      }| j                  �t!        | || ¬«       t        j                  j#                  |t$        «      }| j'                  |«       t(        j+                  d|› �«       |r%| j-                  ||j	                  d«      ¬«       |gS )aq  
        Save an video processor object to the directory `save_directory`, so that it can be re-loaded using the
        [`~video_processing_utils.VideoProcessorBase.from_pretrained`] class method.

        Args:
            save_directory (`str` or `os.PathLike`):
                Directory where the video processor JSON file will be saved (will be created if it does not exist).
            push_to_hub (`bool`, *optional*, defaults to `False`):
                Whether or not to push your model to the Hugging Face model hub after saving it. You can specify the
                repository you want to push to with `repo_id` (will default to the name of `save_directory` in your
                namespace).
            kwargs (`dict[str, Any]`, *optional*):
                Additional key word arguments passed along to the [`~utils.PushToHubMixin.push_to_hub`] method.
        r´   Nrµ   r±   r¶   zProvided path (z#) should be a directory, not a fileT)Úexist_okÚcommit_messageÚrepo_idéÿÿÿÿ)ÚconfigzVideo processor saved in )rÃ   r±   )r<   r·   r¸   r¹   rH   re   ÚosÚpathÚisfileÚAssertionErrorÚmakedirsÚsplitÚsepÚ_create_repoÚ_get_files_timestampsÚ_auto_classr
   Újoinr   Úto_json_filerA   ÚinfoÚ_upload_modified_files)	rJ   r¿   rÀ   r2   r´   rÃ   rÄ   Úfiles_timestampsÚoutput_video_processor_files	            rO   Úsave_pretrainedz"BaseVideoProcessor.save_pretrained(  s„  € ð  Ÿ™Ð$4°dÓ;ˆàÐ%Ü�M‰Mð EÜôð �z‰z˜'Ó"Ð.Ü Ølóð ð -ˆF�7‰Oä�7‰7�>‰>˜.Ô)Ü  ?°>Ð2BÐBeÐ!fÓgÐgä
�‰�N¨TÕ2áØ#ŸZ™ZÐ(8¸$Ó?ˆNØ—j‘j ¨N×,@Ñ,@ÄÇÁÇÁÓ,MÈbÑ,QÓRˆGØ'�d×'Ñ'¨Ñ:°6Ñ:ˆGØ#×9Ñ9¸.ÓIÐð ×ÑÐ'Ü˜t ^¸DÕAô ')§g¡g§l¡l°>ÔCWÓ&XÐ#à×ÑÐ5Ô6Ü�‰Ð/Ð0KÐ/LÐMÔNáØ×'Ñ'ØØØ Ø-Ø—j‘j Ó)ð (ô ð ,Ð,Ð,rU   c                 ó|  — |j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  d	d«      }	|j                  d
d«      }
|j                  dd«      }|j                  dd«      }|j                  dd«      }|�)t        j                  dt        «       |�t	        d«      ‚|}d|dœ}|�||d<   t        «       r|	st        j                  d«       d}	t        |«      }t        j                  j                  |«      }t        j                  j                  |«      r|}d}n]t        |«      r|}t        |«      }nDt        }	 t        t         t"        fD �cg c]  }t%        |||||||	|||
|d¬«      x}	 �|‘Œ  }}|d   }	 t+        |dd¬«      5 }|j-                  «       }ddd«       t/        j0                  «      }|j3                  d|«      }|rt        j                  d"|› �«       ||fS t        j                  d"› d#|› �«       ||fS c c}w # t&        $ r ‚ t(        $ r t'        d|› d|› dt        › d�«      ‚w xY w# 1 sw Y   Œ¡xY w# t.        j4                  $ r t'        d |› d!�«      ‚w xY w)$a  
        From a `pretrained_model_name_or_path`, resolve to a dictionary of parameters, to be used for instantiating a
        video processor of type [`~video_processing_utils.VideoProcessorBase`] using `from_dict`.

        Parameters:
            pretrained_model_name_or_path (`str` or `os.PathLike`):
                The identifier of the pre-trained checkpoint from which we want the dictionary of parameters.
            subfolder (`str`, *optional*, defaults to `""`):
                In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can
                specify the folder name here.

        Returns:
            `tuple[Dict, Dict]`: The dictionary(ies) that will be used to instantiate the video processor object.
        r®   Nr¯   FÚresume_downloadÚproxiesr±   r´   r°   r²   Ú	subfolderÚ Ú_from_pipelineÚ
_from_autorµ   r¶   úvideo processor)Ú	file_typeÚfrom_auto_classÚusing_pipelinez+Offline mode: forcing local_files_only=TrueT)Úfilenamer®   r¯   rÚ   rÙ   r°   r±   Ú
user_agentr²   rÛ   Ú%_raise_exceptions_for_missing_entriesr   z Can't load video processor for 'zœ'. If you were trying to load it from 'https://huggingface.co/models', make sure you don't have a local directory with the same name. Otherwise, make sure 'z2' is the correct path to a directory containing a z fileÚrúutf-8©ÚencodingÚvideo_processorz"It looks like the config file at 'z' is not a valid JSON file.zloading configuration file z from cache at )r<   r·   r¸   r¹   re   r   rA   rÓ   ÚstrrÇ   rÈ   ÚisdirrÉ   r   r   r   r   r   r    ÚOSErrorÚ	ExceptionÚopenÚreadÚjsonÚloadsrH   ÚJSONDecodeError)r¼   r­   r2   r®   r¯   rÙ   rÚ   r±   r´   r°   r²   rÛ   Úfrom_pipelinerá   rä   Úis_localÚresolved_video_processor_fileÚvideo_processor_filerã   Úresolved_fileÚresolved_video_processor_filesÚreaderÚtextr½   s                           rO   rº   z+BaseVideoProcessor.get_video_processor_dicte  s;  € ð$ —J‘J˜{¨DÓ1ˆ	ØŸ™Ð$4°eÓ<ˆØ Ÿ*™*Ð%6¸Ó=ˆØ—*‘*˜Y¨Ó-ˆØ—
‘
˜7 DÓ)ˆØŸ™Ð$4°dÓ;ˆØ!Ÿ:™:Ð&8¸%Ó@ÐØ—:‘:˜j¨$Ó/ˆØ—J‘J˜{¨BÓ/ˆ	àŸ
™
Ð#3°TÓ:ˆØ Ÿ*™* \°5Ó9ˆàÐ%Ü�M‰Mð EÜôð Ð Ü Ølóð ð #ˆEà#4ÈÑYˆ
ØÐ$Ø+8ˆJÐ'Ñ(äÔÑ%5Ü�K‰KÐEÔFØ#Ðä(+Ð,IÓ(JÐ%Ü—7‘7—=‘=Ð!>Ó?ˆÜ�7‰7�>‰>Ð7Ô8Ø,IÐ)Ø‰HÜÐ8Ô9Ø#@Ð Ü,8Ð9VÓ,WÑ)ä#7Ð ð$ô
 &:Ô;OÔQ_Ð$`ö2à ä)4Ø9Ø%-Ø&/Ø+9Ø$+Ø,;Ø-=Ø"'Ø'1Ø%-Ø&/ØBGô*ð ˜ð  ð! ò "ð2Ð.ð 2ð* 1OÈqÑ0QÐ-ð
	äÐ3°SÀ7ÔKð %ÈvØ—{‘{“}�÷%ä#'§:¡:¨dÓ#3Ð Ø#7×#;Ñ#;Ð<MÐOcÓ#dÐ ñ Ü�K‰KÐ5Ð6SÐ5TÐUÔVð
 $ VÐ+Ð+ô �K‰KØ-Ð.BÐ-CÀ?ÐSpÐRqÐrôð $ VÐ+Ð+ùòk2øô, ò ð Üò äØ6Ð7TÐ6Uð V9à9VÐ8Wð X/Ü/CÐ.DÀEðKóð ðú÷%ð %ûô
 ×#Ñ#ò 	ÜØ4Ð5RÐ4SÐSnÐoóð ð	úsB   ÆI Æ"#IÇI ÇJ ÇJÇ,/J ÉI É,J	ÊJÊJ Ê#J;r½   c                 óÂ  — |j                  «       }|j                  dd«      }d|v rd|v r|j                  d«      |d<   d|v rd|v r|j                  d«      |d<    | di |¤Ž}g }|j                  «       D ]0  \  }}t        ||«      sŒt	        |||«       |j                  |«       Œ2 |D ]  }|j                  |d«       Œ t        j                  d|› �«       |r||fS |S )aç  
        Instantiates a type of [`~video_processing_utils.VideoProcessorBase`] from a Python dictionary of parameters.

        Args:
            video_processor_dict (`dict[str, Any]`):
                Dictionary that will be used to instantiate the video processor object. Such a dictionary can be
                retrieved from a pretrained checkpoint by leveraging the
                [`~video_processing_utils.VideoProcessorBase.to_dict`] method.
            kwargs (`dict[str, Any]`):
                Additional parameters from which to initialize the video processor object.

        Returns:
            [`~video_processing_utils.VideoProcessorBase`]: The video processor object instantiated from those
            parameters.
        Úreturn_unused_kwargsFr6   r8   NzVideo processor r‘   )Úcopyr<   r>   Úhasattrr?   ru   rA   rÓ   )r¼   r½   r2   rý   rê   Ú	to_removerK   rL   s           rO   r»   zBaseVideoProcessor.from_dictÛ  s  € ð"  4×8Ñ8Ó:ÐØ%Ÿz™zÐ*@À%ÓHÐð
 �VÑ Ð*>Ñ >Ø+1¯:©:°fÓ+=Ð  Ñ(Ø˜&Ñ  [Ð4HÑ%HØ06·
±
¸;Ó0GÐ  Ñ-áÑ5Ð 4Ñ5ˆð ˆ	Ø Ÿ,™,›.ò 	&‰JˆC�Ü�¨Õ,Ü˜¨¨eÔ4Ø× Ñ  Õ%ð	&ð ò 	"ˆCØ�J‰J�s˜DÕ!ð	"ô 	�‰Ð& Ð&7Ð8Ô9ÙØ" FÐ*Ð*à"Ð"rU   c                 óª   — t        | j                  «      }|j                  dd«       |j                  dd«       | j                  j                  |d<   |S )z¿
        Serializes this instance to a Python dictionary.

        Returns:
            `dict[str, Any]`: Dictionary of all the attributes that make up this video processor instance.
        rG   NÚ_valid_kwargs_namesÚvideo_processor_type)r   Ú__dict__r<   rN   Ú__name__)rJ   Úoutputs     rO   Úto_dictzBaseVideoProcessor.to_dict  sJ   € ô ˜$Ÿ-™-Ó(ˆØ�
‰
Ð0°$Ô7Ø�
‰
Ð(¨$Ô/Ø)-¯©×)@Ñ)@ˆÐ%Ñ&àˆrU   c                 ó  — | j                  «       }|j                  «       D ]3  \  }}t        |t        j                  «      sŒ!|j                  «       ||<   Œ5 |j                  dd«      }|�||d<   t        j                  |dd¬«      dz   S )zÃ
        Serializes this instance to a JSON string.

        Returns:
            `str`: String containing all the attributes that make up this feature_extractor instance in JSON format.
        r=   Nr5   é   T)ÚindentÚ	sort_keysú
)	r  r>   rv   rƒ   r„   Útolistr<   rñ   Údumps)rJ   Ú
dictionaryrK   rL   r=   s        rO   Úto_json_stringz!BaseVideoProcessor.to_json_string  s…   € ð —\‘\“^ˆ
à$×*Ñ*Ó,ò 	1‰JˆC�Ü˜%¤§¡Õ,Ø"'§,¡,£.�
˜3’ð	1ð &Ÿ>™>Ð*<¸dÓCÐØÐ'Ø,<ˆJÐ(Ñ)ä�z‰z˜*¨Q¸$Ô?À$ÑFÐFrU   Újson_file_pathc                 óˆ   — t        |dd¬«      5 }|j                  | j                  «       «       ddd«       y# 1 sw Y   yxY w)zá
        Save this instance to a JSON file.

        Args:
            json_file_path (`str` or `os.PathLike`):
                Path to the JSON file in which this image_processor instance's parameters will be saved.
        Úwrç   rè   N)rï   Úwriter  )rJ   r  Úwriters      rO   rÒ   zBaseVideoProcessor.to_json_file+  s<   € ô �. #°Ô8ð 	0¸FØ�L‰L˜×,Ñ,Ó.Ô/÷	0÷ 	0ñ 	0ús	   � 8¸Ac                 óT   — | j                   j                  › d| j                  «       › �S )Nú )rN   r  r  )rJ   s    rO   Ú__repr__zBaseVideoProcessor.__repr__6  s(   € Ø—.‘.×)Ñ)Ð*¨!¨D×,?Ñ,?Ó,AÐ+BÐCÐCrU   Ú	json_filec                 ó¢   — t        |dd¬«      5 }|j                  «       }ddd«       t        j                  «      } | di |¤ŽS # 1 sw Y   Œ&xY w)aÌ  
        Instantiates a video processor of type [`~video_processing_utils.VideoProcessorBase`] from the path to a JSON
        file of parameters.

        Args:
            json_file (`str` or `os.PathLike`):
                Path to the JSON file containing the parameters.

        Returns:
            A video processor of type [`~video_processing_utils.VideoProcessorBase`]: The video_processor object
            instantiated from that JSON file.
        ræ   rç   rè   Nr‘   )rï   rð   rñ   rò   )r¼   r  rú   rû   r½   s        rO   Úfrom_json_filez!BaseVideoProcessor.from_json_file9  sP   € ô �)˜S¨7Ô3ð 	!°vØ—;‘;“=ˆD÷	!ä#Ÿz™z¨$Ó/ÐÙÑ*Ð)Ñ*Ð*÷	!ð 	!ús   �AÁAc                 ó�   — t        |t        «      s|j                  }ddlmc m} t        ||«      st        |› d�«      ‚|| _        y)a	  
        Register this class with a given auto class. This should only be used for custom video processors as the ones
        in the library are already mapped with `AutoVideoProcessor `.

        <Tip warning={true}>

        This API is experimental and may have some slight breaking changes in the next releases.

        </Tip>

        Args:
            auto_class (`str` or `type`, *optional*, defaults to `"AutoVideoProcessor "`):
                The auto class to register this new video processor with.
        r   Nz is not a valid auto class.)	rv   rë   r  Útransformers.models.autoÚmodelsÚautorÿ   re   rÐ   )r¼   Ú
auto_classÚauto_modules      rO   Úregister_for_auto_classz*BaseVideoProcessor.register_for_auto_classL  sC   € ô  ˜*¤cÔ*Ø#×,Ñ,ˆJç6Ð6ä�{ JÔ/Ü 
˜|Ð+FÐGÓHÐHà$ˆ�rU   Úvideo_url_or_urlsc                 óî   — d}t        «       st        j                  d«       d}t        |t        «      r0t	        t        |D �cg c]  }| j                  ||¬«      ‘Œ c}Ž «      S t        |||¬«      S c c}w )zè
        Convert a single or a list of urls into the corresponding `np.array` objects.

        If a single url is passed, the return value will be a single object. If a list is passed a list of objects is
        returned.
        Ú
torchcodeczÇ`torchcodec` is not installed and cannot be used to decode the video by default. Falling back to `torchvision`. Note that `torchvision` decoding is deprecated and will be removed in future versions. r-   rr   )Úbackendrn   )r   r·   r¸   rv   rC   rs   rz   r&   )rJ   r#  rn   r&  Úxs        rO   rz   zBaseVideoProcessor.fetch_videosf  sx   € ð ˆÜ&Ô(Ü�M‰MðIôð $ˆGäÐ'¬Ô.ÜœÐarÖsÐ\]˜d×/Ñ/°ÐEVÐ/ÕWÒsÐtÓuÐuäÐ/¸ÐTeÔfÐfùò ts   ÁA2)NNrQ   )NFFNÚmain)F)ÚAutoVideoProcessor)Dr  Ú
__module__Ú__qualname__rÐ   ÚresamplerŸ   r    r6   Úsize_divisorr7   r8   r™   r›   rœ   r�   rž   r˜   rm   rc   rb   rl   r�   r   rD   Úmodel_input_namesr   r;   r   rT   r"   r`   r#   r   rg   r   Úfloatrk   ÚdictÚboolr   rC   r   rë   r   rŠ   r   ÚBASE_VIDEO_PROCESSOR_DOCSTRINGrR   r   r   r•   ÚclassmethodrÇ   ÚPathLiker¾   r×   Útupler   rº   r»   r  r  rÒ   r  r  r"  rz   Ú__classcell__)rN   s   @rO   r0   r0   ‘   sˆ  ø„ ð €Kà€HØ€JØ€IØ€DØ€LØÐØ€IØ€IØ€NØ€JØ€NØ€LØ€NØÐØ
€CØ€JØ€NØ€OØ€LØ.Ð/ÐðG ¨Ñ!5ð G¸$õ Gð>1¨Ló 1ðàðð 
óð8 %)Ø+/ñ	3àð3ð ˜S‘Mð3ð �e˜C ˜JÑ'Ñ(ó	3ðr ,0Ø04ñ&&àð&&ð ˜m¨TÐ1Ñ2ð&&ð # 4™.ð	&&ð
 $ HÑ-ð&&ð 
ˆnÑ	ó&&ðV EIØ $ñ	 àð ð $ E¨#Ð/?Ð*?Ñ$@ÑAð ð ˜‘ð	 ð
 
ˆnÑ	ó ñ. Ø&óð&#àð&#ð ˜Ñ&ð&#ð 
ò	&#óð&#ðl <@ñ,hà�^Ñ$ð,hð ð,hð ð	,hð
 ð,hð  Ð 5Ñ6ð,hð ð,hð ð,hð ð,hð ð,hð ð,hð ˜U 5¨$¨u©+Ð#5Ñ6Ñ7ð,hð ˜E %¨¨e©Ð"4Ñ5Ñ6ð,hð !  s¨J Ñ!7Ñ8ð,hð  
ó!,hð\ ð 8<Ø$Ø!&Ø,0Øño=à',¨S°"·+±+Ð-=Ñ'>ðo=ð ˜E # r§{¡{Ð"2Ñ3Ñ4ðo=ð ð	o=ð
 ðo=ð ˜˜c 4˜iÑ(Ñ)ðo=ð òo=ó ðo=ñb;-¨e°C¸¿¹Ð4DÑ.Eð ;-ÐTXó ;-ðz ðs,Ø,1°#°r·{±{Ð2BÑ,Cðs,à	ˆt�C˜�H‰~˜t C¨ H™~Ð-Ñ	.òs,ó ðs,ðj ð*#¨T°#°s°(©^ò *#ó ð*#ðX˜˜c 3˜h™ó ðG ó Gð*	0¨5°°b·k±kÐ1AÑ+Bó 	0òDð ð+ u¨S°"·+±+Ð-=Ñ'>ò +ó ð+ð$ ò%ó ð%ñ2g¨e°C¸¸c¹ÀDÈÈcÉÁOÐ4SÑ.T÷ grU   r0   rß   r)  zvideo processor file)ÚobjectÚobject_classÚobject_files)Drñ   rÇ   r·   rþ   r   Ú	functoolsr   Útypingr   r   r   r   Únumpyrƒ   Údynamic_module_utilsr
   Úimage_processing_utilsr   r   Úimage_processing_utils_fastr   Úimage_utilsr   r   r   Úprocessing_utilsr   r   Úutilsr   r   r   r   r   r   r   r   r   r   r   r   r   Ú	utils.hubr    Úutils.import_utilsr!   Úvideo_utilsr"   r#   r$   r%   r&   r'   r(   r)   r*   rh   Útorchvision.transforms.v2r+   r[   Ú
get_loggerr  rA   r2  r0   rÀ   Ú__doc__Úformatr‘   rU   rO   ú<module>rJ     s:  ðó  Û 	Û Ý Ý ß 1Ó 1ã å 4÷õ @÷ñ ÷
 3÷÷ ÷ õ õ #Ý (÷
÷ 
õ 
ñ ÔÛáÔ Ý9ð 
ˆ×	Ñ	˜HÓ	%€ðA"Ð ñH Ø'Ø"óñ 
Ð,Ô-ôbgÐ/ó bgó .ó	ð
bgñJ "+Ð+=×+IÑ+IÓ!JÐ Ô Ø×!Ñ!×)Ñ)Ð5Ø-?×-KÑ-K×-SÑ-S×-ZÑ-ZØ Ð/CÐRhð .[ó .Ð×"Ñ"Õ*ð 6rU   