Ë
    Gêñi™  ã                   ó¶  — d dl Z d dlZd dlZd dlmZ d dlmZ d dlmZ d dl	Z
d dlmZmZ d dlmZ ddlmZ dd	lmZ dd
lmZ ddlmZmZmZmZ ddlmZmZ ddlmZm Z m!Z!m"Z"m#Z#m$Z$m%Z%m&Z&m'Z'm(Z(m)Z) ddl*m+Z+ ddl,m-Z- ddl.m/Z/m0Z0m1Z1m2Z2m3Z3m4Z4m5Z5m6Z6m7Z7  e%«       rd dl8Z8 e'«       rd dl9m:c m;c m<Z=  e«       rddlm>Z>  e(j~                  e@«      ZAdZB e#deB«       e-d¬«       G d„ de«      «       «       ZC e$eCjˆ                  «      eC_D        eCjˆ                  jŠ                  �8eCjˆ                  jŠ                  j�                  ddd¬«      eCjˆ                  _E        yy)é    N)ÚCallable)Úpartial)ÚAny)Úcreate_repoÚis_offline_mode)Úvalidate_typed_dicté   )Úcustom_object_save)ÚTorchvisionBackend)ÚBatchFeature)ÚChannelDimensionÚSizeDictÚis_vision_availableÚvalidate_kwargs)ÚUnpackÚVideosKwargs)ÚIMAGE_PROCESSOR_NAMEÚPROCESSOR_NAMEÚVIDEO_PROCESSOR_NAMEÚ
TensorTypeÚadd_start_docstringsÚ	copy_funcÚis_torch_availableÚis_torchcodec_availableÚis_torchvision_v2_availableÚloggingÚsafe_load_json_file)Úcached_file)Úrequires)	Ú
VideoInputÚVideoMetadataÚgroup_videos_by_shapeÚinfer_channel_dimension_formatÚis_valid_videoÚ
load_videoÚmake_batched_metadataÚmake_batched_videosÚreorder_videos)ÚPILImageResamplingaÊ  
    Args:
        do_resize (`bool`, *optional*, defaults to `self.do_resize`):
            Whether to resize the video's (height, width) dimensions to the specified `size`. Can be overridden by the
            `do_resize` parameter in the `preprocess` method.
        size (`dict`, *optional*, defaults to `self.size`):
            Size of the output video after resizing. Can be overridden by the `size` parameter in the `preprocess`
            method.
        size_divisor (`int`, *optional*, defaults to `self.size_divisor`):
            The size by which to make sure both the height and width can be divided.
        default_to_square (`bool`, *optional*, defaults to `self.default_to_square`):
            Whether to default to a square video when resizing, if size is an int.
        resample (`PILImageResampling`, *optional*, defaults to `self.resample`):
            Resampling filter to use if resizing the video. Only has an effect if `do_resize` is set to `True`. Can be
            overridden by the `resample` parameter in the `preprocess` method.
        do_center_crop (`bool`, *optional*, defaults to `self.do_center_crop`):
            Whether to center crop the video to the specified `crop_size`. Can be overridden by `do_center_crop` in the
            `preprocess` method.
        crop_size (`dict[str, int]` *optional*, defaults to `self.crop_size`):
            Size of the output video after applying `center_crop`. Can be overridden by `crop_size` in the `preprocess`
            method.
        do_rescale (`bool`, *optional*, defaults to `self.do_rescale`):
            Whether to rescale the video by the specified scale `rescale_factor`. Can be overridden by the
            `do_rescale` parameter in the `preprocess` method.
        rescale_factor (`int` or `float`, *optional*, defaults to `self.rescale_factor`):
            Scale factor to use if rescaling the video. Only has an effect if `do_rescale` is set to `True`. Can be
            overridden by the `rescale_factor` parameter in the `preprocess` method.
        do_normalize (`bool`, *optional*, defaults to `self.do_normalize`):
            Whether to normalize the video. Can be overridden by the `do_normalize` parameter in the `preprocess`
            method. Can be overridden by the `do_normalize` parameter in the `preprocess` method.
        image_mean (`float` or `list[float]`, *optional*, defaults to `self.image_mean`):
            Mean to use if normalizing the video. This is a float or list of floats the length of the number of
            channels in the video. Can be overridden by the `image_mean` parameter in the `preprocess` method. Can be
            overridden by the `image_mean` parameter in the `preprocess` method.
        image_std (`float` or `list[float]`, *optional*, defaults to `self.image_std`):
            Standard deviation to use if normalizing the video. This is a float or list of floats the length of the
            number of channels in the video. Can be overridden by the `image_std` parameter in the `preprocess` method.
            Can be overridden by the `image_std` parameter in the `preprocess` method.
        do_convert_rgb (`bool`, *optional*, defaults to `self.image_std`):
            Whether to convert the video to RGB.
        video_metadata (`VideoMetadata`, *optional*):
            Metadata of the video containing information about total duration, fps and total number of frames.
        do_sample_frames (`int`, *optional*, defaults to `self.do_sample_frames`):
            Whether to sample frames from the video before processing or to process the whole video.
        num_frames (`int`, *optional*, defaults to `self.num_frames`):
            Maximum number of frames to sample when `do_sample_frames=True`.
        fps (`int` or `float`, *optional*, defaults to `self.fps`):
            Target frames to sample per second when `do_sample_frames=True`.
        return_tensors (`str` or `TensorType`, *optional*):
            Returns stacked tensors if set to `pt, otherwise returns a list of tensors.
        data_format (`ChannelDimension` or `str`, *optional*, defaults to `ChannelDimension.FIRST`):
            The channel dimension format for the output video. Can be one of:
            - `"channels_first"` or `ChannelDimension.FIRST`: video in (num_channels, height, width) format.
            - `"channels_last"` or `ChannelDimension.LAST`: video in (height, width, num_channels) format.
            - Unset: Use the channel dimension format of the input video.
        input_data_format (`ChannelDimension` or `str`, *optional*):
            The channel dimension format for the input video. If unset, the channel dimension format is inferred
            from the input video. Can be one of:
            - `"channels_first"` or `ChannelDimension.FIRST`: video in (num_channels, height, width) format.
            - `"channels_last"` or `ChannelDimension.LAST`: video in (height, width, num_channels) format.
            - `"none"` or `ChannelDimension.NONE`: video in (height, width) format.
        device (`torch.device`, *optional*):
            The device to process the videos on. If unset, the device is inferred from the input videos.
        return_metadata (`bool`, *optional*):
            Whether to return video metadata or not.
        z!Constructs a base VideoProcessor.)ÚvisionÚtorchvision)Úbackendsc                   óÀ  ‡ — e Zd ZdZdZdZdZdZdZdZ	dZ
dZdZdZdZdZdZdZdZdZdZdZeZdgZdee   ddfˆ fd„Zdefd	„Zd
ddefd„Z	 	 d?dede dz  de e!z  dz  fd„Z"	 	 d?dedee#z  de$dz  de%dz  de&d   f
d„Z'	 	 d?dede(e)z  dz  de(dz  de&d   fd„Z* e+e,«      dedee   defd„«       Z-	 d@de&d   de$de$de.ddde$d e.d!e$d"e!d#e$d$e!e&e!   z  dz  d%e!e&e!   z  dz  d&e(e/z  dz  defd'„Z0e1	 	 	 	 	 dAd(e(e2jf                  z  d)e(e2jf                  z  dz  d*e$d+e$d,e(e$z  dz  d-e(fd.„«       Z4dBd/e(e2jf                  z  d0e$fd1„Z5e1d(e(e2jf                  z  de6e#e(e7f   e#e(e7f   f   fd2„«       Z8e1d3e#e(e7f   fd4„«       Z9de#e(e7f   fˆ fd5„Z:de(fd6„Z;d7e(e2jf                  z  fd8„Z<d9„ Z=e1d:e(e2jf                  z  fd;„«       Z>e1dCd<„«       Z?d@d=e(e&e(   z  e&e&e(      z  fd>„Z@ˆ xZAS )DÚBaseVideoProcessorNTgp?FÚpixel_values_videosÚkwargsÚreturnc                 ó$   •— t        ‰| �  di |¤Ž y )N© )ÚsuperÚ__init__)Úselfr0   Ú	__class__s     €úe/var/www/pod-logistic/pod-ai/venv/lib/python3.12/site-packages/transformers/video_processing_utils.pyr5   zBaseVideoProcessor.__init__®   s   ø€ Ü‰ÑÑ"˜6Ó"ó    c                 ó(   —  | j                   |fi |¤ŽS ©N)Ú
preprocess)r6   Úvideosr0   s      r8   Ú__call__zBaseVideoProcessor.__call__±   s   € Øˆt�‰˜vÑ0¨Ñ0Ð0r9   Úvideoztorch.Tensorc                 ó  — t        j                  |«      }|j                  d   dk(  s|dddd…dd…f   dk  j                  «       s|S |dddd…dd…f   dz  }d|dddd…dd…f   z
  dz  |dddd…dd…f   |ddd…dd…dd…f   z  z   }|S )zÏ
        Converts a video to RGB format.

        Args:
            video (`"torch.Tensor"`):
                The video to convert.

        Returns:
            `torch.Tensor`: The converted video.
        éýÿÿÿé   .Néÿ   g     ào@r	   )ÚtvFÚgrayscale_to_rgbÚshapeÚany)r6   r?   Úalphas      r8   Úconvert_to_rgbz!BaseVideoProcessor.convert_to_rgb´   s¯   € ô ×$Ñ$ UÓ+ˆØ�;‰;�r‰?˜aÒ¨¨c°1²aº¨lÑ(;¸cÑ(A×'FÑ'FÔ'HØˆLð �c˜1ša¢�lÑ# eÑ+ˆØ�U˜3 ¢aª˜?Ñ+Ñ+¨sÑ2°U¸3ÀÂaÊ¸?Ñ5KÈeÐTWÐY[ÐZ[ÐY[Ò]^Ò`aÐTaÑNbÑ5bÑbˆØˆr9   ÚmetadataÚ
num_framesÚfpsc                 óº  — |�|�t        d«      ‚|�|n| j                  }|�|n| j                  }|j                  }|€6|�4|�|j                  €t        d«      ‚t	        ||j                  z  |z  «      }||kD  rt        d|› d|› d�«      ‚|�*t        j                  d|||z  «      j	                  «       }|S t        j                  d|«      j	                  «       }|S )a%  
        Default sampling function which uniformly samples the desired number of frames between 0 and total number of frames.
        If `fps` is passed along with metadata, `fps` frames per second are sampled uniformty. Arguments `num_frames`
        and `fps` are mutually exclusive.

        Args:
            metadata (`VideoMetadata`):
                Metadata of the video containing information about total duration, fps and total number of frames.
            num_frames (`int`, *optional*):
                Maximum number of frames to sample. Defaults to `self.num_frames`.
            fps (`int` or `float`, *optional*):
                Target frames to sample per second. Defaults to `self.fps`.

        Returns:
            np.ndarray:
                Indices to sample video frames.
        zc`num_frames`, `fps`, and `sample_indices_fn` are mutually exclusive arguments, please use only one!zÈAsked to sample `fps` frames per second but no video metadata was provided which is required when sampling with `fps`. Please pass in `VideoMetadata` object or use a fixed `num_frames` per input videoz(Video can't be sampled. The `num_frames=z` exceeds `total_num_frames=z`. r   )Ú
ValueErrorrK   rL   Útotal_num_framesÚintÚtorchÚarange)r6   rJ   rK   rL   r0   rO   Úindicess          r8   Úsample_framesz BaseVideoProcessor.sample_framesÍ   s  € ð0 ˆ?˜zÐ5ÜØuóð ð $.Ð#9‘Z¸t¿¹ˆ
Ø�_‰c¨$¯(©(ˆØ#×4Ñ4Ðð Ð # /ØÐ 8§<¡<Ð#7Ü ðhóð ô Ð-°·±Ñ<¸sÑBÓCˆJàÐ(Ò(ÜØ:¸:¸,ÐFbÐcsÐbtÐtwÐxóð ð Ð!Ü—l‘l 1Ð&6Ð8HÈ:Ñ8UÓV×ZÑZÓ\ˆGð ˆô —l‘l 1Ð&6Ó7×;Ñ;Ó=ˆGØˆr9   r=   Úvideo_metadataÚdo_sample_framesÚsample_indices_fnc                 óN  — t        |«      }t        ||¬«      }t        |d   «      rW|rUg }g }t        ||«      D ]:  \  }} ||¬«      }	|	|_        |j                  ||	   «       |j                  |«       Œ< |}|}||fS t        |d   «      sŒt        |d   t        «      rc| j                  |«      D �
�cg c]6  }
t        j                  |
D �cg c]  }| j                  |«      ‘Œ c}d¬«      ‘Œ8 }}
}|rt        d«      ‚||fS | j                  ||¬«      \  }}||fS c c}w c c}}
w )zB
        Decode input videos and sample frames if needed.
        )rU   r   )rJ   )ÚdimzUSampling frames from a list of images is not supported! Set `do_sample_frames=False`.©rW   )r'   r&   r$   ÚzipÚframes_indicesÚappendÚ
isinstanceÚlistÚfetch_imagesrQ   ÚstackÚprocess_imagerN   Úfetch_videos)r6   r=   rU   rV   rW   Úsampled_videosÚsampled_metadatar?   rJ   rS   ÚimagesÚimages               r8   Ú_decode_and_sample_videosz,BaseVideoProcessor._decode_and_sample_videos  sX  € ô % VÓ,ˆÜ.¨vÀnÔUˆô ˜& ™)Ô$Ñ)9ØˆNØ!ÐÜ#& v¨~Ó#>ò 2‘��xÙ+°XÔ>�Ø*1�Ô'Ø×%Ñ% e¨G¡nÔ5Ø ×'Ñ'¨Õ1ð	2ð
 $ˆFØ-ˆNð �~Ð%Ð%ô    q¡	Ô*Ü˜& ™)¤TÔ*ð #'×"3Ñ"3°FÓ";÷àô —K‘KÈÖ O¸u ×!3Ñ!3°EÕ!:Ò OÐUVÖWð�ñ ñ $Ü$Øoóð ð �~Ð%Ð%ð *.×):Ñ):¸6ÐUfÐ):Ó)gÑ&�˜à�~Ð%Ð%ùò !Pùós   Â3D!ÃDÃ#D!ÄD!Úinput_data_formatÚdevicec                 ó^  — g }|D ]¥  }t        |t        j                  «      r#t        j                  |«      j                  «       }|€t        |«      }|t        j                  k(  r"|j                  dddd«      j                  «       }|�|j                  |«      }|j                  |«       Œ§ |S )z:
        Prepare the input videos for processing.
        r   rB   r	   é   )r^   ÚnpÚndarrayrQ   Ú
from_numpyÚ
contiguousr#   r   ÚLASTÚpermuteÚtor]   )r6   r=   ri   rj   Úprocessed_videosr?   s         r8   Ú_prepare_input_videosz(BaseVideoProcessor._prepare_input_videos*  s©   € ð ÐØò 	+ˆEä˜%¤§¡Ô,ä×(Ñ(¨Ó/×:Ñ:Ó<�ð !Ð(Ü$BÀ5Ó$IÐ!à Ô$4×$9Ñ$9Ò9ØŸ™ a¨¨A¨qÓ1×<Ñ<Ó>�àÐ!ØŸ™ Ó(�à×#Ñ# EÕ*ð!	+ð"  Ðr9   c           	      ó  — t        |j                  «       t        | j                  j                  j                  «       «      dgz   ¬«       t        | j                  |«       | j                  j                  D ]  }|j                  |t        | |d «      «       Œ! |j                  d«      }|j                  d«      }|j                  d«      }|j                  d«      }|rt        | j                  fi |¤Žnd }| j                  ||||¬«      \  }}| j                  |||¬«      } | j                  di |¤Ž} | j                  di |¤Ž |j                  d	«       |j                  d
«      }	 | j                  dd|i|¤Ž}
|	r||
d<   |
S )NÚreturn_tensors)Úcaptured_kwargsÚvalid_processor_keysri   rV   rj   rU   )rU   rV   rW   )r=   ri   rj   Údata_formatÚreturn_metadatar=   r3   )r   Úkeysr_   Úvalid_kwargsÚ__annotations__r   Ú
setdefaultÚgetattrÚpopr   rT   rh   ru   Ú_standardize_kwargsÚ_validate_preprocess_kwargsÚ_preprocess)r6   r=   r0   Ú
kwarg_nameri   rV   rj   rU   rW   r{   Úpreprocessed_videoss              r8   r<   zBaseVideoProcessor.preprocessG  s–  € ô 	Ø"ŸK™K›MÜ!% d×&7Ñ&7×&GÑ&G×&LÑ&LÓ&NÓ!OÐScÐRdÑ!dõ	
ô 	˜D×-Ñ-¨vÔ6ð ×+Ñ+×;Ñ;ò 	KˆJØ×Ñ˜j¬'°$¸
ÀDÓ*IÕJð	Kð #ŸJ™JÐ':Ó;ÐØ!Ÿ:™:Ð&8Ó9ÐØ—‘˜HÓ%ˆØŸ™Ð$4Ó5ˆáEUœG D×$6Ñ$6ÑA¸&ÒAÐ[_ÐØ!%×!?Ñ!?ØØ)Ø-Ø/ð	 "@ó "
Ñˆ�ð ×+Ñ+°6ÐM^ÐgmÐ+Ónˆà)�×)Ñ)Ñ3¨FÑ3ˆØ(ˆ×(Ñ(Ñ2¨6Ò2ð 	�
‰
�=Ô!Ø Ÿ*™*Ð%6Ó7ˆà.˜d×.Ñ.ÑG°fÐGÀÑGÐÙØ4BÐÐ 0Ñ1Ø"Ð"r9   Údo_convert_rgbÚ	do_resizeÚsizeÚresamplez7PILImageResampling | tvF.InterpolationMode | int | NoneÚdo_center_cropÚ	crop_sizeÚ
do_rescaleÚrescale_factorÚdo_normalizeÚ
image_meanÚ	image_stdrw   c           	      óª  — t        |«      \  }}i }|j                  «       D ]3  \  }}|r| j                  |«      }|r| j                  |||¬«      }|||<   Œ5 t	        ||«      }t        |«      \  }}i }|j                  «       D ]4  \  }}|r| j                  ||«      }| j                  |||	|
||«      }|||<   Œ6 t	        ||«      }t        d|i|¬«      S )N)r‰   rŠ   r/   )ÚdataÚtensor_type)r"   ÚitemsrI   Úresizer(   Úcenter_cropÚrescale_and_normalizer   )r6   r=   r‡   rˆ   r‰   rŠ   r‹   rŒ   r�   rŽ   r�   r�   r‘   rw   r0   Úgrouped_videosÚgrouped_videos_indexÚresized_videos_groupedrF   Ústacked_videosÚresized_videosÚprocessed_videos_groupedrt   s                          r8   r„   zBaseVideoProcessor._preprocessv  s  € ô$ 0EÀVÓ/LÑ,ˆÐ,Ø!#ÐØ%3×%9Ñ%9Ó%;ò 	;Ñ!ˆE�>ÙØ!%×!4Ñ!4°^Ó!D�ÙØ!%§¡¨^À$ÐQY Ó!Z�Ø,:Ð" 5Ò)ð	;ô (Ð(>Ð@TÓUˆô 0EÀ^Ó/TÑ,ˆÐ,Ø#%Ð Ø%3×%9Ñ%9Ó%;ò 	=Ñ!ˆE�>ÙØ!%×!1Ñ!1°.À)Ó!L�à!×7Ñ7Ø 
¨N¸LÈ*ÐV_óˆNð /=Ð$ UÒ+ð	=ô *Ð*BÐDXÓYÐäÐ"7Ð9IÐ!JÐXfÔgÐgr9   Úpretrained_model_name_or_pathÚ	cache_dirÚforce_downloadÚlocal_files_onlyÚtokenÚrevisionc                 óŠ   — ||d<   ||d<   ||d<   ||d<   |�||d<    | j                   |fi |¤Ž\  }} | j                  |fi |¤ŽS )aK  
        Instantiate a type of [`~video_processing_utils.VideoProcessorBase`] from an video processor.

        Args:
            pretrained_model_name_or_path (`str` or `os.PathLike`):
                This can be either:

                - a string, the *model id* of a pretrained video hosted inside a model repo on
                  huggingface.co.
                - a path to a *directory* containing a video processor file saved using the
                  [`~video_processing_utils.VideoProcessorBase.save_pretrained`] method, e.g.,
                  `./my_model_directory/`.
                - a path to a saved video processor JSON *file*, e.g.,
                  `./my_model_directory/video_preprocessor_config.json`.
            cache_dir (`str` or `os.PathLike`, *optional*):
                Path to a directory in which a downloaded pretrained model video processor should be cached if the
                standard cache should not be used.
            force_download (`bool`, *optional*, defaults to `False`):
                Whether or not to force to (re-)download the video processor files and override the cached versions if
                they exist.
            proxies (`dict[str, str]`, *optional*):
                A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
                'http://hostname': 'foo.bar:4012'}.` The proxies are used on each request.
            token (`str` or `bool`, *optional*):
                The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
                the token generated when running `hf auth login` (stored in `~/.huggingface`).
            revision (`str`, *optional*, defaults to `"main"`):
                The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a
                git-based system for storing models and other artifacts on huggingface.co, so `revision` can be any
                identifier allowed by git.


                <Tip>

                To test a pull request you made on the Hub, you can pass `revision="refs/pr/<pr_number>"`.

                </Tip>

            return_unused_kwargs (`bool`, *optional*, defaults to `False`):
                If `False`, then this function returns just the final video processor object. If `True`, then this
                functions returns a `Tuple(video_processor, unused_kwargs)` where *unused_kwargs* is a dictionary
                consisting of the key/value pairs whose keys are not video processor attributes: i.e., the part of
                `kwargs` which has not been used to update `video_processor` and is otherwise ignored.
            subfolder (`str`, *optional*, defaults to `""`):
                In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can
                specify the folder name here.
            kwargs (`dict[str, Any]`, *optional*):
                The values in kwargs of any keys which are video processor attributes will be used to override the
                loaded values. Behavior concerning key/value pairs whose keys are *not* video processor attributes is
                controlled by the `return_unused_kwargs` keyword parameter.

        Returns:
            A video processor of type [`~video_processing_utils.ImagVideoProcessorBase`].

        Examples:

        ```python
        # We can't instantiate directly the base class *VideoProcessorBase* so let's show the examples on a
        # derived class: *LlavaOnevisionVideoProcessor*
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained(
            "llava-hf/llava-onevision-qwen2-0.5b-ov-hf"
        )  # Download video_processing_config from huggingface.co and cache.
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained(
            "./test/saved_model/"
        )  # E.g. video processor (or model) was saved using *save_pretrained('./test/saved_model/')*
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained("./test/saved_model/video_preprocessor_config.json")
        video_processor = LlavaOnevisionVideoProcessor.from_pretrained(
            "llava-hf/llava-onevision-qwen2-0.5b-ov-hf", do_normalize=False, foo=False
        )
        assert video_processor.do_normalize is False
        video_processor, unused_kwargs = LlavaOnevisionVideoProcessor.from_pretrained(
            "llava-hf/llava-onevision-qwen2-0.5b-ov-hf", do_normalize=False, foo=False, return_unused_kwargs=True
        )
        assert video_processor.do_normalize is False
        assert unused_kwargs == {"foo": False}
        ```r    r¡   r¢   r¤   r£   )Úget_video_processor_dictÚ	from_dict)	ÚclsrŸ   r    r¡   r¢   r£   r¤   r0   Úvideo_processor_dicts	            r8   Úfrom_pretrainedz"BaseVideoProcessor.from_pretrained£  st   € ðn (ˆˆ{ÑØ#1ˆÐÑ Ø%5ˆÐ!Ñ"Ø%ˆˆzÑàÐØ#ˆF�7‰Oà'C s×'CÑ'CÐDaÑ'lÐekÑ'lÑ$Ð˜fàˆs�}‰}Ð1Ñ<°VÑ<Ð<r9   Úsave_directoryÚpush_to_hubc           	      ó   — t         j                  j                  |«      rt        d|› d�«      ‚t        j                  |d¬«       |rw|j                  dd«      }|j                  d|j                  t         j                  j                  «      d   «      }t        |fd	di|¤Žj                  }| j                  |«      }| j                  �t        | || ¬
«       t         j                  j                  |t        «      }| j                  |«       t         j#                  d|› �«       |r%| j%                  ||j'                  d«      ¬«       |gS )aq  
        Save an video processor object to the directory `save_directory`, so that it can be re-loaded using the
        [`~video_processing_utils.VideoProcessorBase.from_pretrained`] class method.

        Args:
            save_directory (`str` or `os.PathLike`):
                Directory where the video processor JSON file will be saved (will be created if it does not exist).
            push_to_hub (`bool`, *optional*, defaults to `False`):
                Whether or not to push your model to the Hugging Face model hub after saving it. You can specify the
                repository you want to push to with `repo_id` (will default to the name of `save_directory` in your
                namespace).
            kwargs (`dict[str, Any]`, *optional*):
                Additional key word arguments passed along to the [`~utils.PushToHubMixin.push_to_hub`] method.
        zProvided path (z#) should be a directory, not a fileT)Úexist_okÚcommit_messageNÚrepo_idéÿÿÿÿr®   )ÚconfigzVideo processor saved in r£   )r¯   r£   )ÚosÚpathÚisfileÚAssertionErrorÚmakedirsr�   ÚsplitÚsepr   r°   Ú_get_files_timestampsÚ_auto_classr
   Újoinr   Úto_json_fileÚloggerÚinfoÚ_upload_modified_filesÚget)r6   r«   r¬   r0   r¯   r°   Úfiles_timestampsÚoutput_video_processor_files           r8   Úsave_pretrainedz"BaseVideoProcessor.save_pretrained  s/  € ô �7‰7�>‰>˜.Ô)Ü  ?°>Ð2BÐBeÐ!fÓgÐgä
�‰�N¨TÕ2áØ#ŸZ™ZÐ(8¸$Ó?ˆNØ—j‘j ¨N×,@Ñ,@ÄÇÁÇÁÓ,MÈbÑ,QÓRˆGÜ! 'ÑC°DÐC¸FÑC×KÑKˆGØ#×9Ñ9¸.ÓIÐð ×ÑÐ'Ü˜t ^¸DÕAô ')§g¡g§l¡l°>ÔCWÓ&XÐ#à×ÑÐ5Ô6Ü�‰Ð/Ð0KÐ/LÐMÔNáØ×'Ñ'ØØØ Ø-Ø—j‘j Ó)ð (ô ð ,Ð,Ð,r9   c                 óJ  — |j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  d	d
«      }	|j                  dd«      }
|j                  dd«      }d|dœ}|
�|
|d<   t        «       r|st        j                  d«       d}t	        |«      }t
        j                  j                  |«      }t
        j                  j                  |«      r|}d}d}nXt        }	 t        |t        ||||||||	d¬«      }|t        fD �cg c]  }t        ||||||||||	d¬«      x}	 �|‘Œ }}|r|d   nd}d}|�t        |«      }d|v r|d   }|�|€t        |«      }|€t        d|› d|› d› d�«      ‚|rt        j                  d|› �«       ||fS t        j                  d› d|› �«       ||fS c c}w # t        $ r ‚ t        $ r t        d|› d|› d|› d�«      ‚w xY w)a  
        From a `pretrained_model_name_or_path`, resolve to a dictionary of parameters, to be used for instantiating a
        video processor of type [`~video_processing_utils.VideoProcessorBase`] using `from_dict`.

        Parameters:
            pretrained_model_name_or_path (`str` or `os.PathLike`):
                The identifier of the pre-trained checkpoint from which we want the dictionary of parameters.
            subfolder (`str`, *optional*, defaults to `""`):
                In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can
                specify the folder name here.

        Returns:
            `tuple[Dict, Dict]`: The dictionary(ies) that will be used to instantiate the video processor object.
        r    Nr¡   FÚproxiesr£   r¢   r¤   Ú	subfolderÚ Ú_from_pipelineÚ
_from_autoúvideo processor)Ú	file_typeÚfrom_auto_classÚusing_pipelinez+Offline mode: forcing local_files_only=TrueT)
Úfilenamer    r¡   rÆ   r¢   r£   Ú
user_agentr¤   rÇ   Ú%_raise_exceptions_for_missing_entriesr   z Can't load video processor for 'zœ'. If you were trying to load it from 'https://huggingface.co/models', make sure you don't have a local directory with the same name. Otherwise, make sure 'z2' is the correct path to a directory containing a z fileÚvideo_processorzloading configuration file z from cache at )r�   r   r¾   r¿   Ústrr³   r´   Úisdirrµ   r   r   r   r   ÚOSErrorÚ	Exceptionr   )r¨   rŸ   r0   r    r¡   rÆ   r£   r¢   r¤   rÇ   Úfrom_pipelinerÍ   rÐ   Úis_localÚresolved_video_processor_fileÚresolved_processor_fileÚvideo_processor_filerÏ   Úresolved_fileÚresolved_video_processor_filesr©   Úprocessor_dicts                         r8   r¦   z+BaseVideoProcessor.get_video_processor_dict6  sñ  € ð$ —J‘J˜{¨DÓ1ˆ	ØŸ™Ð$4°eÓ<ˆØ—*‘*˜Y¨Ó-ˆØ—
‘
˜7 DÓ)ˆØ!Ÿ:™:Ð&8¸%Ó@ÐØ—:‘:˜j¨$Ó/ˆØ—J‘J˜{¨BÓ/ˆ	àŸ
™
Ð#3°TÓ:ˆØ Ÿ*™* \°5Ó9ˆà#4ÈÑYˆ
ØÐ$Ø+8ˆJÐ'Ñ(äÔÑ%5Ü�K‰KÐEÔFØ#Ðä(+Ð,IÓ(JÐ%Ü—7‘7—=‘=Ð!>Ó?ˆÜ�7‰7�>‰>Ð7Ô8Ø,IÐ)Ø&*Ð#Ø‰Hä#7Ð ð2ô +6Ø1Ü+Ø'Ø#1Ø#Ø%5ØØ)Ø%Ø'Ø:?ô+Ð'ð &:Ô;OÐ$Pö2à ä)4Ø9Ø%-Ø&/Ø+9Ø$+Ø-=Ø"'Ø'1Ø%-Ø&/ØBGô*ð ˜ð  ð ò "ð2Ð.ð 2ñ* :XÐ2°1Ò5Ð]að .ð&  $ÐØ"Ð.Ü0Ð1HÓIˆNØ  NÑ2Ø'5Ð6GÑ'HÐ$à(Ð4Ð9MÐ9UÜ#6Ð7TÓ#UÐ àÐ'ÜØ2Ð3PÐ2Qð R5à5RÐ4Sð T+Ø+?Ð*@ÀðGóð ñ Ü�K‰KÐ5Ð6SÐ5TÐUÔVð $ VÐ+Ð+ô	 �K‰KØ-Ð.BÐ-CÀ?ÐSpÐRqÐrôð $ VÐ+Ð+ùò2øô. ò ð Üò äØ6Ð7TÐ6Uð V9à9VÐ8Wð X/Ø/CÐ.DÀEðKóð ðús   Ä)$G: Å"G5Å/G: Ç5G: Ç:(H"r©   c           	      ó„  — |j                  «       }|j                  dd«      }|j                  |j                  «       D ��ci c]!  \  }}|| j                  j
                  v sŒ||“Œ# c}}«        | di |¤Ž}g }t        t        |j                  «       «      «      D ]V  }t        ||«      sŒ|| j                  j
                  vsŒ)t        |||j                  |d«      «       |j                  |«       ŒX |r&t        j                  d| j                  › d|› d�«       t        j                  d|› �«       |r||fS |S c c}}w )	aç  
        Instantiates a type of [`~video_processing_utils.VideoProcessorBase`] from a Python dictionary of parameters.

        Args:
            video_processor_dict (`dict[str, Any]`):
                Dictionary that will be used to instantiate the video processor object. Such a dictionary can be
                retrieved from a pretrained checkpoint by leveraging the
                [`~video_processing_utils.VideoProcessorBase.to_dict`] method.
            kwargs (`dict[str, Any]`):
                Additional parameters from which to initialize the video processor object.

        Returns:
            [`~video_processing_utils.VideoProcessorBase`]: The video processor object instantiated from those
            parameters.
        Úreturn_unused_kwargsFNzImage processor z	: kwargs zÍ were applied for backward compatibility. To avoid this warning, add them to valid_kwargs: create a custom TypedDict extending ImagesKwargs with these keys and set it as the `valid_kwargs` class attribute.zVideo processor r3   )Úcopyr�   Úupdater•   r}   r~   Úreversedr_   r|   ÚhasattrÚsetattrr]   r¾   Úwarning_onceÚ__name__r¿   )	r¨   r©   r0   rà   ÚkÚvrÒ   Ú
extra_keysÚkeys	            r8   r§   zBaseVideoProcessor.from_dict´  s7  € ð"  4×8Ñ8Ó:ÐØ%Ÿz™zÐ*@À%ÓHÐØ×#Ñ#°f·l±l³n×$n©d¨a°ÈÈS×M]ÑM]×MmÑMmÒHm Q¨¡TÓ$nÔoÙÑ5Ð 4Ñ5ˆð ˆ
ÜœD §¡£Ó/Ó0ò 	'ˆCÜ�¨Õ,°¸C×<LÑ<L×<\Ñ<\Ò1\Ü˜¨¨f¯j©j¸¸dÓ.CÔDØ×!Ñ! #Õ&ð	'ñ Ü×ÑØ" 3§<¡< .°	¸*¸ð Fað bôô 	�‰Ð& Ð&7Ð8Ô9ÙØ" FÐ*Ð*à"Ð"ùó) %os   Á D<
Á"D<
c                 óz   •— t         ‰| �  «       }|j                  dd«       | j                  j                  |d<   |S )z¿
        Serializes this instance to a Python dictionary.

        Returns:
            `dict[str, Any]`: Dictionary of all the attributes that make up this video processor instance.
        Úimage_processor_typeNÚvideo_processor_type)r4   Úto_dictr�   r7   rç   )r6   Úfiltered_dictr7   s     €r8   rï   zBaseVideoProcessor.to_dictÝ  s=   ø€ ô ™™Ó)ˆØ×ÑÐ0°$Ô7Ø04·±×0GÑ0GˆÐ,Ñ-àÐr9   c                 óä   — | j                  «       }|j                  «       D ]3  \  }}t        |t        j                  «      sŒ!|j                  «       ||<   Œ5 t        j                  |dd¬«      dz   S )zÃ
        Serializes this instance to a JSON string.

        Returns:
            `str`: String containing all the attributes that make up this feature_extractor instance in JSON format.
        rl   T)ÚindentÚ	sort_keysú
)rï   r•   r^   rm   rn   ÚtolistÚjsonÚdumps)r6   Ú
dictionaryrë   Úvalues       r8   Úto_json_stringz!BaseVideoProcessor.to_json_stringê  sb   € ð —\‘\“^ˆ
à$×*Ñ*Ó,ò 	1‰JˆC�Ü˜%¤§¡Õ,Ø"'§,¡,£.�
˜3’ð	1ô �z‰z˜*¨Q¸$Ô?À$ÑFÐFr9   Újson_file_pathc                 óˆ   — t        |dd¬«      5 }|j                  | j                  «       «       ddd«       y# 1 sw Y   yxY w)zá
        Save this instance to a JSON file.

        Args:
            json_file_path (`str` or `os.PathLike`):
                Path to the JSON file in which this image_processor instance's parameters will be saved.
        Úwúutf-8©ÚencodingN)ÚopenÚwriterú   )r6   rû   Úwriters      r8   r½   zBaseVideoProcessor.to_json_fileù  s<   € ô �. #°Ô8ð 	0¸FØ�L‰L˜×,Ñ,Ó.Ô/÷	0÷ 	0ñ 	0ús	   � 8¸Ac                 óT   — | j                   j                  › d| j                  «       › �S )Nú )r7   rç   rú   )r6   s    r8   Ú__repr__zBaseVideoProcessor.__repr__  s(   € Ø—.‘.×)Ñ)Ð*¨!¨D×,?Ñ,?Ó,AÐ+BÐCÐCr9   Ú	json_filec                 ó¢   — t        |dd¬«      5 }|j                  «       }ddd«       t        j                  «      } | di |¤ŽS # 1 sw Y   Œ&xY w)aÌ  
        Instantiates a video processor of type [`~video_processing_utils.VideoProcessorBase`] from the path to a JSON
        file of parameters.

        Args:
            json_file (`str` or `os.PathLike`):
                Path to the JSON file containing the parameters.

        Returns:
            A video processor of type [`~video_processing_utils.VideoProcessorBase`]: The video_processor object
            instantiated from that JSON file.
        Úrrþ   rÿ   Nr3   )r  Úreadrö   Úloads)r¨   r  ÚreaderÚtextr©   s        r8   Úfrom_json_filez!BaseVideoProcessor.from_json_file  sP   € ô �)˜S¨7Ô3ð 	!°vØ—;‘;“=ˆD÷	!ä#Ÿz™z¨$Ó/ÐÙÑ*Ð)Ñ*Ð*÷	!ð 	!ús   �AÁAc                 ó�   — t        |t        «      s|j                  }ddlmc m} t        ||«      st        |› d�«      ‚|| _        y)a	  
        Register this class with a given auto class. This should only be used for custom video processors as the ones
        in the library are already mapped with `AutoVideoProcessor `.

        <Tip warning={true}>

        This API is experimental and may have some slight breaking changes in the next releases.

        </Tip>

        Args:
            auto_class (`str` or `type`, *optional*, defaults to `"AutoVideoProcessor "`):
                The auto class to register this new video processor with.
        r   Nz is not a valid auto class.)	r^   rÓ   rç   Útransformers.models.autoÚmodelsÚautorä   rN   r»   )r¨   Ú
auto_classÚauto_modules      r8   Úregister_for_auto_classz*BaseVideoProcessor.register_for_auto_class  sC   € ô  ˜*¤cÔ*Ø#×,Ñ,ˆJç6Ð6ä�{ JÔ/Ü 
˜|Ð+FÐGÓHÐHà$ˆ�r9   Úvideo_url_or_urlsc                 óî   — d}t        «       st        j                  d«       d}t        |t        «      r0t	        t        |D �cg c]  }| j                  ||¬«      ‘Œ c}Ž «      S t        |||¬«      S c c}w )zè
        Convert a single or a list of urls into the corresponding `np.array` objects.

        If a single url is passed, the return value will be a single object. If a list is passed a list of objects is
        returned.
        Ú
torchcodeczÇ`torchcodec` is not installed and cannot be used to decode the video by default. Falling back to `torchvision`. Note that `torchvision` decoding is deprecated and will be removed in future versions. r+   rZ   )ÚbackendrW   )r   ÚwarningsÚwarnr^   r_   r[   rc   r%   )r6   r  rW   r  Úxs        r8   rc   zBaseVideoProcessor.fetch_videos4  sx   € ð ˆÜ&Ô(Ü�M‰MðIôð $ˆGäÐ'¬Ô.ÜœÐarÖsÐ\]˜d×/Ñ/°ÐEVÐ/ÕWÒsÐtÓuÐuäÐ/¸ÐTeÔfÐfùò ts   ÁA2)NNr;   )NFFNÚmain)F)ÚAutoVideoProcessor)Brç   Ú
__module__Ú__qualname__r»   rŠ   r�   r‘   r‰   Úsize_divisorÚdefault_to_squarerŒ   rˆ   r‹   r�   rŽ   r�   r‡   rV   rL   rK   rU   r{   r   r}   Úmodel_input_namesr   r5   r   r>   r    rI   r!   rP   ÚfloatrT   ÚdictÚboolr   r_   rh   rÓ   r   ru   r   ÚBASE_VIDEO_PROCESSOR_DOCSTRINGr<   r   r   r„   Úclassmethodr³   ÚPathLikerª   rÄ   Útupler   r¦   r§   rï   rú   r½   r  r  r  rc   Ú__classcell__)r7   s   @r8   r.   r.   ‘   s9  ø„ ð €Kà€HØ€JØ€IØ€DØ€LØÐØ€IØ€IØ€NØ€JØ€NØ€LØ€NØÐØ
€CØ€JØ€NØ€OØ€LØ.Ð/Ðð# ¨Ñ!5ð #¸$õ #ð1¨Ló 1ðàðð 
óð8 "&Ø"&ñ	3àð3ð ˜$‘Jð3ð �5‰[˜4Ñó	3ðr )-Ø-1ñ&&àð&&ð &¨Ñ,ð&&ð  ™+ð	&&ð
 $ d™?ð&&ð 
ˆnÑ	ó&&ðV <@Ø!ñ	 àð ð Ð!1Ñ1°DÑ8ð ð �d‘
ð	 ð
 
ˆnÑ	ó ñ: Ø&óð*#àð*#ð ˜Ñ&ð*#ð 
ò	*#óð*#ðt 37ñ+hà�^Ñ$ð+hð ð+hð ð	+hð
 ð+hð Lð+hð ð+hð ð+hð ð+hð ð+hð ð+hð ˜D ™KÑ'¨$Ñ.ð+hð ˜4 ™;Ñ&¨Ñ-ð+hð ˜jÑ(¨4Ñ/ð+hð  
ó!+hðZ ð /3Ø$Ø!&Ø#'Øñ`=à'*¨R¯[©[Ñ'8ð`=ð ˜Ÿ™Ñ$ tÑ+ð`=ð ð	`=ð
 ð`=ð �T‰z˜DÑ ð`=ð ò`=ó ð`=ñD.-¨c°B·K±KÑ.?ð .-Èdó .-ð` ð{,Ø,/°"·+±+Ñ,=ð{,à	ˆt�C˜�H‰~˜t C¨ H™~Ð-Ñ	.ò{,ó ð{,ðz ð&#¨T°#°s°(©^ò &#ó ð&#ðP˜˜c 3˜h™õ ðG ó Gð	0¨3°·±Ñ+<ó 	0òDð ð+ s¨R¯[©[Ñ'8ò +ó ð+ð$ ò%ó ð%ñ2g¨c°D¸±I©oÀÀTÈ#ÁYÁÑ.O÷ gr9   r.   rË   r  zvideo processor file)ÚobjectÚobject_classÚobject_files)Grö   r³   r  Úcollections.abcr   Ú	functoolsr   Útypingr   Únumpyrm   Úhuggingface_hubr   r   Úhuggingface_hub.dataclassesr   Údynamic_module_utilsr
   Úimage_processing_backendsr   Úimage_processing_utilsr   Úimage_utilsr   r   r   r   Úprocessing_utilsr   r   Úutilsr   r   r   r   r   r   r   r   r   r   r   Ú	utils.hubr   Úutils.import_utilsr   Úvideo_utilsr    r!   r"   r#   r$   r%   r&   r'   r(   rQ   Ú$torchvision.transforms.v2.functionalÚ
transformsÚv2Ú
functionalrD   r)   Ú
get_loggerrç   r¾   r'  r.   r¬   Ú__doc__Úformatr3   r9   r8   ú<module>rE     sD  ðó Û 	Û Ý $Ý Ý ã ß 8Ý ;å 4Ý 9Ý 0÷ó ÷ 3÷÷ ÷ ñ õ #Ý (÷
÷ 
õ 
ñ ÔÛáÔ ß6Ó6áÔÝ/ð 
ˆ×	Ñ	˜HÓ	%€ðA"Ð ñH Ø'Ø"óñ 
Ð,Ô-ôp
gÐ+ó p
gó .ó	ð
p
gñf "+Ð+=×+IÑ+IÓ!JÐ Ô Ø×!Ñ!×)Ñ)Ð5Ø-?×-KÑ-K×-SÑ-S×-ZÑ-ZØ Ð/CÐRhð .[ó .Ð×"Ñ"Õ*ð 6r9   