
    ^j:6                        d dl ZddlmZ ddlmZmZ ddlmZm	Z	m
Z
mZ ddlmZmZ ddlmZmZmZ ddlmZ dd	lmZ  e       rd
dlmZmZ  ej4                  e      Z G d de	d      Ze ed       G d de
                    ZdgZy)    N   )
AudioInput)
ImageInputmake_nested_list_of_images)MultiModalDataProcessingKwargsProcessorMixinUnpack)PreTokenizedInput	TextInput)auto_docstringis_vision_availablelogging)requires)
VideoInput   )Gemma4ImageProcessorKwargs get_aspect_ratio_preserving_sizec                   4    e Zd ZU eed<   dddddii ddidZy)Gemma4ProcessorKwargsimages_kwargsT)paddingreturn_mm_token_type_idsdo_convert_rgbreturn_metadata)text_kwargsr   audio_kwargsvideos_kwargsN)__name__
__module____qualname__r   __annotations__	_defaults     w/var/www/ramen.bs-engineer-server.com/venv/lib/python3.12/site-packages/transformers/models/gemma4/processing_gemma4.pyr   r   "   s5    -- (,

 d
 +T2
Ir%   r   F)total)vision)backendsc                   h    e Zd ZeZ	 	 	 	 ddededef fdZ	 	 	 	 ddedz  dee	z  e
e   z  e
e	   z  ded	ef fd
Z	 	 	 	 ddee
e   z  dz  dee	z  e
e   z  e
e	   z  ded	edee   f
 fdZdededefdZdededefdZdededefdZddZdedefdZe fd       Zede
e   fd       Z xZS )Gemma4ProcessorNimage_seq_lengthaudio_seq_lengthaudio_ms_per_tokenc	           	         || _         |j                  | _        |j                  | _        |j                  | _        |j                  | _        |j                  ddgi       d| _        |j                  | j                        | _        || _	        || _
        t        |dd      | _        t        |dd      | _        t        |dd      | _        t        |dd      | _        t!        
| D  d	|||||d|	 y)
u  
        image_seq_length (`int`, *optional*, defaults to 280):
            The number of soft tokens per image used for placeholder expansion.
        audio_seq_length (`int`, *optional*, defaults to 750):
            The maximum number of audio soft tokens per audio segment. Serves as an
            upper-bound cap when dynamic audio token counts are computed.
        audio_ms_per_token (`int`, *optional*, defaults to 40):
            Milliseconds of audio per output soft token. Used to dynamically compute
            the number of audio placeholder tokens as ``ceil(duration_ms / audio_ms_per_token)``.
            The default of 40 comes from the SSCP convolution's 4× time reduction on 10ms frames.
        additional_special_tokensz	<|video|>audio_token_idNaudio_token	boa_token	eoa_token)feature_extractorimage_processor	tokenizervideo_processorchat_templater$   )r,   image_token_id	boi_token	eoi_tokenimage_tokenadd_special_tokensvideo_tokenconvert_tokens_to_idsvideo_token_idr-   r.   getattrr1   r2   r3   r4   super__init__)selfr5   r6   r7   r8   r9   r,   r-   r.   kwargs	__class__s             r&   rD   zGemma4Processor.__init__6   s    . !1'66",,",,$00 	$$&AK=%QR&'==d>N>NO !1 #5%i1A4H"9mTB K> K> 	
/++'	
 	
r%   imagestextvideosaudioc           	         t        |   d||||d|\  }}}}|t        |      }|r7|s5|D cg c]*  }dj                  | j                  gt        |      z        , }}|r|s| j                  gt        |      z  }||||fS c c}w )N)rH   rI   rJ   rK    r$   )rC   prepare_inputs_layoutr   joinr=   lenr2   )rE   rH   rI   rJ   rK   rF   
image_listrG   s          r&   rN   z%Gemma4Processor.prepare_inputs_layoutn   s     ',g&C '
V5'
DJ'
#fe
 /7F $U[\zCHHd../#j/AB\D\$$%E
2DtVU**	 ]s   /BrF   c                    t        
|   d||d| ||t        d      || j                  | j                  | j
                  t        d      ||D cg c]  }|j                  | j                         }}|t        |      t        |      k7  r$t        dt        |       dt        |       d      |D cg c]  }t        |       }	}||	k7  r,t        d| j                   d| d	| j                   d
|	 d	      y |1t        |      r%t        dt        |       d	| j                   d      y y y c c}w c c}w )N)rH   rI   z+You must provide either `text` or `images`.zUAudio inputs were provided, but the tokenizer does not have an `audio_token` defined.z1Received inconsistently sized batches of images (z) and text (z).zThe total number of zP tokens in the prompts should be the same as the number of images passed. Found rM   z tokens and z images per sample.zFound z. tokens in the text but no images were passed.r$   )rC   validate_inputs
ValueErrorr2   r3   r4   countr=   rP   anysum)rE   rH   rI   rJ   rK   rF   samplen_images_in_textsublistn_images_in_imagesrG   s             r&   rS   zGemma4Processor.validate_inputs   s    	CvDCFC<FNJKK!1!1!9T^^=SW[WeWeWmtuuMQR6T-=-= >RR!v;#d)+$KCPVK=Xdehimendooqr  CI%Iwc'l%I"%I#'99$.t/?/?.@ A""2!31T5E5E4FlSeRffy{  :
 C(8$9 S!1231T5E5E4FFtu  %: R &Js   "D:?D?image_inputs	image_idxreturnc                 d    |d   |   }| j                    | j                  |z   | j                   S )Nnum_soft_tokens_per_image)r;   r=   r<   )rE   r\   r]   num_soft_tokenss       r&   replace_image_tokenz#Gemma4Processor.replace_image_token   s;    &'BCIN..!$"2"2_"D!EdnnEUVVr%   video_inputs	video_idxc           
         |d   |   }|d   |   }|j                   t        j                  d       |j                   dn|j                   |_         |j                  D cg c]#  }t	        |dz        ddt	        |dz        d% }}dj                  |D cg c].  }| d| j                   | j                  |z   | j                   0 c}      }|S c c}w c c}w )	Nnum_soft_tokens_per_videovideo_metadataa  Gemma4 requires frame timestamps to construct prompts, but the `fps` of the input video could not be inferred. Probably `video_metadata` was missing from inputs and you passed pre-sampled frames. Defaulting to `fps=24`. Please provide `video_metadata` for more accurate results.   <   02d:rM   )	fpsloggerwarning_once
timestampsintrO   r;   r?   r<   )	rE   rc   rd   ra   metadatasecondstimestamp_strtvideo_replacements	            r&   replace_video_tokenz#Gemma4Processor.replace_video_token   s    &'BCIN 01)<<<e
 &\\1rx|| ]e\o\opQXC2.s31S25Fs4KLppHHbop]^s!DNN#D$4$4$F#GGWXp
 ! 	 qps   (C3Caudio_inputs	audio_idxc                    |d   |   }t        |      }t        d      D ]&  }|dz   dz
  dz  dz   }|d d d   d | }t        |      }( | j                   | j                  t	        |j                               z   | j                   S )Ninput_features_mask   r   r   )rP   ranger3   r2   rp   rW   r4   )rE   rw   rx   maskrt   _t_outs          r&   replace_audio_tokenz#Gemma4Processor.replace_audio_token   s    129= Iq 	AUQY1$q(E!9Ve$DD	A	
 ..!$"2"2S_"D!EdnnEUVVr%   c                 &   t         j                  j                  di       }|j                  |       |j                  dd      xs | j                  j
                  }|j                  dd      xs | j                  j                  }|j                  dd      xs | j                  j                  }||dz  z  }i }	|ig }
|D ]?  }t        |d   |d   |||	      \  }}||z  }||z  }|
j                  ||z  |dz  z         A dgt        |      z  }|	j                  |
|d
       |\t        | j                  dd      }|D cg c]'  }| j                  t        j                  |      |      ) }}|	j                  d|i       t!        di |	S c c}w )av  
        Computes the number of placeholder tokens needed for multimodal inputs with the given sizes.

        Args:
            image_sizes (`list[list[int]]`, *optional*):
                The input sizes formatted as (height, width) per each image.
            audio_lengths (`list[int]`, *optional*):
                The lengths of audio inputs in number of samples. Used to dynamically
                compute per-audio token counts.

        Returns:
            `MultiModalData`: A `MultiModalData` object holding number of tokens per each of the provided
            input modalities, along with other useful data.
        r   
patch_sizeNpooling_kernel_sizemax_soft_tokensr{   r   r   )heightwidthr   max_patchesr   )num_image_tokensnum_image_patchessampling_ratei>  num_audio_tokensr$   )r   r#   getupdater6   r   r   r   r   appendrP   rB   r5   _compute_audio_num_tokensnpzerosr   )rE   image_sizesaudio_lengthsrF   r   r   r   r   r   vision_datar   
image_sizetarget_htarget_wpatch_heightpatch_widthr   r   lengthr   s                       r&   _get_num_multimodal_tokensz*Gemma4Processor._get_num_multimodal_tokens   s     .77;;ORPV$"&&|T:]d>R>R>]>]
3T:fd>R>R>f>f 	 (++,=tDlH\H\HlHl%(;Q(>>"!) 
^
%E%a=$Q-) +(;&"(  (:5&*4 ''{(BFY[\F\(\]
^ "#c+&6 64D[lmn$ $D$:$:OVTM^k TZ..rxx/?O     24DEF,,, s   ,Fr   c                 N   t        |      }| j                  j                  dz   }| j                  j                  dz  }||z   |z
  | j                  j                  z  dz   }|dk  ryd}d\  }}	}
|}t	        |      D ]  }|d|
z  z   |z
  |	z  dz   } t        || j                        S )a  Number of audio soft tokens, replicating the encoder's seq-length arithmetic.

        Mirrors Gemma4AudioFeatureExtractor mel framing + the two stride-2 Conv2d
        subsampling layers in Gemma4AudioSubSampleConvProjection, capped at
        ``audio_seq_length``. Must match ``audio_mask.sum()`` from the audio tower or
        vLLM's ``_merge_multimodal_embeddings`` will raise on a length mismatch.

        Args:
            audio_waveform: A 1-D array or list containing the raw audio samples.
            sampling_rate: The sampling rate of the audio waveform in Hz.

        Returns:
            The number of audio soft tokens to insert as placeholders.
        r   r{   r   )r   r{   r   )rP   r5   frame_length
hop_lengthr|   minr-   )rE   audio_waveformr   num_samplesframe_size_for_unfoldpad_leftnum_mel_framessscp_num_layerssscp_kernelsscp_stridesscp_paddingrt   r~   s                r&   r   z)Gemma4Processor._compute_audio_num_tokens  s     .) !% 6 6 C Ca G))66!;%03HHTMcMcMnMnnqrrQ 18.[,' 	HAQ%%3CaGA	H
 1d++,,r%   c                      t         |   dgz   S )Nmm_token_type_ids)rC   model_input_names)rE   rG   s    r&   r   z!Gemma4Processor.model_input_names,  s    w(,?+@@@r%   c                 
    ddgS )Nr`   rf   r$   )rE   s    r&   unused_input_namesz"Gemma4Processor.unused_input_names0  s    +-HIIr%   )Ni  i  (   )NNNN)NN)r   r    r!   r   valid_processor_kwargsrp   rD   r   r   r   listr   r   rN   r
   r   rS   dictstrrb   rv   r   r   r   propertyr   r   __classcell__)rG   s   @r&   r+   r+   1   s    3  # #"$6
 6
 6
  6
t %)Z^! +T!+ ++d9o=EV@WW+ 	+
 +4 8<Z^! !T*--4! ++d9o=EV@WW! 	!
 ! )*!FW W W W! ! ! !&W W W W5-n&-s &-s &-P A A JDI J Jr%   r+   ) numpyr   audio_utilsr   image_utilsr   r   processing_utilsr   r   r	   r
   tokenization_utils_baser   r   utilsr   r   r   utils.import_utilsr   video_utilsr   image_processing_gemma4r   r   
get_loggerr   rm   r   r+   __all__r$   r%   r&   <module>r      s      % A X X C A A * % e 
		H	%,E  	;Jn J   JD 
r%   