
    ^j6              
           d dl Z d dlZd dlmZ d dlZddlmZmZ ddl	m
Z
 ddlmZ ddlmZmZmZ  ej"                  e      Zdej(                  d	ed
ededej(                  f
dZ G d de
      ZdgZy)    N)Sequence   )mel_filter_bankwindow_function)SequenceFeatureExtractor)BatchFeature)PaddingStrategy
TensorTypeloggingarray	dimensionsizestepreturnc                    | j                   dk7  rt        d      |dk7  r|| j                   dz
  k7  rt        d      | j                  \  }}||z
  |z  dz   }|dk  r$t        j                  |d|f| j
                        S |||f}| j                  d   | j                  d   |z  | j                  d   f}t        j                  j                  j                  | ||      S )	zNA basic NumPy equivalent of PyTorch's unfold for 2D arrays along the last dim.   zFThis unfold implementation currently supports 2D arrays (batch, time).   zFThis unfold implementation only supports unfolding the last dimension.r   dtype)shapestrides)
ndim
ValueErrorr   npzerosr   r   libstride_tricks
as_strided)	r   r   r   r   
batch_sizeoriginal_length
num_framesoutput_shapeoutput_stridess	            /var/www/ramen.bs-engineer-server.com/venv/lib/python3.12/site-packages/transformers/models/gemma4/feature_extraction_gemma4.py_unfoldr&      s    zzQabbB9

Q6abb"'++J!D(T1A5JQxxQ-U[[AA
D1LmmA&a(84(?qAQRN66**5n*]]    c            "           e Zd ZdZddgZ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 d!dedededed	ed
edededededededededee   dz  dee   dz  f  fdZ	de
j                  de
j                  dee
j                  e
j                  f   fdZ	 	 	 	 	 	 d"de
j                  ee   z  ee
j                     z  eee      z  deez  ez  dedz  dededz  deez  dz  dedz  defd Z xZS )#Gemma4AudioFeatureExtractoraR
  An audio feature extractor Universal Speech Models https://huggingface.co/papers/2303.01037.

    Args:
        feature_size (`int`, *optional*, defaults to 128):
            The feature dimension of the extracted features.
        sampling_rate (`int`, *optional*, defaults to 16000):
            The sampling rate at which the audio files should be digitalized expressed in hertz (Hz).
        padding_value (`float`, *optional*, defaults to 0.0):
            Padding value used to pad the audio. Should correspond to silences.
        return_attention_mask (`bool`, *optional*, defaults to `True`):
            Whether to return the attention mask for the generated MEL spectrograms.
        frame_length_ms (`float`, *optional*, defaults to 20.0):
            The length of a frame in milliseconds.
        hop_length_ms (`float`, *optional*, defaults to 10.0):
            Length of the overlapping windows for the STFT used to obtain the Mel Frequency coefficients.
        min_frequency (`float`, *optional*, defaults to 0.0):
            The minimum frequency (in Hz) for the Mel filterbank.
        max_frequency (`float`, *optional*, defaults to 8000.0):
            The maximum frequency (in Hz) for the Mel filterbank.
        preemphasis (`float`, *optional*, defaults to 0.0):
            The preemphasis coefficient.
        preemphasis_htk_flavor (`bool`, *optional*, defaults to `True`):
            Whether to use HTK-style preemphasis.
        fft_overdrive (`bool`, *optional*, defaults to `False`):
            Whether to use FFT overdrive.
        dither (`float`, *optional*, defaults to 0.0):
            Adds dithering. In other words, adds a small Gaussian noise to each frame.
            E.g. use 0.0001 to add dithering with a normal distribution centered
            around 0.0 with standard deviation 0.0001 (assuming [-1,+1] range of raw_speech).
            The value 0.0 means no dithering.
            Dithering has similar effect as `spectrogram(mel_floor=...)`. It reduces
            the high log_mel_fbank values for signals with hard-zero sections,
            when VAD cutoff is present in the signal.
        input_scale_factor (`float`, *optional*, defaults to 1.0):
            Scaling factor applied to the input waveform.
        mel_floor (`float`, *optional*, defaults to 0.001):
            Minimum value for Mel spectrograms to avoid log(0).
        per_bin_mean (`Optional[Sequence[float]]`, *optional*):
            Mean values for per-bin normalization.
        per_bin_stddev (`Optional[Sequence[float]]`, *optional*):
            Standard deviation values for per-bin normalization.
    input_featuresinput_features_maskNfeature_sizesampling_ratepadding_valuereturn_attention_maskframe_length_mshop_length_msmin_frequencymax_frequencypreemphasispreemphasis_htk_flavorfft_overdriveditherinput_scale_factor	mel_floorper_bin_meanper_bin_stddevc           
         t        |   d	||||d| || _        || _        |	| _        |
| _        || _        || _        || _        t        t        ||z  dz              | _        t        t        ||z  dz              | _        t        j                  |t        j                        | _        dt#        j$                  t#        j&                  | j                              z  }| j                  r|dz  }|| _        t+        | j                        j-                  t        j.                        | _        t3        j4                         5  t3        j6                  d       t9        | j(                  dz  dz   |||| j:                  d d      | _        d d d        |,t        j                  |      j?                  dd|      | _         nd | _         |,t        j                  |      j?                  dd|      | _!        y d | _!        y # 1 sw Y   txY w)
N)r,   r-   r.   r/   g     @@r   r   ignorer   htk)num_frequency_binsnum_mel_filtersr2   r3   r-   norm	mel_scale )"super__init__r2   r3   r4   r5   r6   r7   r8   introundframe_length
hop_lengthr   r   float64r9   mathceillog2
fft_lengthr   astypefloat32windowwarningscatch_warningssimplefilterr   r-   mel_filtersreshaper:   r;   )selfr,   r-   r.   r/   r0   r1   r2   r3   r4   r5   r6   r7   r8   r9   r:   r;   kwargsrN   	__class__s                      r%   rE   z$Gemma4AudioFeatureExtractor.__init___   s   ( 	 	
%''"7		

 	
 +*&&<#*"4mo&E&N OPeMM$AF$JKL)2::>$))DIId.?.?$@AA
!OJ$ &d&7&78??

K $$& 
	!!(+.#'??a#7!#; ,++"00 D
	 # " 6 > >q!\ RD $D%"$((>":"B"B1a"VD"&D)
	 
	s   AHH
waveformattention_maskr   c                    |j                   dk(  rt        j                  |d      }| j                  dkD  rO|| j                  t        j                  j
                  |j                   j                  |j                        z  z   }| j                  dk7  r|| j                  z  }| j                  dz  }t        j                  |d|dffd	      }t        j                  ||dfdd
      }| j                  dz   }t        |d|| j                        }| j                  dkD  r| j                  rS|dddf   d| j                  z
  z  }|dddf   | j                  |dddf   z  z
  }t        j                   ||gd      }n*|dddf   | j                  |dddf   z  z
  }n	|dddf   }|| j"                  z  }t        j$                  j'                  || j(                  d      }	t        j*                  |	      }
t        j,                  |
| j.                        }t        j0                  || j2                  z         }| j4                  || j4                  z
  }| j6                  || j6                  z  }|j9                  d      }|j                  d   }t        j:                  |      | j                  z  |z   dz
  }||   j                  t<              }||fS ) r   r   )axis              ?r   )r   r   constant)mode)rb   constant_valuesr   )r   r   r   .N)nr^   )r   r   expand_dimsr7   randomrandnr   rO   r   r8   rH   padr&   rI   r4   r5   concatenaterQ   fftrfftrN   absmatmulrU   logr9   r:   r;   squeezearangebool)rW   rZ   r[   pad_leftframe_size_for_unfoldframes_to_processfirst_in_framerest_in_frameframesstftmagnitude_specmel_speclog_mel_specmel_spectrogramnum_mel_framesframe_end_indicesmasks                    r%   _extract_spectrogramz0Gemma4AudioFeatureExtractor._extract_spectrogram   s   ==A~~hQ7H;;$++		0P0W0WX`XfXf0g"ggH""c)$"9"99H $$)66(Vh]$;*M1J`ab $ 1 1A 5 $HAV]a]l]lmc!**!237!;sTEUEU?U!V 1#qt) <t?O?ORcdgiljlildlRm?m m(GbQ*373d6F6FIZ[^`cac`c[cId6dd&sCRCx0F $++%vv{{6T__2{>99^T-=-=>vvh78('$*;*;;L*'$*=*==L&..q1(..q1
 IIn5GJ__bcc/077=$$r'   
raw_speechpadding
max_length
truncationpad_to_multiple_ofreturn_tensorsc                    t        |t        j                        xr t        |j                        dkD  }	t        |t
              xr# t        |d   t        j                  t
        f      }
|	xs |
}|r.|D cg c]"  }t        j                  |g      j                  $ }}n1|s/t        |t        j                        st        j                  |      }|st        j                  |g      g}| j                  t        d|i      |||||      }g }g }t        |j                  |j                        D ]c  \  }}| j                  |j                  |      \  }}|j                  |j                  t        j                                |j                  |       e t        ||      D cg c]  \  }}||d   z   }}}t        ||d|      S c c}w c c}}w )a  Creates a batch of MEL spectrograms from the provided raw speech.

        This implementation uses a different algorithm for windowing and preemphasis compared to the built-in
        `transformers.audio_utils.spectrogram()` function that _will_ result in different outputs. Consider this
        carefully when selecting an audio feature extractor, especially with pre-trained models.

        Args:
            raw_speech:
                The audio for which MEL spectrograms are created.
            padding (`Union[bool, str, PaddingStrategy]`, *optional*, defaults to `"longest"`):
                The padding strategy to use for batches of audio with different lengths.
            max_length (`int`, *optional*, defaults to 480000):
                If provided, defines the maximum length of the audio to allow. Audio longer than this will be
                truncated if `truncation=True`.
            truncation (`bool`, *optional*, defaults to `True`):
                Whether or not to truncate audio above `max_length`.
            pad_to_multiple_of (`int`, *optional*, defaults to 128):
                When padding, pad to a multiple of this value. The default value is defined for optimal TPU support.
            return_tensors (`Union[str, TensorType]`, *optional*, defaults to `None`):
                The type of tensors to return (e.g., NumPy, or Torch).
            return_attention_mask (`bool`, *optional*, defaults to `True`):
                Whether to return the attention mask for the generated MEL spectrograms.
        r   r   r*   )r   r   r   r   r/   ).N)r*   r+   )tensor_type)
isinstancer   ndarraylenr   r   asarrayTri   r   zipr*   r[   r   appendrO   rP   )rW   r   r   r   r   r   r   r/   rX   is_batched_numpyis_batched_sequence
is_batchedrsbatched_speechprepared_speechprepared_speech_maskspeechr   s                     r%   __call__z$Gemma4AudioFeatureExtractor.__call__   s   F &j"**=[#jFVFVBWZ[B[(X>t:jYZm^`^h^hjr]sCt%<)<
7AB"**bT*,,BJBJz2::$FJ/J**j\23J*J78!!1"7 " 
 ! = =~?\?\] 	.LFD44VXXtDLFD""6==#<= ''-	.
 ILO]qHrs6DO3ss.G[\&
 	
3 C. ts   6'G
%G)   i>  r_   Tg      4@g      $@r_   g     @@r_   TFr_   r`   gMbP?NN)longesti S Tr   NT)__name__
__module____qualname____doc__model_input_namesrF   floatrr   r   rE   r   r   tupler   liststrr	   r
   r   r   __classcell__)rY   s   @r%   r)   r)   1   s   )V *+@A  #"&*!%#"% '+#$'/315#H'H' H' 	H'
  $H' H' H' H' H' H' !%H' H' H' "H' H'  uo,!H'" !$.#H'T8%RZZ 8% 8%X]^`^h^hjljtjt^tXu 8%z 1:!(),26-1D
JJe,tBJJ/??$tE{BSSD
 o-D
 $J	D

 D
  $JD
 j(4/D
  $d{D
 
D
r'   r)   )rK   rR   collections.abcr   numpyr   audio_utilsr   r   !feature_extraction_sequence_utilsr   feature_extraction_utilsr   utilsr	   r
   r   
get_loggerr   loggerr   rF   r&   r)   __all__rC   r'   r%   <module>r      s      $  ; I 4 9 9 
		H	%^2:: ^# ^S ^ ^

 ^&v
": v
r )
)r'   