
    ^j:              
       :   d dl Z ddlmZ ddlmZ ddlmZmZmZm	Z	m
Z
mZmZmZmZmZmZmZmZmZmZmZmZmZmZmZmZmZmZmZmZ ddlm Z  ddl!m"Z" dd	l#m$Z$ dd
l%m&Z& ddl'm(Z( ddl)m*Z* ddl+m,Z, ddl-m.Z. ddl/m0Z0 ddl1m2Z2 ddl3m4Z4 ddl5m6Z6 ddl7m8Z8 ddl9m:Z: ddl;m<Z< ddl=m>Z> ddl?m@Z@ ddlAmBZB ddlCmDZD ddlEmFZF ddlGmHZH ddlImJZJ ddlKmLZL ddlMmNZN ddlOmPZP i d e&d!e*d"e,d#e<d$e"d%eFd&eHd'e6d(e8d)e0d*e>d+e@d,e.d-e2d.eNd/e(d0ePeLe4e4e$eDeBeJe:d1ZQi d ed!e
d"e
d)ed#ed$ed%ed&ed'ed(ed+ed,ed-ed*ed.ed/e	d0eeeeeeeeed1ZReeeeeeeefZS ej                  eU      ZV G d2 d3      ZW G d4 d5      ZXd6eYfd7ZZd8eYfd9Z[d: Z\y);    N   )
AutoConfig)logging)
AqlmConfigAutoRoundConfig	AwqConfigBitNetQuantConfigBitsAndBytesConfigCompressedTensorsConfig
EetqConfigFbgemmFp8ConfigFineGrainedFP8ConfigFourOverSixConfigFPQuantConfigGemmaQuantizationConfig
GPTQConfigHiggsConfig	HqqConfigMetalConfigMxfp4ConfigQuantizationConfigMixinQuantizationMethodQuantoConfigQuarkConfig
SinqConfig
SpQRConfigTorchAoConfig
VptqConfig   )HfQuantizer)AqlmHfQuantizer)AutoRoundQuantizer)AwqQuantizer)BitNetHfQuantizer)Bnb4BitHfQuantizer)Bnb8BitHfQuantizer)CompressedTensorsHfQuantizer)EetqHfQuantizer)FbgemmFp8HfQuantizer)FineGrainedFP8HfQuantizer)FourOverSixHfQuantizer)FPQuantHfQuantizer)GemmaQuantizer)GptqHfQuantizer)HiggsHfQuantizer)HqqHfQuantizer)MetalHfQuantizer)Mxfp4HfQuantizer)QuantoHfQuantizer)QuarkHfQuantizer)SinqHfQuantizer)SpQRHfQuantizer)TorchAoHfQuantizer)VptqHfQuantizerawqbitsandbytes_4bitbitsandbytes_8bitgptqaqlmquantoquarkfouroversixfp_quanteetqhiggshqqzcompressed-tensors
fbgemm_fp8torchaobitnetvptq)spqrfp8mxfp8z
auto-roundmxfp4metalsinqgemmac                   6    e Zd ZdZedefd       Zed        Zy)AutoQuantizationConfigz
    The Auto-HF quantization config class that takes care of automatically dispatching to the correct
    quantization config given a quantization config stored in a dictionary.
    quantization_config_dictc           	      v   |j                  d      }|j                  dd      s|j                  dd      r*|j                  dd      rdnd}t        j                  |z   }n|t        d      |t        vr,t        d| d	t        t        j                                      t        |   }|j                  |      S )
Nquant_methodload_in_8bitFload_in_4bit_4bit_8bitThe model's quantization config from the arguments has no `quant_method` attribute. Make sure that the model has been correctly quantizedUnknown quantization type, got  - supported types are: )	getr   BITS_AND_BYTES
ValueError AUTO_QUANTIZATION_CONFIG_MAPPINGlistAUTO_QUANTIZER_MAPPINGkeys	from_dict)clsrR   rT   suffix
target_clss        g/var/www/ramen.bs-engineer-server.com/venv/lib/python3.12/site-packages/transformers/quantizers/auto.pyrc   z AutoQuantizationConfig.from_dict   s    /33NC#''>BZB^B^_motBu 8 < <^U SWY`F-<<vEL! \  ??1, @/44678: 
 6lC
##$<==    c                     t        j                  |fi |}t        |dd       t        d| d      |j                  }| j                  |      } |j                  di | |S )Nquantization_configz)Did not found a `quantization_config` in z2. Make sure that the model is correctly quantized. )r   from_pretrainedgetattrr^   rj   rc   update)rd   pretrained_model_name_or_pathkwargsmodel_configrR   rj   s         rg   rl   z&AutoQuantizationConfig.from_pretrained   s    !112OZSYZ<!6=E;<Y;Z  [M  N  $0#C#C !mm,DE""",V,""rh   N)__name__
__module____qualname____doc__classmethoddictrc   rl   rk   rh   rg   rQ   rQ      s6    
 > > >( 
# 
#rh   rQ   c                   r    e Zd ZdZedeez  fd       Zed        Zedeez  dedz  fd       Z	e
d        Zy)	AutoHfQuantizerz
     The Auto-HF quantizer class that takes care of automatically instantiating to the correct
    `HfQuantizer` given the `QuantizationConfig`.
    rj   c           	      z   t        |t              rt        j                  |      }|j                  }|t
        j                  k(  r2t        |t              st        d      |j                  r|dz  }n|dz  }|t        vr,t        d| dt        t        j                                      t        |   } ||fi |S )NzZFound `quant_method=bitsandbytes` but `quantization_config` is not a `BitsAndBytesConfig`.rX   rW   rZ   r[   )
isinstancerw   rQ   rc   rT   r   r]   r
   	TypeErrorrU   ra   r^   r`   rb   )rd   rj   rp   rT   rf   s        rg   from_configzAutoHfQuantizer.from_config   s     )40"8"B"BCV"W*77 -<<<13EFp  #//''551, @/44678: 
 ,L9
-888rh   c                 P    t        j                  |fi |}| j                  |      S )N)rQ   rl   r}   )rd   ro   rp   rj   s       rg   rl   zAutoHfQuantizer.from_pretrained   s*    4DDEbmflm233rh   quantization_config_from_argsNc                    |d}nd}t        |t              r;t        |t              rt        j                  |      }nt        j                  |      }|g|j
                  j                  |j
                  j                  k7  r:t        d|j
                  j                   d|j
                  j                   d      t        |t              rgt        |t              rW|j                         }|j                         D ]  \  }}t        |||        |r |dt        |j                                dz  }|dk7  r2t        |t        t        t         f      st#        j$                  |       |S t&        j)                  |       |S )z
        handles situations where both quantization_config from args and quantization_config from model config are present.
        zYou passed `quantization_config` or equivalent parameters to `from_pretrained` but the model you're loading already has a `quantization_config` attribute. The `quantization_config` from the model will be used. zThe model is quantized with z but you are passing a z| config. Please make sure to pass the same quantization config class to `from_pretrained` with different loading attributes.z"However, loading attributes (e.g. z]) will be overwritten with the one you passed to `from_pretrained`. The rest will be ignored.)r{   rw   r   rc   rQ   	__class__rr   r^   LOADING_ATTRIBUTES_CONFIG_TYPESget_loading_attributesitemssetattrr`   rb   r   r   r   warningswarnloggerinfo)rd   rj   r   warning_msgloading_attr_dictattrvals          rg   merge_quantization_configsz*AutoHfQuantizer.merge_quantization_configs   s    )4y 
 K)407I&5&?&?@S&T#&<&F&FGZ&[# *5#--66:W:a:a:j:jj./B/L/L/U/U.VVm  oL  oV  oV  o_  o_  n` `F F 
 )+JKPZ)+JQ
 !> T T V.446 8	c+T378 !!CDIZI_I_IaDbCc  dA   B  B"Z0CkS^`tEu%vMM+& #" KK$""rh   c           	      ^   | j                  dd       }| j                  dd      s| j                  dd      r*| j                  dd      rdnd}t        j                  |z   }n|t        d      |t        vr8t
        j                  d| d	t        t        j                                d
       yy)NrT   rU   FrV   rW   rX   rY   rZ   r[   z~. Hence, we will skip the quantization. To remove the warning, you can delete the quantization_config attribute in config.jsonT)
r\   r   r]   r^   r_   r   warningr`   ra   rb   )rR   rT   re   s      rg   supports_quant_methodz%AutoHfQuantizer.supports_quant_method  s    /33NDI#''>BZB^B^_motBu 8 < <^U SWY`F-<<vEL! \  ??NN1, @/44678 9ii
 rh   )rr   rs   rt   ru   rv   r   rw   r}   rl   r   staticmethodr   rk   rh   rg   ry   ry      s    
 9.E.L 9 98 4 4 /#!$;;/# (?'E/# /#b  rh   ry   methodc                       fd}|S )z-Register a custom quantization configuration.c                 ~    t         v rt        d d      t        | t              st	        d      | t         <   | S )NzConfig '' already registeredz*Config must extend QuantizationConfigMixin)r_   r^   
issubclassr   r|   )rd   r   s    rg   register_config_fnz8register_quantization_config.<locals>.register_config_fn-  sH    55xx/CDEE#67HII36(0
rh   rk   )r   r   s   ` rg   register_quantization_configr   *  s     rh   namec                       fd}|S )zRegister a custom quantizer.c                 ~    t         v rt        d d      t        | t              st	        d      | t         <   | S )NzQuantizer 'r   z!Quantizer must extend HfQuantizer)ra   r^   r   r    r|   )rd   r   s    rg   register_quantizer_fnz1register_quantizer.<locals>.register_quantizer_fn=  sG    )){4&0DEFF#{+?@@'*t$
rh   rk   )r   r   s   ` rg   register_quantizerr   :  s     ! rh   c                 B   t        | dd       xs t        | j                  d      dd       }|d u}|rt        j                  |      sd}|s|G|rt        j	                  ||      | _        n|| _        t        j                  | j
                  |      }nd }||j                  ||       |j                  |      }|j                  |       } |j                  |       } t        |j
                  dd      s&|j
                  j                  }t        |d|      |d	<   || |fS )
Nrj   T)decoderF)pre_quantized)
device_mapweights_only
dequantizevaluequant)rm   get_text_configry   r   r   rj   r}   validate_environmentupdate_device_mapupdate_tp_planupdate_ep_planrT   )	configrj   r   r   
user_agentquantization_params_from_configr   hf_quantizerrT   s	            rg   get_hf_quantizerr   J  sE   &-f6KT&R 'V]t,.CTW# 44?M_BBCbc+7)8)S)S/1D*F& *=F&&22&&' 3 

 ))!% 	* 	
 "33J?
,,V4,,V4 |77uM';;HHL"),"NJw++rh   )]r   models.auto.configuration_autor   utilsr   utils.quantization_configr   r   r   r	   r
   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   baser    quantizer_aqlmr!   quantizer_auto_roundr"   quantizer_awqr#   quantizer_bitnetr$   quantizer_bnb_4bitr%   quantizer_bnb_8bitr&   quantizer_compressed_tensorsr'   quantizer_eetqr(   quantizer_fbgemm_fp8r)   quantizer_finegrained_fp8r*   quantizer_fouroversixr+   quantizer_fp_quantr,   quantizer_gemmar-   quantizer_gptqr.   quantizer_higgsr/   quantizer_hqqr0   quantizer_metalr1   quantizer_mxfp4r2   quantizer_quantor3   quantizer_quarkr4   quantizer_sinqr5   quantizer_spqrr6   quantizer_torchaor7   quantizer_vptqr8   ra   r_   r   
get_loggerrr   r   rQ   ry   strr   r   r   rk   rh   rg   <module>r      s    7       6  + 4 ' / 2 2 F + 6 @ 9 2 + + - ) - - / - + + 1 +	<+ + O	
 O   ) " O  
> 6 & !  !" O#$ $ '$9 >$	9$+$ +$ J	$
 J$ J$ l$ [$ $$ $ 
9$ 1$ /$ [$ }$  !$" J#$$ !!$3$  : 	#  
		H	%&# &#Rl l^  !S ! $,rh   