§
    ‚ŠtjØ  ã                   óÊ   — d dl mZmZ ddlmZ ddlmZ erddlmZ ddl	m
Z
 ddlmZmZmZmZmZ dd	l	mZ  e¦   «         rd d
lZ ej        e¦  «        Z G d„ de¦  «        Zd
S )é    )ÚTYPE_CHECKINGÚOptionalé   )ÚHfQuantizer)Úget_module_from_nameé   )ÚPreTrainedModel)ÚFPQuantConfig)Úis_fp_quant_availableÚis_qutlass_availableÚis_torch_availableÚis_torch_xpu_availableÚlogging)ÚQuantizationConfigMixinNc                   ó¦   ‡ — e Zd ZU dZdZdZded<   defˆ fd„Zd„ Z	dd„Z
ddded
efd„Z	 	 dd„Zedded         fd„¦   «         Zd„ Zd„ Zd„ Zˆ xZS )ÚFPQuantHfQuantizerzŒ
    Quantizer for the FP-Quant method. Enables the loading of prequantized models and in-flight quantization of full-precision models.
    FTr
   Úquantization_configc                 ó<   •—  t          ¦   «         j        |fi |¤Ž d S ©N)ÚsuperÚ__init__)Úselfr   ÚkwargsÚ	__class__s      €úh/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/quantizers/quantizer_fp_quant.pyr   zFPQuantHfQuantizer.__init__+   s)   ø€ Ø�‰ŒÔÐ,Ð7Ð7°Ð7Ð7Ð7Ð7Ð7ó    c                 óR  — t           j                             ¦   «         st          ¦   «         st	          d¦  «        ‚t          ¦   «         s| j        j        st          d¦  «        ‚| j        j        re| j        j	        dk    rUt           j                             ¦   «         r7t           j         
                    ¦   «         d         dk     rt          d¦  «        ‚| j        j        rt                               d¦  «         t          ¦   «         st          d¦  «        ‚|€| j        j        st          d	¦  «        ‚t          |t           ¦  «        rZ| j        j        s)t#          |¦  «        d
k    rd|                     ¦   «         v sd|                     ¦   «         v rt          d¦  «        ‚d S d S )Nz]FPQuant quantization is only supported on GPU or Intel XPU. Please use a different quantizer.a€  Using `fp_quant` with real quantization requires a **Blackwell GPU** and qutlass: `git clone https://github.com/IST-DASLab/qutlass.git && cd qutlass && pip install --no-build-isolation .`. You can use `FPQuantConfig(pseudoquantization=True, ...)` to use Triton-based pseudo-quantization. It doesn't provide any speedups but emulates the quantization behavior of the real quantization.Únvfp4r   é	   zäNVFP4 pseudoquantization requires a GPU with compute capability >= 9.0 (Hopper or newer) because the Triton kernel uses the `fp8e4nv` type. Please use `forward_dtype='mxfp4'` instead, or use a GPU with compute capability >= 9.0.zŠUsing pseudo-quantization for FP-Quant. This doesn't provide any speedups but emulates the quantization behavior of the real quantization.zGUsing `fp_quant` quantization requires fp_quant: `pip install fp_quant`zyYou are attempting to load a FPQuant model without setting device_map. Please set device_map comprised of 'cuda' devices.r   ÚcpuÚdiskz±You are attempting to load a FPQuant model with a device_map that contains a CPU or disk device. This is not supported. Please remove the CPU or disk device from the device_map.)ÚtorchÚcudaÚis_availabler   ÚNotImplementedErrorr   r   ÚpseudoquantizationÚImportErrorÚforward_dtypeÚget_device_capabilityÚ
ValueErrorÚloggerÚwarningr   Ú
isinstanceÚdictÚlenÚvalues)r   Ú
device_mapr   s      r   Úvalidate_environmentz'FPQuantHfQuantizer.validate_environment.   sÝ  € ÝŒz×&Ò&Ñ(Ô(ð 	Õ1GÑ1IÔ1Ið 	Ý%Øoñô ð õ $Ñ%Ô%ð 	¨dÔ.FÔ.Yð 	Ýð Sñô ð ð
 Ô$Ô7ð
	àÔ(Ô6¸'ÒAÐAÝ”
×'Ò'Ñ)Ô)ð Bå”
×0Ò0Ñ2Ô2°1Ô5¸Ò9Ð9åð?ñô ð ð Ô#Ô6ð 	Ý�NŠNð ]ñô ð õ %Ñ&Ô&ð 	iÝÐgÑhÔhÐhàÐ dÔ&>Ô&QÐÝðFñô ð õ ˜
¥DÑ)Ô)ð 
	àÔ,Ô?ð	å˜
‘O”O aÒ'Ð'Ø˜Z×.Ò.Ñ0Ô0Ð0Ð0Ø˜Z×.Ò.Ñ0Ô0Ð0Ð0å ðhñô ð ð
	ð 
	ð
 1Ð0r   Údtypeútorch.dtypeÚreturnc                 óz   — |t           j        k    r*t                               d|› d�¦  «         t           j        }|S )NzSetting dtype to zP, but only bfloat16 is supported right now. Overwriting torch_dtype to bfloat16.)r"   Úbfloat16r+   Úwarning_once)r   r3   s     r   Úupdate_dtypezFPQuantHfQuantizer.update_dtype^   sC   € Ø•E”NÒ"Ð"Ý×ÒØ{ EÐ{Ð{Ð{ñô ð õ ”NˆEØˆr   Úmodelr	   Ú
param_namec                 ód   — ddl m} t          ||¦  «        \  }}t          ||¦  «        r|dv rdS dS )Nr   )ÚFPQuantLinear)ÚweightÚqweightÚdqweightTF)Úfp_quantr=   r   r-   )r   r:   r;   r   r=   ÚmoduleÚtensor_names          r   Úparam_needs_quantizationz+FPQuantHfQuantizer.param_needs_quantizationf   sO   € Ø*Ð*Ð*Ð*Ð*Ð*å2°5¸*ÑEÔEÑˆ�Ý�f˜mÑ,Ô,ð 	°Ð@aÐ1aÐ1aà�4à�5r   c                 óT   — ddl m} ddlm}  || || j        ¦  «        ¬¦  «         d S )Nr   )Úreplace_with_fp_quant_linearr   )Úadapt_fp_quant_config)Úfp_quant_linear_config)rA   rF   Úintegrations.fp_quantrG   r   )r   r:   r   rF   rG   s        r   Ú$_process_model_before_weight_loadingz7FPQuantHfQuantizer._process_model_before_weight_loadingp   s`   € ð
 	:Ð9Ð9Ð9Ð9Ð9àAÐAÐAÐAÐAÐAà$Ð$ØØ#8Ð#8¸Ô9QÑ#RÔ#Rð	
ñ 	
ô 	
ð 	
ð 	
ð 	
r   Nc                 óV   — | j         j        }|st                               d¦  «         |S )Nz²You are attempting to train a model with FPQuant quantization. This is only supported when `store_master_weights=True`. Please set `store_master_weights=True` to train the model.)r   Ústore_master_weightsr+   r,   )r   r:   Ú	trainables      r   Úis_trainablezFPQuantHfQuantizer.is_trainable~   s9   € àÔ,ÔAˆ	Øð 	Ý�NŠNð Eñô ð ð Ðr   c                 ó   — dS )NT© )r   s    r   Úis_serializablez"FPQuantHfQuantizer.is_serializable‡   s   € Øˆtr   c                 ó$   — ddl m}  || ¦  «        S )Nr   )ÚFpQuantQuantize)rI   rS   )r   rS   s     r   Úget_quantize_opsz#FPQuantHfQuantizer.get_quantize_opsŠ   s$   € Ø;Ð;Ð;Ð;Ð;Ð;àˆ˜tÑ$Ô$Ð$r   c                 ó¬   — ddl m} ddlm} | j        r@| j        j        r |dgd || ¦  «        g¬¦  «        gS  |dgd || ¦  «        g¬¦  «        gS g S )Nr   )ÚWeightConverter)ÚFpQuantDeserializez	.dqweight)Úsource_patternsÚtarget_patternsÚ
operationsz.qweight)Úcore_model_loadingrV   rI   rW   Úpre_quantizedr   r&   )r   rV   rW   s      r   Úget_weight_conversionsz)FPQuantHfQuantizer.get_weight_conversions�   s¶   € Ø8Ð8Ð8Ð8Ð8Ð8Ø>Ð>Ð>Ð>Ð>Ð>àÔð 	ØÔ'Ô:ð à#�OØ)4¨Ø(3Ø$6Ð$6°tÑ$<Ô$<Ð#=ðñ ô ðð ð $�OØ)3¨Ø(2Ø$6Ð$6°tÑ$<Ô$<Ð#=ðñ ô ðð ð ˆ	r   )r3   r4   r5   r4   )r:   r	   r   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úrequires_calibrationÚis_qat_trainableÚ__annotations__r   r   r2   r9   ÚstrÚboolrD   rJ   Úpropertyr   rN   rQ   rT   r]   Ú__classcell__)r   s   @r   r   r   "   s5  ø€ € € € € € ðð ð !ÐØÐØ(Ð(Ð(Ñ(ð8Ð,Cð 8ð 8ð 8ð 8ð 8ð 8ð.ð .ð .ð`ð ð ð ðÐ.?ð ÈSð Ð_cð ð ð ð ð
à ð
ð 
ð 
ð 
ð ðð  (Ð+<Ô"=ð ð ð ñ „Xððð ð ð%ð %ð %ð
ð ð ð ð ð ð r   r   )Útypingr   r   Úbaser   Úquantizers_utilsr   Úmodeling_utilsr	   Úutils.quantization_configr
   Úutilsr   r   r   r   r   r   r"   Ú
get_loggerr^   r+   r   rP   r   r   ú<module>rp      s  ðð +Ð *Ð *Ð *Ð *Ð *Ð *Ð *à Ð Ð Ð Ð Ð Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2ð ð :Ø0Ð0Ð0Ð0Ð0Ð0Ø9Ð9Ð9Ð9Ð9Ð9à tÐ tÐ tÐ tÐ tÐ tÐ tÐ tÐ tÐ tÐ tÐ tÐ tÐ tØ ?Ð ?Ð ?Ð ?Ð ?Ð ?ð ÐÑÔð Ø€L€L€Là	ˆÔ	˜HÑ	%Ô	%€ðBð Bð Bð Bð B˜ñ Bô Bð Bð Bð Br   