§
    ‚Štj  ã                   óº   — d dl mZ d dlmZmZ d dlmZmZ  e¦   «         r
ddlZddl	m
Z
  ej        e¦  «        Z G d„ de¦  «        Z	 	 d
dee         dz  fd	„ZdS )é   )ÚConversionOps)Úget_module_from_nameÚshould_convert_module)Úis_torch_availableÚloggingé    Nc                   óª   — e Zd Zd„ Z	 	 	 d	deeeej                 f         dej	        j
        dz  dedz  dee         dz  deeej        f         f
d„ZdS )
ÚQuantoQuantizec                 ó   — || _         d S )N)Úhf_quantizer)Úselfr   s     ú^/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/integrations/quanto.pyÚ__init__zQuantoQuantize.__init__   s   € Ø(ˆÔÐÐó    NÚ
input_dictÚmodelÚfull_layer_nameÚmissing_keysÚreturnc                 óX  — t          |                     ¦   «         ¦  «        d         \  }}|d         }ddlm}  ||||¦  «         t	          ||¦  «        \  }	}t          j        |	j        j        ¦  «        |	_        t          j        |	j	        j        ¦  «        |	_	        |	 
                    ¦   «          d|	j        _        d|	_        |                     dd¦  «        d         }
|                     |
› d�¦  «         |                     |
› d	�¦  «         |                     |
› d
�¦  «         i S )Nr   r   )Ú_load_parameter_into_modelFTú.é   z.weightz.input_scalez.output_scale)ÚtupleÚitemsÚmodeling_utilsr   r   ÚtorchÚonesÚinput_scaleÚshapeÚoutput_scaleÚfreezeÚweightÚrequires_gradÚ_is_hf_initializedÚrsplitÚdiscard)r   r   r   r   r   ÚkwargsÚ_Úvaluer   ÚmoduleÚmodule_names              r   ÚconvertzQuantoQuantize.convert   s0  € õ ˜×)Ò)Ñ+Ô+Ñ,Ô,¨QÔ/‰ˆˆ5Ø�a”ˆà?Ð?Ð?Ð?Ð?Ð?à"Ð" 5¨/¸5ÑAÔAÐAÝ(¨°Ñ@Ô@‰	ˆ�å"œZ¨Ô(:Ô(@ÑAÔAˆÔÝ#œj¨Ô)<Ô)BÑCÔCˆÔà�Š‰ŒˆØ&+ˆŒÔ#Ø$(ˆÔ!ð &×,Ò,¨S°!Ñ4Ô4°QÔ7ˆØ×Ò Ð4Ð4Ð4Ñ5Ô5Ð5Ø×Ò Ð9Ð9Ð9Ñ:Ô:Ð:Ø×Ò Ð:Ð:Ð:Ñ;Ô;Ð;Øˆ	r   )NNN)Ú__name__Ú
__module__Ú__qualname__r   ÚdictÚstrÚlistr   ÚTensorÚnnÚModuler-   © r   r   r
   r
      sª   € € € € € ð)ð )ð )ð )-Ø&*Ø)-ðð à˜˜d 5¤<Ô0Ð0Ô1ðð ŒxŒ Ñ%ðð ˜t™ð	ð
 ˜3”i $Ñ&ðð 
ˆc�5”<ÐÔ	 ðð ð ð ð ð r   r
   Úmodules_to_not_convertc                 óÎ  — ddl m}m}m}m}m}m} ||||dœ}	d||dœ}
d}|                      ¦   «         D �]\  }}t          ||¦  «        sŒt          j
        d¦  «        5  d}t          |t          j        ¦  «        rC ||j        |j        |j        du|j        j        |	|j                 |
|j                 ¬¦  «        }nWt          |t          j        j        ¦  «        r8|j        �1 ||j        |j        |j        |j        du|
|j                 ¬	¦  «        }|�d
}|                      ||¦  «         ddd¦  «         n# 1 swxY w Y   �Œ|st4                               d¦  «         | S )a½  
    Public method that recursively replaces the Linear layers of the given model with Quanto quantized layers.
    Returns the converted model and a boolean that indicates if the conversion has been successful or not.

    Args:
        model (`torch.nn.Module`):
            The model to convert, can be any `torch.nn.Module` instance.
        quantization_config (`QuantoConfig`, defaults to `None`):
            The quantization config object that contains the quantization parameters.
        modules_to_not_convert (`list`, *optional*, defaults to `None`):
            A list of modules to not convert. If a module name is in the list (e.g. `lm_head`), it will not be
            converted.
    r   )Ú
QLayerNormÚQLinearÚqfloat8Úqint2Úqint4Úqint8)Úfloat8Úint8Úint4Úint2N)Nr@   rA   FÚmeta)Úin_featuresÚout_featuresÚbiasÚdtypeÚweightsÚactivations)rJ   Tz½You are loading your model using quanto but no linear modules were found in your model. Please double check your model architecture, or submit an issue on github if you think this is a bug.)Úoptimum.quantor:   r;   r<   r=   r>   r?   Únamed_modulesr   r   ÚdeviceÚ
isinstancer5   ÚLinearrE   rF   rG   r#   rH   rI   rJ   Ú	LayerNormÚnormalized_shapeÚepsÚelementwise_affineÚset_submoduleÚloggerÚwarning)r   Úquantization_configr8   r:   r;   r<   r=   r>   r?   Ú	w_mappingÚ	a_mappingÚhas_been_replacedr,   r+   Ú
new_modules                  r   Úreplace_with_quanto_layersr\   >   s  € ð$ QÐPÐPÐPÐPÐPÐPÐPÐPÐPÐPÐPÐPÐPÐPÐPà"¨E¸5È%ÐPÐP€IØ w¸Ð>Ð>€IàÐØ$×2Ò2Ñ4Ô4ð =ñ =Ñˆ�VÝ$ [Ð2HÑIÔIð 	ØÝŒ\˜&Ñ!Ô!ð 	=ð 	=ØˆJÝ˜&¥"¤)Ñ,Ô,ð Ø$˜WØ &Ô 2Ø!'Ô!4Øœ¨DÐ0Ø œ-Ô-Ø%Ð&9Ô&AÔBØ )Ð*=Ô*IÔ Jðñ ô �
�
õ ˜F¥E¤HÔ$6Ñ7Ô7ð Ð<OÔ<[Ð<gØ'˜ZØÔ+Ø”JØÔ-Ø”K tÐ+Ø )Ð*=Ô*IÔ Jðñ ô �
ð Ð%Ø$(Ð!Ø×#Ò# K°Ñ<Ô<Ð<ð+	=ð 	=ð 	=ñ 	=ô 	=ð 	=ð 	=ð 	=ð 	=ð 	=ð 	=øøøð 	=ð 	=ð 	=ð 	=ùð. ð 
Ý�Šðñ	
ô 	
ð 	
ð €Ls   ÁCD<Ä<E 	ÅE 	)NN)Úcore_model_loadingr   Úquantizers.quantizers_utilsr   r   Úutilsr   r   r   Útorch.nnr5   Ú
get_loggerr.   rU   r
   r3   r2   r\   r7   r   r   ú<module>rb      sò   ðð /Ð .Ð .Ð .Ð .Ð .Ø UÐ UÐ UÐ UÐ UÐ UÐ UÐ UØ /Ð /Ð /Ð /Ð /Ð /Ð /Ð /ð ÐÑÔð Ø€L€L€LØÐÐÐÐÐà	ˆÔ	˜HÑ	%Ô	%€ð ð  ð  ð  ð  �]ñ  ô  ð  ðJ Ø/3ð9ð 9ð ! œI¨Ñ,ð9ð 9ð 9ð 9ð 9ð 9r   