§
    ‚Štj{1 ã                   óT  — U d dl Z d dlZd dlZd dlZd dlZd dlZd dlZd dlZd dlZd dl	m
Z
 d dl mZ d dlmZmZ d dlmZ d dlmZmZ d dlmZmZ d dlmZ d d	lmZ d d
lmZmZmZmZmZ d dl m!Z! d dl"Z"d dl#m$Z$m%Z% d dl&m'Z' d dl(m)Z) d dl*m+Z, d dl*m-Z. d dl"m/Z/m0Z0 d dl1m2Z2 d dl3m4Z4 ddl5m6Z7 ddl8m9Z9 ddl:m;Z; ddl<m=Z=m>Z>m?Z?m@Z@ ddlAmBZB ddlCmDZD ddlEmFZFmGZG ddlHmIZImJZJmKZKmLZLmMZM ddlNmOZOmPZPmQZQmRZRmSZSmTZTmUZU ddlVmWZW ddlXmYZY dd lZm[Z[ dd!l\m]Z] dd"l^m_Z_ dd#l`maZa dd$lbmcZcmdZdmeZe dd%lfmgZg dd&lhmiZi dd'ljmkZk dd(llmmZm dd)lnmoZompZpmqZqmrZrmsZsmtZtmuZu dd*lvmwZw dd+lxmyZymzZzm{Z{m|Z| dd,l}m~Z~ dd-lm€Z€m�Z� dd.l‚mƒZƒ dd/l„m…Z… dd0l†m‡Z‡ dd1lˆm‰Z‰ dd2lŠm‹Z‹ dd3lŒm�Z�mŽZŽm�Z�m�Z�m‘Z‘m’Z’m“Z“m”Z”m•Z•m–Z–m—Z—m˜Z˜m™Z™mšZšm›Z›mœZœm�Z�mžZžmŸZŸm Z m¡Z¡ dd4l¢m£Z£m¤Z¤m¥Z¥ dd5l¦m§Z§m¨Z¨m©Z©mªZª dd6l«m¬Z¬m­Z­m®Z®m¯Z¯m°Z°m±Z±m²Z² dd7l³m´Z´mµZµ dd8l¶m·Z·m¸Z¸ dd9l¹mºZº  eš¦   «         rd d:l»m¼Z¼ d d;l½m¾Z¾ erd d<l¿mÀZÀ dd=lÁmÂZÂ e"jA         Ã                    ¦   «         ZÄ e°¦   «         r2d dlÅmÆc m"ZÇ d d>lÈmÉZÊ  e'jË        eÊ¦  «         e'jË        d?¦  «        k    ZÌnd@ZÌ e¡jÍ        eÎ¦  «        ZÏejÐ         Ñ                    dAdB¦  «         Ò                    ¦   «         ZÓejÐ         Ñ                    dCdB¦  «         Ò                    ¦   «         ZÔ edDdE¬F¦  «        ZÕd@aÖd@a× edG¬H¦  «         G dI„ dJ¦  «        ¦   «         ZØdKeÙfdL„ZÚdKeÛfdM„ZÜdN„ ZÝedO„ ¦   «         ZÞedP„ ¦   «         Zßed”dQe"jà        dReádz  fdS„¦   «         ZâdT„ ZãdU„ Zäe"jÙ        e"jå        e"jæ        e"jç        e"jè        e"jé        e"jê        e"jë        e"jì        e"jí        e"jî        e"jï        e"jð        e"jñ        e"jò        dVœZódWdXdKeÙfdY„Zô	 	 	 d•d[eáejõ        z  d\eáe"jö        z  d]eÙd^eÙdz  dKe÷eáe"j/        f         f
d_„Zød`e"j/        dKeÛfda„Zùdbe0jú        dKeûeá         fdc„Züddeûeýeá                  dee÷eáe"j/        f         dKeþeûeýeá                  eûeá         f         fdf„Zÿddeûeýeá                  dee÷eáe"j/        f         dKeþeûeýeá                  eûeýeá                  f         fdg„�Z dee÷eáe"j/        f         dhdEdKe÷eáe"j/        f         fdi„�ZdhdEdjeád`e"j/        fdk„�Zd”dleádmeádz  dKeáfdn„�Z	 	 	 d–doeáejõ        z  dz  dmeádz  dpeádz  dqeÙdz  dre÷dz  dseÙdteádz  due§dz  dv�edz  dKeþeûeá         dz  e÷dz  f         fdw„�Z	 d”dQeáe"jà        z  e÷z  dz  dxeûeá         dz  dye9dze÷dz  dee÷dz  d]eÙd{e…dz  dKeþe9e"jà        f         fd|„�Z G d}„ d~¦  «        �Z G d„ d€¦  «        �Z G d�„ dEe0jú        �e�ee•eI¦  «        �Z	 e˜�e	�j
        ¦  «        �e	�_
        �e	�j
        �j        �3�e	�j
        �j        �                     dhd‚dƒ¬„¦  «        �e	�j
        �_        ed—dh�e	d…eÙdK�e	fd†„¦   «         �Zed—dhe0jú        d…eÙdKe0jú        fd‡„¦   «         �Zd—dhe0jú        d…eÙdKe0jú        fdˆ„�Zd‰eáeÛz  e"jö        z  dKeÙfdŠ„�Z	 d”dh�e	d‹e÷d{e…dz  fdŒ„�Zdh�e	d�e÷d{e…dz  fdŽ„�Z G d�„ d�e£¦  «        �Z �e¦   «         �Z�e�ed‘<    G d’„ d“�e	¦  «        �ZdS )˜é    N)Úabstractmethod)Údefaultdict)ÚCallableÚIterator)Úcontextmanager)Ú	dataclassÚfield)ÚpartialÚwraps)Úcycle)ÚThread)ÚTYPE_CHECKINGÚAnyÚTypeVarÚget_type_hintsÚoverload)Ú
is_zipfile)Úis_offline_modeÚ"split_torch_state_dict_into_shards)Úversion)Ú	safe_open)Úload)Ú	save_file)ÚTensorÚnn)Úconstraints)Ú
checkpointé   )Úinitialization)ÚPreTrainedConfig)Úget_model_conversion_mapping)ÚWeightConverterÚWeightRenamingÚ$convert_and_load_state_dict_in_modelÚrevert_weight_conversion)ÚDistributedConfig)Úcustom_object_save)ÚCompileConfigÚGenerationConfig)ÚPeftAdapterMixinÚdeepspeed_configÚhub_kernelsÚis_deepspeed_zero3_enabledÚis_fsdp_enabled)Ú_get_device_mapÚaccelerate_disk_offloadÚaccelerate_dispatchÚcheck_and_set_device_mapÚexpand_device_mapÚ
get_deviceÚload_offloaded_parameter)Ú!_load_state_dict_into_zero3_model)Úeager_paged_attention_forward)ÚALL_FP8_EXPERTS_FUNCTIONS)Úflash_attention_forward)Úpaged_attention_forward)Úflex_attention_forward)Úallow_all_hub_kernelsÚ	is_kernelÚ	kernelize)ÚALL_EXPERTS_FUNCTIONS)Úmaybe_load_adapters)Úsdpa_attention_forward)Úsdpa_attention_paged_forward)ÚALL_PARALLEL_STYLESÚ_get_parameter_tp_planÚdistribute_modelÚgather_state_dict_for_saveÚinitialize_tensor_parallelismÚshard_and_distribute_moduleÚverify_tp_plan)ÚLOSS_MAPPING)Ú$FLASH_ATTENTION_COMPATIBILITY_MATRIXÚFLASH_ATTN_KERNEL_FALLBACKÚlazy_import_flash_attentionÚ!lazy_import_paged_flash_attention)ÚROPE_INIT_FUNCTIONS)Úapply_patchesÚpatch_output_recorders)Úid_tensor_storage)ÚHfQuantizer)Úget_hf_quantizer)Úget_module_from_name)Úauto_conversion)ÚADAPTER_SAFE_WEIGHTS_NAMEÚDUMMY_INPUTSÚSAFE_WEIGHTS_INDEX_NAMEÚSAFE_WEIGHTS_NAMEÚWEIGHTS_INDEX_NAMEÚWEIGHTS_NAMEÚContextManagersÚKernelConfigÚPushToHubMixinÚcached_fileÚcheck_torch_load_is_safeÚ	copy_funcÚhas_fileÚis_accelerate_availableÚis_bitsandbytes_availableÚis_env_variable_trueÚis_kernels_availableÚis_torch_flex_attn_availableÚis_torch_npu_availableÚis_torch_xpu_availableÚlogging)ÚGeneralInterfaceÚis_flash_attention_requestedÚsplit_attention_implementation)ÚDownloadKwargsÚcreate_and_tag_model_cardÚget_checkpoint_shard_filesÚhf_api)ÚKERNELS_MAX_VERSIONÚKERNELS_MIN_VERSIONÚis_flash_attn_greater_or_equalÚ#is_huggingface_hub_greater_or_equalÚis_sagemaker_mp_enabledÚis_torch_cuda_availableÚ
is_tracing)ÚLoadStateDictInfoÚlog_state_dict_report)Ú_CAN_RECORD_REGISTRYÚOutputRecorder)ÚQuantizationMethod)Úadd_hook_to_module)Úextract_model_from_parallel)ÚMode)ÚDeviceMeshLike)Ú__version__z1.10FÚXLA_USE_BF16Ú0ÚXLA_DOWNCAST_BF16ÚSpecificPreTrainedModelTypeÚPreTrainedModel)ÚboundT)Úfrozenc                   ó‚  — e Zd ZU dZdZedz  ed<    ee¬¦  «        Z	edz  ed<   dZ
edz  ed<   dZeed<   dZedz  ed	<   dZedz  ed
<   dZedz  ed<   dZeed<   dZej        dz  ed<    ee¬¦  «        Zeed<   dZedz  ed<   dZded<   dZeed<   dZeeez           dz  ed<   dZedz  ed<   edefd„¦   «         ZdS )ÚLoadStateDictConfigze
    Config for loading weights. This allows bundling arguments that are just
    passed around.
    NÚpretrained_model_name_or_path)Údefault_factoryÚdownload_kwargsÚuse_safetensorsFÚignore_mismatched_sizesÚsharded_metadataÚ
device_mapÚdisk_offload_folderÚoffload_buffersÚdtypeÚ
dtype_planÚhf_quantizerúDeviceMeshLike | NoneÚdevice_meshTÚweights_onlyÚweight_mappingÚdisable_mmapÚreturnc                 ó   — | j         d uS ©N)r˜   ©Úselfs    úY/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/modeling_utils.pyÚis_quantizedz LoadStateDictConfig.is_quantizedÁ   s   € àÔ ¨Ð,Ð,ó    ) Ú__name__Ú
__module__Ú__qualname__Ú__doc__r�   ÚstrÚ__annotations__r	   ro   r�   r�   Úboolr‘   r’   Údictr“   r”   r•   r–   Útorchr—   r˜   rS   rš   r›   rœ   Úlistr"   r#   r�   Úpropertyr¤   © r¥   r£   rŒ   rŒ   ª   sŽ  € € € € € € ðð ð
 15Ð! 3¨¡:Ð4Ð4Ñ4Ø-2¨UÀ>Ð-RÑ-RÔ-R€O�^ dÑ*ÐRÐRÑRØ#'€O�T˜D‘[Ð'Ð'Ñ'Ø$)Ð˜TÐ)Ð)Ñ)Ø$(Ð�d˜T‘kÐ(Ð(Ñ(Ø"€J��t‘Ð"Ð"Ñ"Ø&*Ð˜˜t™Ð*Ð*Ñ*Ø!€O�TÐ!Ð!Ñ!Ø $€Eˆ5Œ;˜ÑÐ$Ð$Ñ$Ø�u¨TÐ2Ñ2Ô2€J�Ð2Ð2Ñ2Ø'+€L�+ Ñ$Ð+Ð+Ñ+Ø+/€KÐ(Ð/Ð/Ñ/Ø€L�$ÐÐÑØDH€N�D˜¨>Ñ9Ô:¸TÑAÐHÐHÑHØ $€L�$˜‘+Ð$Ð$Ñ$àð-˜dð -ð -ð -ñ „Xð-ð -ð -r¥   rŒ   rž   c                  ó€   — t           o7t          t          j        d¦  «        ot          j                             ¦   «         S )NÚis_initialized)Ú_torch_distributed_availableÚhasattrr®   Údistributedr³   r±   r¥   r£   Ú!_is_torch_distributed_initializedr·   Æ   s7   € å$ð 	/Ý•EÔ%Ð'7Ñ8Ô8ð	/åÔ×,Ò,Ñ.Ô.ðr¥   c                  ó’   — t          ¦   «         rt          t          j        d¦  «        sdS t          j                             ¦   «         S )NÚget_world_sizer   )r·   rµ   r®   r¶   r¹   r±   r¥   r£   Ú!_get_torch_distributed_world_sizerº   Î   s?   € Ý,Ñ.Ô.ð µg½eÔ>OÐQaÑ6bÔ6bð ØˆqÝÔ×+Ò+Ñ-Ô-Ð-r¥   c                  ó€   — t          ¦   «         o0t          t          j                             dd¦  «        ¦  «        dk    S )NÚ
LOCAL_RANKz-1r   )r·   ÚintÚosÚenvironÚgetr±   r¥   r£   Úis_local_dist_rank_0rÁ   Ô   s2   € Ý,Ñ.Ô.Ð_µ3µr´z·~²~ÀlÐTXÑ7YÔ7YÑ3ZÔ3ZÐ^_Ò3_Ð_r¥   c               #   ó*   K  — da 	 d V — da d S # da w xY w©NTF)Ú_is_quantizedr±   r¥   r£   Úset_quantized_staterÅ   Ø   s5   è è € ð €MðØˆˆˆàˆˆˆø˜ˆÐÐÐÐó   † Žc               #   ó*   K  — da 	 d V — da d S # da w xY wrÃ   )Ú_is_ds_init_calledr±   r¥   r£   Úset_zero3_staterÉ   å   s:   è è € ð Ðð#Øˆˆˆà"ÐÐÐø˜UÐÐ"Ð"Ð"Ð"rÆ   r–   Úmodel_class_namec              #   ó
  K  — | j         s |�	|› d| › d�}nd| › d�}t          |¦  «        ‚t          j        ¦   «         }	 t          j        | ¦  «         dV — t          j        |¦  «         dS # t          j        |¦  «         w xY w)zà
    Locally change the torch default dtype to `dtype`, and restore the old one upon exiting the context.
    If `model_class_name` is provided, it's used to provide a more helpful error message if `dtype` is not valid.
    Nz% cannot be instantiated under `dtype=z$` as it's not a floating-point dtypezCannot set `z7` as torch's default as it's not a floating-point dtype)Úis_floating_pointÚ
ValueErrorr®   Úget_default_dtypeÚset_default_dtype)r–   rÊ   Úerror_messageÚoriginal_dtypes       r£   Úlocal_torch_dtyperÒ   ï   s°   è è € ð Ô"ð (ØÐ'à#ÐuÐuÈ%ÐuÐuÐuð ˆMð j¨5ÐiÐiÐiˆMÝ˜Ñ'Ô'Ð'åÔ,Ñ.Ô.€Nð0ÝÔ Ñ&Ô&Ð&ØˆˆˆåÔ Ñ/Ô/Ð/Ð/Ð/ø�Ô Ñ/Ô/Ð/Ð/øøøs   ¾A, Á,Bc                  ó¢   — t          j        g ¦  «        j        } t          j        ¦   «         }| |k    r|t          j        d¦  «        k    r|S dS | S )zì
    Test if a device context manager is currently in use, or if it is not the case, check if the default device
    is not "cpu". This is used to infer the correct device to load the model on, in case `device_map` is not provided.
    ÚcpuN)r®   ÚtensorÚdeviceÚget_default_device)Údevice_in_contextÚdefault_devices     r£   Ú*get_torch_context_manager_or_global_devicerÚ     sV   € õ
 œ RÑ(Ô(Ô/ÐÝÔ-Ñ/Ô/€Nà˜NÒ*Ð*Ø�Uœ\¨%Ñ0Ô0Ò0Ð0Ø!Ð!ØˆtØÐr¥   c                 óf  — |                       ¦   «         D ]K}|                     ¦   «         r5dt          |j        ¦  «        vrdt          |j        ¦  «        vr	|j        c S ŒLt	          | ¦  «        dk    rt
          j        S t          t          |                       ¦   «         ¦  «        ¦  «        j        S )zt
    Returns the first found floating dtype in `state_dict` if there is one, otherwise returns the first dtype.
    Úfloat8_Úfloat4_r   )	ÚvaluesrÌ   rª   r–   Úlenr®   Úfloat32ÚnextÚiter)Ú
state_dictÚts     r£   Úget_state_dict_dtyperå     s¨   € ð ×ÒÑ Ô ð ð ˆà×ÒÑ Ô ð 	 Yµc¸!¼'±l´lÐ%BÐ%BÀyÕX[Ð\]Ô\cÑXdÔXdÐGdÐGdØ”7ˆNˆNˆNøõ ˆ:�„˜!ÒÐÝŒ}ÐÝ•�Z×&Ò&Ñ(Ô(Ñ)Ô)Ñ*Ô*Ô0Ð0r¥   )ÚBOOLÚU8ÚI8ÚI16ÚU16ÚF16ÚBF16ÚI32ÚU32ÚF32ÚF64ÚI64ÚU64ÚF8_E4M3ÚF8_E5M2Úpathzstr | os.PathLikec                 óþ  — t           j                             d¦  «        sdS 	 t          j                             t          j        | ¦  «        ¦  «        }t          dd¬¦  «        5 }t          d„ d„ |D ¦   «         D ¦   «         d„ d	¬
¦  «        }ddd¦  «         n# 1 swxY w Y   |D ]>\  }}||k    s+|                     | 	                    d¦  «        dz   ¦  «        r|dk    c S Œ?n# t          t          f$ r Y nw xY wdS )a  True if `path` lives on an hf-mount FUSE filesystem (device string 'hf-mount').

    hf-mount's mmap + readahead interaction deadlocks under parallel page-faults,
    so callers should load the file into memory instead. Linux-only; returns False
    on other platforms.
    ÚlinuxFz/proc/mountsúutf-8©Úencodingc              3   ó\   K  — | ]'}t          |¦  «        d k    ¯|d         |d         fV — Œ(dS )é   r   r   N©rß   )Ú.0Úps     r£   ú	<genexpr>z"_is_on_hf_mount.<locals>.<genexpr>E  s8   è è € ÐNÐN !Å#ÀaÁ&Ä&ÈAÂ+À+�!�A”$˜˜!œ�À+À+À+À+ÐNÐNr¥   c              3   ó>   K  — | ]}|                      ¦   «         V — Œd S r    )Úsplit)rþ   Úls     r£   r   z"_is_on_hf_mount.<locals>.<genexpr>E  s*   è è € Ð'>Ð'>°a¨¯ª©	¬	Ð'>Ð'>Ð'>Ð'>Ð'>Ð'>r¥   c                 ó,   — t          | d         ¦  «        S )Nr   rý   )Úes    r£   ú<lambda>z!_is_on_hf_mount.<locals>.<lambda>F  s   € �c ! A¤$™iœi€ r¥   T)ÚkeyÚreverseNú/zhf-mount)ÚsysÚplatformÚ
startswithr¾   rõ   ÚrealpathÚfspathÚopenÚsortedÚrstripÚOSErrorrÍ   )rõ   ÚrealÚfhÚentriesÚdevÚmps         r£   Ú_is_on_hf_mountr  8  si  € õ Œ<×"Ò" 7Ñ+Ô+ð ØˆuðÝŒw×Ò¥¤	¨$¡¤Ñ0Ô0ˆÝ�.¨7Ð3Ñ3Ô3ð 	°rÝØNÐNÐ'>Ð'>¸2Ð'>Ñ'>Ô'>ÐNÑNÔNØ'Ð'Øðñ ô ˆGð	ð 	ð 	ñ 	ô 	ð 	ð 	ð 	ð 	ð 	ð 	øøøð 	ð 	ð 	ð 	ð ð 	)ð 	)‰GˆC�Ø�rŠzˆz˜TŸ_š_¨R¯YªY°s©^¬^¸cÑ-AÑBÔBˆzØ˜jÒ(Ð(Ð(Ð(ð ð	)øõ •ZÐ ð ð ð Øˆðøøøàˆ5s=   £AC& Á%(BÂC& ÂBÂC& Â BÂ!AC& Ã$C& Ã&C:Ã9C:rÔ   Úcheckpoint_fileÚmap_locationr›   r�   c                 óÔ  ‡— t          j        | ¦  «        }|€t          |¦  «        }|                     d¦  «        �rm|rw‰dk    rqt	          |d¦  «        5 }t          |                     ¦   «         ¦  «        }ddd¦  «         n# 1 swxY w Y   ‰dk    r ˆfd„|                     ¦   «         D ¦   «         }|S t          |d¬¦  «        5 }i }| 	                    ¦   «         D ]²}‰dk    r| 
                    |¦  «        }	|	                     ¦   «         }
|
t          v rt          |
         }nt          d	|
› �¦  «        ‚t          j        |	                     ¦   «         |d¬
¦  «        ||<   Œ‡|                     |¦  «                             ‰¦  «        ||<   Œ³|cddd¦  «         S # 1 swxY w Y   |rt'          ¦   «          i }‰dk    rt)          |¦  «        rddi}t          j        |f‰|dœ|¤ŽS )aW  
    Reads a `safetensor` or a `.bin` checkpoint file. We load the checkpoint on "cpu" by default.

    When `disable_mmap` is True, safetensors files are read fully into memory instead of
    being memory-mapped. When `disable_mmap` is None (default), it is auto-detected to True
    on hf-mount FUSE filesystems (see `_is_on_hf_mount`).
    Nú.safetensorsÚmetaÚrbrÔ   c                 óB   •— i | ]\  }}||                      ‰¦  «        “ŒS r±   )Úto)rþ   ÚkÚvr  s      €r£   ú
<dictcomp>z#load_state_dict.<locals>.<dictcomp>g  s+   ø€ ÐSÐSÐS¹¸¸1˜a §¢ lÑ!3Ô!3ÐSÐSÐSr¥   Úpt)Ú	frameworkz)Cannot load safetensors of unknown dtype )Úsizer–   rÖ   ÚmmapT©r  r›   )r¾   r  r  Úendswithr  Ú_safe_load_bytesÚreadÚitemsr   ÚkeysÚ	get_sliceÚ	get_dtypeÚstr_to_torch_dtyperÍ   r®   ÚemptyÚ	get_shapeÚ
get_tensorr   ra   r   r   )r  r  r›   r�   Úcheckpoint_pathÚ_fhrã   Úfr!  Ú_sliceÚk_dtyper–   Ú
extra_argss    `           r£   Úload_state_dictr:  Q  sŠ  ø€ õ ”i Ñ0Ô0€OØÐÝ& Ñ7Ô7ˆà×Ò Ñ/Ô/ñ Øð 	˜L¨FÒ2Ð2Ý�o tÑ,Ô,ð :°Ý-¨c¯hªh©j¬jÑ9Ô9�
ð:ð :ð :ñ :ô :ð :ð :ð :ð :ð :ð :øøøð :ð :ð :ð :à˜uÒ$Ð$ØSÐSÐSÐSÀ
×@PÒ@PÑ@RÔ@RÐSÑSÔS�
ØÐÝ�°$Ð7Ñ7Ô7ð 	¸1ØˆJØ—V’V‘X”Xð 
Eð 
E�Ø 6Ò)Ð)ØŸ[š[¨™^œ^�FØ$×.Ò.Ñ0Ô0�GØÕ"4Ð4Ð4Ý 2°7Ô ;˜˜å(Ð)^ÐU\Ð)^Ð)^Ñ_Ô_Ð_Ý$)¤K°V×5EÒ5EÑ5GÔ5GÈuÐ]cÐ$dÑ$dÔ$d�J˜q‘M�Mà$%§L¢L°¡O¤O×$6Ò$6°|Ñ$DÔ$D�J˜q‘M�MØð	ð 	ð 	ð 	ñ 	ô 	ð 	ð 	ð 	ð 	ð 	ð 	øøøð 	ð 	ð 	ð 	ð  ð #Ý Ñ"Ô"Ð"Ø€Jà�vÒÐ¥*¨_Ñ"=Ô"=ÐØ˜d�^ˆ
åŒ:�oÐj°LÈ|ÐjÐjÐ_iÐjÐjÐjs%   Á"BÂBÂ
BÃCFÆF#Æ&F#rÕ   c                 óÜ   — |                       ¦   «         rC|                      d¦  «        d                              ¦   «         |                      ¦   «         z   }n|                      ¦   «         }|S )Néÿÿÿÿ)ÚnelementÚviewÚdata_ptrÚelement_size)rÕ   Ústops     r£   Ú_end_ptrrB  ƒ  s[   € à‡‚ÑÔð !Ø�{Š{˜2‰Œ˜rÔ"×+Ò+Ñ-Ô-°×0CÒ0CÑ0EÔ0EÑEˆˆà�ŠÑ Ô ˆØ€Kr¥   Úmodulec                 óÌ   ‡— g }|                       ¦   «         D ]K\  Š}t          |di ¦  «        pi }|                     ˆfd„|                     ¦   «         D ¦   «         ¦  «         ŒL|S )NÚ_tied_weights_keysc                 ó&   •— g | ]}‰r‰› d |› �n|‘ŒS ©ú.r±   ©rþ   r!  Únames     €r£   ú
<listcomp>z)_get_tied_weight_keys.<locals>.<listcomp>�  s,   ø€ Ð SÐ SÐ SÀ!°$Ð!= D  ¨1   ¸AÐ SÐ SÐ Sr¥   )Únamed_modulesÚgetattrÚextendr-  )rC  Útied_weight_keysÚ	submoduleÚtiedrJ  s       @r£   Ú_get_tied_weight_keysrR  Œ  s{   ø€ Ø"$ÐØ!×/Ò/Ñ1Ô1ð Uð U‰ˆˆiÝ�yÐ"6¸Ñ;Ô;ÐA¸rˆØ×ÒÐ SÐ SÐ SÐ SÀtÇyÂyÁ{Ä{Ð SÑ SÔ SÑTÔTÐTÐTØÐr¥   Útensorsrã   c                 óª  — g }| D ]ò}t          |¦  «        dk     r|                     |¦  «         Œ+g }|D ]A}||         }|                     |                     ¦   «         t          |¦  «        |f¦  «         ŒB|                     ¦   «          |d         \  }}}	|                     |	h¦  «         |dd …         D ]@\  }
}}|
|k    r|                     |h¦  «         n|d                              |¦  «         |}ŒAŒóg }g }|D ]R} t          | ¦  «        dk    r(|                     |                      ¦   «         ¦  «         Œ=|                     | ¦  «         ŒS||fS )Nrü   r   r   r<  )rß   Úappendr?  rB  ÚsortÚaddÚpop)rS  rã   Úfiltered_tensorsÚsharedÚareasrJ  rÕ   Ú_Ú	last_stopÚ	last_nameÚstartrA  Údisjoint_tensorsÚshared_tensorss                 r£   Ú_find_disjointrb  ”  s”  € ØÐØð ð ˆÝˆv‰;Œ;˜Š?ˆ?Ø×#Ò# FÑ+Ô+Ð+ØàˆØð 	Fð 	FˆDØ Ô%ˆFØ�LŠL˜&Ÿ/š/Ñ+Ô+­X°fÑ-=Ô-=¸tÐDÑEÔEÐEÐEØ�
Š
‰Œˆà"'¨¤(Ñˆˆ9�iØ×Ò  Ñ,Ô,Ð,Ø!& q r r¤ð 	ð 	ÑˆE�4˜Ø˜	Ò!Ð!Ø ×'Ò'¨¨Ñ/Ô/Ð/Ð/à  Ô$×(Ò(¨Ñ.Ô.Ð.ØˆIˆIð	ð ÐØ€NØ#ð +ð +ˆÝˆw‰<Œ<˜1ÒÐØ×#Ò# G§K¢K¡M¤MÑ2Ô2Ð2Ð2à×!Ò! 'Ñ*Ô*Ð*Ð*ØÐ+Ð+Ð+r¥   c                 ó”  — g }g }| D ]¾}t          |¦  «        dk     rŒt          j        t          ¦  «        }|D ]N}||         }|j        |                     ¦   «         t          |¦  «        f}||                              |¦  «         ŒOt          |¦  «        dk    r|                     |¦  «         Œ©|                     |¦  «         Œ¿||fS )Nrü   r   )	rß   Úcollectionsr   ÚsetrÖ   r?  rB  rW  rU  )	rS  rã   ra  Ú	identicalrZ  r[  rJ  rÕ   Úareas	            r£   Ú_find_identicalrh  ³  sÙ   € ð €NØ "€IØð *ð *ˆÝˆv‰;Œ;˜Š?ˆ?ØåÔ'­Ñ,Ô,ˆØð 	"ð 	"ˆDØ Ô%ˆFØ”M 6§?¢?Ñ#4Ô#4µh¸vÑ6FÔ6FÐGˆDØ�$ŒK�OŠO˜DÑ!Ô!Ð!Ð!Ýˆu‰:Œ:˜Š?ˆ?Ø×Ò˜VÑ$Ô$Ð$Ð$à×!Ò! &Ñ)Ô)Ð)Ð)Ø˜9Ð$Ð$r¥   Úmodelc                 ó\  ‡— t          j        t          ¦  «        }|                      ¦   «         D ]¾\  Š}t	          |t
          j        ¦  «        s)|t          |¦  «                                      ‰¦  «         ŒH|j	        j
        dk    r>|                     ‰¦  «        }|t          |¦  «                                      ‰¦  «         Œ–|t          |¦  «                                      ‰¦  «         Œ¿d„ |                     ¦   «         D ¦   «         }t          t          |¦  «        ¦  «        }g }t          ¦   «         }|�y|                     ¦   «         D ]d}d}	t!          |¦  «        D ]PŠt#          ˆfd„|D ¦   «         ¦  «        }
|
r1‰| v r-|	dz  }	|	t%          |¦  «        k     r|                     ‰¦  «         ŒQŒet)          |                     ¦   «         | ¦  «        \  }}|D ]Š| ‰                              ¦   «         | ‰<   Œ t-          || ¦  «        \  }}|D ]\}|                     |¦  «        }|D ]Š| ‰= Œ|                     |¦  «        }t%          |¦  «        dk    r|                     |¦  «         Œ]|r|                     |¦  «         t%          |¦  «        dk    rt5          d|› d|› d	�¦  «        ‚| S )
aX  
    Remove all tied weights from the given `state_dict`, making sure to keep only the main weight that `model`
    will expect when reloading (even if we now tie weights symmetrically, it's better to keep the intended one).
    This is because `safetensors` does not allow tensor aliasing - so we're going to remove aliases before saving.
    r  c                 ó@   — i | ]\  }}t          |¦  «        d k    ¯||“ŒS )r   rý   )rþ   ÚptrÚnamess      r£   r#  z7remove_tied_weights_from_state_dict.<locals>.<dictcomp>ä  s)   € ÐOÐOÐO¡* # uÅÀEÁ
Ä
ÈQÂÀ�3˜ÀÀÀr¥   Nr   c              3   óB   •K  — | ]}t          j        |‰¦  «        V — Œd S r    ©ÚreÚsearch)rþ   ÚpatrJ  s     €r£   r   z6remove_tied_weights_from_state_dict.<locals>.<genexpr>ð  s/   øè è € Ð%fÐ%f¸s¥b¤i°°TÑ&:Ô&:Ð%fÐ%fÐ%fÐ%fÐ%fÐ%fr¥   r   z8The weights trying to be saved contained shared tensors z\ which are not properly defined. We found all the potential target tied weights keys to be: zo.
This can also just mean that the module's tied weight keys are wrong vs the actual tied weights in the model.)rd  r   r¯   r,  Ú
isinstancer®   r   ÚidrU  rÖ   ÚtypeÚget_parameter_or_bufferrR   re  rR  rÞ   r  Úanyrß   rW  rb  Úclonerh  ÚintersectionÚ
differencerN  ÚRuntimeError)rã   ri  ÚptrsrÕ   Úshared_ptrsÚall_potential_tied_weights_keysÚerror_namesÚto_delete_namesrm  ÚfoundÚmatches_patternÚshared_namesÚdisjoint_namesÚidentical_namesÚinamesÚknownÚunknownrJ  s                    @r£   Ú#remove_tied_weights_from_state_dictr‰  È  s  ø€ õ Ô"¥4Ñ(Ô(€DØ"×(Ò(Ñ*Ô*ð 9ð 9‰ˆˆfÝ˜&¥%¤,Ñ/Ô/ð 	9ð •�F‘”Ô×#Ò# DÑ)Ô)Ð)Ð)àŒ]Ô 6Ò)Ð)ð ×2Ò2°4Ñ8Ô8ˆFØ•�F‘”Ô×#Ò# DÑ)Ô)Ð)Ð)ð Õ" 6Ñ*Ô*Ô+×2Ò2°4Ñ8Ô8Ð8Ð8àOÐO°·
²
±´ÐOÑOÔO€Kõ '*Õ*?ÀÑ*FÔ*FÑ&GÔ&GÐ#Ø€KÝ‘e”e€Oð 'Ð2Ø ×'Ò'Ñ)Ô)ð 	2ð 	2ˆEØˆEÝ˜u™œð 2ð 2�Ý"%Ð%fÐ%fÐ%fÐ%fÐFeÐ%fÑ%fÔ%fÑ"fÔ"f�Ø"ð 2 t¨zÐ'9Ð'9Ø˜Q‘J�EØ�s 5™zœzÒ)Ð)Ø'×+Ò+¨DÑ1Ô1Ð1øð2õ $2°+×2DÒ2DÑ2FÔ2FÈ
Ñ#SÔ#SÑ €L�.ð ð 4ð 4ˆØ% dÔ+×1Ò1Ñ3Ô3ˆ
�4ÑÐõ %4°LÀ*Ñ$MÔ$MÑ!€L�/à!ð (ð (ˆØ×#Ò# OÑ4Ô4ˆØð 	!ð 	!ˆDØ˜4Ð Ð Ø×#Ò# OÑ4Ô4ˆÝˆw‰<Œ<˜!ÒÐØ×Ò˜wÑ'Ô'Ð'øàð )Ø×Ò˜<Ñ(Ô(Ð(å
ˆ;ÑÔ˜!ÒÐÝð|À{ð |ð |ØJið|ð |ð |ñ
ô 
ð 	
ð Ðr¥   Ú
param_namec                 óä   — t          | |¦  «        \  }}||j        v rBt          |t          j        ¦  «        s(t          j        ||                     ¦   «         ¬¦  «        }t          |||¦  «         dS )zUCast a single parameter or buffer `param_name` into the `model`, with value `tensor`.)Úrequires_gradN)rU   Ú_parametersrs  r   Ú	ParameterrÌ   Úsetattr)ri  rŠ  rÕ   ÚparentÚ
param_types        r£   Ú_load_parameter_into_modelr’    so   € å-¨e°ZÑ@Ô@Ñ€FˆJØ�VÔ'Ð'Ð'µ
¸6Å2Ä<Ñ0PÔ0PÐ'Ý”˜f°F×4LÒ4LÑ4NÔ4NÐOÑOÔOˆõ ˆF�J Ñ'Ô'Ð'Ð'Ð'r¥   Úweights_nameÚvariantc                 óP   — |�#|                       dd¦  «        \  }}|› d|› d|› �} | S )NrH  r   )Úrsplit)r“  r”  rõ   rJ  s       r£   Ú_add_variantr—  "  sB   € ØÐØ!×(Ò(¨¨aÑ0Ô0‰
ˆˆdØÐ1Ð1 Ð1Ð1¨4Ð1Ð1ˆØÐr¥   r�   Ú	gguf_filer�   Ú
user_agentÚis_remote_codeÚtransformers_explicit_filenamer�   Ú
tqdm_classc	                 óØ  — |pt          ¦   «         }|                     d¦  «        }	|                     dd¦  «        }
|                     d¦  «        }|                     dd¦  «        }|                     d¦  «        }|                     d¦  «        pd}|                     d	d
¦  «        }|                     d¦  «        }|�B|                     d¦  «        s-|                     d¦  «        s|dk    rt          d|› �¦  «        ‚d}| ��÷|�€ôt	          | ¦  «        } t
          j                             | ¦  «        }|�rz|�ât
          j                             | |¦  «        }t
          j                             ||¦  «        }	 t
          j         	                    |¦  «        }t
          j         	                    |¦  «        }t
          j         
                    ||g¦  «        |k    }n# t          $ r d}Y nw xY w|st          d|› �¦  «        ‚|                     d¦  «        }�nœ|dur‡t
          j                             t
          j                             | |t          t          |¦  «        ¦  «        ¦  «        r6t
          j                             | |t          t          |¦  «        ¦  «        }�n|dur‰t
          j                             t
          j                             | |t          t          |¦  «        ¦  «        ¦  «        r8t
          j                             | |t          t          |¦  «        ¦  «        }d}�n„|s‡t
          j                             t
          j                             | |t          t          |¦  «        ¦  «        ¦  «        r6t
          j                             | |t          t          |¦  «        ¦  «        }�nû|s‰t
          j                             t
          j                             | |t          t           |¦  «        ¦  «        ¦  «        r8t
          j                             | |t          t           |¦  «        ¦  «        }d}�np|r)t#          dt          t          |¦  «        › d| › d�¦  «        ‚t#          dt          t          |¦  «        › dt          t          |¦  «        › d| › d�¦  «        ‚t
          j                             t
          j                             || ¦  «        ¦  «        r| }d}�nÃ|�|}|                     d¦  «        }n/|durt          t          |¦  «        }nt          t          |¦  «        }||||	|dœ}|
||dd||dœ|¥}t%          ¦   «          ot'          d¦  «         o| o|d
k    }	 t)          | |fi |¤Ž}|€Ã|t          t          |¦  «        k    rªt)          | t          t          |¦  «        fi |¤Ž}|�d}n„|r_|dk    r|rt+          | fi |¤Ž\  }}}||d<   |€>t#          | › dt          t          |¦  «        › dt          t          |¦  «        › d�¦  «        ‚n#t          t          |¦  «        }t)          | |fi |¤Ž}|€>|t          t          |¦  «        k    r%t)          | t          t           |¦  «        fi |¤Ž}|�d}|�`|rt          nt          }|t          t           fv r?t-          | |fi |¤Žs1|r/t/          t*          | fddi|¥d¬ ¦  «                             ¦   «          n~|�>t-          | t          fi |¤Žr+t#          | › dt          t          |¦  «        › d!|› d"�¦  «        ‚t#          | › dt          t          |¦  «        › dt          t          |¦  «        › d�¦  «        ‚nI# t"          $ r ‚ t2          $ r2}t#          d#| › d$| › d%t          t          |¦  «        › d�¦  «        |‚d}~ww xY w|r t4                               d&|› �¦  «         |}nat4                               d&|› d'|› �¦  «         n@|r>t
          j                             |¦  «        r|}n|	|
||||||dd|d(œ}t)          | |fi |¤Ž}d}|rt9          | ||	|
||||||||¬)¦  «        \  } }n| �|gnd} | |fS )*zÀGet all the checkpoint filenames based on `pretrained_model_name_or_path`, and optional metadata if the
    checkpoints are sharded.
    This function will download the data if necessary.
    Ú	cache_dirÚforce_downloadFÚproxiesÚlocal_files_onlyÚtokenÚrevisionÚmainÚ	subfolderÚ Úcommit_hashNr  z.safetensors.index.jsonzadapter_model.binz¥The transformers file in the config seems to be incorrect: it is neither a safetensors file (*.safetensors) nor a safetensors index file (*.safetensors.index.json): zM`transformers_weights` must reference a file inside the model directory, got TzError no file named z found in directory rH  z, or z, found in directory )r£  r   r¢  rž  r¡  )rŸ  r™  r¥  Ú _raise_exceptions_for_gated_repoÚ%_raise_exceptions_for_missing_entriesÚ_commit_hashrœ  ÚDISABLE_SAFETENSORS_CONVERSIONz& does not appear to have a file named z or zX and thus cannot be loaded with `safetensors`. Please do not set `use_safetensors=True`.Úignore_errors_during_conversionzThread-auto_conversion)ÚtargetÚargsÚkwargsrJ  z) but there is a file without the variant z;. Use `variant=None` to load this model from those weights.zCan't load the model for 'zœ'. If you were trying to load it from 'https://huggingface.co/models', make sure you don't have a local directory with the same name. Otherwise, make sure 'z=' is the correct path to a directory containing a file named zloading weights file z from cache at )rž  rŸ  r   r¡  r¢  r™  r£  r¥  r¨  r©  rª  )
rž  rŸ  r   r¡  r¢  r™  r£  r¥  rª  rœ  )ro   rÀ   r)  rÍ   rª   r¾   rõ   ÚisdirÚjoinÚabspathÚ
commonpathÚisfiler—  rZ   rY   r\   r[   r  r   rf   r`   rV   rc   r   r_  Ú	ExceptionÚloggerÚinforq   )!r�   r”  r˜  r�   r™  rš  r›  r�   rœ  rž  rŸ  r   r¡  r¢  r£  r¥  r§  Ú
is_shardedÚis_localÚbase_dirÚarchive_fileÚabsolute_base_dirÚabsolute_archive_fileÚ	containedÚfilenameÚhas_file_kwargsÚcached_file_kwargsÚcan_auto_convertÚresolved_archive_fileÚsafe_weights_namer  r’   Úcheckpoint_filess!                                    r£   Ú_get_resolved_checkpoint_filesrÆ  )  så
  € ð &Ð9­Ñ)9Ô)9€OØ×#Ò# KÑ0Ô0€IØ$×(Ò(Ð)9¸5ÑAÔA€NØ×!Ò! )Ñ,Ô,€GØ&×*Ò*Ð+=¸uÑEÔEÐØ×Ò Ñ(Ô(€EØ×"Ò" :Ñ.Ô.Ð8°&€HØ×#Ò# K°Ñ4Ô4€IØ!×%Ò% mÑ4Ô4€KØ%Ð1Ø-×6Ò6°~ÑFÔFð 	ÐOm×OvÒOvØ%ñP
ô P
ð 	ð .Ð1DÒDÐDÝ ð8à5ð8ð 8ñô ð ð €Jà$Ñ0°YÑ5FÝ(+Ð,IÑ(JÔ(JÐ%Ý”7—=’=Ð!>Ñ?Ô?ˆàñ z	Ø-Ð9åœ7Ÿ<š<Ð(EÀyÑQÔQ�Ý!œwŸ|š|¨HÐ6TÑUÔU�ð&Ý(*¬¯ª¸Ñ(AÔ(AÐ%Ý,.¬G¯OªO¸LÑ,IÔ,IÐ)Ý "¤× 2Ò 2Ð4EÐG\Ð3]Ñ ^Ô ^ÐbsÒ s�I�IøÝ!ð &ð &ð &Ø %�I�I�Ið&øøøà ð Ý$ð Ið  iGð  Ið  Iñô ð ð <×DÒDÐE^Ñ_Ô_�
‘
Ø ¨Ð-Ð-µ"´'·.².Ý”—’Ð:¸IÅ|ÕTeÐgnÑGoÔGoÑpÔpñ3ô 3Ð-õ  "œwŸ|š|Ø1°9½lÕK\Ð^eÑ>fÔ>fñ ô  �‘ð !¨Ð-Ð-µ"´'·.².Ý”—’Ð:¸IÅ|ÕTkÐmtÑGuÔGuÑvÔvñ3ô 3Ð-õ  "œwŸ|š|Ø1°9½lÕKbÐdkÑ>lÔ>lñ ô  �ð "�
‘
Ø$ð ­¬¯ªÝ”—’Ð:¸IÅ|ÕT`ÐbiÑGjÔGjÑkÔkñ*ô *ð õ  "œwŸ|š|Ø1°9½lÍ<ÐY`Ñ>aÔ>añ ô  �‘ð %ð ­¬¯ªÝ”—’Ð:¸IÅ|ÕTfÐhoÑGpÔGpÑqÔqñ*ô *ð õ  "œwŸ|š|Ø1°9½lÕK]Ð_fÑ>gÔ>gñ ô  �ð "�
‘
Ø ð 	Ýð9­<Õ8IÈ7Ñ+SÔ+Sð 9ð 9Ø5ð9ð 9ð 9ñô ð õ
 ðL­<Õ8IÈ7Ñ+SÔ+Sð Lð LÕZfÕgsÐu|ÑZ}ÔZ}ð Lð LØ+HðLð Lð Lñô ð õ ŒW�^Š^�BœGŸLšL¨Ð4QÑRÔRÑSÔSð @	Ø8ˆLØˆH‰Hð .Ð9Ø9�Ø;×DÒDÐE^Ñ_Ô_�
�
Ø ¨Ð-Ð-Ý'Õ(9¸7ÑCÔC��å'­°gÑ>Ô>�ð %Ø"ØØ&Ø$4ðð ˆOð #1Ø(Ø&Ø49Ø9>Ø +Ø(ð	"ð 	"ð "ð	"Ðõ $Ñ%Ô%Ð%ð $å,Ð-MÑNÔNÐNð$ð 'Ð&ð$ð  ’Oð ðYõ )4Ð4QÐS[Ð(rÐ(rÐ_qÐ(rÐ(rÐ%ð )Ð0°XÅÕN_ÐahÑAiÔAiÒ5iÐ5iå,7Ø5Ý$Õ%<¸gÑFÔFð-ð -ð -ð-ð -Ð)ð
 -Ð8Ø%)˜
˜
Ø(ð Ø# vÒ-Ð-Ð2BÐ-ÝJYØ =ðKð KØASðKð KÑGÐ1°8¸Zð :BÐ*¨:Ñ6Ø0Ð8Ý")Ø#@ð !zð !zÝ$0Õ1BÀGÑ$LÔ$Lð!zð !zÝR^Õ_vÐxñ  SAô  SAð!zð !zð !zñ#ô #ð ð 9õ $0µ¸gÑ#FÔ#F˜Ý0;Ø9¸8ð1ð 1ØGYð1ð 1Ð-ð
 )Ð0°XÅÍlÐ\cÑAdÔAdÒ5dÐ5då,7Ø5Ý$Õ%7¸ÑAÔAð-ð -ð -ð-ð -Ð)ð
 -Ð8Ø%)˜
ð )Ð4ØCMÐ(dÕ(?Ð(?ÕSdÐ%à ¥\Õ3EÐ$FÐFÐFÝ (Ð)FÐHYÐ mÐ mÐ]lÐ mÐ mð Gà,ð Gõ Ý#2Ø"?Ð!AØ$EÀtÐ#bÐOaÐ#bØ!9ð	ñ ô ÷
  š%™'œ'˜'øð
 Ð*­xØ5µ|ð0ð 0ØGVð0ð 0Ð*õ &Ø<ð eð eÝ ,­\¸7Ñ CÔ Cðeð eà 'ðeð eð eñô ð õ &Ø<ð uð uÝ ,­\¸7Ñ CÔ Cðuð uÝIUÕVgÐipÑIqÔIqðuð uð uñô ð øøõ
 ð ð ð ð Ýð ð ð åðaÐ1Nð að aà9Vðað aõ ;GÅ|ÐU\Ñ:]Ô:]ðað að añô ð
 ðøøøøðøøøð ð 	bÝ�KŠKÐ>°Ð>Ð>Ñ?Ô?Ð?Ø$0Ð!Ð!å�KŠKÐ`°Ð`Ð`ÐI^Ð`Ð`ÑaÔaÐaÐaà	ð påŒ7�>Š>˜)Ñ$Ô$ð 	pØ$-Ð!Ð!ð
 'Ø"0Ø"Ø$4ØØ(Ø$Ø&Ø49Ø9>Ø +ð"ð "Ðõ %0Ð0MÈyÐ$oÐ$oÐ\nÐ$oÐ$oÐ!ð ÐØð jÝ-GØ)Ø!ØØ)ØØ-ØØ!ØØØ$Ø!ð.
ñ .
ô .
Ñ*ÐÐ*Ð*ð 7TÐ6_Ð1Ð2Ð2ÐeiÐàÐ-Ð-Ð-s,   Å>A#G" Ç"G1Ç0G1Õ>G3]2 Ý2^8Þ-^3Þ3^8rÅ  Úconfigr’   r˜   c                 óÜ  — |du}| ��…t          | t          ¦  «        �r;| dk    rÎt          |d¦  «        r-|j        �&|j        } t                               d| › d�¦  «         nË|rd|v r	|d         } nc|�t          |¦  «        } nQ|�(|d                              d¦  «        rt          j	        } n't          |d         d|¬	¦  «        }t          |¦  «        } t                               d
| › d�¦  «         n:t          t          | ¦  «        rt          t          | ¦  «        } nt          d¦  «        ‚t          | t          ¦  «        rt          t          | ¦  «        n| } nGt          | t          t          j        f¦  «        st          d| › �¦  «        ‚nt          j        ¦   «         } |�|                     | ¦  «        } t          | t          ¦  «        rr|                      dt          j        ¦   «         ¦  «        }t          |t          ¦  «        rt          t          |¦  «        n|}t                               d|› d�¦  «         n| }||_        |j        D ]}	t          ||	¦  «        x}
�||
_        Œ||fS )a®  Find the correct `dtype` to use based on provided arguments. Also update the `config` based on the
    inferred dtype. We do the following:
    1. If dtype is "auto", we try to read the config, else auto-detect dtype from the loaded state_dict, by checking
    its first weights entry that is of a floating type - we assume all floating dtype weights are of the same dtype
    2. Else, use the dtype provided as a dict or str
    NÚautor–   zWill use dtype=z$ as defined in model's config objectr   z.ggufr  r(  zTSince the `dtype` attribute can't be found in model's config object, will use dtype=z  as derived from model's weightsze`dtype` provided as a `str` can only be `'auto'`, or a string representation of a valid `torch.dtype`z¨`dtype` can be one of: `torch.dtype`, `'auto'`, a string of a valid `torch.dtype` or a `dict` with valid `dtype` for each sub-config in composite configs, but received r¦  zÅUsing different dtypes per module is deprecated and will be removed in future versions Setting different dtypes per backbone model might cause device errors downstream, therefore setting the dtype=z for all modules.)rs  rª   rµ   r–   r¶  r·  rå   r)  r®   rà   r:  rM  rÍ   r­   rÎ   Úupdate_dtyperÀ   Úwarning_onceÚsub_configs)r–   rÅ  rÇ  r’   rã   r›   r˜   r¸  Ú
main_dtypeÚsub_config_keyÚ
sub_configs              r£   Ú
_get_dtyperÐ  B  sÖ  € ð "¨Ð-€JàÑÝ�e�SÑ!Ô!ñ "	Ø˜ŠˆÝ˜6 7Ñ+Ô+ð °´Ð0HØ"œL�EÝ—K’KÐ ]°%Ð ]Ð ]Ð ]Ñ^Ô^Ð^Ð^à!ð 
A gÐ1AÐ&AÐ&AØ 0°Ô 9˜˜Ø#Ð/Ý 4°ZÑ @Ô @˜˜Ø)Ð5Ð:JÈ1Ô:M×:VÒ:VÐW^Ñ:_Ô:_Ð5Ý %¤˜˜å%4Ø,¨QÔ/¸fÐS_ð&ñ &ô &˜
õ !5°ZÑ @Ô @˜Ý—K’KðRØ*/ðRð Rð Rñô ð ð õ � Ñ&Ô&ð Ý¥ uÑ-Ô-��å Ø{ñô ð õ
 .8¸½sÑ-CÔ-CÐN•G�E 5Ñ)Ô)Ð)ÈˆEˆEÝ˜E¥D­%¬+Ð#6Ñ7Ô7ð 	ÝðRØJOðRð Rñô ð ð	õ Ô'Ñ)Ô)ˆàÐØ×)Ò)¨%Ñ0Ô0ˆõ �%�ÑÔð Ø—Y’Y˜r¥5Ô#:Ñ#<Ô#<Ñ=Ô=ˆ
Ý3=¸jÍ#Ñ3NÔ3NÐ^•W�U JÑ/Ô/Ð/ÐT^ˆ
å×Òð?à!+ð?ð ?ð ?ñ	
ô 	
ð 	
ð 	
ð ˆ
ð €F„LØ Ô,ð *ð *ˆÝ! &¨.Ñ9Ô9Ð9ˆJÐFØ)ˆJÔøà�:ÐÐr¥   c                   óê   — e Zd ZdZedddej        fd„¦   «         Zedddej        fd„¦   «         Zdddedefd„Z	e
d	„ ¦   «         Z	 ddddedeedf         dej        d
z  def
d„Zddddededefd„Zd
S )ÚModuleUtilsMixinzH
    A few utilities for `torch.nn.Modules`, to be used as a mixin.
    r¢   rˆ   rž   c                 óX   — t          d„ |                      ¦   «         D ¦   «         ¦  «        S )z�
        `torch.device`: The device on which the module is (assuming that all the module parameters are on the same
        device).
        c              3   ó$   K  — | ]}|j         V — Œd S r    ©rÖ   ©rþ   Úparams     r£   r   z*ModuleUtilsMixin.device.<locals>.<genexpr>   s$   è è € Ð@Ð@ U�E”LÐ@Ð@Ð@Ð@Ð@Ð@r¥   ©rá   Ú
parametersr¡   s    r£   rÖ   zModuleUtilsMixin.deviceš  s+   € õ Ð@Ð@¨d¯oªoÑ.?Ô.?Ð@Ñ@Ô@Ñ@Ô@Ð@r¥   c                 óX   — t          d„ |                      ¦   «         D ¦   «         ¦  «        S )zw
        `torch.dtype`: The dtype of the module (assuming that all the module parameters have the same dtype).
        c              3   óL   K  — | ]}|                      ¦   «         ¯|j        V — Œ d S r    )rÌ   r–   rÖ  s     r£   r   z)ModuleUtilsMixin.dtype.<locals>.<genexpr>§  s5   è è € Ð\Ð\ EÀ%×BYÒBYÑB[ÔB[Ð\�E”KÐ\Ð\Ð\Ð\Ð\Ð\r¥   rØ  r¡   s    r£   r–   zModuleUtilsMixin.dtype¢  s+   € õ
 Ð\Ð\¨T¯_ª_Ñ->Ô->Ð\Ñ\Ô\Ñ\Ô\Ð\r¥   Úencoder_attention_maskc                 ó\  — t                                d¦  «         |                     ¦   «         dk    r|dd…ddd…dd…f         }|                     ¦   «         dk    r|dd…dddd…f         }|                     | j        ¬¦  «        }d|z
  t          j        | j        ¦  «        j        z  }|S )zè
        Invert an attention mask (e.g., switches 0. and 1.).

        Args:
            encoder_attention_mask (`torch.Tensor`): An attention mask.

        Returns:
            `torch.Tensor`: The inverted attention mask.
        z¡Detected the usage of `invert_attention_mask`: This function is deprecated and will be removed in v5.12.0. Please use the new API in `transformers.masking_utils`é   Nrü   ©r–   ç      ð?)r¶  rË  Údimr   r–   r®   ÚfinfoÚmin)r¢   rÜ  Úencoder_extended_attention_masks      r£   Úinvert_attention_maskz&ModuleUtilsMixin.invert_attention_mask©  sÛ   € õ 	×ÒðEñ	
ô 	
ð 	
ð
 "×%Ò%Ñ'Ô'¨1Ò,Ð,Ø.DÀQÀQÀQÈÈaÈaÈaÐQRÐQRÐQRÀ]Ô.SÐ+Ø!×%Ò%Ñ'Ô'¨1Ò,Ð,Ø.DÀQÀQÀQÈÈdÐTUÐTUÐTUÐEUÔ.VÐ+ð +J×*LÒ*LÐSWÔS]Ð*LÑ*^Ô*^Ð'Ø+.Ð1PÑ+PÕTYÔT_Ð`dÔ`jÑTkÔTkÔToÑ*oÐ'à.Ð.r¥   c                 ó2  — t                                d¦  «         |j        }| \  }}t          j        ||¬¦  «        }|d d d d …f                              ||d¦  «        |d d d …d f         k    }|                     |j        ¦  «        }|j        d         |j        d         k     rP|j        d         |j        d         z
  }t          j	        t          j
        |||f||j        ¬¦  «        |gd¬¦  «        }|d d …d d d …d d …f         |d d …d d d d …f         z  }|S )Nz¶Detected the usage of `create_extended_attention_mask_for_decoder`: This function is deprecated and will be removed in v5.12.0. Please use the new API in `transformers.masking_utils`rÕ  r   ©rÖ   r–   r<  ©Úaxis)r¶  rË  rÖ   r®   ÚarangeÚrepeatr   r–   ÚshapeÚcatÚones)	Úinput_shapeÚattention_maskrÖ   Ú
batch_sizeÚ
seq_lengthÚseq_idsÚcausal_maskÚprefix_seq_lenÚextended_attention_masks	            r£   Ú*create_extended_attention_mask_for_decoderz;ModuleUtilsMixin.create_extended_attention_mask_for_decoderÄ  s\  € å×ÒðEñ	
ô 	
ð 	
ð
  Ô&ˆØ!,Ñˆ
�JÝ”,˜z°&Ð9Ñ9Ô9ˆØ˜d D¨!¨!¨!˜mÔ,×3Ò3°JÀ
ÈAÑNÔNÐRYÐZ^Ð`aÐ`aÐ`aÐcgÐZgÔRhÒhˆà!—n’n ^Ô%9Ñ:Ô:ˆàÔ˜QÔ .Ô"6°qÔ"9Ò9Ð9Ø+Ô1°!Ô4°{Ô7HÈÔ7KÑKˆNÝœ)å”J 
¨J¸ÐGÐPVÐ^iÔ^oÐpÑpÔpØðð ðñ ô ˆKð #.¨a¨a¨a°°q°q°q¸!¸!¸!¨mÔ"<¸~ÈaÈaÈaÐQUÐW[Ð]^Ð]^Ð]^ÐN^Ô?_Ñ"_ÐØ&Ð&r¥   Nrð  rï  .r–   c                 óø  — t                                d¦  «         |€| j        }|                     ¦   «         dk    r|dd…ddd…dd…f         }nv|                     ¦   «         dk    rCt	          | j        dd¦  «        rt                               ||¦  «        }n,|dd…dddd…f         }nt          d|› d|j	        › d�¦  «        ‚| 
                    |¬	¦  «        }d
|z
  t          j        |¦  «        j        z  }|S )aâ  
        Makes broadcastable attention and causal masks so that future and masked tokens are ignored.

        Arguments:
            attention_mask (`torch.Tensor`):
                Mask with ones indicating tokens to attend to, zeros for tokens to ignore.
            input_shape (`tuple[int]`):
                The shape of the input to the model.

        Returns:
            `torch.Tensor` The extended attention mask, with a the same dtype as `attention_mask.dtype`.
        z§Detected the usage of `get_extended_attention_mask`: This function is deprecated and will be removed in v5.12.0. Please use the new API in `transformers.masking_utils`NrÞ  rü   Ú
is_decoderz!Wrong shape for input_ids (shape z) or attention_mask (shape ú)rß  rà  )r¶  rË  r–   rá  rM  rÇ  rÒ  r÷  rÍ   rì  r   r®   râ  rã  )r¢   rð  rï  r–   rö  s        r£   Úget_extended_attention_maskz,ModuleUtilsMixin.get_extended_attention_maskß  sD  € õ$ 	×ÒðEñ	
ô 	
ð 	
ð
 ˆ=Ø”JˆEð ×ÒÑÔ 1Ò$Ð$Ø&4°Q°Q°Q¸¸a¸a¸aÀÀÀ°]Ô&CÐ#Ð#Ø×ÒÑ!Ô! QÒ&Ð&õ �t”{ L°$Ñ7Ô7ð KÝ*:×*eÒ*eØ ñ+ô +Ð'Ð'ð +9¸¸¸¸DÀ$ÈÈÈÐ9IÔ*JÐ'Ð'åØs°KÐsÐsÐ\jÔ\pÐsÐsÐsñô ð ð #:×"<Ò"<À5Ð"<Ñ"IÔ"IÐØ#&Ð)@Ñ#@ÅEÄKÐPUÑDVÔDVÔDZÑ"ZÐØ&Ð&r¥   FÚonly_trainableÚexclude_embeddingsc                 óö  — |rd„ |                       ¦   «         D ¦   «         }t          | dd¦  «        }|rddl}d}|                      ¦   «         D ]ª\  }}|r||v rŒ|j        s|s•|r|t          ||j        j        ¦  «        rbt          |d¦  «        r| 	                    ¦   «         }	nt          |d¦  «        r|j
        j        }	nd}	||                     ¦   «         d	z  |	z  z  }Œ“||                     ¦   «         z  }Œ«|S )
aé  
        Get number of (optionally, trainable or non-embeddings) parameters in the module.

        Args:
            only_trainable (`bool`, *optional*, defaults to `False`):
                Whether or not to return only the number of trainable parameters

            exclude_embeddings (`bool`, *optional*, defaults to `False`):
                Whether or not to return only the number of non-embeddings parameters

        Returns:
            `int`: The number of parameters.
        c                 óR   — g | ]$\  }}t          |t          j        ¦  «        ¯|› d �‘Œ%S )z.weight)rs  r   Ú	Embedding)rþ   rJ  Úmodule_types      r£   rK  z3ModuleUtilsMixin.num_parameters.<locals>.<listcomp>%  sJ   € ð %ð %ð %Ù%6 T¨;ÕR\Ð]hÕjlÔjvÑRwÔRwð%ØÐ Ð Ð ð%ð %ð %r¥   Úis_loaded_in_4bitFr   Nr@  Úquant_storager   rü   )rL  rM  ÚbitsandbytesÚnamed_parametersrŒ  rs  r   Ú
Params4bitrµ   r@  r  ÚitemsizeÚnumel)
r¢   rü  rý  Úembedding_param_namesr  ÚbnbÚtotal_paramsrJ  r×  Ú	num_bytess
             r£   Únum_parameterszModuleUtilsMixin.num_parameters  sL  € ð ð 	ð%ð %Ø:>×:LÒ:LÑ:NÔ:Nð%ñ %ô %Ð!õ $ DÐ*=¸uÑEÔEÐØð 	'Ø&Ð&Ð&Ð&àˆØ×0Ò0Ñ2Ô2ð 	2ð 	2‰KˆD�%Ø!ð  dÐ.CÐ&CÐ&CØØÔ"ð 2¨.ð 2ð %ð 	2­°E¸3¼6Ô;LÑ)MÔ)Mð 	2Ý˜u nÑ5Ô5ð &Ø$)×$6Ò$6Ñ$8Ô$8˜	˜	Ý  ¨Ñ8Ô8ð &Ø$)Ô$7Ô$@˜	˜	à$%˜	Ø  E§K¢K¡M¤M°AÑ$5¸	Ñ$AÑA�L�Là  E§K¢K¡M¤MÑ1�LøàÐr¥   r    ©FF)r¦   r§   r¨   r©   r°   r®   rÖ   r–   r   rå  Ústaticmethodr÷  Útupler½   rû  r¬   r  r±   r¥   r£   rÒ  rÒ  •  sf  € € € € € ðð ð ðAÐ&ð A¨5¬<ð Að Að Añ „XðAð ð]Ð%ð ]¨%¬+ð ]ð ]ð ]ñ „Xð]ð/Ð$5ð /Èvð /ÐZ`ð /ð /ð /ð /ð6 ð'ð 'ñ „\ð'ð< %)ð	4'ð 4'Øð4'àð4'ð ˜3 ˜8”_ð4'ð Œ{˜TÑ!ð	4'ð
 
ð4'ð 4'ð 4'ð 4'ðl*ð *Ð.ð *Àð *Ðbfð *Ðsvð *ð *ð *ð *ð *ð *r¥   rÒ  c                   óN   — e Zd ZdZdZdej        fd„Zdej        fd„Zd„ Z	d„ Z
d	S )
ÚEmbeddingAccessMixinzÔ
    Base utilities to regroup getters and setters for embeddings.
    Introduces the `input_layer_embed` attribute, which indicates
    where the input embeddings come from and where they
    should be set.
    Úembed_tokensrž   c                 ó@  — t          | dd¦  «        }t          | |d¦  «        x}�|S t          | dd¦  «        }|� t          ||¦  «        rt          ||¦  «        S t          | dd¦  «        }|� t          ||¦  «        rt          ||¦  «        S t          | dd¦  «        }|�(|| ur$t          |d¦  «        r|                     ¦   «         S t          | dd¦  «        }|�(|| ur$t          |d¦  «        r|                     ¦   «         S t          d	| j        j        › d
�¦  «        ‚)z–
        Returns the model's input embeddings.

        Returns:
            `nn.Module`: A torch module mapping vocabulary to hidden states.
        Ú_input_embed_layerr  NÚ
embeddingsri  Úlanguage_modelÚget_input_embeddingsÚ
base_modelu.   `get_input_embeddings` not autoâ€‘handled for ú"; please override in the subclass.)rM  rµ   r  ÚNotImplementedErrorÚ	__class__r¦   )r¢   rJ  Údefault_embeddingr  ri  r  r  s          r£   r  z)EmbeddingAccessMixin.get_input_embeddingsL  sV  € õ �tÐ1°>ÑBÔBˆõ ")¨¨t°TÑ!:Ô!:Ð:ÐÐGØ$Ð$å˜T <°Ñ6Ô6ˆ
ØÐ!¥g¨j¸$Ñ&?Ô&?Ð!Ý˜: tÑ,Ô,Ð,å˜˜g tÑ,Ô,ˆØÐ¥¨°Ñ!5Ô!5ÐÝ˜5 $Ñ'Ô'Ð'å  Ð'7¸Ñ>Ô>ˆàÐ&Ø dÐ*Ð*Ý˜Ð(>Ñ?Ô?ð +ð "×6Ò6Ñ8Ô8Ð8õ ˜T <°Ñ6Ô6ˆ
ØÐ! j¸Ð&<Ð&<ÅÈÐUkÑAlÔAlÐ&<Ø×2Ò2Ñ4Ô4Ð4å!Øx¸T¼^Ô=TÐxÐxÐxñ
ô 
ð 	
r¥   Úvaluec                 ót  — t          | dd¦  «        }t          | |¦  «        rt          | ||¦  «         dS t          | dd¦  «        x}�#t          ||¦  «        rt          |||¦  «         dS t          | dd¦  «        x}�#t          ||¦  «        rt          |||¦  «         dS t          | dd¦  «        x}�+|| ur't          |d¦  «        r|                     |¦  «         dS t          | dd¦  «        x}�+|| ur't          |d¦  «        r|                     |¦  «         dS t	          d	| j        j        › d
�¦  «        ‚)aù  Fallback setter that handles **~70%** of models in the code-base.

        Order of attempts:
        1. `self.<_input_embed_layer>` (direct attribute)
        2. `self.embeddings.<_input_embed_layer>` (nested embeddings for vision/audio models)
        3. `self.model.<_input_embed_layer>` (encoder/decoder models)
        4. delegate to the *base model* if one exists
        5. otherwise raise `NotImplementedError` so subclasses still can (and
            should) override for exotic layouts.
        r  r  r  Nri  r  Úset_input_embeddingsr  u.   `set_input_embeddings` not autoâ€‘handled for r  )rM  rµ   r�  r   r  r  r¦   )r¢   r  rJ  r  ri  r  r  s          r£   r   z)EmbeddingAccessMixin.set_input_embeddingss  s‹  € õ �tÐ1°>ÑBÔBˆå�4˜ÑÔð 	Ý�D˜$ Ñ&Ô&Ð&Ð&Ð&å# D¨,¸Ñ=Ô=Ð=ˆjÐJÍwÐWaÐcgÑOhÔOhÐJÝ�J  eÑ,Ô,Ð,Ð,Ð,å˜t W¨dÑ3Ô3Ð3ˆeÐ@ÅWÈUÐTXÑEYÔEYÐ@Ý�E˜4 Ñ'Ô'Ð'Ð'Ð'õ  ' tÐ-=¸tÑDÔDÐDˆ^ÐQØ dÐ*Ð*Ý˜Ð(>Ñ?Ô?ð +ð ×/Ò/°Ñ6Ô6Ð6Ð6Ð6õ # 4¨°tÑ<Ô<Ð<ˆZÐIØ $Ð&Ð&Ý˜
Ð$:Ñ;Ô;ð 'ð ×+Ò+¨EÑ2Ô2Ð2Ð2Ð2å%Ø|ÀÄÔAXÐ|Ð|Ð|ñô ð r¥   c                 ó‚   — t          | d¦  «        sd S 	 |                      ¦   «          n# t          $ r Y d S w xY w| j        S )NÚlm_head)rµ   r  r  r"  r¡   s    r£   Úget_output_embeddingsz*EmbeddingAccessMixin.get_output_embeddingsœ  s_   € Ý�t˜YÑ'Ô'ð 	Ø�4ð	ð ×%Ò%Ñ'Ô'Ð'Ð'øÝ"ð 	ð 	ð 	Ø�4�4ð	øøøàŒ|Ðs   ”) ©
7¶7c                 ó8   — t          | d¦  «        r	|| _        dS dS )ze
        Sets the model's output embedding, defaulting to setting new_embeddings to lm_head.
        r"  N)rM  r"  )r¢   Únew_embeddingss     r£   Úset_output_embeddingsz*EmbeddingAccessMixin.set_output_embeddings§  s+   € õ �4˜Ñ#Ô#ð 	*Ø)ˆDŒLˆLˆLð	*ð 	*r¥   N)r¦   r§   r¨   r©   r  r   ÚModuler  r   r#  r&  r±   r¥   r£   r  r  B  s€   € € € € € ðð ð (Ðð%
 b¤ið %
ð %
ð %
ð %
ðN'¨"¬)ð 'ð 'ð 'ð 'ðR	ð 	ð 	ð*ð *ð *ð *ð *r¥   r  c                   ó  ‡ — e Zd ZU dZdZee         dz  ed<   eZ	ee         ed<   dZ
dZeed<   dZeed<   dZee         dz  ed	<   d
Zeed<   dZeee         z  ed<   dZee         ee         z  dz  ed<   dZee         ee         z  dz  ed<   dZee         ee         z  dz  ed<   dZee         ee         z  dz  ed<   dZeeef         ed<   dZee         ee         z  dz  ed<   dZee         ee         z  dz  ed<   dZee         ee         z  dz  ed<   dZeed<   dZeed<   dZeed<   dZ ee         dz  ed<   dZ!eeef         ed<   dZ"dZ#eee$eef         f         ed<   dZ%eeef         ed<   dZ&eeef         ed<   dZ'eed<   dZ(eed<   dZ)eed <   dZ*edz  ed!<   e+e,j-        j.        d"eee/f         fd#„¦   «         ¦   «         Z0e+d"eee,j1        f         fd$„¦   «         Z2ˆ fd%„Z3d&efˆ fd'„Z4d(„ Z5e+d"eeef         fd)„¦   «         Z6e+d"eeef         fd*„¦   «         Z7e+d"eee$eef         f         fd+„¦   «         Z8e6j9        d,eeef         dz  fd-„¦   «         Z6e8j9        d,eee$eef         f         dz  fd.„¦   «         Z8dÌd/„Z:d0„ Z;d1ee         ez  d"dfd2„Z<e=d3„ ¦   «         Z>e+d"e?j@        fd4„¦   «         ZAe=d"efd5„¦   «         ZB	 	 dÍd7eCd8eDd9eDd:e$e$eDef         d;f         d<e$e$eDef         d;f         d=eCdz  fd>„ZEdÎd7eCd?ed"efd@„ZFdÎd?ed"efdA„ZGd"efdB„ZHdÎd?ed"efdC„ZI	 dÏdDedz  d?edEed"efdF„ZJdGedz  d"efdH„ZKdÎdIedz  d?ed"efdJ„ZLdKedz  d"efdL„ZMe=d"efdM„¦   «         ZNe=d"efdN„¦   «         ZOdÎdDeez  dEefdO„ZPd"eeedz  f         fdP„ZQdGeez  fdQ„ZRdR„ ZSdS„ ZTdÌdTedz  fdU„ZUdÌdTedz  fdV„ZVdW„ ZWdX„ ZX e,jY        ¦   «         dY„ ¦   «         ZZdÎdZefd[„Z[ e,jY        ¦   «          e\j]        ¦   «         d\„ ¦   «         ¦   «         Z^dÎd]ed"efd^„Z_dÐd`ee         dz  daefdb„Z`dc„ Za	 	 	 dÑddeCdz  deeCdz  dfed"e?jb        fdg„ZcdÐdh„Zd	 	 	 dÑdie?jb        ddeCdz  deeCdz  dfed"e?jb        f
dj„Ze	 	 	 dÒdke?jf        ddeCdz  dledfed"e?jf        f
dm„Zgdn„ Zh	 dÎdlefdo„Zidp„ Zjdq„ ZkdreCfds„Zld"e?jb        e$e?jb                 z  fdt„Zmdu„ ZndÌdv„Zod_epfdwedxeDfdy„Zqdz„ Zre+d"efd{„¦   «         Zs	 	 	 	 	 	 	 	 dÓd}eetju        z  d~ededz  d€ed�eCez  d‚edz  dƒeez  dz  d„ed…efd†„Zv ewexjy        ¦  «        ˆ fd‡„¦   «         ZydÔdˆ„Zz ewe,j?        j@        j{        ¦  «        ˆ fd‰„¦   «         Z{ ewe,j?        j@        j|        ¦  «        ˆ fdŠ„¦   «         Z|ˆ fd‹„Z}ˆ fdŒ„Z~e=d�e,j        dŽed�edEedz  fd�„¦   «         Z€d�e,j        d"efd‘„Z�dÕd’e‚dz  d“d”fd•„Zƒe=ddddddd–dd_ddd—œd˜ee„         d™eetju        z  dz  d&eez  etju        z  dz  dšeetju        z  dz  d›edœed�edƒeez  dz  džedŸedz  d ed¡eeeeee…f         z  f         dz  d¢edz  d"e„fd£„¦   «         Z†e‡	 dÌd¤d dedz  d¥ee         dz  d¦eˆd§ee         dz  d"e$e‰ef         fd¨„¦   «         ZŠe‡d¦eˆd©e‰d"e‰fdª„¦   «         Z‹dÏd«„ZŒe=dÖd­„¦   «         Z�d®„ ZŽe+d¯„ ¦   «         Z�e+d°„ ¦   «         Z�e+d±„ ¦   «         Z‘e+d²„ ¦   «         Z’e’j9        d³„ ¦   «         Z’e+d"efd´„¦   «         Z“e“j9        dµed"dfd¶„¦   «         Z“d"e”fd·„Z•d¸e”dz  d"eDfd¹„Z–e=dº„ ¦   «         Z—d`ee         d»edz  d¼d½d¾e˜dz  d"df
d¿„Z™dŽed"dfdÀ„Zšd©e‰d"dfdÁ„Z›dÂ„ ZœdÃefdÄ„Z�	 d×dÅedÆed"eže$ee,j1        f                  fdÇ„ZŸdÔd“efˆ fdÈ„Z dÉ„ Z¡e=d"efdÊ„¦   «         Z¢e=d"efdË„¦   «         Z£ˆ xZ¤S )Ørˆ   a¡  
    Base class for all models.

    [`PreTrainedModel`] takes care of storing the configuration of the models and handles methods for loading,
    downloading and saving models as well as a few methods common to all models to:

        - resize the input embeddings

    Class attributes (overridden by derived classes):

        - **config_class** ([`PreTrainedConfig`]) -- A subclass of [`PreTrainedConfig`] to use as configuration class
          for this model architecture.
        - **base_model_prefix** (`str`) -- A string indicating the attribute associated to the base model in derived
          classes of the same architecture adding modules on top of the base model.
        - **main_input_name** (`str`) -- The name of the principal input to the model (often `input_ids` for NLP
          models, `pixel_values` for vision models and `input_values` for speech models).
        - **can_record_outputs** (dict):
    NÚconfig_classÚgeneration_config_classr¦  Úbase_model_prefixFÚ_is_statefulÚ
model_tagsÚ	input_idsÚmain_input_nameÚtextÚinput_modalitiesÚ_no_split_modulesÚ_skip_keys_device_placementÚ_keep_in_fp32_modulesÚ_keep_in_fp32_modules_strictrE  Ú_keys_to_ignore_on_load_missingÚ"_keys_to_ignore_on_load_unexpectedÚ_keys_to_ignore_on_saveÚ_supports_sdpaÚ_supports_flash_attnÚ_supports_flex_attnÚ!_compatible_flash_implementationsÚ_tp_planÚ_pp_planÚ_ep_planÚ
_fsdp_planÚsupports_gradient_checkpointingÚ_can_compile_fullgraphÚ_supports_attention_backendÚ_can_record_outputsrž   c                 ó   — | j         pi S )a  
         Maps output names (e.g., "attentions", "hidden_states")
         to either:
             - A module class (e.g., `LlamaDecoderLayer`), using default index conventions:
                 * index = 0 for a key that contains "hidden_states" (e.g. "hidden_states" or "vision_hidden_states")
                 * index = 1 for any other key: "attentions", "cross_attentions", etc.
             - A class name as a string, when the class is not importable at declaration time.
             - An `OutputRecorder(...)` with `target_class`, optional `index`, and `layer_name`.
             - A list of any of the above, to record outputs from several module types under one key.

         Examples:
             These two are equivalent:

         ```python
             _can_record_outputs = {
                 "attentions": LlamaAttention,
                 "hidden_states": LlamaDecoderLayer
             }

             _can_record_outputs = {
                 "attentions": OutputRecorder(LlamaAttention, index=1),
                 "hidden_states": OutputRecorder(LlamaDecoderLayer, index=0)
             }
        ```

         This means you can record outputs from the same class, by specifying a layer name. Before
         collecting outputs, we check that they come from this layer.

         If you have cross attention that come from `LlamaAttention` and self attention that also
         come from `LlamaAttention` but from `self_attn` you can do this:

         ```python
         class LlamaModel(PreTrainedModel):
             _can_record_outputs = {
                 "attentions": OutputRecorder(LlamaAttention, index=1, layer_name="self_attn"),
                 "cross_attentions": OutputRecorder(LlamaAttention, index=1, layer_name="cross_attn")
             }

        ```
        )rD  r¡   s    r£   Úcan_record_outputsz"PreTrainedModel.can_record_outputs	  s   € ðV Ô'Ð-¨2Ð-r¥   c                 ó8   — dt          j        t          ¦  «        iS )z^
        `dict[str, torch.Tensor]`: Dummy inputs to do a forward pass in the network.
        r.  )r®   rÕ   rX   r¡   s    r£   Údummy_inputszPreTrainedModel.dummy_inputs6  s   € ð
 �Uœ\­,Ñ7Ô7Ð8Ð8r¥   c                 ól  •—  t          ¦   «         j        di |¤Ž t          j        | ¦  «                             dd ¦  «        }| j                             dd ¦  «        }t          | ¦  «                             dd ¦  «        }| j        }|�	|| _        d S |�	|| _        d S |�	|| _        d S |�	|| _        d S d S )NrÇ  r)  r±   )ÚsuperÚ__init_subclass__ÚinspectÚget_annotationsrÀ   Ú__dict__r   r)  )Úclsr¯  Úchild_annotationÚchild_attributeÚfull_annotationÚfull_attributer  s         €r£   rK  z!PreTrainedModel.__init_subclass__=  sÛ   ø€ Ø!�‰ŒÔ!Ð+Ð+ FÐ+Ð+Ð+õ #Ô2°3Ñ7Ô7×;Ò;¸HÀdÑKÔKÐØœ,×*Ò*¨>¸4Ñ@Ô@ˆõ )¨Ñ-Ô-×1Ò1°(¸DÑAÔAˆØÔ)ˆð Ð&Ø.ˆCÔÐÐØÐ)Ø/ˆCÔÐÐØÐ'Ø-ˆCÔÐÐØÐ(Ø.ˆCÔÐÐð )Ð(r¥   rÇ  c                 ó¤  •— t          ¦   «                              ¦   «          t          |t          ¦  «        s*t	          d| j        j        › d| j        j        › d�¦  «        ‚|| _        |j        | _        d | _	        |  
                    | j        j        dt          j        ¬¦  «        | j        _        |                      | j        j        ¦  «        | j        _        |                      ¦   «         rJ	 | j                             |¦  «        | _        n)# t,          $ r |                      ¦   «         | _        Y nw xY w| j        j        }|t.          vr[dd                     t.          ¦  «        › d�}t3          j        || j        j        ¦  «        }t7          |¦  «        d	k    r	|d	         }nd }|| _        | j        t<          t?          | j        ¦  «        <   d S )
NzParameter config in `zt(config)` should be an instance of class `PreTrainedConfig`. To create a model from a pretrained model use `model = z(.from_pretrained(PRETRAINED_MODEL_NAME)`T©Úis_init_checkÚallow_all_kernelsú(ú|rú  r   ) rJ  Ú__init__rs  r    Ú	TypeErrorr  r¦   rÇ  Úname_or_pathÚkernel_configÚ%_check_and_adjust_attn_implementationÚ_attn_implementationr,   ÚALLOW_ALL_KERNELSÚ_attn_implementation_internalÚ(_check_and_adjust_experts_implementationÚ_experts_implementationÚ _experts_implementation_internalÚcan_generater*  Úfrom_model_configÚgeneration_configr  rJ   r±  rp  Úfindallrß   Ú	loss_typerD  r|   rª   )r¢   rÇ  Úinputsr¯  ri  Úloss_groupsr  s         €r£   rZ  zPreTrainedModel.__init__U  sè  ø€ Ý‰Œ×ÒÑÔÐÝ˜&Õ"2Ñ3Ô3ð 	Ýð^¨¬Ô(?ð ^ð ^à œNÔ3ð^ð ^ð ^ñô ð ð
 ˆŒØ"Ô/ˆÔð "ˆÔð 59×4^Ò4^ØŒKÔ,Øå)Ô;ð	 5_ñ 5
ô 5
ˆŒÔ1ð 8<×7dÒ7dØŒKÔ/ñ8
ô 8
ˆŒÔ4ð ×ÒÑÔð 	HðHØ)-Ô)E×)WÒ)WÐX^Ñ)_Ô)_�Ô&Ð&øÝ&ð Hð Hð HØ)-×)EÒ)EÑ)GÔ)G�Ô&Ð&Ð&ðHøøøð ”NÔ+ˆ	Ø�LÐ(Ð(Ø7˜cŸhšh¥|Ñ4Ô4Ð7Ð7Ð7ˆKÝœ
 ;°´Ô0GÑHÔHˆIÝ�9‰~Œ~ Ò!Ð!Ø% aœL�	�	à �	Ø"ˆŒà48Ô4LÕ�S ¤Ñ0Ô0Ñ1Ð1Ð1s   Ã/D Ä#D5Ä4D5c                 óŠ
  ‡— t          | j        pi ¦  «        | _        t          | j        pi ¦  «        | _        t          | j        pi ¦  «        | _        t          | j        pi ¦  «        | _        | j        | u r˜| j                             | j        j        pi ¦  «         | j                             | j        j	        pi ¦  «         | j                             | j        j
        pi ¦  «         | j                             | j        j        pi ¦  «         |                      d¬¦  «        | _        t          | j        pg ¦  «        | _        t          | j        pg ¦  «        | _        t          | j        pg ¦  «        | _        t          | j        pg ¦  «        | _        t          | j        pg ¦  «        | _        t          | j        pg ¦  «        | _        t          | j        pg ¦  «        | _        |                      ¦   «         D �]\  Š}t/          |dd¦  «        x}rJ| j                             ˆfd„|                     ¦   «                              ¦   «         D ¦   «         ¦  «         t/          |dd¦  «        x}rJ| j                             ˆfd„|                     ¦   «                              ¦   «         D ¦   «         ¦  «         t/          |dd¦  «        x}rJ| j                             ˆfd	„|                     ¦   «                              ¦   «         D ¦   «         ¦  «         t/          |d
d¦  «        x}rJ| j                             ˆfd„|                     ¦   «                              ¦   «         D ¦   «         ¦  «         t/          |dd¦  «        x}rJ| j                             ˆfd„|                     ¦   «                              ¦   «         D ¦   «         ¦  «         t/          |dd¦  «        x}r| j                             |¦  «         t/          |dd¦  «        x}r| j                             |¦  «         t/          |dd¦  «        x}r| j                             |¦  «         t/          |dd¦  «        x}r| j                             |¦  «         t/          |dd¦  «        x}r| j                             |¦  «         t/          |dd¦  «        x}	r| j                             |	¦  «         t/          |dd¦  «        x}
r&| j                             ˆfd„|
D ¦   «         ¦  «         �Œ |                      ¦   «          |                      ¦   «          dS )aÏ  
        A method executed at the end of each Transformer model initialization, to execute code that needs the model's
        modules properly initialized (such as weight initialization).
        It is also used to obtain all correct static properties (parallelism plans, tied_weights_keys, _keep_in_fp32_modules, etc)
        correctly in the case of composite models (that is, the top level model should know about those properties from its children).
        F©Úall_submodelsr?  Nc                 ó&   •— i | ]\  }}‰› d |› �|“ŒS rG  r±   ©rþ   r!  r"  rJ  s      €r£   r#  z-PreTrainedModel.post_init.<locals>.<dictcomp>«  ó)   ø€ Ð%WÐ%WÐ%W¹4¸1¸a¨ m m° m m°QÐ%WÐ%WÐ%Wr¥   r=  c                 ó&   •— i | ]\  }}‰› d |› �|“ŒS rG  r±   rp  s      €r£   r#  z-PreTrainedModel.post_init.<locals>.<dictcomp>­  rq  r¥   r>  c                 ó&   •— i | ]\  }}‰› d |› �|“ŒS rG  r±   rp  s      €r£   r#  z-PreTrainedModel.post_init.<locals>.<dictcomp>¯  rq  r¥   r@  c                 ó&   •— i | ]\  }}‰› d |› �|“ŒS rG  r±   rp  s      €r£   r#  z-PreTrainedModel.post_init.<locals>.<dictcomp>±  s)   ø€ Ð'YÐ'YÐ'Y¹T¸QÀ¨4¨¨°!¨¨°qÐ'YÐ'YÐ'Yr¥   Úall_tied_weights_keysc                 ó0   •— i | ]\  }}‰› d |› �‰› d |› �“ŒS rG  r±   rp  s      €r£   r#  z-PreTrainedModel.post_init.<locals>.<dictcomp>´  s6   ø€ Ð2uÐ2uÐ2uÑTXÐTUÐWX°d°=°=¸Q°=°=ÀTÀ-À-ÈAÀ-À-Ð2uÐ2uÐ2ur¥   r4  r5  r2  r3  r7  r6  r8  c                 ó   •— h | ]	}‰› d |› �’Œ
S rG  r±   rI  s     €r£   ú	<setcomp>z,PreTrainedModel.post_init.<locals>.<setcomp>Ç  s#   ø€ Ð4XÐ4XÐ4XÀq¸°]°]¸q°]°]Ð4XÐ4XÐ4Xr¥   )r­   r=  r?  r>  r@  r  ÚupdaterÇ  Úbase_model_pp_planÚbase_model_tp_planÚbase_model_ep_planÚbase_model_fsdp_planÚget_expanded_tied_weights_keysru  re  r4  r5  r2  r3  r7  r6  r8  Únamed_childrenrM  Úcopyr,  Úinit_weightsÚ._backward_compatibility_gradient_checkpointing)r¢   rC  ÚplanÚ	tied_keysÚ	keep_fp32Úkeep_fp32_strictÚno_splitÚ	skip_keysÚignore_unexpectedÚignore_missingÚignore_saverJ  s              @r£   Ú	post_initzPreTrainedModel.post_init„  s>  ø€ õ ˜Tœ]Ð0¨bÑ1Ô1ˆŒÝ˜Tœ]Ð0¨bÑ1Ô1ˆŒÝ˜Tœ]Ð0¨bÑ1Ô1ˆŒÝ˜tœÐ4°"Ñ5Ô5ˆŒàŒ?˜dÐ"Ð"ØŒM× Ò  ¤Ô!?Ð!EÀ2ÑFÔFÐFØŒM× Ò  ¤Ô!?Ð!EÀ2ÑFÔFÐFØŒM× Ò  ¤Ô!?Ð!EÀ2ÑFÔFÐFØŒO×"Ò" 4¤;Ô#CÐ#IÀrÑJÔJÐJà%)×%HÒ%HÐW\Ð%HÑ%]Ô%]ˆÔ"å%(¨Ô)CÐ)IÀrÑ%JÔ%JˆÔ"Ý,/°Ô0QÐ0WÐUWÑ,XÔ,XˆÔ)å!$ TÔ%;Ð%A¸rÑ!BÔ!BˆÔÝ+.¨tÔ/OÐ/UÐSUÑ+VÔ+VˆÔ(å25°dÔ6]Ð6cÐacÑ2dÔ2dˆÔ/Ý/2°4Ô3WÐ3]Ð[]Ñ/^Ô/^ˆÔ,Ý'*¨4Ô+GÐ+MÈ2Ñ'NÔ'NˆÔ$ð !×/Ò/Ñ1Ô1ð 	Zñ 	Z‰LˆD�&å˜v z°4Ñ8Ô8Ð8ˆtð YØ”×$Ò$Ð%WÐ%WÐ%WÐ%WÀ4Ç9Â9Á;Ä;×CTÒCTÑCVÔCVÐ%WÑ%WÔ%WÑXÔXÐXÝ˜v z°4Ñ8Ô8Ð8ˆtð YØ”×$Ò$Ð%WÐ%WÐ%WÐ%WÀ4Ç9Â9Á;Ä;×CTÒCTÑCVÔCVÐ%WÑ%WÔ%WÑXÔXÐXÝ˜v z°4Ñ8Ô8Ð8ˆtð YØ”×$Ò$Ð%WÐ%WÐ%WÐ%WÀ4Ç9Â9Á;Ä;×CTÒCTÑCVÔCVÐ%WÑ%WÔ%WÑXÔXÐXÝ˜v |°TÑ:Ô:Ð:ˆtð [Ø”×&Ò&Ð'YÐ'YÐ'YÐ'YÀTÇYÂYÁ[Ä[×EVÒEVÑEXÔEXÐ'YÑ'YÔ'YÑZÔZÐZå# FÐ,CÀTÑJÔJÐJˆyð wØÔ*×1Ò1Ð2uÐ2uÐ2uÐ2uÐ\e×\jÒ\jÑ\lÔ\l×\rÒ\rÑ\tÔ\tÐ2uÑ2uÔ2uÑvÔvÐvå# FÐ,CÀTÑJÔJÐJˆyð =ØÔ*×1Ò1°)Ñ<Ô<Ð<Ý#*¨6Ð3QÐSWÑ#XÔ#XÐXÐð KØÔ1×8Ò8Ð9IÑJÔJÐJå" 6Ð+>ÀÑEÔEÐEˆxð 8ØÔ&×-Ò-¨hÑ7Ô7Ð7Ý# FÐ,IÈ4ÑPÔPÐPˆyð CØÔ0×7Ò7¸	ÑBÔBÐBõ %,¨FÐ4XÐZ^Ñ$_Ô$_Ð_Ð ð RØÔ7×>Ò>Ð?PÑQÔQÐQÝ!(¨Ð1RÐTXÑ!YÔ!YÐYˆ~ð LØÔ4×;Ò;¸NÑKÔKÐKå% fÐ.GÈÑNÔNÐNˆ{ð ZØÔ,×3Ò3Ð4XÐ4XÐ4XÐ4XÈKÐ4XÑ4XÔ4XÑYÔYÐYùð 	×ÒÑÔÐØ×;Ò;Ñ=Ô=Ð=Ð=Ð=r¥   c                 ó²   — t          | j        d¦  «        r<| j        j        j        r+| j        st          d| j        j        › d�¦  «        ‚| j        S | j        S )z:
        The full tp plan for the model's modules
        Údistributed_configzGExpert parallelism was requested (`enable_expert_parallel=True`), but `zs` does not define an expert-parallel plan. Add a `base_model_ep_plan` to its config, or disable expert parallelism.)	rµ   rÇ  rŽ  Úenable_expert_parallelr?  rÍ   r  r¦   r=  r¡   s    r£   Útp_planzPreTrainedModel.tp_planÍ  sv   € õ
 �4”;Ð 4Ñ5Ô5ð 	!¸$¼+Ô:XÔ:oð 	!Ø”=ð Ý ðZØœÔ/ðZð Zð Zñô ð ð
 ”=Ð ØŒ}Ðr¥   c                 ó   — | j         S r    )r@  r¡   s    r£   Ú	fsdp_planzPreTrainedModel.fsdp_planÜ  s
   € àŒÐr¥   c                 ó   — | j         S r    )r>  r¡   s    r£   Úpp_planzPreTrainedModel.pp_planà  s
   € àŒ}Ðr¥   rƒ  c                 ó.  — |€	i | _         d S t          |t          ¦  «        st          d¦  «        ‚|                     ¦   «         D ]D\  }}|t
          vr6t          d|› d|› dt          t          j        ¦   «         ¦  «        › �¦  «        ‚ŒEd„ |                      ¦   «         D ¦   «         }|                     ¦   «         D ]R}| 	                    dd¦  «        }d}|D ]}t          j        ||¦  «        rd	} nŒ|st          j        d
|› d�¦  «         ŒS|| _         d S )Nz&Can only set a dictionary as `tp_plan`z#Unsupported tensor parallel style 'z' for layer 'z'. Supported styles are c                 ó   — g | ]\  }}|‘ŒS r±   r±   )rþ   rJ  r\  s      r£   rK  z+PreTrainedModel.tp_plan.<locals>.<listcomp>ö  s   € ÐIÐIÐI¡g d¨A˜TÐIÐIÐIr¥   Ú*z\d+FTzLayer pattern 'z�' does not match any parameters in the model. This rule may not be applied during tensor parallelization, or may lead to dimension mismatches)r=  rs  r­   rÍ   r,  rC   r¯   r-  r  Úreplacerp  ÚmatchÚwarningsÚwarn)r¢   rƒ  Úlayer_patternÚparallel_styleÚmodel_param_namesÚregex_patternÚpattern_matchedrŠ  s           r£   r�  zPreTrainedModel.tp_planä  s†  € àˆ<ØˆDŒMØˆFÝ˜$¥Ñ%Ô%ð 	GÝÐEÑFÔFÐFð .2¯ZªZ©\¬\ð 	ð 	Ñ)ˆM˜>ØÕ%8Ð8Ð8Ý ðO¸.ð Oð OÐWdð Oð OÝ,0Õ1DÔ1IÑ1KÔ1KÑ,LÔ,LðOð Oñô ð ð 9ð JÐI°×1FÒ1FÑ1HÔ1HÐIÑIÔIÐØ!ŸYšY™[œ[ð 	ð 	ˆMà)×1Ò1°#°vÑ>Ô>ˆMØ#ˆOØ/ð ð �
Ý”8˜M¨:Ñ6Ô6ð Ø&*�OØ�Eðð #ð Ý”ðd mð dð dð dñô ð øð ˆŒˆˆr¥   c                 ór   — |€	i | _         d S t          |t          ¦  «        st          d¦  «        ‚|| _         d S )Nz&Can only set a dictionary as `pp_plan`)r>  rs  r­   rÍ   )r¢   rƒ  s     r£   r”  zPreTrainedModel.pp_plan  s@   € àˆ<ØˆDŒMØˆFÝ˜$¥Ñ%Ô%ð 	GÝÐEÑFÔFÐFàˆŒˆˆr¥   c                 ót   — t          | dd¦  «        }|€t          d¦  «        ‚|                     | |¬¦  «        S )zŽ
        Potentially dequantize the model in case it has been quantized by a quantization method that support
        dequantization.
        r˜   Nz?You need to first quantize your model in order to dequantize itrß  )rM  rÍ   Ú
dequantize)r¢   r–   r˜   s      r£   r£  zPreTrainedModel.dequantize  sC   € õ
 ˜t ^°TÑ:Ô:ˆàÐÝÐ^Ñ_Ô_Ð_à×&Ò& t°5Ð&Ñ9Ô9Ð9r¥   c                 óš   — | j         rAt          | j        dd¦  «        r-|                      ¦   «          t	          | j        d¦  «         d S d S d S )NÚgradient_checkpointingF)rA  rM  rÇ  Úgradient_checkpointing_enableÚdelattrr¡   s    r£   r‚  z>PreTrainedModel._backward_compatibility_gradient_checkpointing  sc   € ØÔ/ð 	;µG¸D¼KÐIaÐchÑ4iÔ4ið 	;Ø×.Ò.Ñ0Ô0Ð0å�D”KÐ!9Ñ:Ô:Ð:Ð:Ð:ð	;ð 	;ð 	;ð 	;r¥   Útagsc                 ó¢   — t          |t          ¦  «        r|g}| j        €g | _        |D ]%}|| j        vr| j                             |¦  «         Œ&dS )a\  
        Add custom tags into the model that gets pushed to the Hugging Face Hub. Will
        not overwrite existing tags in the model.

        Args:
            tags (`Union[list[str], str]`):
                The desired tags to inject in the model

        Examples:

        ```python
        from transformers import AutoModel

        model = AutoModel.from_pretrained("google-bert/bert-base-cased")

        model.add_model_tags(["custom", "custom-bert"])

        # Push the model to your namespace with the name "my-custom-bert".
        model.push_to_hub("my-custom-bert")
        ```
        N)rs  rª   r-  rU  )r¢   r¨  Útags      r£   Úadd_model_tagszPreTrainedModel.add_model_tags$  si   € õ, �d�CÑ Ô ð 	Ø�6ˆDàŒ?Ð"Ø ˆDŒOàð 	,ð 	,ˆCØ˜$œ/Ð)Ð)Ø”×&Ò& sÑ+Ô+Ð+øð	,ð 	,r¥   c                 ó¾  — |                      d|j        ¦  «        }|                      dd¦  «        x}�)t                               d¦  «         ||j        k    r|n|}t	          |t
          ¦  «        rt          t          |¦  «        }||_        |j        D ]}t          ||¦  «        x}�||_        Œd|v r|                      d¦  «        |_	        d|v r|                      d¦  «        |_
        |                     dd¦  «        }t          ¦   «         g}|�(|                     t          || j        ¦  «        ¦  «         |r!|                     t!          ¦   «         ¦  «         t#          ¦   «         ot$           ot&           }	|	rxt                               d	¦  «         d
dl}
|                     t/          j        ¦   «         |
j                             t7          ¦   «         ¬¦  «        t9          ¦   «         g¦  «         t;          |¦  «        5   | |fi |¤Ž}t=          |¦  «         ddd¦  «         n# 1 swxY w Y   |	r%ddlm }  ||¦  «         | !                    ¦   «          |S )zê
        All context managers that the model should be initialized under go here.

        Args:
            dtype (`torch.dtype`, *optional*):
                Override the default `dtype` and load the model under this dtype.
        r–   Útorch_dtypeNz1`torch_dtype` is deprecated! Use `dtype` instead!Úattn_implementationÚexperts_implementationrW  Fú@Detected DeepSpeed ZeRO-3: activating zero.init() for this modelr   ©Úconfig_dict_or_pathr   )Úinitialize_weights_zero3)"rX  r–   r¶  rË  rs  rª   rM  r®   rÌ  r_  rc  rÀ   rP   rU  rÒ   r¦   r<   r-   rÄ   rÈ   r·  Ú	deepspeedrN  ÚinitÚno_init_weightsÚzeroÚInitr+   rÉ   r]   rQ   Úintegrations.deepspeedr³  Útie_weights)rO  rÇ  r¯  r–   r­  rÎ  rÏ  rW  Úinit_contextsÚneeds_zero3_initr´  ri  r³  s                r£   Ú_from_configzPreTrainedModel._from_configD  sÊ  € ð —
’
˜7 F¤LÑ1Ô1ˆØ!Ÿ:š: m°TÑ:Ô:Ð:ˆKÐGÝ×ÒÐ SÑTÔTÐTà" f¤lÒ2Ð2�E�E¸ˆEÝ�e�SÑ!Ô!ð 	*Ý�E 5Ñ)Ô)ˆEð
 ˆŒØ$Ô0ð 	)ð 	)ˆNÝ% f¨nÑ=Ô=Ð=�
ÐJØ#(�
Ô øð ! FÐ*Ð*Ø*0¯*ª*Ð5JÑ*KÔ*KˆFÔ'ð $ vÐ-Ð-Ø-3¯ZªZÐ8PÑ-QÔ-QˆFÔ*ð #ŸJšJÐ':¸EÑBÔBÐå&™œÐ)ˆØÐØ× Ò Õ!2°5¸#¼,Ñ!GÔ!GÑHÔHÐHØð 	:Ø× Ò Õ!6Ñ!8Ô!8Ñ9Ô9Ð9å5Ñ7Ô7ÐhÅÐ<MÐhÕVhÐRhÐØð 	Ý�KŠKÐZÑ[Ô[Ð[ð ÐÐÐà× Ò åÔ(Ñ*Ô*Ø”N×'Ò'Õ<LÑ<NÔ<NÐ'ÑOÔOÝ#Ñ%Ô%ðñô ð õ ˜]Ñ+Ô+ð 	*ð 	*Ø�C˜Ð)Ð) &Ð)Ð)ˆEÝ" 5Ñ)Ô)Ð)ð	*ð 	*ð 	*ñ 	*ô 	*ð 	*ð 	*ð 	*ð 	*ð 	*ð 	*øøøð 	*ð 	*ð 	*ð 	*ð ð 	 ØHÐHÐHÐHÐHÐHà$Ð$ UÑ+Ô+Ð+Ø×ÒÑÔÐàˆs   ÈH+È+H/È2H/c                 ó.   — t          | | j        | ¦  «        S )z@
        `torch.nn.Module`: The main body of the model.
        )rM  r+  r¡   s    r£   r  zPreTrainedModel.base_modelŽ  s   € õ
 �t˜TÔ3°TÑ:Ô:Ð:r¥   c                 ó   — dt          | j        ¦  «        v rdS | j        D ];}t          |d¦  «        sŒdt          |¦  «        vr|                     ¦   «         r dS Œ<t          | d¦  «        r"t                               | j        › d�¦  «         dS )aÓ  
        Returns whether this model can generate sequences with `.generate()`, from the `GenerationMixin`
        or one with a similar interface.

        Under the hood, on classes where this function returns True, some generation-specific changes are triggered:
        for instance, the model instance will have a populated `generation_config` attribute.

        Returns:
            `bool`: Whether this model can generate sequences with `.generate()`.
        ÚGenerationMixinTre  rˆ   Úprepare_inputs_for_generationu6   has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From ðŸ‘‰v4.50ðŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
  - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.F)rª   Ú	__bases__rµ   re  r¶  Úwarningr¦   )rO  Úbases     r£   re  zPreTrainedModel.can_generate•  s´   € ð ¥ C¤MÑ 2Ô 2Ð2Ð2Ø�4à”Mð 	ð 	ˆDÝ˜4 Ñ0Ô0ð ØØ ­¨D©	¬	Ð1Ð1°d×6GÒ6GÑ6IÔ6IÐ1Ø�t�tøõ �3Ð7Ñ8Ô8ð 	Ý�NŠNØ”<ð 	 ð 	 ð 	 ñô ð ð ˆur¥   r±   Úflash_attn_versionÚgeneral_availability_checkÚpkg_availability_checkÚsupported_devices.Úcustom_supported_devicesÚcuda_min_major_versionc                 ó~  — |D ],\  }} |¦   «         rt                                |¦  «          dS Œ- |¦   «         sýd|› d�}	 |¦   «         st          |	› d|› d�¦  «        ‚|dk    r$t          d¦  «        st          |	› d|› d	�¦  «        ‚t	          |Ž \  }
}t          d
„ |
D ¦   «         ¦  «        st          |	› d|› d|› d�¦  «        ‚|�qt          ¦   «         ret          j         	                    ¦   «         \  }}||k     r@t          |	› d|› d|› dt          j         	                    ¦   «         › d|› d�
¦  «        ‚dS dS dS dS )aG  
        Checks whether the specified Flash Attention version is supported and if not, searches for the specific reason
        on why it failed - package import and/or device incompatibility issues.

        Args:
            flash_attn_version (`int`):
                The requested version of Flash Attention.
            general_availability_check (`Callable`):
                Checks whether our `is_available` function detects the specific FA version. Failing reasons
                are then checked for one-by-one.
            pkg_availability_check (`Callable`):
                Checks whether the package could theoretically be detected in the environment by the init structures.
                This is not a sure-fire check as device compatibility with FA is just as important.
            supported_devices (`tuple[tuple[Callable, str]]`):
                Essentially a list (for mutable kwargs reasons a tuple) of the supported devices in the format of
                `(device_availability_check, device_name)`, i.e. a pair of the associated device's name and whether
                it is available in the environment.
            custom_supported_devices (`tuple[tuple[Callable, str]]`, *optional*, defaults to `()`):
                Essentially a list (for mutable kwargs reasons a tuple) of the custom supported devices in the format of
                `(device_availability_check, info_message)`. These custom devices have custom logic outside the torch
                ecosystem either via kernels or other packages and hence have early checks for availability.
            cuda_min_major_version (`int`, *optional*):
                The minimum major cuda version supported for this version of Flash Attention. This is mostly
                affecting more recent versions which are more specialized to the features of new hardware.
        NÚFlashAttentionzG has been toggled on, but it cannot be used due to the following error:z the package for FlashAttentionz doesn't seem to be installed.rü   z2.3.3z FlashAttentionz# requires at least version `2.3.3`.c              3   ó*   K  — | ]} |¦   «         V — Œd S r    r±   )rþ   Údevice_availability_checks     r£   r   z;PreTrainedModel._flash_attn_import_error.<locals>.<genexpr>ò  s-   è è € ÐsÐsÐ;TÐ4Ð4Ñ6Ô6ÐsÐsÐsÐsÐsÐsr¥   zT is not available on CPU. Please make sure you are on any of the supported devices: rH  z  requires compute capability >= z, but found z with compute capability z.x)
r¶  r·  ÚImportErrorru   Úziprw  rx   r®   ÚcudaÚget_device_capability)r¢   rÅ  rÆ  rÇ  rÈ  rÉ  rÊ  rÎ  Úinfo_messageÚprefaceÚdevice_availability_checksÚdevice_namesÚmajorr\  s                 r£   Ú_flash_attn_import_errorz(PreTrainedModel._flash_attn_import_error¼  sS  € ðF 8Pð 	ð 	Ñ3Ð% |Ø(Ð(Ñ*Ô*ð Ý—’˜LÑ)Ô)Ð)Ø��ðð *Ð)Ñ+Ô+ð 	ð CÐ'9ð  Cð  Cð  CˆGð *Ð)Ñ+Ô+ð Ý!ØÐqÐqÐ?QÐqÐqÐqñô ð ð $ qÒ(Ð(Õ1OÐPWÑ1XÔ1XÐ(Ý! WÐ"tÐ"tÐ=OÐ"tÐ"tÐ"tÑuÔuÐuõ <?Ð@QÐ;RÑ8Ð*¨LÝÐsÐsÐXrÐsÑsÔsÑsÔsð 
Ý%Ø"ð  kð  kÐ3Eð  kð  kð  \hð  kð  kð  kñô ð ð ,Ð7Õ<SÑ<UÔ<UÐ7Ý$œz×?Ò?ÑAÔA‘H�E˜1ØÐ5Ò5Ð5Ý)Ø&ð  Vð  VÐ7Ið  Vð  Vð  lBð  Vð  Võ  PUô  PZ÷  Ppò  Ppñ  Prô  Prð  Vð  Vð  MRð  Vð  Vð  Vñô ð ð-	ð 	ð& 8Ð7Ð7Ð7à5Ð5r¥   rV  c                 ó°  — | j         s,t          | j        j        › d|› d| j        j        › d�¦  «        ‚|dvrt          d|› d�¦  «        ‚ | j        di t          |         ¤Ž |dk    rCt          | j        d¦  «        r.| j        j	        d	k    rt                               d
|› d�¦  «         | j        j        }|€t                               d
|› d�¦  «         nM|�K|t          j        t          j        fvr1t                               d|› d| j        j        › d|› d|› d�	¦  «         |s®t!          d„ |                      ¦   «         D ¦   «         ¦  «        }t%          |¦  «        dk    rp|d	         j        dk    r_d}t          |         d         D ]4\  }} |¦   «         r%d}t                               d
|› d|› d�¦  «          nŒ5|st          d
|› d�¦  «        ‚dS )a®  
        Check the availability of Flash Attention for a given model.

        Args:
            flash_attn_version (`int`):
                The requested version of Flash Attention.
            is_init_check (`bool`, *optional*):
                Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
                fully instantiated. This is needed as we also check the devices of the weights, which are only available
                later after __init__. This allows to raise proper exceptions early before instantiating the full models
                if we know that the model does not support the requested attention.
        z" does not support Flash Attention zm yet. Please request to add support where the model is hosted, on its model hub page: https://huggingface.co/zk/discussions/new or in the Transformers GitHub repo: https://github.com/huggingface/transformers/issues/new)rü   rÞ  é   zRequested Flash Attention z which is not supported.rü   Úattention_dropoutr   z*You are attempting to use Flash Attention zv with dropout. This might lead to unexpected behaviour as this is not supported on recent versions of Flash Attention.NzD without specifying a dtype. This might lead to unexpected behaviourzFlash Attention zP only supports torch.float16 and torch.bfloat16 dtypes, but the current dype in z is a&  . You should run training or inference using Automatic Mixed-Precision via the `with torch.autocast(device_type='torch_device'):` decorator, or load the model with the `dtype` argument. Example: `model = AutoModel.from_pretrained("meta-llama/Llama-3.2-1B", attn_implementation="flash_attention_z", dtype=torch.float16)`c                 ó   — h | ]	}|j         ’Œ
S r±   rÕ  rÖ  s     r£   rx  z;PreTrainedModel._flash_attn_can_dispatch.<locals>.<setcomp>0  s   € Ð!NÐ!NÐ!N°5 %¤,Ð!NÐ!NÐ!Nr¥   r   rÔ   FrÈ  Tzá with a model not initialized on GPU. Please make sure to have access to a GPU and either initialise the model on a GPU by passing a device_map or initialising the model on CPU and then moving it to GPU, e.g. with `model.to('z')`.a    with a model not initialized on GPU and with no GPU available. This is not supported yet. Please make sure to have access to a GPU and either initialise the model on a GPU by passing a device_map or initialising the model on CPU and then moving it to GPU.r±   )r:  rÍ   r  r¦   rÇ  Ú_name_or_pathrØ  rK   rµ   rÛ  r¶  rË  r–   r®   Úfloat16Úbfloat16r¯   rÙ  rß   ru  )r¢   rÅ  rV  r–   Úparam_devicesÚfound_devicerÎ  Údevice_names           r£   Ú_flash_attn_can_dispatchz(PreTrainedModel._flash_attn_can_dispatchþ  sü  € ð Ô(ð 	ÝØ”>Ô*ð nð nÐN`ð nð nØW[ÔWbÔWpðnð nð nñô ð ð  YÐ.Ð.ÝÐfÐ:LÐfÐfÐfÑgÔgÐgð 	&ˆÔ%ÐaÐaÕ(LÐM_Ô(`ÐaÐaÐað  Ò!Ð!Ý�t”{Ð$7Ñ8Ô8ð ¸T¼[Ô=ZÐ]^Ò=^Ð=^Ý×#Ò#ð~ÐASð ~ð ~ð ~ñô ð ð ”Ô!ˆØˆ=Ý×Òð VÐ=Oð  Vð  Vð  Vñô ð ð ð Ð 5µ´ÅÄÐ0OÐ#OÐ#OÝ×ÒðZÐ#5ð Zð ZØ(,¬Ô(?ðZð ZØEJðZð Zð n@ðZð Zð Zñô ð ð ð 	Ý Ð!NÐ!N¸D¿OºOÑ<MÔ<MÐ!NÑ!NÔ!NÑOÔOˆMÝ�=Ñ!Ô! QÒ&Ð&¨=¸Ô+;Ô+@ÀEÒ+IÐ+IØ$�Ý>bÐcuÔ>vØ'ô?ð 
ð 
Ñ:Ð-¨{ð 1Ð0Ñ2Ô2ð Ø'+˜Ý×+Ò+ðXÐI[ð Xð XàFQðXð Xð Xñô ð ð
 ˜ðð $ð Ý$ðVÐEWð Vð Vð Vñô ð ð ˆtr¥   c                 ó–  — | j         st          | j        j        › d�¦  «        ‚t          j        j        �”t          j                             ¦   «         dk    rrt          j	        t          j
        ¦  «        t          j	        d¦  «        k     r>t                               d¦  «         t          j        j                             d¦  «         dS )aA  
        Check the availability of SDPA for a given model.

        Args:
            is_init_check (`bool`, *optional*):
                Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
                fully instantiated. This is needed as we also check the devices of the weights, which are only available
                later after __init__. This allows to raise proper exceptions early before instantiating the full models
                if we know that the model does not support the requested attention.
        aâ   does not support an attention implementation through torch.nn.functional.scaled_dot_product_attention yet. Please request the support for this architecture: https://github.com/huggingface/transformers/issues/28005. If you believe this error is a bug, please open an issue in Transformers GitHub repository and load your model with the argument `attn_implementation="eager"` meanwhile. Example: `model = AutoModel.from_pretrained("openai/whisper-tiny", attn_implementation="eager")`Nr   z2.4.1z¦Using the `SDPA` attention implementation on multi-gpu setup with ROCM may lead to performance issues due to the FA backend. Disabling it to use alternative backends.FT)r9  rÍ   r  r¦   r®   r   ÚhiprÑ  Údevice_countÚparserƒ   r¶  rË  ÚbackendsÚenable_flash_sdp©r¢   rV  s     r£   Ú_sdpa_can_dispatchz"PreTrainedModel._sdpa_can_dispatchI  sÁ   € ð Ô"ð 	ÝØ”>Ô*ð Oð Oð Oñô ð õ ŒMÔÐ)Ý”
×'Ò'Ñ)Ô)¨AÒ-Ð-Ý”�eÔ/Ñ0Ô0µ7´=ÀÑ3IÔ3IÒIÐIå×Òð yñô ð õ ŒNÔ×0Ò0°Ñ7Ô7Ð7àˆtr¥   c                 óf   — |                       ¦   «         st          | j        j        › d�¦  «        ‚dS )zI
        Check the availability of Grouped MM for a given model.
        z1 does not support setting experts implementation.T)Ú_can_set_experts_implementationrÍ   r  r¦   r¡   s    r£   Ú_grouped_mm_can_dispatchz(PreTrainedModel._grouped_mm_can_dispatchg  s<   € ð
 ×3Ò3Ñ5Ô5ð 	lÝ ¤Ô 7ÐjÐjÐjÑkÔkÐkð ˆtr¥   c                 ó†   — | j         st          | j        j        › d�¦  «        ‚t	          ¦   «         st          d¦  «        ‚dS )aK  
        Check the availability of Flex Attention for a given model.

        Args:
            is_init_check (`bool`, *optional*):
                Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
                fully instantiated. This is needed as we also check the devices of the weights, which are only available
                later after __init__. This allows to raise proper exceptions early before instantiating the full models
                if we know that the model does not support the requested attention.
        aÄ   does not support an attention implementation through torch's flex_attention. Please request the support for this architecture: https://github.com/huggingface/transformers/issues/34809. If you believe this error is a bug, please open an issue in Transformers GitHub repository and load your model with the argument `attn_implementation="eager"` meanwhile. Example: `model = AutoModel.from_pretrained("openai/whisper-tiny", attn_implementation="eager")`z]PyTorch Flex Attention requirements in Transformers are not met. Please install torch>=2.5.0.T)r;  rÍ   r  r¦   rh   rÏ  rê  s     r£   Ú_flex_attn_can_dispatchz'PreTrainedModel._flex_attn_can_dispatchr  sg   € ð Ô'ð 	ÝØ”>Ô*ð tð tð tñô ð õ ,Ñ-Ô-ð 	ÝØoñô ð ð
 ˆtr¥   r®  rW  c           	      ó  — t          |¦  «        \  }}|t          | dd¦  «        pg v rd}|�bt          | dd¦  «        }t          |¬¦  «        rA|�?||vr;|rd|d         › �n|d         }t                               d|› d|› d	|› d
�¦  «         |}t          |¦  «        \  }}|}d}	t          |¬¦  «        r=t          j        ¦   «         D ])}
|d|
› �k    rt          |
         d         ¦   «         sd}	 nŒ*| j        rH|	rFt          ¦   «         r8t          ¦   «         s*t          |         }t          ¦   «         r|dk    rd}	|rd|› �}t          |¦  «        r‰	 |rt          ||¬¦  «         nt          ||¬¦  «         |	rt                               d|› d�¦  «         nw# t          $ r5}|	r,t!          |d         ¦  «        }
|                      |
|¬¦  «         |‚d}~ww xY w|                      ||¦  «        }t          |¬¦  «        rt          |¦  «         |S )aÈ  
        Check that the `attn_implementation` exists and is supported by the models, and try to get the kernel from hub if
        it matches hf kernels pattern.

        Args:
            attn_implementation (`str` or `None`):
                The attention implementation to check for existence/validity.
            is_init_check (`bool`, *optional*):
                Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
                fully instantiated. This is needed as we also check the devices of the weights, which are only available
                later after __init__. This allows to raise proper exceptions early before instantiating the full models
                if we know that the model does not support the requested attention.
            allow_all_kernels (`bool`, optional):
                Whether to load kernels from unverified hub repos, if `attn_implementation` is a custom kernel outside
                of the `kernels-community` hub repository.

        Returns:
            `str`: The final attention implementation to use, including potential fallbacks from sdpa to eager, or from
            None to sdpa (to potentially eager).
        r<  NT©Ú"requested_attention_implementationzpaged|r   zNThis model is compatible with the following flash attention implementations: `z"`. Automatically falling back to `z` instead of `z`.FÚflash_attention_rÆ  Úflash_attention_2)rW  z/You do not have `flash_attn` installed, using `z%` from the `kernels` library instead!r<  ©rÅ  rV  )rn   rM  rm   r¶  rË  rK   r-  r:  rg   ri   rL   rj   r=   rN   rM   rµ  r½   rã  Úget_correct_attn_implementation)r¢   r®  rV  rW  Úis_pagedÚbase_implementationÚ compatible_flash_implementationsÚdefault_flash_implementationÚapplicable_attn_implementationÚrequested_original_flash_attnÚ
fa_versionr  s               r£   r^  z5PreTrainedModel._check_and_adjust_attn_implementation�  sU  € õ. )GÐGZÑ([Ô([Ñ%ˆÐ%ð ¥7¨4Ð1TÐVZÑ#[Ô#[Ð#aÐ_aÐbÐbØ $Ðð Ð*Ý/6°tÐ=`ÐbfÑ/gÔ/gÐ,å,ÐPcÐdÑdÔdðCà4Ð@Ø'Ð/OÐOÐOð GOÐwÐBÐ=¸aÔ@ÐBÐBÐBÐTtÐuvÔTwð -õ ×#Ò#ðzð  fFð zð zØ6Rðzð zØbuðzð zð zñô ð ð 'CÐ#å(FÐGZÑ([Ô([Ñ%ˆÐ%à)<Ð&à(-Ð%Ý'ÐK^Ð_Ñ_Ô_ð 		åBÔGÑIÔIð ð �
ð (Ð+J¸jÐ+JÐ+JÒJÐJÝ@ÀÔLÐMiÔjÑlÔlð Kð 59Ð1Ø�Eøð Ô%ð	[à-ð	[õ %Ñ&Ô&ð	[õ +Ñ,Ô,ð		[õ .HÐH[Ô-\Ð*å%Ñ'Ô'ð 6Ð,?ÐCVÒ,VÐ,Vð 16Ð-àð [Ø1ZÐ:XÐ1ZÐ1ZÐ.åÐ3Ñ4Ô4ð 	Lðàð uÝ5Ø6ÐJ[ðñ ô ð ð õ 0Ð0NÐbsÐtÑtÔtÐtð 1ð Ý×'Ò'ð>ÐJhð >ð >ð >ñô ð øøõ ð ð ð à0ð nÝ!$Ð%8¸Ô%<Ñ!=Ô!=�JØ×1Ò1ÀZÐ_lÐ1ÑmÔmÐmð �øøøøðøøøð .2×-QÒ-QØ.°ñ.ô .Ð*õ
 ,ÐOmÐnÑnÔnð LÝ+Ð,JÑKÔKÐKà-Ð-s   ÅAF Æ
GÆ 0GÇGr¯  c                 ó0   — |                       |¦  «        }|S )a>  
        Check that the `experts_implementation` exists and is supported by the models.

        Args:
            experts_implementation (`str` or `None`):
                The experts implementation to check for existence/validity.
        Returns:
            `str`: The final experts implementation to use.
        )Ú"get_correct_experts_implementation)r¢   r¯  Ú!applicable_experts_implementations      r£   rb  z8PreTrainedModel._check_and_adjust_experts_implementation   s   € ð -1×,SÒ,SÐTjÑ,kÔ,kÐ)Ø0Ð0r¥   Úrequested_attentionc                 ó¾  — |€dn|}|dgt                                ¦   «         z   vryd|› d�}| j        st          | dd¦  «        r1|dz  }t	          j        ¦   «         D ]}|d|› d	|› d
�z  }Œ|d d…         }| j        r|dz  }| j        r|dz  }t          |dz   ¦  «        ‚t          |¬¦  «        rQt          j        d|¦  «        x}r:t          |                     d¦  «        ¦  «        }|                      ||¬¦  «         n]d|v r|                      |¦  «         nCd|v r?	 |                      |¦  «         n(# t          t"          f$ r}|�d|v r|‚d}Y d }~nd }~ww xY w|S )NÚsdpaÚeagerú Specified `attn_implementation="zc"` is not supported. The only possible arguments are `attn_implementation="eager"`, `"paged|eager"`Ú_supports_flash_attn_2Fú, z&`"attn_implementation=flash_attention_z0"`, `"attn_implementation=paged|flash_attention_z"`, éþÿÿÿzB, `"attn_implementation=sdpa"`, `"attn_implementation=paged|sdpa"`z(, `"attn_implementation=flex_attention"`rH  rò  z^flash_attention_(\d)$r   rö  Úflex_attention)ÚALL_ATTENTION_FUNCTIONSÚ
valid_keysr:  rM  rK   r-  r9  r;  rÍ   rm   rp  rq  r½   Úgrouprã  rð  rë  rÏ  )r¢   r  rV  Úapplicable_attentionÚmessagerþ  Ú
fa_matchedr  s           r£   r÷  z/PreTrainedModel.get_correct_attn_implementation  s*  € Ø)<Ð)D˜v˜vÐJ]ÐØ¨ yÕ3J×3UÒ3UÑ3WÔ3WÑ'WÐWÐWðAÐ3Gð Að Að Að ð
 Ô(ð '­G°DÐ:RÐTYÑ,ZÔ,Zð 'Ø˜4‘�Ý"FÔ"KÑ"MÔ"Mð Uð U�JØð   UÈ
ð   Uð   Uð  EOð   Uð   Uð   Uñ  U�G�GØ! # 2 #œ,�ØÔ"ð `ØÐ_Ñ_�ØÔ'ð FØÐEÑE�Ý˜W s™]Ñ+Ô+Ð+õ (ÐK_Ð`Ñ`Ô`ð 	/Ýœ)Ð$=Ð?SÑTÔTÐTˆJð	/õ ˜Z×-Ò-¨aÑ0Ô0Ñ1Ô1ˆJØ×)Ò)¸ZÐWdÐ)ÑeÔeÐeÐeØÐ!5Ð5Ð5Ø×(Ò(¨Ñ7Ô7Ð7Ð7ØÐ+Ð+Ð+ð/Ø×'Ò'¨Ñ6Ô6Ð6Ð6øÝ¥Ð,ð /ð /ð /Ø&Ð2°vÐATÐ7TÐ7TØ�GØ'.Ð$Ð$Ð$Ð$Ð$Ð$øøøøð/øøøð
 $Ð#s   ÄD5 Ä5EÅ
EÅEÚrequested_expertsc                 óÔ  — |€dn|}dgt          t          t          j        ¦   «         ¦  «        t          t	          j        ¦   «         ¦  «        z  ¦  «        z   }d„ |D ¦   «         }d|d         z   |d<   d                     |¦  «        }||vrd|› d|› d	�}t          |¦  «        ‚|dk    r>	 |                      ¦   «          n(# t          t          f$ r}|dk    r|‚d}Y d }~nd }~ww xY w|S )
NÚ
grouped_mmr  c                 ó   — g | ]}d |› d�‘Œ	S )z`experts_implementation="z"`r±   )rþ   Úfns     r£   rK  zFPreTrainedModel.get_correct_experts_implementation.<locals>.<listcomp>6  s$   € Ð!`Ð!`Ð!`ÈÐ"D¸bÐ"DÐ"DÐ"DÐ!`Ð!`Ð!`r¥   zand r<  r  z#Specified `experts_implementation="z5"` is not supported. The only possible arguments are rH  )	r¯   re  r?   r-  r8   r±  rÍ   rî  rÏ  )r¢   r  Úapplicable_expertsÚbase_experts_fnsÚvalid_experts_str_listÚvalid_experts_strr  r  s           r£   r   z2PreTrainedModel.get_correct_experts_implementation3  sL  € Ø->Ð-F˜\˜\ÐL]ÐØ#˜9¥t­CÕ0EÔ0JÑ0LÔ0LÑ,MÔ,MÕPSÕTmÔTrÑTtÔTtÑPuÔPuÑ,uÑ'vÔ'vÑvÐØ!`Ð!`ÐO_Ð!`Ñ!`Ô!`ÐØ%+Ð.DÀRÔ.HÑ%HÐ˜rÑ"Ø ŸIšIÐ&<Ñ=Ô=ÐØÐ%5Ð5Ð5ð(Ð6Hð (ð (Ø$ð(ð (ð (ð õ ˜WÑ%Ô%Ð%ð  Ò-Ð-ð-Ø×-Ò-Ñ/Ô/Ð/Ð/øÝ¥Ð,ð -ð -ð -Ø$¨Ò4Ð4Ø�GØ%,Ð"Ð"Ð"Ð"Ð"Ð"øøøøð-øøøð
 "Ð!s   Â+C  Ã C%Ã
C Ã C%c                 óh  — t          | dd¦  «        }t          |t          ¦  «        r|S t          j                             | j        ¦  «        }|€dS 	 t          j        |¦  «        }n# t          t          f$ r Y dS w xY wt          j        d|t          j        ¦  «        rd|v }nd}|| _        | j        S )a  Detect whether the class supports setting its attention implementation dynamically. Inspects the module
        source as a heuristic, which avoids maintaining yet another property flag. Instead, the flag is set dynamically
        on the first succesful call.
        Ú)_can_set_attn_implementation_cached_valueNFz%^class \w*Attention\w*\(nn\.Module\):z&ALL_ATTENTION_FUNCTIONS.get_interface(T)rM  rs  r¬   r
  ÚmodulesrÀ   r§   rL  Ú	getsourcer  r[  rp  rq  Ú	MULTILINEr  ©rO  Úcached_valueÚclass_moduleÚcodeÚcan_sets        r£   Ú_can_set_attn_implementationz,PreTrainedModel._can_set_attn_implementationK  sÌ   € õ ˜sÐ$OÐQUÑVÔVˆÝ�l¥DÑ)Ô)ð 	 ØÐå”{—’ s¤~Ñ6Ô6ˆàÐØ�5ð	ÝÔ$ \Ñ2Ô2ˆDˆDøÝ�Ð#ð 	ð 	ð 	Ø�5�5ð	øøøõ Œ9Ð=¸tÅRÄ\ÑRÔRð 	Ø>À$ÐFˆGˆGð ˆGà8?ˆÔ5ØÔ<Ð<ó   ÁA' Á'A<Á;A<c                 ó  — t          | dd¦  «        }t          |t          ¦  «        r|S t          j                             | j        ¦  «        }|€dS 	 t          j        |¦  «        }n# t          t          f$ r Y dS w xY wd|v }|| _        |S )a  Detect whether the class supports setting its experts implementation dynamically. Inspects the module source
        as a heuristic, which avoids maintaining yet another property flag. Instead, the flag is set dynamically
        on the first succesful call.
        Ú,_can_set_experts_implementation_cached_valueNFz@use_experts_implementation)rM  rs  r¬   r
  r  rÀ   r§   rL  r  r  r[  r'  r  s        r£   rí  z/PreTrainedModel._can_set_experts_implementationh  s¦   € õ ˜sÐ$RÐTXÑYÔYˆÝ�l¥DÑ)Ô)ð 	 ØÐå”{—’ s¤~Ñ6Ô6ˆàÐØ�5ð	ÝÔ$ \Ñ2Ô2ˆDˆDøÝ�Ð#ð 	ð 	ð 	Ø�5�5ð	øøøð 0°4Ð7ˆØ;BˆÔ8Øˆr%  c                 óÄ  — t          |t          ¦  «        s|n|                     d| j        j        ¦  «        }|| j        j        k    r`|                      ¦   «         s(t                               | j        j	        › d�¦  «         n$|  
                    |d|¬¦  «        }|| j        _        |                      ¦   «         D �]}|| u�rt          |t          ¦  «        rü|j        j        | j        j        k    rât          |j        d¦  «        sÍ|                     ¦   «         s(t                               |j        j	        › d�¦  «         n…|}t          |t          ¦  «        rM| j        j        D ]@}t!          | j        |¦  «        |j        u r"|                     ||j        j        ¦  «        } nŒA|                     |¦  «        }||j        _        d|j        _        �Œ| j        j        D ]ü}t!          | j        |¦  «        x}�ãt          |t          ¦  «        s|n|                     ||j        ¦  «        }t          |d¦  «        s�||j        k    r„|dgt&                               ¦   «         z   vr<t+          d	|› d
|› dt-          t&                               ¦   «         ¦  «        › �¦  «        ‚||_        t                               d|› d|› d�¦  «         Œêt          |d¦  «        r|`ŒýdS )a£  
        Set the requested `attn_implementation` for this model.

        Args:
            attn_implementation (`str` or `dict`):
                The attention implementation to set for this model. It can be either a `str`, in which case it will be
                dispatched to all submodels if relevant, or a `dict` where keys are the sub_configs name, in which case each
                submodel will dispatch the corresponding value.
            allow_all_kernels (`bool`, optional):
                Whether to load kernels from unverified hub repos, if `attn_implementation` is a custom kernel outside
                of the `kernels-community` hub repository.
        r¦  zØ does not support setting its attention implementation dynamically, because it does not follow the functional approach based on AttentionInterface (see https://huggingface.co/docs/transformers/en/attention_interface)FrU  Ú_attn_was_changedTNr  r  z"` is not supported for zd. The only possible arguments are "eager" (manual attention implementation)or one of the following: z8We set the attention implementation for the sub-config `z` to `zŽ` without finding the associated sub-model. For this reason we could not check if the model supports it. You may encounter undefined behavior.)rs  r­   rÀ   rÇ  r_  r$  r¶  rÃ  r  r¦   r^  ra  r  rˆ   rµ   rÌ  rM  r÷  r)  r  r  rÍ   r¯   )r¢   r®  rW  Úrequested_implementationrP  Úsub_implementationÚsubconfig_keyÚ	subconfigs           r£   Úset_attn_implementationz'PreTrainedModel.set_attn_implementation€  sœ  € õ Ð1µ4Ñ8Ô8ðOÐÐà$×(Ò(¨¨T¬[Ô-MÑNÔNð 	!ð $ t¤{Ô'GÒGÐGà×4Ò4Ñ6Ô6ð UÝ—’Ø”~Ô.ð \ð \ð \ñô ð ð ð ,0×+UÒ+UØ,¸EÐUfð ,Vñ ,ô ,Ð(ð =U�”Ô9ð Ÿš™œð "	:ñ "	:ˆIð  Ð%Ñ%Ý˜y­/Ñ:Ô:ð &àÔ$Ô.°$´+Ô2GÒGÐGå 	Ô 0Ð2EÑFÔFð Hð
 !×=Ò=Ñ?Ô?ð XÝ—N’NØ$Ô.Ô7ð `ð `ð `ñô ð ð ð *BÐ&Ý!Ð"5µtÑ<Ô<ð &Ø-1¬[Ô-Dð &ð &˜Må& t¤{°MÑBÔBÀiÔFVÐVÐVØ5H×5LÒ5LØ$1°9Ô3CÔ3Xñ6"ô 6"Ð 2ð !& ð	  Wð *3×)RÒ)RÐSeÑ)fÔ)fÐ&ØEW�IÔ$ÔBð 6:�	Ô Ô2ùð "œ[Ô4ð 	8ð 	8ˆMÝ$ T¤[°-Ñ@Ô@Ð@�	ÐMõ &Ð&9½4Ñ@Ô@ð`Ð,Ð,à,×0Ò0°À	Ô@^Ñ_Ô_ð #õ   	Ð+>Ñ?Ô?ð8ð +¨iÔ.LÒLÐLà)°'°Õ=T×=_Ò=_Ñ=aÔ=aÑ1aÐaÐaÝ(ðeÐ?Qð eð eÐkxð eð eå8<Õ=T×=_Ò=_Ñ=aÔ=aÑ8bÔ8bðeð eñô ð ð
 ?Q�IÔ;Ý—N’Nð@ÐS`ð @ð @Ðhzð @ð @ð @ñô ð ð õ ˜yÐ*=Ñ>Ô>ð 8Ø%Ð7øð9	8ð 	8r¥   c                 ó„   — d| j         j        i}| j         j        D ]$}t          | j         |d¦  «        }|�
|j        ||<   Œ%|S )aU  
        Return the experts implementation of this model and its submodels, as a `dict` in the form accepted by
        `set_experts_implementation` (`""` for this model, and one entry per sub_config). This is the counterpart of
        `set_experts_implementation`, e.g. to snapshot the current implementation and later restore it.
        r¦  N)rÇ  rc  rÌ  rM  )r¢   r¯  r,  r-  s       r£   Úget_experts_implementationz*PreTrainedModel.get_experts_implementationæ  sX   € ð #% d¤kÔ&IÐ!JÐØ!œ[Ô4ð 	Zð 	ZˆMÝ ¤¨]¸DÑAÔAˆIØÐ$Ø8AÔ8YÐ& }Ñ5øØ%Ð%r¥   c                 ó   — t          |t          ¦  «        s|n|                     d| j        j        ¦  «        }| j        j        }d||fv r||k    rt          d|›d|›d�¦  «        ‚|| j        j        k    r5|                      |¦  «        }|                      ¦   «         r|| j        _        |  	                    ¦   «         D ]Î}|| urÈt          |t          ¦  «        r³|j        j        | j        j        k    r™|                     ¦   «         r…|}t          |t          ¦  «        rM| j        j        D ]@}t          | j        |¦  «        |j        u r"|                     ||j        j        ¦  «        } nŒA|                     |¦  «        }||j        _        ŒÏdS )aÃ  
        Set the requested `experts_implementation` for this model.

        Args:
            experts_implementation (`str` or `dict`):
                The experts implementation to set for this model. It can be either a `str`, in which case it will be
                dispatched to all submodels if relevant, or a `dict` where keys are the sub_configs name, in which case each
                submodel will dispatch the corresponding value.
        r¦  Údeepgemm_megamoez*Cannot switch experts implementation from ú to z at runtime: `deepgemm_megamoe` is a load-time choice. Reload via `from_pretrained(..., experts_implementation=...)` to switch.N)rs  r­   rÀ   rÇ  rc  r{  rb  rí  rd  r  rˆ   r  rÌ  rM  r   )r¢   r¯  r*  ÚcurrentrP  r+  r,  s          r£   Úset_experts_implementationz*PreTrainedModel.set_experts_implementationó  sñ  € õ Ð4µdÑ;Ô;ðUÐ"Ð"à'×+Ò+¨B°´Ô0SÑTÔTð 	!ð ”+Ô5ˆØ 'Ð+CÐ!DÐDÐDÈÐTlÒIlÐIlÝðP¸Wð Pð PÐLdð Pð Pð Pñô ð ð $ t¤{Ô'JÒJÐJØ'+×'TÒ'TÐUmÑ'nÔ'nÐ$ð ×3Ò3Ñ5Ô5ð Xà?W�”Ô<ð Ÿš™œð 	Wð 	WˆIð  Ð%Ð%Ý˜y­/Ñ:Ô:ð &àÔ$Ô.°$´+Ô2GÒGÐGØ×=Ò=Ñ?Ô?ð Hð &>Ð"ÝÐ4µdÑ;Ô;ð "Ø)-¬Ô)@ð "ð "˜å" 4¤;°Ñ>Ô>À)ÔBRÐRÐRØ1G×1KÒ1KØ -¨yÔ/?Ô/Wñ2ô 2Ð.ð "˜Eð	 Sð &/×%QÒ%QÐRdÑ%eÔ%eÐ"ØDV�	Ô ÔAøð+	Wð 	Wr¥   c                 óD  — d„ }g }t          ¦   «         }d}|                      ¦   «         D ]´}t          |t          ¦  «        rt	          |d¦  «        sŒ(	 |                     ¦   «         }n# t          $ r Y ŒJw xY w|�t	          |d¦  «        sŒat          |¦  «        }||v rŒu|                     |¦  «         | 	                    | 
                    |¦  «        ¦  «         d}Œµ|| _        |r|d         | _        |s)t                               | j        j        › d�¦  «         dS dS )	zŸ
        Enables the gradients for the input embeddings. This is useful for fine-tuning adapter weights while keeping
        the model weights fixed.
        c                 ó0   — |                      d¦  «         d S ©NT)Úrequires_grad_)rC  ÚinputÚoutputs      r£   Úmake_inputs_require_gradszMPreTrainedModel.enable_input_require_grads.<locals>.make_inputs_require_grads5	  s   € Ø×!Ò! $Ñ'Ô'Ð'Ð'Ð'r¥   Fr  NÚregister_forward_hookTr   a   does not expose input embeddings. Gradients cannot flow back to the token embeddings when using adapters or gradient checkpointing. Override `get_input_embeddings` to fully support those features, or set `_input_embed_layer` to the attribute name that holds the embeddings.)re  r  rs  rˆ   rµ   r  r  rt  rW  rU  r=  Ú_require_grads_hooksÚ_require_grads_hookr¶  rË  r  r¦   )r¢   r<  ÚhooksÚseen_modulesÚfound_embeddingsrC  Úinput_embeddingsÚembedding_ids           r£   Úenable_input_require_gradsz*PreTrainedModel.enable_input_require_grads/	  s€  € ð	(ð 	(ð 	(ð ˆÝ‘u”uˆØ Ðà—l’l‘n”nð 	$ð 	$ˆFÝ˜v¥Ñ7Ô7ð ½GÀFÐLbÑ<cÔ<cð ØðØ#)×#>Ò#>Ñ#@Ô#@Ð Ð øÝ&ð ð ð Ø�ðøøøð  Ð'­wÐ7GÐI`Ñ/aÔ/aÐ'ØåÐ.Ñ/Ô/ˆLØ˜|Ð+Ð+Øà×Ò˜\Ñ*Ô*Ð*Ø�LŠLÐ)×?Ò?Ð@YÑZÔZÑ[Ô[Ð[Ø#ÐÐà$)ˆÔ!Øð 	0à',¨Q¤xˆDÔ$Øð 	Ý×ÒØ”>Ô*ð wð wð wñô ð ð ð ð	ð 	s   ÁA(Á(
A5Á4A5c                 ó˜   — t          | dd¦  «        }|sdS |D ]}|                     ¦   «          Œg | _        t          | d¦  «        r| `dS dS )z4
        Removes the `_require_grads_hook`.
        r>  Nr?  )rM  Úremover>  rµ   r?  )r¢   r@  Úhooks      r£   Údisable_input_require_gradsz+PreTrainedModel.disable_input_require_grads[	  sr   € õ ˜Ð4°dÑ;Ô;ˆØð 	ØˆFàð 	ð 	ˆDØ�KŠK‰MŒMˆMˆMà$&ˆÔ!Ý�4Ð.Ñ/Ô/ð 	)ØÐ(Ð(Ð(ð	)ð 	)r¥   Úmodalityc                 ó:  — |dv rg d¢}n$|dk    rg d¢}n|€ddg}nt          d|› �¦  «        ‚|D ]$}t          | |¦  «        rt          | |¦  «        c S Œ%| j        | ur=t          | j        d	¦  «        r(| j                             |¬
¦  «        }|| j        k    r|S | S )ai  
        Best-effort lookup of the *encoder* module. If provided with `modality` argument,
        it looks for a modality-specific encoder in multimodal models (e.g. "image_encoder")
        By default the function returns model's text encoder if any, and otherwise returns `self`.

        Possible `modality` values are "image", "video" and "audio".
        ©ÚimageÚvideo©Úvision_towerÚvisualÚvision_modelÚvision_encoderÚimage_towerÚaudio)Úaudio_towerÚaudio_encoderÚspeech_encoderNÚtext_encoderÚencoderúHUnnrecognized modality, has to be "image", "video" or "audio" but found Úget_encoder©rJ  )rÍ   rµ   rM  r  r\  )r¢   rJ  Úpossible_module_namesrJ  Úbase_encoders        r£   r\  zPreTrainedModel.get_encoderj	  sõ   € ð Ð)Ð)Ð)Ø$oÐ$oÐ$oÐ!Ð!Ø˜Ò Ð Ø$VÐ$VÐ$VÐ!Ð!ØÐØ%3°YÐ$?Ð!Ð!åÐrÐhpÐrÐrÑsÔsÐsà)ð 	+ð 	+ˆDÝ�t˜TÑ"Ô"ð +Ý˜t TÑ*Ô*Ð*Ð*Ð*ð+ð Œ? $Ð&Ð&­7°4´?ÀMÑ+RÔ+RÐ&Øœ?×6Ò6ÀÐ6ÑIÔIˆLð ˜tœÒ.Ð.Ø#Ð#ð ˆr¥   c                 ó<  — |dv rg d¢}n$|dk    rddg}n|€ddg}nt          d	|› �¦  «        ‚|D ]&}t          | |¦  «        rt          | ||¦  «          dS Œ'| j        | ur<t          | j        d
¦  «        r| j                             ||¬¦  «         dS || _        dS dS )zS
        Symmetric setter. Mirrors the lookup logic used in `get_encoder`.
        rL  rO  rU  rV  rW  NrY  rZ  r[  Úset_encoderr]  )rÍ   rµ   r�  r  ra  ri  )r¢   rZ  rJ  r^  rJ  s        r£   ra  zPreTrainedModel.set_encoderŠ	  sú   € ð Ð)Ð)Ð)Ø$oÐ$oÐ$oÐ!Ð!Ø˜Ò Ð Ø%2°OÐ$DÐ!Ð!ØÐØ%3°YÐ$?Ð!Ð!åÐrÐhpÐrÐrÑsÔsÐsà)ð 	ð 	ˆDÝ�t˜TÑ"Ô"ð Ý˜˜d GÑ,Ô,Ð,Ø��ðð Œ? $Ð&Ð&Ý�t”¨Ñ6Ô6ð %Ø”×+Ò+¨G¸hÐ+ÑGÔGÐGÐGÐGà$�”
�
�
ð	 'Ð&r¥   c                 óÊ   — g d¢}|D ]$}t          | |¦  «        rt          | |¦  «        c S Œ%| j        | ur.t          | j        d¦  «        r| j                             ¦   «         S | S )aœ  
        Best-effort lookup of the *decoder* module.

        Order of attempts (covers ~85 % of current usages):

        1. `self.decoder/self.language_model/self.text_model`
        2. `self.base_model`                  (many wrappers store the decoder here)
        3. `self.base_model.get_decoder()`    (nested wrappers)
        4. fallback: raise for the few exotic models that need a bespoke rule
        )r  Ú
text_modelÚdecoderÚtext_decoderÚget_decoder)rµ   rM  r  rf  )r¢   r^  rJ  s      r£   rf  zPreTrainedModel.get_decoder¤	  sˆ   € ð !\Ð [Ð [ÐØ)ð 	+ð 	+ˆDÝ�t˜TÑ"Ô"ð +Ý˜t TÑ*Ô*Ð*Ð*Ð*ð+ð Œ? $Ð&Ð&­7°4´?ÀMÑ+RÔ+RÐ&Ø”?×.Ò.Ñ0Ô0Ð0ð ˆr¥   c                 óæ   — g d¢}|D ]&}t          | |¦  «        rt          | ||¦  «          dS Œ'| j        | ur:t          | j        d¦  «        r| j                             |¦  «         dS || _        dS dS )zS
        Symmetric setter. Mirrors the lookup logic used in `get_decoder`.
        )r  rc  rd  NÚset_decoder)rµ   r�  r  rh  ri  )r¢   rd  r^  rJ  s       r£   rh  zPreTrainedModel.set_decoder»	  s¢   € ð
 !LÐ KÐ KÐØ)ð 	ð 	ˆDÝ�t˜TÑ"Ô"ð Ý˜˜d GÑ,Ô,Ð,Ø��ðð Œ? $Ð&Ð&Ý�t”¨Ñ6Ô6ð %Ø”×+Ò+¨GÑ4Ô4Ð4Ð4Ð4à$�”
�
�
ð	 'Ð&r¥   c           	      óÈ  — t          | j        d¦  «        r| j        j        pd}nlt          | j        d¦  «        r| j        j        }nJt          | j        d¦  «        r| j        j        }n(t          | j                             ¦   «         dd¦  «        }t          |t          j	        t          j
        t          j        t          j        t          j        t          j        f¦  «        rQt          |dd¦  «        �t          j        |j        d|¬¦  «         |j        �t          j        |j        ¦  «         dS dS t          |t          j        ¦  «        rN|                     ¦   «         D ]7\  }}d|v rt          j        |¦  «         Œd	|v rt          j        |d¦  «         Œ8dS t          |t          j        ¦  «        rct          j        |j        d|¬¦  «         |j        �<t          |j        d
d¦  «        s(t          j        |j        |j                 ¦  «         dS dS dS t          |t          j        ¦  «        r|                     ¦   «          dS t          |t          j        t          j        t          j        t          j        f¦  «        sd|j         j!        v sd|j         j!        v r´t          |dd¦  «        �t          j"        |j        ¦  «         t          |d	d¦  «        �t          j        |j        ¦  «         t          |dd¦  «        �Mt          j        |j#        ¦  «         t          j"        |j$        ¦  «         t          j        |j%        ¦  «         dS dS d|j         j!        v r}t          |d¦  «        ro|j&        dk    rtN          |j&                 n|j(        } ||j        ¦  «        \  }}t          j)        |j*        |¦  «         t          j)        |j+        |¦  «         dS dS dS )ad  
        Initialize the weights. This is quite general on purpose, in the spirit of what we usually do. For more complex
        initialization scheme, it should be overridden by the derived `PreTrainedModel` class. In case a model adds an explicit
        `nn.Parameter`, this method should also be overridden in order to initialize it correctly.
        Úinitializer_rangeg{®Gáz”?Úinit_stdÚinitializer_factorÚweightNg        ©ÚmeanÚstdÚbiasÚ_is_hf_initializedFÚ	LayerNormÚRMSNormÚrunning_meanÚRotaryEmbeddingÚoriginal_inv_freqÚdefault),rµ   rÇ  rj  rk  rl  rM  Úget_text_configrs  r   ÚLinearÚConv1dÚConv2dÚConv3dÚConvTranspose1dÚConvTranspose2drµ  Únormal_rm  rq  Úzeros_ÚLSTMr  Úxavier_uniform_Ú	constant_r   Úpadding_idxÚMultiheadAttentionÚ_reset_parametersÚ	GroupNormÚBatchNorm1dÚBatchNorm2dÚBatchNorm3dr  r¦   Úones_ru  Úrunning_varÚnum_batches_trackedÚ	rope_typerO   Úcompute_default_rope_parametersÚcopy_Úinv_freqrw  )r¢   rC  rp  rJ  r×  Úrope_fnÚbuffer_valuer\  s           r£   Ú_init_weightszPreTrainedModel._init_weightsÌ	  s¾  € õ �4”;Ð 3Ñ4Ô4ð 	TØ”+Ô/Ð7°4ˆCˆCÝ�T”[ *Ñ-Ô-ð 	TØ”+Ô&ˆCˆCÝ�T”[Ð"6Ñ7Ô7ð 	TØ”+Ô0ˆCˆCõ ˜$œ+×5Ò5Ñ7Ô7Ð9LÈdÑSÔSˆCå�f�rœy­"¬)µR´YÅÄ	Í2ÔK]Õ_aÔ_qÐrÑsÔsð -	?Ý�v˜x¨Ñ.Ô.Ð:Ý”˜Vœ]°¸#Ð>Ñ>Ô>Ð>ØŒ{Ð&Ý”˜FœKÑ(Ô(Ð(Ð(Ð(ð 'Ð&å˜¥¤Ñ(Ô(ð (	?Ø%×6Ò6Ñ8Ô8ð /ð /‘��eØ˜tÐ#Ð#ÝÔ(¨Ñ/Ô/Ð/Ð/Ø˜t�^�^Ý”N 5¨#Ñ.Ô.Ð.øð	/ð /õ
 ˜¥¤Ñ-Ô-ð "	?ÝŒL˜œ¨S°cÐ:Ñ:Ô:Ð:àÔ!Ð-µg¸f¼mÐMaÐchÑ6iÔ6iÐ-Ý”˜FœM¨&Ô*<Ô=Ñ>Ô>Ð>Ð>Ð>ð .Ð-Ð-Ð-å˜¥Ô 5Ñ6Ô6ð 	?à×$Ò$Ñ&Ô&Ð&Ð&Ð&õ �v¥¤­b¬n½b¼nÍbÌnÐ]Ñ^Ô^ð	?à˜fÔ.Ô7Ð7Ð7Ø˜FÔ,Ô5Ð5Ð5õ �v˜x¨Ñ.Ô.Ð:Ý”
˜6œ=Ñ)Ô)Ð)Ý�v˜v tÑ,Ô,Ð8Ý”˜FœKÑ(Ô(Ð(å�v˜~¨tÑ4Ô4Ð@Ý”˜FÔ/Ñ0Ô0Ð0Ý”
˜6Ô-Ñ.Ô.Ð.Ý”˜FÔ6Ñ7Ô7Ð7Ð7Ð7ð AÐ@ð
  &Ô"2Ô";Ð;Ð;ÅÈÐPcÑ@dÔ@dÐ;ð Ô# yÒ0Ð0õ $ FÔ$4Ô5Ð5àÔ;ð ð
 &˜g f¤mÑ4Ô4‰OˆL˜!ÝŒJ�v”¨Ñ5Ô5Ð5ÝŒJ�vÔ/°Ñ>Ô>Ð>Ð>Ð>ð <Ð;Ð;Ð;r¥   Úis_custom_codec                 ó.  — t          |dd¦  «        rdS |rct          d„ |                     d¬¦  «        D ¦   «         ¦  «        r6t          d„ |                     d¬¦  «        D ¦   «         ¦  «        r	d|_        dS |                      |¦  «         d|_        dS )zM
        Initialize the weights if they are not already initialized.
        rr  FNc              3   ó8   K  — | ]}t          |d d¦  «        V — ŒdS )rr  FN©rM  rÖ  s     r£   r   z6PreTrainedModel._initialize_weights.<locals>.<genexpr>
  s/   è è € ÐnÐnÀE•G˜EÐ#7¸Ñ?Ô?ÐnÐnÐnÐnÐnÐnr¥   )Úrecursec              3   ó<   K  — | ]}|®t          |dd¦  «        V — Œd S )Nrr  Fr™  )rþ   Úbuffers     r£   r   z6PreTrainedModel._initialize_weights.<locals>.<genexpr>
  sA   è è € ð ð àØÐ%õ ˜Ð 4°eÑ<Ô<à%Ð%Ð%Ð%ðð r¥   T)rM  ÚallrÙ  Úbuffersrr  r•  )r¢   rC  r–  s      r£   Ú_initialize_weightsz#PreTrainedModel._initialize_weights
  sÈ   € õ �6Ð/°Ñ7Ô7ð 	ØˆFð ð
	åÐnÐnÈV×M^ÒM^ÐglÐM^ÑMmÔMmÐnÑnÔnÑnÔnð
	õ ð ð à$Ÿnšn°U˜nÑ;Ô;ðñ ô ñ ô ð
	ð )-ˆFÔ%ØˆFà×Ò˜6Ñ"Ô"Ð"Ø$(ˆÔ!Ð!Ð!r¥   c                 ó^  ‡— t          t          j        j        d¦  «        sYdt          j        dt          t          j        t
          gdf         dt
          fˆfd„Št          t          j        j        d‰¦  «         t          | d¦  «        } || j        |  	                    ¦   «         ¦  «         dS )a  
        This is equivalent to calling `self.apply(self._initialize_weights)`, but correctly handles composite models.
        This function dynamically dispatches the correct `init_weights` function to the modules as we advance in the
        module graph along the recursion. It can handle an arbitrary number of sub-models. Without it, every composite
        model would have to recurse a second time on all sub-models explicitly in the outer-most `_init_weights`, which
        is extremely error prone and inefficient.
        Úsmart_applyrC  r  Nr–  c                 ó¸   •— |                       ¦   «         D ]7}t          |t          ¦  «        r ‰||j        |¦  «         Œ* ‰|||¦  «         Œ8 || |¦  «         | S r    )Úchildrenrs  rˆ   rŸ  )rC  r  r–  Úchildr¡  s       €r£   r¡  z7PreTrainedModel.initialize_weights.<locals>.smart_apply2
  sv   ø€ Ø#Ÿ_š_Ñ.Ô.ð ?ð ?�Eå! %­Ñ9Ô9ð ?Ø#˜ E¨5Ô+DÀnÑUÔUÐUÐUà#˜ E¨2¨~Ñ>Ô>Ð>Ð>Ø��6˜>Ñ*Ô*Ð*Ø�r¥   )
rµ   r®   r   r'  r   r¬   r�  rM  rŸ  r–  )r¢   Úsmart_apply_fnr¡  s     @r£   Úinitialize_weightsz"PreTrainedModel.initialize_weights%
  s³   ø€ õ •u”x”¨Ñ6Ô6ð 	Að¥B¤Ið µ8½R¼YÍÐ<MÈtÐ<SÔ3Tð Õfjð ð ð ð ð ð õ •E”H”O ]°KÑ@Ô@Ð@õ !  }Ñ5Ô5ˆàˆ�tÔ/°×1DÒ1DÑ1FÔ1FÑGÔGÐGÐGÐGr¥   rn  c                 óæ  ‡‡‡‡— |r†i }|                       d¬¦  «        D ]k\  Š}t          |t          ¦  «        rQ|                     d¬¦  «        }‰dk    r ˆfd„|                     ¦   «         D ¦   «         }|                     |¦  «         Œl|S | j        }t          | j        dd¦  «        }|si S |€i S t          j
        d¦  «        Št          ˆfd	„|                     ¦   «         |                     ¦   «         z  D ¦   «         ¦  «        r|                     ¦   «         S i }d
„ |                      d¬¦  «        D ¦   «         d„ |                      d¬¦  «        D ¦   «         z  }|                     ¦   «         D ]ý\  ŠŠd‰z   Šd‰z   Št#          t%          ˆfd„|¦  «        ¦  «        }t#          t%          ˆfd„|¦  «        ¦  «        }	t'          |¦  «        dk    r6t'          |	¦  «        dk    r#t'          |	¦  «        t'          |¦  «        z  dk    rt)          d‰› d‰› d|› d|	› �¦  «        ‚t+          |	t-          |¦  «        ¦  «        D ],\  }
}||                     ¦   «         v r||         ||
<   Œ'|||
<   Œ-Œþ|S )aÿ
  
        Return the expanded tied weight keys (in case they contain modules or regex patterns) for only the current
        model, or recursively for all submodels if `all_submodels=True` (i.e. it will re-check the config values for all
        submodels).

        For almost all models, we only require to tie the embeddings, so the model has an internal property
        `_tied_weights_keys = {"lm_head.weight": "model.embed_tokens.weight"}`. In this case, the mapping is already
        "expanded", i.e. it already contains full parameters, and this function will simply return a copy of the property.
        For more complex patterns, e.g. for `DFineForObjectDetection`, we have the following attribute
        ```
        _tied_weights_keys = {
            r"bbox_embed.(?![0])\d+": "bbox_embed.0",
            r"class_embed.(?![0])\d+": "class_embed.0",
            "model.decoder.class_embed": "class_embed",
            "model.decoder.bbox_embed": "bbox_embed",
        }
        ```
        In this case, the function looks up all the model's parameters and buffers, and matches all the params,
        returning the following:
        ```
        {
            'bbox_embed.1.layers.0.bias': 'bbox_embed.0.layers.0.bias',
            'bbox_embed.1.layers.0.weight': 'bbox_embed.0.layers.0.weight',
            'bbox_embed.1.layers.1.bias': 'bbox_embed.0.layers.1.bias',
            'bbox_embed.1.layers.1.weight': 'bbox_embed.0.layers.1.weight',
            'bbox_embed.1.layers.2.bias': 'bbox_embed.0.layers.2.bias',
            'bbox_embed.1.layers.2.weight': 'bbox_embed.0.layers.2.weight',
            'bbox_embed.2.layers.0.bias': 'bbox_embed.0.layers.0.bias',
            'bbox_embed.2.layers.0.weight': 'bbox_embed.0.layers.0.weight',
            ...
            'class_embed.1.bias': 'class_embed.0.bias',
            'class_embed.1.weight': 'class_embed.0.weight',
            'class_embed.2.bias': 'class_embed.0.bias',
            'class_embed.2.weight': 'class_embed.0.weight',
            ...
            'model.decoder.class_embed.0.bias': 'class_embed.0.bias',
            'model.decoder.class_embed.0.weight': 'class_embed.0.weight',
            'model.decoder.class_embed.1.bias': 'class_embed.0.bias',
            'model.decoder.class_embed.1.weight': 'class_embed.0.weight',
            ...
            'model.decoder.bbox_embed.0.layers.0.bias': 'bbox_embed.0.layers.0.bias',
            'model.decoder.bbox_embed.0.layers.0.weight': 'bbox_embed.0.layers.0.weight',
            'model.decoder.bbox_embed.0.layers.1.bias': 'bbox_embed.0.layers.1.bias',
            'model.decoder.bbox_embed.0.layers.1.weight': 'bbox_embed.0.layers.1.weight',
            ...
        }
        ```
        i.e. all the parameters matching the regex and modules patterns in `_tied_weights_keys`
        F)Úremove_duplicaterm  r¦  c                 ó0   •— i | ]\  }}‰› d |› �‰› d |› �“ŒS rG  r±   )rþ   r!  r"  Úprefixs      €r£   r#  zBPreTrainedModel.get_expanded_tied_weights_keys.<locals>.<dictcomp>|
  s@   ø€ ð 1ð 1ð 1ÙAEÀÀA˜v˜O˜O¨˜O˜O°¨_¨_¸¨_¨_ð1ð 1ð 1r¥   Útie_word_embeddingsNz ^[A-Za-z0-9_\.]+(weight)|(bias)$c              3   óB   •K  — | ]}‰                      |¦  «        V — Œd S r    )r™  )rþ   r!  Úcommon_case_regexs     €r£   r   zAPreTrainedModel.get_expanded_tied_weights_keys.<locals>.<genexpr>�
  s2   øè è € Ð_Ð_¨aÐ ×&Ò& qÑ)Ô)Ð_Ð_Ð_Ð_Ð_Ð_r¥   c                 ó   — h | ]\  }}|’ŒS r±   r±   ©rþ   r!  r\  s      r£   rx  zAPreTrainedModel.get_expanded_tied_weights_keys.<locals>.<setcomp>•
  s   € ÐWÐWÐW¡  A˜1ÐWÐWÐWr¥   c                 ó   — h | ]\  }}|’ŒS r±   r±   r¯  s      r£   rx  zAPreTrainedModel.get_expanded_tied_weights_keys.<locals>.<setcomp>•
  s/   € ð [
ð [
ð [
Ù�!�QˆAð[
ð [
ð [
r¥   ú^c                 ó.   •— t          j        ‰| ¦  «        S r    ro  )ÚxÚsource_names    €r£   r  z@PreTrainedModel.get_expanded_tied_weights_keys.<locals>.<lambda>œ
  ó   ø€ µB´I¸kÈ1Ñ4MÔ4M€ r¥   c                 ó.   •— t          j        ‰| ¦  «        S r    ro  )r³  Útarget_names    €r£   r  z@PreTrainedModel.get_expanded_tied_weights_keys.<locals>.<lambda>�
  rµ  r¥   r   zAThere is an issue with your definition of `tie_weights_keys` for ú:z. We found z to tie into )rL  rs  rˆ   r~  r,  ry  rE  rM  rÇ  rp  Úcompiler�  r-  rÞ   r€  r  Únamed_buffersr  Úfilterrß   rÍ   rÐ  r   )r¢   rn  Úexpanded_tied_weightsrP  Úsubmodel_tied_weightsÚtied_mappingr«  Úall_param_namesÚsource_paramsÚtarget_paramsÚtarget_nÚsource_nr­  rª  r´  r·  s               @@@@r£   r~  z.PreTrainedModel.get_expanded_tied_weights_keysC
  sK  øøøø€ ðd ð 	)Ø$&Ð!Ø%)×%7Ò%7ÈÐ%7Ñ%OÔ%Oð Hð HÑ!�˜	Ý˜i­Ñ9Ô9ð Hà,5×,TÒ,TÐchÐ,TÑ,iÔ,iÐ)Ø ’|�|ð1ð 1ð 1ð 1ØI^×IdÒIdÑIfÔIfð1ñ 1ô 1Ð-ð *×0Ò0Ð1FÑGÔGÐGøØ(Ð(àÔ.ˆõ & d¤kÐ3HÈ%ÑPÔPÐØ"ð 	ØˆIàÐ!ØˆIõ œJÐ'JÑKÔKÐÝÐ_Ð_Ð_Ð_°<×3DÒ3DÑ3FÔ3FÈ×I\ÒI\ÑI^ÔI^Ñ3^Ð_Ñ_Ô_Ñ_Ô_ð 	'Ø×$Ò$Ñ&Ô&Ð&ð !#ÐØWÐW¨×)>Ò)>ÐPUÐ)>Ñ)VÔ)VÐWÑWÔWð [
ð [
Ø×,Ò,¸eÐ,ÑDÔDð[
ñ [
ô [
ñ 
ˆð )5×(:Ò(:Ñ(<Ô(<ð 	?ð 	?Ñ$ˆK˜Ø Ñ+ˆKØ Ñ+ˆKå"¥6Ð*MÐ*MÐ*MÐ*MÈÑ#_Ô#_Ñ`Ô`ˆMÝ"¥6Ð*MÐ*MÐ*MÐ*MÈÑ#_Ô#_Ñ`Ô`ˆMå˜Ñ&Ô&¨Ò*Ð*Ý˜=Ñ)Ô)¨AÒ-Ð-Ý�}Ñ%Ô%­¨MÑ(:Ô(:Ñ:¸aÒ?Ð?å ðLÐXcð Lð LÐfqð Lð LØ -ðLð LØ<IðLð Lñô ð õ
 '*¨-½¸}Ñ9MÔ9MÑ&NÔ&Nð 	?ð 	?Ñ"�˜(ð Ð4×9Ò9Ñ;Ô;Ð;Ð;à6KÈHÔ6UÐ)¨(Ñ3Ð3ð 7?Ð)¨(Ñ3Ð3ð	?ð %Ð$r¥   TÚmissing_keysÚrecompute_mappingc                 óÔ  — |s| j         }n|                      d¬¦  «        }t          |                     ¦   «         ¦  «        }t	          |¦  «        D �]•\  }\  }}|�þd}||v}||v}	|rŸ|	r�|                      |¦  «        }
|                      |¦  «        }|
j        j        dk    r|j        j        dk    rŒdt          j	        |
|¦  «        s<t                               d|› d|› d�¦  «         | j                              |¦  «         ŒµnS|s|	r||}}nJ|sH|	sF||dz   d…         D ]\  }}||k    r
||v}|r|} n$Œd	}t                               d
|› d|› d�¦  «         |                      |¦  «        }
d|v r/|                     dd¦  «        \  }}|                      |¦  «        }n|}| }t!          |||
¦  «         |                      ||
¦  «         |�|r|                     |¦  «         �Œ—dS )aM  
        Tie the model weights. If `recompute_mapping=False` (default when called internally), it will rely on the
        `model.all_tied_weights_keys` attribute, containing the `{target: source}` mapping for the tied params.
        If `recompute_mapping=True`, it will re-check all internal submodels and their config to determine the params
        that need to be tied. This is the default when `model.tie_weights()` is called on its own, outside of
        `__init__`, and `from_pretrained`, in case the config values were changed somewhere.

        Note that during `from_pretrained`, tying is *symmetric*: if the mapping says "tie target -> source" but
        `source` is missing in the checkpoint while `target` exists, we *swap* source and target so we can still
        tie everything to the parameter that actually exists.
        Trm  Nr  zDThe tied weights mapping and config for this model specifies to tie r3  z°, but both are present in the checkpoints with different values, so we will NOT tie them. You should update the config with `tie_word_embeddings=False` to silence this warning.r   FzYThis checkpoint seem corrupted. The tied weights mapping for this model specifies to tie zk, but both are absent from the checkpoint, and we could not find another related tied weight for those keysrH  )ru  r~  r¯   r,  Ú	enumerateÚget_parameterrÖ   ru  r®   Úequalr¶  rÃ  rX  rv  r–  Úget_submoduler�  Ú_adjust_biasÚdiscard)r¢   rÄ  rÅ  r„  ÚiÚtarget_param_nameÚsource_param_nameÚremove_from_missingÚsource_is_thereÚtarget_is_thereÚsource_paramÚtarget_paramÚtarget_backupÚsource_backupÚtarget_backup_is_thereÚparent_namerJ  r�  s                     r£   rº  zPreTrainedModel.tie_weightsµ
  sÚ  € ð !ð 	PØÔ2ˆIˆIà×;Ò;È$Ð;ÑOÔOˆIå˜ŸšÑ*Ô*Ñ+Ô+ˆ	Ý9BÀ9Ñ9MÔ9Mð F	8ñ F	8Ñ5ˆAÑ5Ð!Ð#4àÐ'Ø&*Ð#Ø"3¸<Ð"G�Ø"3¸<Ð"G�ð #ð / ð /Ø#'×#5Ò#5Ð6GÑ#HÔ#H�LØ#'×#5Ò#5Ð6GÑ#HÔ#H�Lð $Ô*Ô/°6Ò9Ð9¸lÔ>QÔ>VÐZ`Ò>`Ð>`Ø õ !œ; |°\ÑBÔBð 	!ÝŸšðÐctð ð Ø0ðð ð ñô ð ð Ô2×6Ò6Ð7HÑIÔIÐIà ð	!ð )ð ¨_ð Ø;LÐN_Ð'8Ð%Ð%à(ð °ð Ø8AÀ!ÀaÁ%À'À'Ô8Jð ð Ñ4˜ }ð )Ð,=Ò=Ð=Ø5BÈ,Ð5VÐ2ð  6ð &Ø4AÐ 1Ø % øð /4Ð+ÝŸšð_Ø0ð_ð _Ø6Gð_ð _ð _ñô ð ð  ×7Ò7Ð8IÑJÔJˆLØÐ'Ð'Ð'Ø$5×$<Ò$<¸SÀ!Ñ$DÔ$DÑ!�˜TØ×+Ò+¨KÑ8Ô8��à(�Ø�å�F˜D ,Ñ/Ô/Ð/Ø×Ò˜f lÑ3Ô3Ð3àÐ'Ð,?Ð'Ø×$Ò$Ð%6Ñ7Ô7Ð7ùðMF	8ð F	8r¥   c                 ób  — t          |dd ¦  «        �mt          |d¦  «        r]|j        j        }t          j                             |j        j        d|d         |j        j        d         z
  fdd¦  «        |j        _        t          |d¦  «        rt          |d¦  «        r|j	        |_
        d S d S d S )Nrq  rm  r   ÚconstantÚout_featuresÚnum_embeddings)rM  rµ   rm  rì  r   Ú
functionalÚpadrq  ÚdatarÜ  rÛ  )r¢   Úoutput_embeddingsrC  Úweight_shapes       r£   rË  zPreTrainedModel._adjust_bias  sÏ   € ÝÐ$ f¨dÑ3Ô3Ð?ÅGÐL]Ð_gÑDhÔDhÐ?Ø,Ô3Ô9ˆLÝ*,¬-×*;Ò*;Ø!Ô&Ô+Ø�L ”OÐ&7Ô&<Ô&BÀ1Ô&EÑEÐFØØñ	+ô +ÐÔ"Ô'õ Ð$ nÑ5Ô5ð 	M½'ÐBRÐTdÑ:eÔ:eð 	MØ-=Ô-LÐÔ*Ð*Ð*ð	Mð 	Mð 	Mð 	Mr¥   Únew_num_tokensÚpad_to_multiple_ofÚmean_resizingc                 óÈ  — |                       |||¦  «        }|€|€|S t          | d¦  «        o| j        du}t          ¦   «         rR|sPddl}|j                             |j        d¬¦  «        5  |j        j        d         }ddd¦  «         n# 1 swxY w Y   n|j        j        d         }|| j	         
                    ¦   «         _        || _        |                      ¦   «          |S )a$	  
        Resizes input token embeddings matrix of the model if `new_num_tokens != config.vocab_size`.

        Takes care of tying weights embeddings afterwards if the model class has a `tie_weights()` method.

        Arguments:
            new_num_tokens (`int`, *optional*):
                The new number of tokens in the embedding matrix. Increasing the size will add newly initialized
                vectors at the end. Reducing the size will remove vectors from the end. If not provided or `None`, just
                returns a pointer to the input tokens `torch.nn.Embedding` module of the model without doing anything.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the embedding matrix to a multiple of the provided value.If `new_num_tokens` is set to
                `None` will just pad the embedding to a multiple of `pad_to_multiple_of`.

                This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
                `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128. For more
                details about this, or help on choosing the correct value for resizing, refer to this guide:
                https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tc
            mean_resizing (`bool`):
                Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
                covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

                Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
                where the generated tokens' probabilities won't be affected by the added embeddings because initializing the new embeddings with the
                old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
                Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html

        Return:
            `torch.nn.Embedding`: Pointer to the input tokens Embeddings Module of the model.
        Nr˜   r   ©Úmodifier_rank)Ú_resize_token_embeddingsrµ   r˜   r-   r´  r·  ÚGatheredParametersrm  rì  rÇ  ry  Ú
vocab_sizerº  )r¢   râ  rã  rä  Úmodel_embedsr¤   r´  rê  s           r£   Úresize_token_embeddingsz'PreTrainedModel.resize_token_embeddings  s@  € ðH ×4Ò4°^ÐEWÐYfÑgÔgˆØÐ!Ð&8Ð&@ØÐõ ˜t ^Ñ4Ô4ÐV¸Ô9JÐRVÐ9VˆÝ%Ñ'Ô'ð 	6°ð 	6ØÐÐÐà”×2Ò2°<Ô3FÐVZÐ2Ñ[Ô[ð :ð :Ø)Ô0Ô6°qÔ9�
ð:ð :ð :ñ :ô :ð :ð :ð :ð :ð :ð :øøøð :ð :ð :ð :øð &Ô,Ô2°1Ô5ˆJð 4>ˆŒ×#Ò#Ñ%Ô%Ô0Ø$ˆŒð 	×ÒÑÔÐàÐs   Á,BÂBÂBc                 ó   — |                       ¦   «         }|                      ||||¦  «        }t          |d¦  «        r|j        }t	          ||¦  «         |j        j        }|                     |¦  «         |                      |¦  «         t          | d¦  «        o| j	        d u}|�rt          ¦   «         rR|sPdd l}	|	j                             |j        d ¬¦  «        5  |j        j        d         }d d d ¦  «         n# 1 swxY w Y   n|j        j        d         }|                      ¦   «         �Á|                      ¦   «         }
t!          |
t"          j        j        ¦  «        r|                      |
||¬¦  «        }n|                      |
||¬¦  «        }t          |
d¦  «        r|
j        }t	          ||¦  «         |
j        j        }|                     |¦  «         |                      |¦  «         |                       ¦   «         S )NÚ_hf_hookr˜   r   ræ  )rä  )r  Ú_get_resized_embeddingsrµ   rî  r   rm  rŒ  r9  r   r˜   r-   r´  r·  ré  rì  r#  rs  r®   r   r   Ú_get_resized_lm_headr&  )r¢   râ  rã  rä  Úold_embeddingsr%  rH  Úold_embeddings_requires_gradr¤   r´  Úold_lm_headÚnew_lm_headÚold_lm_head_requires_grads                r£   rè  z(PreTrainedModel._resize_token_embeddingsW  si  € Ø×2Ò2Ñ4Ô4ˆØ×5Ò5Ø˜NÐ,>Àñ
ô 
ˆõ �> :Ñ.Ô.ð 	5Ø!Ô*ˆDÝ˜~¨tÑ4Ô4Ð4Ø'5Ô'<Ô'JÐ$Ø×%Ò%Ð&BÑCÔCÐCØ×!Ò! .Ñ1Ô1Ð1Ý˜t ^Ñ4Ô4ÐV¸Ô9JÐRVÐ9Vˆð Ð)Ý)Ñ+Ô+ð @°Lð @Ø Ð Ð Ð à”^×6Ò6°~Ô7LÐ\`Ð6ÑaÔað Dð DØ%3Ô%:Ô%@ÀÔ%C�NðDð Dð Dñ Dô Dð Dð Dð Dð Dð Dð Døøøð Dð Dð Dð Døð "0Ô!6Ô!<¸QÔ!?�ð ×%Ò%Ñ'Ô'Ð3Ø×4Ò4Ñ6Ô6ˆKÝ˜+¥u¤xÔ'9Ñ:Ô:ð rØ"×:Ò:¸;ÈÐfsÐ:ÑtÔt��à"×7Ò7¸À^ÐcpÐ7ÑqÔq�Ý�{ JÑ/Ô/ð 6Ø"Ô+�Ý" ;°Ñ5Ô5Ð5Ø(3Ô(:Ô(HÐ%Ø×&Ò&Ð'@ÑAÔAÐAØ×&Ò& {Ñ3Ô3Ð3à×(Ò(Ñ*Ô*Ð*s   ÃC9Ã9C=Ä C=rñ  c           	      ó0  — |�Kt          |t          ¦  «        st          d|› d�¦  «        ‚|€|j        j        d         }||z   dz
  |z  |z  }nt
                               d|› d�¦  «         |€|S t          | d¦  «        o| j        du}t          ¦   «         r\|sZddl
}|j                             |j        d¬	¦  «        5  |j                             ¦   «         \  }}ddd¦  «         n# 1 swxY w Y   n|j                             ¦   «         \  }}||k    rt          ¦   «         s	||_        |S t          |t          j        ¦  «        s:t#          d
t%          |¦  «        › dt          j        › dt          j        › d�¦  «        ‚t          j        |||j        j        |j        j        ¬¦  «        }	||k    r|s|                      |	¦  «         n¦||k    r |ržt
                               d¦  «         ||z
  }
t          ¦   «         rY|sWddl
}|j                             |j        gd¬	¦  «        5  |                      ||	||
¦  «         ddd¦  «         n# 1 swxY w Y   n|                      ||	||
¦  «         t1          ||¦  «        }t          ¦   «         rt|srddl
}|j        |	j        g}|j                             |d¬	¦  «        5  |j        j        d|…dd…f         |	j        j        d|…dd…f<   ddd¦  «         n# 1 swxY w Y   n+|j        j        d|…dd…f         |	j        j        d|…dd…f<   t          ¦   «         r�|s‹ddl
}|j        |	j        g}|j                             |d¬	¦  «        5  |	j        |_        |	j        j        j        d         |_        |j        �|dz
  |j        k     rd|_        ddd¦  «         n# 1 swxY w Y   nN|	j        j        |j        _        |	j        j        j        d         |_        |j        �|dz
  |j        k     rd|_        |S )aÂ	  
        Build a resized Embedding Module from a provided token Embedding Module. Increasing the size will add newly
        initialized vectors at the end. Reducing the size will remove vectors from the end

        Args:
            old_embeddings (`torch.nn.Embedding`):
                Old embeddings to be resized.
            new_num_tokens (`int`, *optional*):
                New number of tokens in the embedding matrix.

                Increasing the size will add newly initialized vectors at the end. Reducing the size will remove
                vectors from the end. If not provided or `None`, just returns a pointer to the input tokens
                `torch.nn.Embedding` module of the model without doing anything.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the embedding matrix to a multiple of the provided value. If `new_num_tokens` is set to
                `None` will just pad the embedding to a multiple of `pad_to_multiple_of`.

                This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
                `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128. For more
                details about this, or help on choosing the correct value for resizing, refer to this guide:
                https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tc
            mean_resizing (`bool`):
                Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
                covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

                Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
                where the generated tokens' probabilities will not be affected by the added embeddings because initializing the new embeddings with the
                old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
                Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html


        Return:
            `torch.nn.Embedding`: Pointer to the resized Embedding Module or the old Embedding Module if
            `new_num_tokens` is `None`
        Nz5Asking to pad the embedding matrix to a multiple of `z@`, which is not and integer. Please make sure to pass an integerr   r   z�You are resizing the embedding layer without providing a `pad_to_multiple_of` parameter. This means that the new embedding dimension will be a.  . This might induce some performance reduction as *Tensor Cores* will not be available. For more details about this, or help on choosing the correct value for resizing, refer to this guide: https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tcr˜   ræ  zOld embeddings are of type ú, which is not an instance of zj. You should either use a different resize function or make sure that `old_embeddings` are an instance of rH  rç  zýThe new embeddings will be initialized from a multivariate normal distribution that has old embeddings' mean and covariance. As described in this article: https://nlp.stanford.edu/~johnhew/vocab-expansion.html. To disable this, use `mean_resizing=False`)rs  r½   rÍ   rm  rì  r¶  r·  rµ   r˜   r-   r´  r·  ré  r&  rÜ  r   r   r[  ru  rÖ   r–   r•  rË  Ú(_init_added_embeddings_weights_with_meanrã  rß  r…  )r¢   rñ  râ  rã  rä  r¤   r´  Úold_num_tokensÚold_embedding_dimr%  Úadded_num_tokensÚnÚparamss                r£   rï  z'PreTrainedModel._get_resized_embeddings~  s*  € ðV Ð)ÝÐ0µ#Ñ6Ô6ð Ý ð aÐL^ð  að  að  añô ð ð Ð%Ø!/Ô!6Ô!<¸QÔ!?�Ø-Ð0BÑBÀQÑFÐK]Ñ]ÐasÑsˆNˆNå�KŠKðDØ&4ðDð Dð Dñô ð ð Ð!Ø!Ð!å˜t ^Ñ4Ô4ÐV¸Ô9JÐRVÐ9VˆÝ%Ñ'Ô'ð 	M°ð 	MØÐÐÐà”×2Ò2°>Ô3HÐX\Ð2Ñ]Ô]ð Qð QØ4BÔ4I×4NÒ4NÑ4PÔ4PÑ1�Ð 1ðQð Qð Qñ Qô Qð Qð Qð Qð Qð Qð Qøøøð Qð Qð Qð Qøð 1?Ô0E×0JÒ0JÑ0LÔ0LÑ-ˆNÐ-à˜^Ò+Ð+Õ4NÑ4PÔ4PÐ+Ø,:ˆNÔ)Ø!Ð!å˜.­"¬,Ñ7Ô7ð 	Ýð$­d°>Ñ.BÔ.Bð $ð $ÕbdÔbnð $ð $å”Lð$ð $ð $ñô ð õ œØØØ!Ô(Ô/Ø Ô'Ô-ð	
ñ 
ô 
ˆð ˜NÒ*Ð*°=Ð*à×Ò˜~Ñ.Ô.Ð.Ð.à˜nÒ,Ð,°Ð,õ ×Òð=ñô ð ð  .°Ñ>ÐÝ)Ñ+Ô+ð 
°Lð 
Ø Ð Ð Ð à”^×6Ò6¸Ô8MÐ7NÐ^bÐ6ÑcÔcð ð Ø×AÒAØ&¨¸ÐHXñô ð ðð ð ñ ô ð ð ð ð ð ð øøøð ð ð ð øð
 ×=Ò=Ø" N°NÐDTñô ð õ � Ñ/Ô/ˆå%Ñ'Ô'ð 	R°ð 	RØÐÐÐà$Ô+¨^Ô-BÐCˆFØ”×2Ò2°6ÈÐ2ÑKÔKð Vð VØ4BÔ4IÔ4NÈrÐPQÈrÐSTÐSTÐSTÈuÔ4U�Ô%Ô*¨2¨A¨2¨q¨q¨q¨5Ñ1ðVð Vð Vñ Vô Vð Vð Vð Vð Vð Vð Vøøøð Vð Vð Vð Vøð 1?Ô0EÔ0JÈ2ÈAÈ2ÈqÈqÈqÈ5Ô0QˆNÔ!Ô& r¨ r¨1¨1¨1 uÑ-õ
 &Ñ'Ô'ð 	2°ð 	2ØÐÐÐà$Ô+¨^Ô-BÐCˆFØ”×2Ò2°6ÈÐ2ÑKÔKð 6ð 6Ø(6Ô(=�Ô%Ø0>Ô0EÔ0JÔ0PÐQRÔ0S�Ô-ð "Ô-Ð9¸~ÐPQÑ?QÐUcÔUoÒ>oÐ>oØ15�NÔ.ð6ð 6ð 6ñ 6ô 6ð 6ð 6ð 6ð 6ð 6ð 6øøøð 6ð 6ð 6ð 6øð *8Ô)>Ô)CˆNÔ!Ô&Ø,:Ô,AÔ,FÔ,LÈQÔ,OˆNÔ)ØÔ)Ð5¸>ÈAÑ;MÐQ_ÔQkÒ:kÐ:kØ-1�Ô*àÐsI   Â>C'Ã'C+Ã.C+È'IÉIÉIÊ>,K6Ë6K:Ë=K:Í+AN<Î<O ÏO ró  Ú
transposedc           	      óÔ  — |€|S t          | d¦  «        o| j        du}t          ¦   «         r‰|s‡ddl}|j                             |j        d¬¦  «        5  |s|j                             ¦   «         n*|j                             ¦   «                              ¦   «         \  }}ddd¦  «         n# 1 swxY w Y   nI|s|j                             ¦   «         n*|j                             ¦   «                              ¦   «         \  }}||k    rt          ¦   «         s	||_	        |S t          |t          j        ¦  «        s:t          dt          |¦  «        › dt          j        › dt          j        › d�¦  «        ‚|s||fn||f}	|j        du}
t          j        |	|
|j        j        |j        j        d	œŽ}||k    r|s|                      |¦  «         në||k    rå|rãt&                               d
¦  «         ||z
  }t          ¦   «         rƒ|s�ddl}|j        g}|
r||j        gz  }|j                             |d¬¦  «        5  |                      ||||||¦  «         |
r|                      |||¦  «         ddd¦  «         n# 1 swxY w Y   n3|                      ||||||¦  «         |
r|                      |||¦  «         t/          ||¦  «        }t          ¦   «         rn|slddl}|j        |j        |j        |j        g}|j                             |d¬¦  «        5  |                      |||||
¦  «         ddd¦  «         n# 1 swxY w Y   n|                      |||||
¦  «         t3          |dd¦  «         |S )a¦  
        Build a resized Linear Module from a provided old Linear Module. Increasing the size will add newly initialized
        vectors at the end. Reducing the size will remove vectors from the end

        Args:
            old_lm_head (`torch.nn.Linear`):
                Old lm head liner layer to be resized.
            new_num_tokens (`int`, *optional*):
                New number of tokens in the linear matrix.

                Increasing the size will add newly initialized vectors at the end. Reducing the size will remove
                vectors from the end. If not provided or `None`, just returns a pointer to the input tokens
                `torch.nn.Linear` module of the model without doing anything. transposed (`bool`, *optional*, defaults
                to `False`): Whether `old_lm_head` is transposed or not. If True `old_lm_head.size()` is `lm_head_dim,
                vocab_size` else `vocab_size, lm_head_dim`.
            mean_resizing (`bool`):
                Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
                covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

                Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
                where the generated tokens' probabilities will not be affected by the added embeddings because initializing the new embeddings with the
                old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
                Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html

        Return:
            `torch.nn.Linear`: Pointer to the resized Linear Module or the old Linear Module if `new_num_tokens` is
            `None`
        Nr˜   r   ræ  z#Old language model head is of type r÷  zg. You should either use a different resize function or make sure that `old_lm_head` are an instance of rH  )rq  rÖ   r–   a  The new lm_head weights will be initialized from a multivariate normal distribution that has old embeddings' mean and covariance. As described in this article: https://nlp.stanford.edu/~johnhew/vocab-expansion.html. To disable this, use `mean_resizing=False`rr  T)rµ   r˜   r-   r´  r·  ré  rm  r&  rä   rÛ  rs  r   rz  r[  ru  rq  rÖ   r–   r•  r¶  rË  Ú%_init_added_lm_head_weights_with_meanÚ"_init_added_lm_head_bias_with_meanrã  Ú!_copy_lm_head_original_to_resizedr�  )r¢   ró  râ  rþ  rä  r¤   r´  rù  Úold_lm_head_dimÚnew_lm_head_shapeÚhas_new_lm_head_biasrô  rû  rý  Únum_tokens_to_copys                  r£   rð  z$PreTrainedModel._get_resized_lm_head  s  € ðH Ð!ØÐå˜t ^Ñ4Ô4ÐV¸Ô9JÐRVÐ9VˆÝ%Ñ'Ô'ð 
	°ð 
	ØÐÐÐà”×2Ò2°;Ô3EÐUYÐ2ÑZÔZð ð à5?Ðb�KÔ&×+Ò+Ñ-Ô-Ð-À[ÔEW×EYÒEYÑE[ÔE[×E`ÒE`ÑEbÔEbñ 0� ðð ð ñ ô ð ð ð ð ð ð øøøð ð ð ð øð 2<Ð^�Ô"×'Ò'Ñ)Ô)Ð)ÀÔAS×AUÒAUÑAWÔAW×A\ÒA\ÑA^ÔA^ñ ,ˆN˜Oð ˜^Ò+Ð+Õ4NÑ4PÔ4PÐ+Ø'5ˆKÔ$ØÐå˜+¥r¤yÑ1Ô1ð 	Ýð!µd¸;Ñ6GÔ6Gð !ð !ÕgiÔgpð !ð !å”Ið!ð !ð !ñô ð ð FPÐv˜_¨nÐ=Ð=ÐVdÐfuÐUvÐØ*Ô/°tÐ;Ðõ ”iØØ%ØÔ%Ô,ØÔ$Ô*ð	
ð 
ð 
ˆð ˜NÒ*Ð*°=Ð*à×Ò˜{Ñ+Ô+Ð+Ð+à˜nÒ,Ð,°Ð,õ ×Òð=ñô ð ð  .°Ñ>ÐÝ)Ñ+Ô+ð h°Lð hØ Ð Ð Ð à%Ô,Ð-�Ø'ð 1Ø˜{Ô/Ð0Ñ0�FØ”^×6Ò6°vÈTÐ6ÑRÔRð lð lØ×>Ò>Ø# [°/À>ÐScÐeoñô ð ð ,ð lØ×?Ò?ÀÈ[ÐZjÑkÔkÐkðlð lð lñ lô lð lð lð lð lð lð løøøð lð lð lð løð ×:Ò:Ø ¨o¸~ÐO_Ðakñô ð ð (ð hØ×;Ò;¸KÈÐVfÑgÔgÐgå  °Ñ@Ô@Ðå%Ñ'Ô'ð 	°ð 	ØÐÐÐà!Ô(¨+Ô*:¸KÔ<NÐP[ÔP`ÐaˆFØ”×2Ò2°6ÈÐ2ÑKÔKð ð Ø×6Ò6Ø Ð.@À*ÐNbñô ð ðð ð ñ ô ð ð ð ð ð ð øøøð ð ð ð øð
 ×2Ò2Ø˜[Ð*<¸jÐJ^ñô ð õ 	�Ð1°4Ñ8Ô8Ð8ØÐs7   ÁA
B)Â)B-Â0B-È34I3É3I7É:I7ÌL2Ì2L6Ì9L6c                 ó¨  — |j         j                             t          j        ¦  «        }t          j        |d¬¦  «        }||z
  }|j        |z  |z  }d}	t          j         	                    |	|z  ¦  «         
                    ¦   «         }
|
rut          j        j                             ||	|z  ¬¦  «        }|                     |f¬¦  «                             |j         j        ¦  «        |j         j        d|z  d …d d …f<   d S |d d d …f                              |d¦  «                             |j         j        ¦  «        |j         j        d|z  d …d d …f<   d S )Nr   rè  ç•Ö&è.>)Úcovariance_matrix)Úsample_shaper<  r   )rm  rß  r   r®   rà   ro  ÚTr   Úpositive_definiteÚcheckr�  ÚdistributionsÚmultivariate_normalÚMultivariateNormalÚsampler–   rë  )r¢   rñ  r%  rù  rû  Úold_embeddings_weightÚmean_embeddingsÚold_centered_embeddingsÚ
covarianceÚepsilonÚis_covariance_psdÚdistributions               r£   rø  z8PreTrainedModel._init_added_embeddings_weights_with_mean   s{  € ð !/Ô 5Ô :× =Ò =½e¼mÑ LÔ LÐÝœ*Ð%:ÀÐCÑCÔCˆØ"7¸/Ñ"IÐØ,Ô.Ð1HÑHÈ>ÑYˆ
ð ˆÝ'Ô9×?Ò?ÀÈ*Ñ@TÑUÔU×YÒYÑ[Ô[ÐØð 	å Ô.ÔB×UÒUØ°7¸ZÑ3Gð Vñ ô ˆLð FR×EXÒEXØ.Ð0ð FYñ Fô FçŠb�Ô&Ô,Ñ-Ô-ð Ô!Ô& rÐ,<Ñ'<Ð'>Ð'>ÀÀÀÐ'AÑBÐBÐBð    a a a Ô(×/Ò/Ð0@À!ÑDÔD×GÒGÈÔH]ÔHcÑdÔdð Ô!Ô& rÐ,<Ñ'<Ð'>Ð'>ÀÀÀÐ'AÑBÐBÐBr¥   c                 ó  — |r6|j         j        j        |j         _        |j         j        j        |j         _        |                      ||||¦  «         |r8|j         j        j        |j         _        |j         j        j        |j         _        d S d S r    )rm  rß  r  rø  )r¢   ró  rô  r  rù  rû  rþ  s          r£   r   z5PreTrainedModel._init_added_lm_head_weights_with_mean¹  s–   € ð ð 	@à&1Ô&8Ô&=Ô&?ˆKÔÔ#Ø&1Ô&8Ô&=Ô&?ˆKÔÔ#ð 	×5Ò5°kÀ;ÐP^Ð`pÑqÔqÐqàð 	@à&1Ô&8Ô&=Ô&?ˆKÔÔ#Ø&1Ô&8Ô&=Ô&?ˆKÔÔ#Ð#Ð#ð	@ð 	@r¥   c                 ó4  — t          j        |j        j        dt           j        ¬¦  «        }t          j        |j        j        d¬¦  «                             t           j        ¦  «        }|j        j        d|z  d …                              |d|z  ¬¦  «         d S )Nr   )ré  r–   rè  r<  r  rn  )r®   ro  rq  rß  rà   rp  r   r€  )r¢   ró  rô  rû  Ú	bias_meanÚbias_stds         r£   r  z2PreTrainedModel._init_added_lm_head_bias_with_meanÏ  s‡   € Ý”J˜{Ô/Ô4¸1ÅEÄMÐRÑRÔRˆ	Ý”9˜[Ô-Ô2¸Ð;Ñ;Ô;×>Ò>½u¼}ÑMÔMˆØÔÔ˜bÐ#3Ñ3Ð5Ð5Ô6×>Ò>ÀIÐSWÐZbÑSbÐ>ÑcÔcÐcÐcÐcr¥   c                 ó  — |s,|j         j        d |…d d …f         |j         j        d |…d d …f<   n+|j         j        d d …d |…f         |j         j        d d …d |…f<   |r%|j        j        d |…         |j        j        d |…<   d S d S r    )rm  rß  rq  )r¢   rô  ró  r  rþ  r  s         r£   r  z1PreTrainedModel._copy_lm_head_original_to_resizedÔ  sÒ   € ð ð 	nØ>IÔ>PÔ>UÐViÐWiÐViÐklÐklÐklÐVlÔ>mˆKÔÔ#Ð$7Ð%7Ð$7¸¸¸Ð$:Ñ;Ð;à>IÔ>PÔ>UÐVWÐVWÐVWÐYlÐZlÐYlÐVlÔ>mˆKÔÔ# A A AÐ':Ð(:Ð':Ð$:Ñ;ð  ð 	dØ9DÔ9IÔ9NÐObÐPbÐObÔ9cˆKÔÔ!Ð"5Ð#5Ð"5Ñ6Ð6Ð6ð	dð 	dr¥   Únew_num_position_embeddingsc           	      ó\   — t          d| j        › d| j        › d| j        j        › d�¦  «        ‚)Nz4`resize_position_embeddings` is not implemented for úB`. To implement it, you should overwrite this method in the class ú in `modeling_ú.py`©r  r  r§   )r¢   r  s     r£   Úresize_position_embeddingsz*PreTrainedModel.resize_position_embeddingsá  sV   € Ý!ðpÀ4Ä>ð pð pØ26´.ðpð pØPTÔP^ÔPiðpð pð pñ
ô 
ð 	
r¥   c           	      ó\   — t          d| j        › d| j        › d| j        j        › d�¦  «        ‚)Nz1`get_position_embeddings` is not implemented for r   r!  r"  r#  r¡   s    r£   Úget_position_embeddingsz'PreTrainedModel.get_position_embeddingsç  sV   € Ý!ðpÀÄð pð pØ26´.ðpð pØPTÔP^ÔPiðpð pð pñ
ô 
ð 	
r¥   c                 ó¢   — t          ¦   «         t          j        d¦  «        k    r|                      ¦   «          |                      d¬¦  «         dS )z«
        Initialize and tie the weights if needed. If using a custom `PreTrainedModel`, you need to implement any
        initialization logic in `_init_weights`.
        r  F)rÅ  N)rÚ   r®   rÖ   r¦  rº  r¡   s    r£   r�  zPreTrainedModel.init_weightsí  sN   € õ 6Ñ7Ô7½5¼<ÈÑ;OÔ;OÒOÐOà×#Ò#Ñ%Ô%Ð%à×Ò¨5ÐÑ1Ô1Ð1Ð1Ð1r¥   c                 óì  — | j         st          | j        j        › d�¦  «        ‚|€ddi}t	          j        t          fi |¤Ž}dt          j        | j	        ¦  «        j
        v }|s|  	                    d|¬¦  «         nC|                      t          | j	        d¬¦  «        ¦  «         t                               d	¦  «         | j        d
k    }|pt          | dd¦  «        }|r|                      ¦   «          dS dS )að  
        Activates gradient checkpointing for the current model.

        We pass the `__call__` method of the modules instead of `forward` because `__call__` attaches all the hooks of
        the module. https://discuss.pytorch.org/t/any-different-between-model-input-and-model-forward-input/3690/2

        Args:
            gradient_checkpointing_kwargs (dict, *optional*):
                Additional keyword arguments passed along to the `torch.utils.checkpoint.checkpoint` function.
        z) does not support gradient checkpointing.NÚuse_reentrantFr  T)ÚenableÚgradient_checkpointing_func©r  áV  You are using an old version of the checkpointing format that is deprecated (We will also silently ignore `gradient_checkpointing_kwargs` in case you passed it).Please update to the new format on your modeling file. To use the new format, you need to completely remove the definition of the method `_set_gradient_checkpointing` in your model.r.  Ú_hf_peft_config_loaded)rA  rÍ   r  r¦   Ú	functoolsr
   r   rL  Ú	signatureÚ_set_gradient_checkpointingrÙ  Úapplyr¶  rÃ  r/  rM  rE  )r¢   Úgradient_checkpointing_kwargsr+  Ú_is_using_old_formatÚneeds_embedding_gradsÚenable_input_gradss         r£   r¦  z-PreTrainedModel.gradient_checkpointing_enableù  s-  € ð Ô3ð 	dÝ ¤Ô 7ÐbÐbÐbÑcÔcÐcà(Ð0Ø-<¸eÐ,DÐ)å&/Ô&7½
Ð&dÐ&dÐFcÐ&dÐ&dÐ#ð  '­'Ô*;¸DÔ<\Ñ*]Ô*]Ô*hÐhÐà#ð 	Ø×,Ò,°DÐVqÐ,ÑrÔrÐrÐrà�JŠJ•w˜tÔ?ÀtÐLÑLÔLÑMÔMÐMÝ�NŠNðHñô ð ð
 !%Ô 4¸Ò CÐà2Ðdµg¸dÐD\Ð^cÑ6dÔ6dÐØð 	.ð
 ×+Ò+Ñ-Ô-Ð-Ð-Ð-ð	.ð 	.r¥   r*  r+  c                 ó  — d}t          | d¦  «        r|| _        || _        d}|                      ¦   «         D ]6}t          |d¦  «        r$t	          |d|¦  «         t	          |d|¦  «         d}Œ7|st          | j        j        › d�¦  «        ‚d S )NFr¥  TÚ_gradient_checkpointing_funczÂ is not compatible with gradient checkpointing. Make sure all the architecture support it by setting a boolean attribute `gradient_checkpointing` to modules of the model that uses checkpointing.)rµ   r8  r¥  r  r�  rÍ   r  r¦   )r¢   r*  r+  Úis_gradient_checkpointing_setrC  s        r£   r1  z+PreTrainedModel._set_gradient_checkpointing#  sÉ   € Ø(-Ð%õ �4Ð1Ñ2Ô2ð 	1Ø0KˆDÔ-Ø*0ˆDÔ'Ø,0Ð)à—l’l‘n”nð 	5ð 	5ˆFÝ�vÐ7Ñ8Ô8ð 5Ý˜Ð >Ð@[Ñ\Ô\Ð\Ý˜Ð 8¸&ÑAÔAÐAØ04Ð-øà,ð 	ÝØ”>Ô*ð ]ð ]ð ]ñô ð ð	ð 	r¥   c                 óZ  — | j         r|dt          j        | j        ¦  «        j        v }|s|                      d¬¦  «         nCt
                               d¦  «         |                      t          | j        d¬¦  «        ¦  «         t          | dd¦  «        r|  
                    ¦   «          dS dS )zK
        Deactivates gradient checkpointing for the current model.
        r  F)r*  r-  r,  r.  N)rA  rL  r0  r1  rÙ  r¶  rÃ  r2  r
   rM  rI  )r¢   r4  s     r£   Úgradient_checkpointing_disablez.PreTrainedModel.gradient_checkpointing_disable9  sÈ   € ð Ô/ð 	Sð $+­gÔ.?ÀÔ@`Ñ.aÔ.aÔ.lÐ#lÐ Ø'ð SØ×0Ò0¸Ð0Ñ>Ô>Ð>Ð>å—’ðLñô ð ð —
’
�7 4Ô#CÈ5ÐQÑQÔQÑRÔRÐRå�4Ð1°5Ñ9Ô9ð 	/Ø×,Ò,Ñ.Ô.Ð.Ð.Ð.ð	/ð 	/r¥   c                 óX   — t          d„ |                      ¦   «         D ¦   «         ¦  «        S )zT
        Whether gradient checkpointing is activated for this model or not.
        c              3   óD   K  — | ]}t          |d ¦  «        o|j        V — ŒdS )r¥  N)rµ   r¥  )rþ   Úms     r£   r   z<PreTrainedModel.is_gradient_checkpointing.<locals>.<genexpr>R  s6   è è € ÐmÐmÐYZ•7˜1Ð6Ñ7Ô7ÐT¸AÔ<TÐmÐmÐmÐmÐmÐmr¥   )rw  r  r¡   s    r£   Úis_gradient_checkpointingz)PreTrainedModel.is_gradient_checkpointingM  s.   € õ
 ÐmÐmÐ^b×^jÒ^jÑ^lÔ^lÐmÑmÔmÑmÔmÐmr¥   Ú50GBÚsave_directoryÚis_main_processrã   Úpush_to_hubÚmax_shard_sizer”  r¢  Úsave_peft_formatÚsave_original_formatc
           	      ó>  ‡2— |�||
d<   t          | dd¦  «        }t          | dd¦  «        }|duo(t          |t          ¦  «        o|                     ¦   «         }|�!|s|st	          d|j        j        › d�¦  «        ‚| j        �t          d¦  «        st          d	¦  «        ‚t          j                             |¦  «        r t                               d
|› d�¦  «         dS t          j        |d¬¦  «         t          j        |¦  «        }|r |
                     dd¦  «        }|
                     d|                     t          j        j        ¦  «        d         ¦  «        }|
                     dd¦  «        } t)          ¦   «         j        |fddi|
¤Žj        }|                      |¦  «        }i }|�|                     | ¦  «        \  }}d|d<   t3          | ¦  «        }|j        }t7          |¦  «                             d¦  «        d         |j        _        |j        j                             d¦  «        g|j        _         |  !                    ¦   «         rtE          | || j        ¬¦  «         |�r|s|j         #                    |¦  «         |  $                    ¦   «         r|j%         #                    |¦  «         |rÒt           &                    d¦  «         | '                    |¬¦  «        }|r@t           &                    d¦  «         i }| (                    ¦   «         D ]\  }}||d|› �<   Œ|}|  )                    ¦   «         }tU          |¦  «        dk    rt	          d¦  «        ‚|d         }| j+        |         }| #                    |¦  «         |€| ,                    ¦   «         }d}t[          | d¦  «        rƒtU          t]          | j/         0                    ¦   «         ¦  «        ¦  «        dk    rLd | j/         0                    ¦   «         v sd!| j/         0                    ¦   «         v rd}tc          j2        d"¦  «         tf          r'th          j5        j6        j7        D ]\  }} ||¦  «        }Œ| j8        �)tU          | j8        ¦  «        dk    r| j8        D ]	}||v r||= Œ
| j        �!ts          || j:        | j;        | j        ¦  «        }ty          ||¦  «        }|	r|s|st{          ||¦  «        }|st|          }t          ||¦  «        }nt€          }| A                    d#d$¦  «         A                    d%d&¦  «        } t…          || |¬'¦  «        }!t          jC        |¦  «        D ]ò}"t          j         D                    ||"¦  «        }#| A                    d#d(¦  «         A                    d%d(¦  «        }$|" A                    d#d(¦  «         A                    d%d(¦  «        }%t‹          jF        d)¦  «        }&|" G                    |$¦  «        rSt          j                             |#¦  «        r4|"|!jH        vr+|r)|& I                    |%¦  «        �t          jJ        |#¦  «         Œód}'|!jK        r|r|	r|r|!jL        ni }'t›          jN        |!jH         (                    ¦   «         d*¬+¦  «        D ]ü\  Š2}(t          j         D                    |‰2¦  «        }"i })|(D ]P}*|                     |*¦  «        }+|r |+jO        jP        d,k    rt£          ||*¦  «        }+|+ R                    ¦   «         |)|*<   ŒQ|rm|	rk|si	 t{          ||)¦  «        })|!jK        r3|' S                    ˆ2fd-„|) T                    ¦   «         D ¦   «         ¦  «         n# tª          $ r t­          d.¦  «        ‚w xY wt¯          |)|"|¬/¦  «         ~)Œýd},|!jK        r d0|  X                    ¦   «         i|!jY        ¥|'d1œ},|,€>t          j         D                    ||¦  «        }-t           &                    d2|-› �¦  «         nÄt´          }.t          j         D                    |t          |.|¦  «        ¦  «        }.t·          |.d3d4¬5¦  «        5 }/t¹          j]        |,d6d¬7¦  «        d8z   }0|/ ^                    |0¦  «         ddd¦  «         n# 1 swxY w Y   t           &                    d9|› d:tU          |!jH        ¦  «        › d;|.› d�¦  «         |rgt¿          || j`        |¬<¦  «        }1|1 a                    t          j         D                    |d=¦  «        ¦  «         |  b                    ||||||¬>¦  «         dS dS )?až  
        Save a model and its configuration file to a directory, so that it can be re-loaded using the
        [`~PreTrainedModel.from_pretrained`] class method.

        Arguments:
            save_directory (`str` or `os.PathLike`):
                Directory to which to save. Will be created if it doesn't exist.
            is_main_process (`bool`, *optional*, defaults to `True`):
                Whether the process calling this is the main process or not. Useful when in distributed training like
                TPUs and need to call this function on all processes. In this case, set `is_main_process=True` only on
                the main process to avoid race conditions.
            state_dict (nested dictionary of `torch.Tensor`):
                The state dictionary of the model to save. Will default to `self.state_dict()`, but can be used to only
                save parts of the model or if special precautions need to be taken when recovering the state dictionary
                of a model (like when using model parallelism).
            push_to_hub (`bool`, *optional*, defaults to `False`):
                Whether or not to push your model to the Hugging Face model hub after saving it. You can specify the
                repository you want to push to with `repo_id` (will default to the name of `save_directory` in your
                namespace).
            max_shard_size (`int` or `str`, *optional*, defaults to `"50GB"`):
                The maximum size for a checkpoint before being sharded. Checkpoints shard will then be each of size
                lower than this size. If expressed as a string, needs to be digits followed by a unit (like `"5MB"`).

                <Tip warning={true}>

                If a single weight of the model is bigger than `max_shard_size`, it will be in its own checkpoint shard
                which will be bigger than `max_shard_size`.

                </Tip>

            variant (`str`, *optional*):
                If specified, weights are saved in the format model.<variant>.safetensors.
            token (`str` or `bool`, *optional*):
                The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
                the token generated when running `hf auth login` (stored in `~/.huggingface`).
            save_peft_format (`bool`, *optional*, defaults to `True`):
                For backward compatibility with PEFT library, in case adapter weights are attached to the model, all
                keys of the state dict of adapters needs to be prepended with `base_model.model`. Advanced users can
                disable this behaviours by setting `save_peft_format` to `False`.
            save_original_format (`bool`, *optional*, defaults to `True`):
                For backward compatibility with the previous versions of `transformers` you can save the checkpoint with
                its reverse mapping. The reverse mapping needs to exists even if the model was loaded from a None legacy
                checkpoint.
            kwargs (`dict[str, Any]`, *optional*):
                Additional key word arguments passed along to the [`~utils.PushToHubMixin.push_to_hub`] method.
        Nr¢  r.  Fr˜   zThe model is quantized with z˜ and is not serializable - check out the warnings from the logger on the traceback to understand the reason why the quantized model is not serializable.z0.31.4z[Saving a model with tensor parallelism requires `huggingface_hub` version 0.31.4 or higher.zProvided path (z#) should be a directory, not a fileT)Úexist_okÚcommit_messageÚrepo_idr<  Ú	create_prrH  r$  ÚformatrH  r   ÚFSDP)rÇ  zhDetected adapters on the model, saving the model in the PEFT format, only adapter weights will be saved.)rã   zƒTo match the expected format of the PEFT library, all keys of the state dict of adapters will be prepended with `base_model.model`.zbase_model.model.zßMultiple active adapters detected, saving multiple active adapters is not supported yet. You can save adapters separately one by one by iteratively calling `model.set_adapter(adapter_name)` then `model.save_pretrained(...)`r   Úhf_device_maprÔ   Údiskz}Attempting to save a model with offloaded modules. Ensure that unallocated cpu memory exceeds the `shard_size` (50GB default)z.binz{suffix}.binr  z{suffix}.safetensors)Úfilename_patternrD  r¦  z(.*?)-\d{5}-of-\d{5}zWriting model shards)Údescr  c              3   óZ   •K  — | ]%}|t           j                             ‰¦  «        iV — Œ&d S r    )r¾   rõ   Úbasename)rþ   r!  Ú
shard_files     €r£   r   z2PreTrainedModel.save_pretrained.<locals>.<genexpr>[  s9   øè è € Ð)mÐ)mÐPQ¨1­b¬g×.>Ò.>¸zÑ.JÔ.JÐ*KÐ)mÐ)mÐ)mÐ)mÐ)mÐ)mr¥   zôWe could not revert some weight conversions because of offlading, and several weights needed for a single conversion operation living in different shard files. Try reducing `max_shard_size` a bit, or worst case set `save_original_format=False`.)ÚmetadataÚtotal_parameters)rU  Ú
weight_mapzModel weights saved in Úwrø   rù   rü   )ÚindentÚ	sort_keysú
z:The model is bigger than the maximum size per checkpoint (z) and is going to be split in z^ checkpoint shards. You can find where each parameters has been saved in the index located at )r¢  z	README.md)rI  r¢  rK  )crM  rs  rS   Úis_serializablerÍ   Úquantization_configÚquant_methodÚ_tp_sizerv   rÏ  r¾   rõ   r´  r¶  ÚerrorÚmakedirsr  rX  r  Úseprr   Úcreate_reporJ  Ú_get_files_timestampsÚget_state_dict_and_metadataÚunwrap_modelr–   rª   rÇ  r  r¦   ÚremoveprefixÚarchitecturesrš  r'   Úsave_pretrainedre  rg  r·  Úget_adapter_state_dictr,  Úactive_adaptersrß   Úpeft_configrã   rµ   re  rN  rÞ   rš  r›  ÚIS_SAGEMAKER_MP_POST_1_10ÚsmpÚstateÚmodule_managerÚtranslate_functionsr8  rF   r=  Ú_device_meshr‰  r%   rZ   r—  rW   r˜  r   Úlistdirr±  rp  r¹  r  Úfilename_to_tensorsÚ	fullmatchrG  r¸  Útensor_to_filenamerk   ÚtqdmrÖ   ru  r5   Ú
contiguousry  r-  rµ  r{  Úsafe_save_filer  rU  rY   r  ÚjsonÚdumpsÚwriterp   r-  ÚsaveÚ_upload_modified_files)3r¢   rA  rB  rã   rC  rD  r”  r¢  rE  rF  r¯  r.  r˜   Úquantization_serializableÚsave_directory_pathrI  rJ  rK  Úfiles_timestampsrU  Úmodel_to_saver–   Úpeft_state_dictr  r  Úactive_adapterÚcurrent_peft_configÚis_offloadedÚ	smp_to_hfr\  Ú
ignore_keyr“  rP  Ústate_dict_splitr¿  Úfull_filenameÚweights_no_suffixÚfilename_no_suffixÚregrW  Útensor_namesÚshard_state_dictÚtensor_namerÕ   ÚindexÚpath_to_weightsÚsave_index_filer6  ÚcontentÚ
model_cardrT  s3                                                     @r£   ri  zPreTrainedModel.save_pretrainedT  s²
  ø€ ðv ÐØ#ˆF�7‰Oå!(¨Ð/GÈÑ!OÔ!OÐå˜t ^°TÑ:Ô:ˆà Ð$Ðq­°LÅ+Ñ)NÔ)NÐqÐS_×SoÒSoÑSqÔSqð 	"ð Ð#Ð,BÐ#ÐKdÐ#Ýðu¨|Ô/OÔ/\ð uð uð uñô ð ð Œ=Ð$Õ-PÐQYÑ-ZÔ-ZÐ$ÝØmñô ð õ Œ7�>Š>˜.Ñ)Ô)ð 	Ý�LŠLÐ^¨>Ð^Ð^Ð^Ñ_Ô_Ð_ØˆFå
Œ�N¨TÐ2Ñ2Ô2Ð2Ý œi¨Ñ7Ô7Ðàð 	JØ#ŸZšZÐ(8¸$Ñ?Ô?ˆNØ—j’j Ð,?×,EÒ,EÅbÄgÄkÑ,RÔ,RÐSUÔ,VÑWÔWˆGØŸ
š
 ;°Ñ6Ô6ˆIØ*•f‘h”hÔ*¨7ÐLÐL¸TÐLÀVÐLÐLÔTˆGØ#×9Ò9¸.ÑIÔIÐàˆØÐ#Ø#/×#KÒ#KÈDÑ#QÔ#QÑ ˆJ˜Ø!ˆ�Ñõ % TÑ*Ô*ˆð Ô#ˆÝ%(¨¡Z¤Z×%5Ò%5°cÑ%:Ô%:¸1Ô%=ˆÔÔ"ð /<Ô.EÔ.N×.[Ò.[Ð\bÑ.cÔ.cÐ-dˆÔÔ*ð ×ÒÑ Ô ð 	IÝ˜t ^¸D¼KÐHÑHÔHÐHð ñ 	DØ)ð EØÔ$×4Ò4°^ÑDÔDÐDØ× Ò Ñ"Ô"ð PØÔ/×?Ò?ÀÑOÔOÐOà%ð DÝ—’Ø~ñô ð ð +×AÒAÈZÐAÑXÔX�
à#ð 1Ý—K’Kð ^ñô ð ð ')�OØ&0×&6Ò&6Ñ&8Ô&8ð Kð K™
˜˜UØEJ˜Ð(A¸CÐ(AÐ(AÑBÐBØ!0�Jà!%×!5Ò!5Ñ!7Ô!7�å�~Ñ&Ô&¨Ò*Ð*Ý$ðuñô ð ð "0°Ô!2�à&*Ô&6°~Ô&FÐ#Ø#×3Ò3°NÑCÔCÐCð ÐØ&×1Ò1Ñ3Ô3ˆJð ˆå�D˜/Ñ*Ô*ð		å•C˜Ô*×1Ò1Ñ3Ô3Ñ4Ô4Ñ5Ô5¸Ò9Ð9Ø˜$Ô,×3Ò3Ñ5Ô5Ð5Ð5¸À4ÔCU×C\ÒC\ÑC^ÔC^Ð9^Ð9^àˆLÝŒMð:ñô ð õ %ð 	3Ý #¤	Ô 8Ô Lð 3ð 3‘�	˜1Ø&˜Y zÑ2Ô2�
�
ð Ô'Ð3½¸DÔ<XÑ8YÔ8YÐ\]Ò8]Ð8]Ø"Ô:ð /ð /�
Ø Ð+Ð+Ø" :Ð.øð Œ=Ð$Ý3°JÀÄÈtÔO`ÐbfÔboÑpÔpˆJõ 9¸À]ÑSÔSˆ
ð  ð 	M¨ð 	MÐ=Sð 	MÝ1°-ÀÑLÔLˆJð &ð 	5Ý,ˆLÝ'¨°gÑ>Ô>ˆLˆLå4ˆLà'×/Ò/°¸ÑGÔG×OÒOÐP^Ð`vÑwÔwÐÝ=ØÐ)9È.ð
ñ 
ô 
Ðõ
 œ
 >Ñ2Ô2ð 	)ð 	)ˆHÝœGŸLšL¨¸ÑBÔBˆMð !-× 4Ò 4°V¸RÑ @Ô @× HÒ HÈÐY[Ñ \Ô \Ðð "*×!1Ò!1°&¸"Ñ!=Ô!=×!EÒ!EÀnÐVXÑ!YÔ!YÐÝ”*Ð4Ñ5Ô5ˆCð ×#Ò#Ð$5Ñ6Ô6ð)å”G—N’N =Ñ1Ô1ð)ð Ð$4Ô$HÐHÐHØ#ð Ià—M’MÐ"4Ñ5Ô5ÐAå”	˜-Ñ(Ô(Ð(øð ˆ
ØÔ&ð 	ð
 %ðØ)=ðØF\ðÐ Ô3Ð3àð õ )0¬ØÔ0×6Ò6Ñ8Ô8Ð?Uð)
ñ )
ô )
ð (	!ð (	!Ñ$ˆJ˜õ ”w—|’| N°JÑ?Ô?ˆHØ!ÐØ+ð Dð D�à#Ÿš¨Ñ4Ô4�ð
  ð R F¤MÔ$6¸&Ò$@Ð$@Ý5°mÀ[ÑQÔQ�Fð 17×0AÒ0AÑ0CÔ0CÐ  Ñ-Ð-ð ð Ð 4ð Ð=Sð ð
Ý'?ÀÐO_Ñ'`Ô'`Ð$à'Ô2ð nØ"×)Ò)Ð)mÐ)mÐ)mÐ)mÐUe×UjÒUjÑUlÔUlÐ)mÑ)mÔ)mÑmÔmÐmøøÝ ð ð ð Ý&ð<ñô ð ðøøøõ Ð+¨XÀÐIÑIÔIÐIà Ð ð ˆØÔ&ð 	à/°×1DÒ1DÑ1FÔ1FÐdÐJZÔJcÐdØ(ðð ˆEð
 ˆ=Ý œgŸlšl¨>¸<ÑHÔHˆOÝ�KŠKÐC°/ÐCÐCÑDÔDÐDÐDå5ˆOÝ œgŸlšl¨>½<ÈÐY`Ñ;aÔ;aÑbÔbˆOå�o s°WÐ=Ñ=Ô=ð !ÀÝœ* U°1ÀÐEÑEÔEÈÑL�Ø—’˜Ñ Ô Ð ð!ð !ð !ñ !ô !ð !ð !ð !ð !ð !ð !øøøð !ð !ð !ð !õ �KŠKð7È^ð 7ð 7ÝÐ 0Ô DÑEÔEð7ð 7à$3ð7ð 7ð 7ñô ð ð ð 	å2°7¸D¼OÐSXÐYÑYÔYˆJð �OŠO�BœGŸLšL¨¸ÑEÔEÑFÔFÐFà×'Ò'ØØØ Ø-ØØ#ð (ñ ô ð ð ð ð	ð 	s   ÜA
]Ý]0à70a3á3a7á:a7c                 óü   •— | j         �| j         ng }|                     dg ¦  «        }t          |t          ¦  «        r|g}|D ]}||vr|                     |¦  «         Œ|r||d<    t          ¦   «         j        |i |¤ŽS )Nr¨  )r-  rÀ   rs  rª   rU  rJ  rC  )r¢   r®  r¯  r¨  Útags_kwargsrª  r  s         €r£   rC  zPreTrainedModel.push_to_hub’  s™   ø€ à"&¤/Ð"=ˆtŒˆÀ2ˆà—j’j ¨Ñ,Ô,ˆÝ�k¥3Ñ'Ô'ð 	(Ø&˜-ˆKàð 	!ð 	!ˆCØ˜$ˆˆØ—’˜CÑ Ô Ð øàð 	"Ø!ˆF�6‰NØ"�u‰wŒwÔ" DÐ3¨FÐ3Ð3Ð3r¥   c                 óÀ   — t          d„ |                      ¦   «         D ¦   «         ¦  «        }|r0t          d„ |                      ¦   «         D ¦   «         ¦  «        }||z   }|S )a  
        Get the memory footprint of a model. This will return the memory footprint of the current model in bytes.
        Useful to benchmark the memory footprint of the current model and design some tests. Solution inspired from the
        PyTorch discussions: https://discuss.pytorch.org/t/gpu-memory-that-model-uses/56822/2

        Arguments:
            return_buffers (`bool`, *optional*, defaults to `True`):
                Whether to return the size of the buffer tensors in the computation of the memory footprint. Buffers
                are tensors that do not require gradients and not registered as parameters. E.g. mean and std in batch
                norm layers. Please see: https://discuss.pytorch.org/t/what-pytorch-means-by-buffers/120266/2
        c              3   óh   K  — | ]-}|                      ¦   «         |                     ¦   «         z  V — Œ.d S r    ©r=  r@  rÖ  s     r£   r   z7PreTrainedModel.get_memory_footprint.<locals>.<genexpr>®  s=   è è € ÐYÐY¸e�%—.’.Ñ"Ô" U×%7Ò%7Ñ%9Ô%9Ñ9ÐYÐYÐYÐYÐYÐYr¥   c              3   óh   K  — | ]-}|                      ¦   «         |                     ¦   «         z  V — Œ.d S r    rš  )rþ   Úbufs     r£   r   z7PreTrainedModel.get_memory_footprint.<locals>.<genexpr>°  s;   è è € ÐYÐYÀ3˜3Ÿ<š<™>œ>¨C×,<Ò,<Ñ,>Ô,>Ñ>ÐYÐYÐYÐYÐYÐYr¥   )ÚsumrÙ  rž  )r¢   Úreturn_buffersÚmemÚmem_bufss       r£   Úget_memory_footprintz$PreTrainedModel.get_memory_footprint¢  sd   € õ ÐYÐYÀtÇÂÑGXÔGXÐYÑYÔYÑYÔYˆØð 	!ÝÐYÐYÈ$Ï,Ê,É.Ì.ÐYÑYÔYÑYÔYˆHØ˜‘.ˆCØˆ
r¥   c                 ó  •— t          | dd ¦  «        t          j        k    r�ddlm}  t          ¦   «         j        |i |¤Ž |                      ¦   «         D ]Y}t          ||¦  «        rGt          |¦  «        dk    r	|d         }n| 
                    dd¦  «        }|                     |¦  «         ŒZ| S t          | dd ¦  «        t          j        k    r t          | dd¦  «        rt          d¦  «        ‚ t          ¦   «         j        |i |¤ŽS )	NÚquantization_methodr   ©Ú	HQQLinearrÖ   rÑ  Úis_loaded_in_8bitFzœCalling `cuda()` is not supported for `8-bit` quantized models.  Please use the model as it is, since the model has already been set to the correct devices.)rM  r~   ÚHQQÚhqq.core.quantizer¥  rJ  rÑ  r  rs  rß   rÀ   ÚBITS_AND_BYTESrÍ   )r¢   r®  r¯  r¥  rC  rÖ   r  s         €r£   rÑ  zPreTrainedModel.cuda´  s$  ø€ å�4Ð.°Ñ5Ô5Õ9KÔ9OÒOÐOØ3Ð3Ð3Ð3Ð3Ð3ð �E‰GŒGŒL˜$Ð) &Ð)Ð)Ð)ØŸ,š,™.œ.ð (ð (�Ý˜f iÑ0Ô0ð (Ý˜4‘y”y 1’}�}Ø!% a¤˜˜à!'§¢¨H°fÑ!=Ô!=˜Ø—K’K Ñ'Ô'Ð'øØˆKõ �4Ð.°Ñ5Ô5Õ9KÔ9ZÒZÐZÝ�tÐ0°%Ñ8Ô8ð Ý ðsñô ð ð �u‰wŒwŒ|˜TÐ, VÐ,Ð,Ð,r¥   c                 ód  •— d|v }|s#|D ] }t          |t          j        ¦  «        rd} nŒ!t          | dd ¦  «        t          j        k    r�ddlm}  t          ¦   «         j	        |i |¤Ž |  
                    ¦   «         D ]Y}t          ||¦  «        rGd|v r	|d         }n|d         }d|v r	|d         }n|r|}nd }|�||_        |                     |¦  «         ŒZ| S |r.t          | dd ¦  «        t          j        k    rt          d¦  «        ‚t          | dd ¦  «        t          j        k    rA|rt          d¦  «        ‚t          | d	d
¦  «        rt!          d¦  «        st          d¦  «        ‚n0t          | dd ¦  «        t          j        k    r|rt          d¦  «        ‚ t          ¦   «         j	        |i |¤ŽS )Nr–   Tr£  r   r¤  rÖ   zBCasting a Quark quantized model to a new `dtype` is not supported.z­You cannot cast a bitsandbytes model in a new `dtype`. Make sure to load the model using `from_pretrained` using the desired `dtype` by passing the correct `dtype` argument.r¦  Fz0.48zsYou need to install `pip install bitsandbytes>=0.48.0` if you want to move a 8-bit model across devices using to().z¥You cannot cast a GPTQ model in a new `dtype`. Make sure to load the model using `from_pretrained` using the desired `dtype` by passing the correct `dtype` argument.)rs  r®   r–   rM  r~   r§  r¨  r¥  rJ  r   r  Úcompute_dtyperÑ  ÚQUARKrÍ   r©  re   ÚGPTQ)
r¢   r®  r¯  Údtype_present_in_argsÚargr¥  rC  rÖ   r–   r  s
            €r£   r   zPreTrainedModel.toÎ  s@  ø€ ð !(¨6Ð 1Ðà$ð 	Øð ð �Ý˜c¥5¤;Ñ/Ô/ð Ø,0Ð)Ø�Eðõ �4Ð.°Ñ5Ô5Õ9KÔ9OÒOÐOØ3Ð3Ð3Ð3Ð3Ð3ð �E‰GŒGŒJ˜Ð' Ð'Ð'Ð'ØŸ,š,™.œ.ð (ð (�Ý˜f iÑ0Ô0ð (Ø 6Ð)Ð)Ø!'¨Ô!1˜˜à!% a¤˜Ø &Ð(Ð(Ø & w¤˜˜Ø.ð %Ø #˜˜à $˜ð Ð(Ø/4˜Ô,Ø—K’K Ñ'Ô'Ð'øØˆKà ð 	c¥W¨TÐ3HÈ$Ñ%OÔ%OÕSeÔSkÒ%kÐ%kÝÐaÑbÔbÐbõ �4Ð.°Ñ5Ô5Õ9KÔ9ZÒZÐZØ$ð Ý ðPñô ð õ
 �tÐ0°%Ñ8Ô8ð ÕAZÐ[aÑAbÔAbð Ý ð Jñô ð øõ �TÐ0°$Ñ7Ô7Õ;MÔ;RÒRÐRØ$ð Ý ðHñô ð ð �u‰wŒwŒz˜4Ð* 6Ð*Ð*Ð*r¥   c                 óp   •— t          | dd¦  «        rt          d¦  «        ‚ t          ¦   «         j        |Ž S )Nr¤   FzŽ`.half()` is not supported for quantized model. Please use the model as it is, since the model has already been casted to the correct `dtype`.)rM  rÍ   rJ  Úhalf©r¢   r®  r  s     €r£   r±  zPreTrainedModel.half  sD   ø€ å�4˜¨Ñ/Ô/ð 	'ÝðIñô ð ð
  •5‘7”7”< Ð&Ð&r¥   c                 óp   •— t          | dd¦  «        rt          d¦  «        ‚ t          ¦   «         j        |Ž S )Nr¤   Fz�`.float()` is not supported for quantized model. Please use the model as it is, since the model has already been casted to the correct `dtype`.)rM  rÍ   rJ  Úfloatr²  s     €r£   r´  zPreTrainedModel.float  sD   ø€ å�4˜¨Ñ/Ô/ð 	(ÝðIñô ð ð
 !•5‘7”7”= $Ð'Ð'r¥   r–   r¤   rÈ   c                 ó°  — t          || j        ¦  «        t          j        ¦   «         t	          ¦   «         g}|r!|                     t          ¦   «         ¦  «         t          ¦   «         rµdd l}|sw|sut           
                    d¦  «         |                     t          j        ¦   «         |j                             t          ¦   «         ¬¦  «        t!          ¦   «         g¦  «         nr|r5|                     t#          j        d¦  «        t'          ¦   «         g¦  «         n:|                     t#          j        d¦  «        t          j        ¦   «         g¦  «         |S )Nr   r°  r±  r  )rÒ   r¦   rµ  Úno_tie_weightsrP   rU  r<   r-   r´  r¶  r·  rN  r¶  r·  r¸  r+   rÉ   r®   rÖ   rÅ   Úmeta_device_safe_creation_ops)rO  r–   r¤   rÈ   rW  r»  r´  s          r£   Úget_init_contextz PreTrainedModel.get_init_context  sO  € õ
 +¨5°#´,Ñ?Ô?ÅÔATÑAVÔAVÕXeÑXgÔXgÐhˆàð 	:Ø× Ò Õ!6Ñ!8Ô!8Ñ9Ô9Ð9Ý%Ñ'Ô'ð 	_ØÐÐÐð  ð 
TÐ(:ð 
TÝ—’Ð^Ñ_Ô_Ð_Ø×$Ò$åÔ,Ñ.Ô.Ø!œ×+Ò+Õ@PÑ@RÔ@RÐ+ÑSÔSÝ'Ñ)Ô)ðñô ð ð ð ð TØ×$Ò$¥e¤l°6Ñ&:Ô&:Õ<OÑ<QÔ<QÐ%RÑSÔSÐSøð
 × Ò ¥%¤,¨vÑ"6Ô"6½Ô8ZÑ8\Ô8\Ð!]Ñ^Ô^Ð^àÐr¥   c                 ón  — i }| j         �M|t          j        k    r=|                     t                               | j         t          j        ¦  «        ¦  «         | j        �W|t          j        t          j        fv r=|                     t                               | j        t          j        ¦  «        ¦  «         |S )z\Create the dtype_plan describing modules/parameters that should use the `keep_in_fp32` flag.)	r4  r®   rÞ  ry  r­   Úfromkeysrà   r5  rß  )r¢   r–   r—   s      r£   Ú_get_dtype_planzPreTrainedModel._get_dtype_plan?  s’   € àˆ
ð
 Ô%Ð1°e½u¼}Ò6LÐ6LØ×Ò�dŸmšm¨DÔ,FÍÌÑVÔVÑWÔWÐWð Ô,Ð8¸UÅuÄ}ÕV[ÔVdÐFeÐ=eÐ=eØ×Ò�dŸmšm¨DÔ,MÍuÌ}Ñ]Ô]Ñ^Ô^Ð^àÐr¥   r]  ÚmodezMode | Nonec           	      ó˜  — |rÀt          ¦   «         s(t          dt          › dt          › dt          › d�¦  «        ‚ddlm}  |¦   «          |�et          |t          ¦  «        st          dt          |¦  «        › �¦  «        ‚|| _	        | 
                    | ¦  «         |                     | ¦  «         t          | |¬	¦  «         dS d
| _        dS )a‰  
        Set whether or not to use the `kernels` library to kernelize some layers of the model.
        Args:
            use_kernels (`bool`):
                Whether or not to use the `kernels` library to kernelize some layers of the model.
            kernel_config (`KernelConfig`, *optional*):
                The kernel configuration to use to kernelize the model. If `None`, the default kernel mapping will be used.
            mode (`Mode`, *optional*):
                The mode that should be applied during `kernelize`. Optional, defaults to either training or inference mode
                based on the internal `training` flag.
        z@Kernels are not available. Please install a compatible version (z <= version < z), e.g. `pip install kernels==ú`r   )Ú$register_kernel_mapping_transformersNz=Expeced `kernel_config` to be of type `KernelConfig` but got )r¼  F)rg   rÍ   rt   rs   Úintegrations.hub_kernelsr¿  rs  r^   ru  r]  Úsanitize_kernel_mappingÚcreate_compatible_mappingr>   Ú_use_kernels)r¢   Úuse_kernelsr]  r¼  r¿  s        r£   Úset_use_kernelszPreTrainedModel.set_use_kernelsO  s%  € ð ð 	&Ý'Ñ)Ô)ð Ý ðIÝ<OðIð IÝ_rðIð Iå2EðIð Ið Iñô ð ð
 WÐVÐVÐVÐVÐVà0Ð0Ñ2Ô2Ð2àÐ(Ý! -µÑ>Ô>ð Ý$ØmÕX\Ð]jÑXkÔXkÐmÐmñô ð ð
 &3�Ô"ð ×5Ò5°dÑ;Ô;Ð;ð ×7Ò7¸Ñ=Ô=Ð=å�d Ð&Ñ&Ô&Ð&Ð&Ð&à %ˆDÔÐÐr¥   r¤  )rÇ  rž  r‘   rŸ  r¡  r¢  r£  r�   r›   Úfusion_configr�   rO  r�   rž  r‘   rŸ  r¡  r£  r�   r›   rÆ  r�   c                ó¦  — |                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      d	d¦  «        }|                      d
d¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|                      di ¦  «        pi                      ¦   «         }|                      dd¦  «        } |                      dd¦  «        }!|                      dd¦  «        }"|                      dd¦  «        }#|                      dd¦  «        }$|                      dd¦  «        }%|                      dd¦  «        }&|                      dd¦  «        }'|                      dd¦  «        }(|                      dd¦  «        })|                      d d¦  «        }*|                      d!d¦  «        }+|%�|#€d"}#d#D ]},|                      |,d¦  «        }-Œ|�|�|n|}|€d"}t          ¦   «         r|sd$}|||||||d%œ}.i |.¥d&|i¥}/|�|€|"�t          d'¦  «        ‚|d"k    rGt	          t
          j                             d(d)¦  «        ¦  «        rt           	                    d*¦  «         |#€|$�t          |#|$|&|¬+¦  «        \  }}&}$|"�t          ¦   «         st          d,¦  «        ‚|€i }t          ||/fi |¤Ž\  }0}}t          |¦  «        }d-d.|d/œ}1|�||1d0<   t          |t          ¦  «        sn|�|n|}2| j        }3|3€t          | j        › d1�¦  «        ‚ |3j        |2fd$|"||d2œ|.¤|¤Ž\  }}4d|4v r|4                      d¦  «         |4                      d|¦  «        }n't          j        |¦  «        }|}4t)          |d|¦  «        }||/d&<   d3|v r|                      d3¦  «        |_        d4|v r|                      d4¦  «        |_        t/          ||||
|1¦  «        \  }5}}|"rQ|5�t          d5¦  «        ‚|�>t          |t0          ¦  «        rd6|                     ¦   «         v sd6|v rt5          d7¦  «        ‚|*�|)st                               d8¦  «         d$})t9          |||"|	|/|1|                      ¦   «         t)          |d9d¦  «        |¬:¦	  «	        \  }6}7|5du}8t=          ||6||7||
|5¦  «        \  }}|"rXd;d<lm }9 tC          j"        d=¦  «        5   | |¦  «        }:ddd¦  «         n# 1 swxY w Y    |9|6d>         d$|:|¬?¦  «        d@         }||_#        |�t          j        |¦  «        |_$        t)          |dAd¦  «        }|�d;dBl%m&};  |;| ||¦  «         |*�O|)rMd;dCl'm(}< |(rtS          ¦   «         gng }=tU          |=¦  «        5   |<| ||*¦  «         ddd¦  «         n# 1 swxY w Y   |  +                    ||8tX          |(¦  «        }>t          j        |¦  «        }tU          |>¦  «        5   | |g|¢R i |4¤Ž}?t[          |?¦  «         |5�|5 .                    |?|||6|)¬D¦  «         ddd¦  «         n# 1 swxY w Y   |? /                    |¦  «        }@ta          |?|+|5¦  «        }Atb          r|&�te          |?|#|%|&|$¦  «        }?|�tg          |?|||5¦  «        }ti          |||7|||||@|5|&|
|A|	|.|¬E¦  «        }B|  5                    |?||6|B¦  «        \  }C}D|  6                    |?|B|C¦  «        }C|? 7                    ¦   «          |? 8                    |)|*¦  «         |? 9                    ¦   «         r)tu          |?dF¦  «        r|"s |?j;        |!|||fi |.¤d|'i¤|¤Ž |�ity          t{          |                     ¦   «         ¦  «        ¦  «        d;k    s#d6t{          |                     ¦   «         ¦  «        v rt}          |?|5|||D|¦  «         |5�|5|?_?        |5 @                    |?¦  «         |0� |�||dG<   |? A                    |0| |B|¬H¦  «        }C|r|?|C B                    ¦   «         fS |?S )Ia´@  
        Instantiate a pretrained pytorch model from a pre-trained model configuration.

        The model is set in evaluation mode by default using `model.eval()` (Dropout modules are deactivated). To train
        the model, you should first set it back in training mode with `model.train()`.

        The warning *Weights from XXX not initialized from pretrained model* means that the weights of XXX do not come
        pretrained with the rest of the model. It is up to you to train those weights with a downstream fine-tuning
        task.

        The warning *Weights from XXX not used in YYY* means that the layer XXX is not used by YYY, therefore those
        weights are discarded.

        Parameters:
            pretrained_model_name_or_path (`str` or `os.PathLike`, *optional*):
                Can be either:

                    - A string, the *model id* of a pretrained model hosted inside a model repo on huggingface.co.
                    - A path to a *directory* containing model weights saved using
                      [`~PreTrainedModel.save_pretrained`], e.g., `./my_model_directory/`.
                    - `None` if you are both providing the configuration and state dictionary (resp. with keyword
                      arguments `config` and `state_dict`).
            model_args (sequence of positional arguments, *optional*):
                All remaining positional arguments will be passed to the underlying model's `__init__` method.
            config (`Union[PreTrainedConfig, str, os.PathLike]`, *optional*):
                Can be either:

                    - an instance of a class derived from [`PreTrainedConfig`],
                    - a string or path valid as input to [`~PreTrainedConfig.from_pretrained`].

                Configuration for the model to use instead of an automatically loaded configuration. Configuration can
                be automatically loaded when:

                    - The model is a model provided by the library (loaded with the *model id* string of a pretrained
                      model).
                    - The model was saved using [`~PreTrainedModel.save_pretrained`] and is reloaded by supplying the
                      save directory.
                    - The model is loaded by supplying a local directory as `pretrained_model_name_or_path` and a
                      configuration JSON file named *config.json* is found in the directory.
            state_dict (`dict[str, torch.Tensor]`, *optional*):
                A state dictionary to use instead of a state dictionary loaded from saved weights file.

                This option can be used if you want to create a model from a pretrained configuration but load your own
                weights. In this case though, you should check if using [`~PreTrainedModel.save_pretrained`] and
                [`~PreTrainedModel.from_pretrained`] is not a simpler option.
            cache_dir (`Union[str, os.PathLike]`, *optional*):
                Path to a directory in which a downloaded pretrained model configuration should be cached if the
                standard cache should not be used.
            ignore_mismatched_sizes (`bool`, *optional*, defaults to `False`):
                Whether or not to raise an error if some of the weights from the checkpoint do not have the same size
                as the weights of the model (if for instance, you are instantiating a model with 10 labels from a
                checkpoint with 3 labels).
            force_download (`bool`, *optional*, defaults to `False`):
                Whether or not to force the (re-)download of the model weights and configuration files, overriding the
                cached versions if they exist.
            proxies (`dict[str, str]`, *optional*):
                A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
                'http://hostname': 'foo.bar:4012'}`. The proxies are used on each request.
            output_loading_info(`bool`, *optional*, defaults to `False`):
                Whether or not to also return a dictionary containing missing keys, unexpected keys and error messages.
            local_files_only(`bool`, *optional*, defaults to `False`):
                Whether or not to only look at local files (i.e., do not try to download the model).
            token (`str` or `bool`, *optional*):
                The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
                the token generated when running `hf auth login` (stored in `~/.huggingface`).
            revision (`str`, *optional*, defaults to `"main"`):
                The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a
                git-based system for storing models and other artifacts on huggingface.co, so `revision` can be any
                identifier allowed by git.

                <Tip>

                To test a pull request you made on the Hub, you can pass `revision="refs/pr/<pr_number>"`.

                </Tip>
            attn_implementation (`str`, *optional*):
                The attention implementation to use in the model (if relevant). Can be any of
                    - `"eager"` (manual implementation of the attention)
                    - `"sdpa"` (using [`F.scaled_dot_product_attention`](https://pytorch.org/docs/master/generated/torch.nn.functional.scaled_dot_product_attention.html))
                    - `"flash_attention_2"` (using [Dao-AILab/flash-attention](https://github.com/Dao-AILab/flash-attention))
                    - `"flash_attention_3"` (using [Dao-AILab/flash-attention/hopper](https://github.com/Dao-AILab/flash-attention/tree/main/hopper))
                    - `"flash_attention_4"` (using [Dao-AILab/flash-attention/flash_attn/cute](https://github.com/Dao-AILab/flash-attention/tree/main/flash_attn/cute)).
                By default, if available, SDPA will be used. The default is otherwise the manual `"eager"` implementation.

                Accept HF kernel references in the form:
                  <namespace>/<repo_name>[@<revision>][:<kernel_name>]

                - <namespace> and <repo_name> are any non-"/" and non-":" sequences.
                - "@<revision>" is optional (branch, tag, or commit-ish), e.g. "@main", "@v1.2.0", "@abc123".
                - ":<kernel_name>" is optional and selects a function inside the kernel repo.
                - Both options can appear together and in this order only: @revision first, then :kernel_name.
                - We intentionally allow a leading "<wrapper>|" prefix (e.g., "flash|...") because the code
                  strips it before loading; '|' is not excluded in the character classes here.

                Examples that match:
                  "org/model"
                  "org/model@main"
                  "org/model:custom_kernel"
                  "org/model@v1.2.3:custom_kernel"
            experts_implementation (`str`, *optional*):
                The experts implementation to use in the model (if relevant). Can be any of:

                - `"eager"` (sequential implementation of the experts matrix multiplications).
                - `"batched_mm"` (using [`torch.bmm`](https://pytorch.org/docs/stable/generated/torch.bmm.html)).
                - `"grouped_mm"` (using [`torch.nn.functional.grouped_mm`](https://docs.pytorch.org/docs/main/generated/torch.nn.functional.grouped_mm.html)).

                By default, if the model supports it, `"grouped_mm"` will be used. The default is otherwise the manual `"eager"` implementation.

            > Parameters for big model inference

            dtype (`str` or `torch.dtype`, *optional*, defaults to `"auto"`):
                Override the default `torch_dtype` and load the model under a specific `dtype`. The different options
                are:

                1. `torch.float16` or `torch.bfloat16` or `torch.float`: load in a specified
                  `dtype`, ignoring the model's `config.dtype` if one exists. If not specified
                  - the model will get loaded in `torch.float` (fp32).

                2. `"auto"` - A `dtype` or `torch_dtype` entry in the `config.json` file of the model will be
                  attempted to be used. If this entry isn't found then next check the `dtype` of the first weight in
                  the checkpoint that's of a floating point type and use that as `dtype`. This will load the model
                  using the `dtype` it was saved in at the end of the training. It can't be used as an indicator of how
                  the model was trained. Since it could be trained in one of half precision dtypes, but saved in fp32.

                3. A string that is a valid `torch.dtype`. E.g. "float32" loads the model in `torch.float32`, "float16" loads in `torch.float16` etc.

                <Tip>

                For some models the `dtype` they were trained in is unknown - you may try to check the model's paper or
                reach out to the authors and ask them to add this information to the model's card and to insert the
                `dtype` or `torch_dtype` entry in `config.json` on the hub.

                </Tip>

            device_map (`str` or `dict[str, Union[int, str, torch.device]]` or `int` or `torch.device`, *optional*):
                A map that specifies where each submodule should go. It doesn't need to be refined to each
                parameter/buffer name, once a given module name is inside, every submodule of it will be sent to the
                same device. If we only pass the device (*e.g.*, `"cpu"`, `"cuda:1"`, `"mps"`, or a GPU ordinal rank
                like `1`) on which the model will be allocated, the device map will map the entire model to this
                device. Passing `device_map = 0` means put the whole model on GPU 0.

                To have Accelerate compute the most optimized `device_map` automatically, set `device_map="auto"`. For
                more information about each option see [designing a device
                map](https://hf.co/docs/accelerate/main/en/usage_guides/big_modeling#designing-a-device-map).
            max_memory (`Dict`, *optional*):
                A dictionary device identifier to maximum memory if using `device_map`. Will default to the maximum memory available for each
                GPU and the available CPU RAM if unset.
            tp_plan (`Optional[Union[dict, str]]`, *optional*):
                A torch tensor parallel plan, see [here](https://pytorch.org/tutorials/intermediate/TP_tutorial.html). Use `tp_plan="auto"` to
                use the predefined plan based on the model. If it's a dict, then it should match between module names and desired layout.
                Note that if you use it, you should launch your script accordingly with `torchrun [args] script.py`. This will be much
                faster than using a `device_map`, but has limitations.
            tp_size (`str`, *optional*):
                A torch tensor parallel degree. If not provided would default to world size.
            device_mesh (`torch.distributed.DeviceMesh`, *optional*):
                A torch device mesh. If not provided would default to world size. Used only for tensor parallel for now.
                If provided, it has to contain dimension named `"tp"` in case it's > 1 dimensional, this dimension will be used for tensor parallelism
            offload_folder (`str` or `os.PathLike`, *optional*):
                If the `device_map` contains any value `"disk"`, the folder where we will offload weights.
            offload_buffers (`bool`, *optional*):
                Whether or not to offload the buffers with the model parameters.
            quantization_config (`Union[QuantizationConfigMixin,Dict]`, *optional*):
                A dictionary of configuration parameters or a QuantizationConfigMixin object for quantization (e.g
                bitsandbytes, gptq).
            subfolder (`str`, *optional*, defaults to `""`):
                In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can
                specify the folder name here.
            variant (`str`, *optional*):
                If specified load weights from `variant` filename, *e.g.* pytorch_model.<variant>.bin.
            use_safetensors (`bool`, *optional*, defaults to `None`):
                Whether or not to use `safetensors` checkpoints. Defaults to `None`. If not specified and `safetensors`
                is not installed, it will be set to `False`.
            weights_only (`bool`, *optional*, defaults to `True`):
                Indicates whether unpickler should be restricted to loading only tensors, primitive types,
                dictionaries and any types added via torch.serialization.add_safe_globals().
                When set to False, we can load wrapper tensor subclass weights.
            disable_mmap (`bool`, *optional*):
                Whether to disable memory mapping when loading safetensors checkpoints. When `None` (default),
                it is auto-detected to `True` when the checkpoint lives on an `hf-mount` FUSE filesystem
                (used by HF Spaces/Endpoints), where mmap + parallel page-faults can deadlock. When `True`,
                files are read fully into memory and parsed with `safetensors.torch.load`. When `False`, the
                default memory-mapped loader is always used.
            fusion_config (`dict[str, bool | dict[str, Any]]`, *optional*):
                Optional fusion configuration applied before model instantiation. Each key enables a fusion family and
                its value can either be `True` to enable that fusion with default options or a dictionary of
                family-specific options. For example, `{"patch_embeddings": True}` enables patch embedding fusion.
                This should only be used as an inference optimization, as it can slightly change outputs. If omitted,
                `from_pretrained()` falls back to `config.fusion_config` when available. Refer to the fusion mapping
                guide in `docs/source/en/fusion_mapping.md` for more details.
            key_mapping (`dict[str, str], *optional*):
                A potential mapping of the weight names if using a model on the Hub which is compatible to a Transformers
                architecture, but was not converted accordingly.
            kwargs (remaining dictionary of keyword arguments, *optional*):
                Can be used to update the configuration object (after it being loaded) and initiate the model (e.g.,
                `output_attentions=True`). Behaves differently depending on whether a `config` is provided or
                automatically loaded:

                    - If a configuration is provided with `config`, `**kwargs` will be directly passed to the
                      underlying model's `__init__` method (we assume all relevant updates to the configuration have
                      already been done)
                    - If a configuration is not provided, `kwargs` will be first passed to the configuration class
                      initialization function ([`~PreTrainedConfig.from_pretrained`]). Each key of `kwargs` that
                      corresponds to a configuration attribute will be used to override said attribute with the
                      supplied `kwargs` value. Remaining keys that do not correspond to any configuration attribute
                      will be passed to the underlying model's `__init__` function.

        <Tip>

        Activate the special ["offline-mode"](https://huggingface.co/transformers/installation.html#offline-mode) to
        use this method in a firewalled environment.

        </Tip>

        Examples:

        ```python
        >>> from transformers import BertConfig, BertModel

        >>> # Download model and configuration from huggingface.co and cache.
        >>> model = BertModel.from_pretrained("google-bert/bert-base-uncased")
        >>> # Model was saved using *save_pretrained('./test/saved_model/')* (for example purposes, not runnable).
        >>> model = BertModel.from_pretrained("./test/saved_model/")
        >>> # Update configuration during loading.
        >>> model = BertModel.from_pretrained("google-bert/bert-base-uncased", output_attentions=True)
        >>> assert model.config.output_attentions == True
        ```
        rã   Nr   rœ  Úoutput_loading_infoFÚ_from_pipelineÚ
_from_autor–   r­  r“   Ú
max_memoryÚoffload_folderr•   r]  r¥  r¦  rª  r”  Úadapter_kwargsÚadapter_namerx  rg  r˜  r�  Útp_sizerŽ  rš   Útrust_remote_coderW  rÄ  r]  Úkey_mappingrÉ  )ÚmirrorÚ
_fast_initÚlow_cpu_mem_usageÚfrom_tfÚ	from_flaxÚoffload_state_dictT)rž  rŸ  r   r¡  r¢  r£  r¥  r§  zq`state_dict` cannot be passed together with a model name or a `gguf_file`. Use one of the two loading strategies.Ú
WORLD_SIZEr…   a  You've set device_map=`auto` while triggering a distributed run with torchrun. This might lead to unexpected behavior. If your plan is to load the model on each device, you should set device_map={: PartialState().process_index} where PartialState comes from accelerate library)rÏ  rš   r“   zIaccelerate is required when loading a GGUF file `pip install accelerate`.ri  Úpytorch)Ú	file_typer%  Úfrom_auto_classÚusing_pipelinezN does not define `config_class`; pass an explicit config to `from_pretrained`.)Úreturn_unused_kwargsr˜  rÊ  rÉ  r®  r¯  zÂYou cannot combine Quantization and loading a model from a GGUF file, try again by making sure you did not passed a `quantization_config` or that you did not load a quantized model from the Hub.rO  zxOne or more modules is configured to be mapped to disk. Disk offload is not supported for models loaded from GGUF files.zœA kernel_config was provided but use_kernels is False; setting use_kernels=True automatically. To suppress this warning, explicitly set use_kernels to True.Útransformers_weights)	r�   r”  r˜  r�   r�   r™  rš  r›  rœ  r   )Úload_gguf_checkpointr  r   )Úreturn_tensorsÚmodel_to_loadr­  rS  rÆ  )Úregister_fusion_patches)Ú(register_kernel_replacements_and_fusions)ri  r–   r“   rÅ  rÄ  )r�   r‘   r’   r“   r”   r•   r–   r—   r˜   rš   r›   rœ   r�   r�   r�   Úadjust_generation_fnr¢  )rÎ  Úload_configrÍ  )CrX  r€  r   rÍ   r½   r¾   r¿   rÀ   r¶  r·  rG   rd   r@   r2   rs  r    r)  r¦   Úfrom_pretrainedÚdeepcopyrM  r_  rc  rT   r­   rÞ   r{  rË  rÆ  rš  rÐ  Úmodeling_gguf_pytorch_utilsrß  r®   rÖ   r\  rÆ  Úfusion_mappingrâ  rÀ  rã  r<   r]   r¸  rÈ   rQ   Úpreprocess_modelr»  r!   r´   rE   r/   rŒ   Ú_load_pretrained_modelÚ_finalize_model_loadingÚevalrÅ  re  rµ   rä  rß   re  r1   r˜   Úpostprocess_modelÚload_adapterÚto_dict)ErO  r�   rÇ  rž  r‘   rŸ  r¡  r¢  r£  r�   r›   rÆ  r�   Ú
model_argsr¯  rã   r   rœ  rÈ  Úfrom_pipelinerÛ  r–   r­  r“   rË  rÌ  r•   r]  r¥  r§  r”  rÍ  rÎ  rg  r˜  r�  rÏ  rŽ  rš   rÐ  rW  rÄ  r]  rÑ  rJ  r\  r�   Údownload_kwargs_with_commitÚ_adapter_model_pathr™  Úconfig_pathr)  Úmodel_kwargsr˜   rÅ  r’   r¤   rß  Údummy_modelrâ  rã  Úallow_all_kernels_contextÚmodel_init_contextri  r—   Úweight_conversionsrå  Úloading_infoÚdisk_offload_indexsE                                                                        r£   ræ  zPreTrainedModel.from_pretrainedy  s  € ðj —Z’Z ¨dÑ3Ô3ˆ
Ø—*’*˜Y¨Ñ-Ô-ˆØ—Z’Z ¨dÑ3Ô3ˆ
Ø$ŸjšjÐ)>ÀÑFÔFÐØŸ
š
Ð#3°TÑ:Ô:ˆØ Ÿ*š* \°5Ñ9Ô9ˆØ—
’
˜7 DÑ)Ô)ˆØ—j’j °Ñ5Ô5ˆØ—Z’Z ¨dÑ3Ô3ˆ
Ø—Z’Z ¨dÑ3Ô3ˆ
ØŸšÐ$4°dÑ;Ô;ˆØ Ÿ*š*Ð%6¸Ñ>Ô>ˆØ$ŸjšjÐ)>ÀÑEÔEÐØ—J’J˜{¨BÑ/Ô/ˆ	Ø—j’j °Ñ6Ô6ˆØ—*’*˜Y¨Ñ-Ô-ˆØ Ÿ*š*Ð%5°rÑ:Ô:Ð@¸b×FÒFÑHÔHˆØ—z’z .°)Ñ<Ô<ˆØ"ŸJšJÐ':¸DÑAÔAÐØ—J’J˜{¨DÑ1Ô1ˆ	Ø—*’*˜Y¨Ñ-Ô-ˆØ—*’*˜Y¨Ñ-Ô-ˆØ06·
²
Ð;OÐQUÑ0VÔ0VÐØ—j’j °Ñ5Ô5ˆØ"ŸJšJÐ':¸DÑAÔAÐØ"ŸJšJÐ':¸EÑBÔBÐØ—j’j °Ñ6Ô6ˆØŸ
š
 ?°DÑ9Ô9ˆØ—j’j °Ñ5Ô5ˆàÐ)¨g¨oØˆGð pð 	'ð 	'ˆDØ—
’
˜4 Ñ&Ô&ˆAˆAð Ð"Ø"Ð.�E�E°KˆEØˆ=ØˆEåÑÔð 	$Ð%5ð 	$Ø#Ðð #Ø,ØØ 0ØØ Ø"ð
ð 
ˆð 'V¨Ð&U¸-ÈÐ&UÐ&UÐ#àÐ!Ð'DÐ'PÐT]ÐTiÝð Dñô ð ð ˜ÒÐ¥C­¬
¯ª°|ÀSÑ(IÔ(IÑ$JÔ$JÐÝ�KŠKðcñô ð ð Ð 'Ð"5Ý/LØ °kÈjð0ñ 0ô 0Ñ,ˆJ˜ Wð Ð Õ)@Ñ)BÔ)BÐ ÝÐhÑiÔiÐiàÐ!ØˆNåM`Ø)Ø'ðN
ð N
ð ðN
ð N
ÑJÐÐ:¸Nõ
 .¨jÑ9Ô9ˆ
à#*¸ÐWfÐgÐgˆ
ØÐ$Ø+8ˆJÐ'Ñ(õ ˜&Õ"2Ñ3Ô3ð 	GØ$*Ð$6˜&˜&Ð<YˆKØÔ+ˆLØÐ#Ý Ø”|ÐsÐsÐsñô ð ð $@ <Ô#?Øð$à%)Ø#Ø*Ø,ð$ð $ð "ð$ð ð$ð $Ñ ˆF�Lð ˜lÐ*Ð*Ø× Ò  Ñ-Ô-Ð-Ø&×*Ò*¨>¸;ÑGÔGˆKˆKå”] 6Ñ*Ô*ˆFØ!ˆLÝ! &¨.¸+ÑFÔFˆKà5@Ð# MÑ2ð ! FÐ*Ð*Ø*0¯*ª*Ð5JÑ*KÔ*KˆFÔ'à# vÐ-Ð-Ø-3¯ZªZÐ8PÑ-QÔ-QˆFÔ*å+;ØÐ'¨°\À:ñ,
ô ,
Ñ(ˆ�f˜jð ð 	ØÐ'Ý ð Yñô ð ð Ð%Ý˜J­Ñ-Ô-ð &Ø28¸J×<MÒ<MÑ<OÔ<OÐ2OÐ2OÐTZÐ^hÐThÐThå"ð.ñô ð ð
 Ð$¨[Ð$Ý×Òð oñô ð ð ˆKå-KØ*GØØØ+Ø7Ø!Ø×-Ò-Ñ/Ô/Ý+2°6Ð;QÐSWÑ+XÔ+XØ!ð
.
ñ 
.
ô 
.
Ñ*ÐÐ*ð $¨4Ð/ˆõ #ØÐ# VÐ-=¸zÈ<ÐYeñ
ô 
‰ˆ�ð ð 
	ØIÐIÐIÐIÐIÐIõ ”˜fÑ%Ô%ð *ð *Ø!˜c &™kœk�ð*ð *ð *ñ *ô *ð *ð *ð *ð *ð *ð *øøøð *ð *ð *ð *ð .Ð-Ø  Ô#°DÈÐafðñ ô àôˆJð <ˆÔð Ð$Ý#'¤=°Ñ#?Ô#?ˆFÔ õ   ¨¸Ñ>Ô>ˆØÐ$Ø?Ð?Ð?Ð?Ð?Ð?à#Ð# C¨°Ñ?Ô?Ð?ð Ð$¨Ð$ØZÐZÐZÐZÐZÐZð FWÐ(^Õ)>Ñ)@Ô)@Ð(AÐ(AÐ\^Ð%Ý Ð!:Ñ;Ô;ð Uð UØ8Ð8¸¸fÀmÑTÔTÐTðUð Uð Uñ Uô Uð Uð Uð Uð Uð Uð Uøøøð Uð Uð Uð Uð !×1Ò1°%¸ÕGYÐ[lÑmÔmÐå”˜vÑ&Ô&ˆÝÐ/Ñ0Ô0ð 	ð 	Ø�C˜Ð< Ð<Ð<Ð<¨|Ð<Ð<ˆEÝ" 5Ñ)Ô)Ð)àÐ'Ø×-Ò-ØØØ)Ø%5Ø +ð .ñ ô ð ð	ð 	ð 	ñ 	ô 	ð 	ð 	ð 	ð 	ð 	ð 	øøøð 	ð 	ð 	ð 	ð ×*Ò*¨5Ñ1Ô1ˆ
õ :¸%ÀÈlÑ[Ô[Ðå'ð 	_¨KÐ,CÝ$ U¨GÐ5GÈÐV]Ñ^Ô^ˆEð Ð!Ý(¨°
¸JÈÑUÔUˆJõ *Ø*GØ$;Ø-Ø!Ø .Ø+ØØ!Ø%Ø#Ø%Ø-Ø+Ø+Ø%ð
ñ 
ô 
ˆð" ,/×+EÒ+EÀeÈZÐYiÐkvÑ+wÔ+wÑ(ˆÐ(Ø×2Ò2°5¸+À|ÑTÔTˆØ�
Š
‰ŒˆØ×Ò˜k¨=Ñ9Ô9Ð9ð ×ÒÑÔð 		¥G¨EÐ3IÑ$JÔ$Jð 		ÐS\ð 		Ø&ˆEÔ&Ø!ØØØ-ð	ð ð
 "ðð ð #4ðð ð ðð ð ð Ð!¥s­3¨z×/@Ò/@Ñ/BÔ/BÑ+CÔ+CÑ'DÔ'DÀqÒ'HÐ'HÈFÕVYÐZd×ZkÒZkÑZmÔZmÑVnÔVnÐLnÐLnÝ  |°ZÀÐQcÐetÑuÔuÐuàÐ#Ø!-ˆEÔØ×*Ò*Øñô ð ð Ð*ØÐ Ø*/�˜wÑ'Ø ×-Ò-Ø#Ø)Ø'Ø-ð	 .ñ ô ˆLð ð 	1Ø˜,×.Ò.Ñ0Ô0Ð0Ð0Øˆs6   ÖV3Ö3V7Ö:V7ÙY'Ù'Y+Ù.Y+Ú28[6Û6[:Û=[:ri  rÅ  rå  Úexpected_keysc           	      ó   — |j         }|j        }|duo#|j        j        t          j        t          j        hv }|€3t          |                      ¦   «          	                    ¦   «         ¦  «        n|}t          j        t          j        k    rt          |t          | dd¦  «        ¦  «         d}|j        �Cd|j                             ¦   «         v r(t%          | |j        ||j        |j        |j        ¦  «        }|j        �-|s+t-          |j        |¦  «        }	t/          | |	|j         ¦  «         g }
t1          ¦   «         r|s}|€9i }|D ]2}|                     t5          |d|j        |j        ¬¦  «        ¦  «         Œ3|}t;          | ||¦  «        \  }
}t=          ||
t?          ¦   «         t?          ¦   «         i ¬¦  «        }�nÆt?          ¦   «         }|�|}�nz|��5|d                               d¦  «        �r|�€i }|D �]}|j        stC          |¦  «        r]tE          |d	¦  «        5 }|                     tG          | $                    ¦   «         ¦  «        ¦  «         ddd¦  «         n# 1 swxY w Y   Œv|j        duo/tK          d
„ |j                             ¦   «         D ¦   «         ¦  «        }|rdnd\  }}tM          |d||¬¦  «        }| '                    |¦  «         | 	                    ¦   «         D ]}| (                    |¦  «        ||<   Œ�ŒnB|�1i }|D ]+}|                     t5          ||j        ¬¦  «        ¦  «         Œ,ntS          d¦  «        ‚tU          | ||| j+        |¬¦  «        \  }}|D ]}| ,                    ddd¦  «         Œ||fS )zzPerform the actual loading of some checkpoints into a `model`, by reading them from disk and dispatching them accordingly.Nr=  rO  rÔ   )r  r›   r�   )rÄ  Ú
error_msgsÚunexpected_keysÚmismatched_keysÚconversion_errorsr   r  r  c              3   ód   K  — | ]+}t          |t          j        ¦  «        r|j        n|d k    V — Œ,dS )ÚmpsN)rs  r®   rÖ   ru  )rþ   Úds     r£   r   z9PreTrainedModel._load_pretrained_model.<locals>.<genexpr>Ý  sZ   è è € ð Hð Hàõ $.¨aµ´Ñ#>Ô#>ÐE˜œ˜ÀAÈ%ÒOðHð Hð Hð Hð Hð Hr¥   )Úpreadr  )r'  rÔ   r$  )r%  rÖ   Úbackend)r�   z5Neither a state dict nor checkpoint files were found.)ri  rã   rå  r�  rü  )-r˜   r¤   r]  r^  r~   r§  r¬  r¯   rã   r-  r¶  Úlevelrk   ÚWARNINGrI   rM  r“   rÞ   r0   r”   r’   rœ   r3   Úcaching_allocator_warmupr-   ry  r:  r›   r�   r6   rz   re  r)  r  r  r*  r+  rw  r   rW  r.  rÍ   r$   r�  Ú__exit__)ri  rã   rÅ  rå  rý  r˜   r¤   Úis_hqq_or_quarkrü  Úexpanded_device_maprÿ  Úmerged_state_dictÚ	ckpt_filerÄ  rû  Úall_pointerÚfiler5  Úis_mpsr  rÖ   Úfile_pointerr!  s                          r£   rë  z&PreTrainedModel._load_pretrained_model’  s‘  € ð #Ô/ˆØ"Ô/ˆØ&¨dÐ2ð 
°|Ô7WÔ7dÝÔ"ÝÔ$ði
ð 8
ˆð <IÐ;P�˜U×-Ò-Ñ/Ô/×4Ò4Ñ6Ô6Ñ7Ô7Ð7ÐVcˆåŒ<�7œ?Ò*Ð*Ý˜=­'°%¸ÀTÑ*JÔ*JÑKÔKÐKð "ÐàÔ!Ð-°&¸KÔ<R×<YÒ<YÑ<[Ô<[Ð2[Ð2[Ý!8ØØÔ/Ø ØÔ&ØÔ,ØÔ*ñ"ô "Ðð Ô!Ð-°oÐ-Ý"3°KÔ4JÈMÑ"ZÔ"ZÐÝ$ UÐ,?ÀÔAYÑZÔZÐZàˆ
å%Ñ'Ô'ð <	-°ð <	-ØÐ!Ø$&Ð!Ø!1ð ð �IØ%×,Ò,Ý'Ø%Ø).Ø)4Ô)AØ)4Ô)Að	ñ ô ñô ð ð ð /�
Ý'HÈÐPZÐ\gÑ'hÔ'hÑ$ˆJ˜å,Ø)Ø%Ý #¡¤Ý #¡¤Ø"$ðñ ô ˆL‰Lõ ™%œ%ˆKØÐ%Ø$.Ð!Ñ!Ø!Ñ-Ð2BÀ1Ô2E×2NÒ2NÈ~Ñ2^Ô2^Ñ-ÐcmÑcuØ$&Ð!Ø,ð Iñ I�DØ"Ô/ð !µ?À4Ñ3HÔ3Hð !Ý! $¨Ñ-Ô-ð S°Ø-×4Ò4Õ5EÀcÇhÂhÁjÄjÑ5QÔ5QÑRÔRÐRðSð Sð Sñ Sô Sð Sð Sð Sð Sð Sð Søøøð Sð Sð Sð Sà Ø(Ô3¸4Ð?ð ÅCð Hð Hà!,Ô!7×!>Ò!>Ñ!@Ô!@ðHñ Hô Hñ Eô E�Fð ;AÐ&UÐ&6Ð&6Ào‘O�G˜VÝ#,¨T¸TÈ&ÐZaÐ#bÑ#bÔ#b�LØ—O’O LÑ1Ô1Ð1Ø)×.Ò.Ñ0Ô0ð Ið I˜Ø/;×/EÒ/EÀaÑ/HÔ/HÐ)¨!Ñ,Ð,ñIðIð "Ð-Ø$&Ð!Ø!1ð pð p�IØ%×,Ò,­_¸YÐU`ÔUmÐ-nÑ-nÔ-nÑoÔoÐoÐoðpõ !Ð!XÑYÔYÐYå/SØØ,Ø'ØœØ#5ð0ñ 0ô 0Ñ,ˆLÐ,ð !ð -ð -�Ø—
’
˜4  tÑ,Ô,Ð,Ð,àÐ/Ð/Ð/s   È5IÉI	É!I	rû  c           	      óÒ  — 	 |                       |¦  «         |                      |                     ¦   «         |j        |j        |j        ¦  «         |                      |j        ¦  «         |                      |j	        d¬¦  «         |  
                    |¦  «         t          | |j        |j        |t          ¬¦  «         n(# t          | |j        |j        |t          ¬¦  «         w xY w|S )a$  Perform all post processing operations after having loaded some checkpoints into a model, such as moving
        missing keys from meta device to their expected device, reinitializing missing weights according to proper
        distributions, tying the weights and logging the loading report.F)rÄ  rÅ  )ri  r�   r‘   rû  r¶  )Ú mark_tied_weights_as_initializedÚ&_move_missing_keys_from_meta_to_deviceÚmissing_and_mismatchedr“   rš   r˜   Ú_initialize_missing_keysr¤   rº  rÄ  Ú#_adjust_missing_and_unexpected_keysr{   r�   r‘   r¶  )ri  rå  rû  s      r£   rì  z'PreTrainedModel._finalize_model_loadingü  s  € ð	à×2Ò2°<Ñ@Ô@Ð@ð ×8Ò8Ø×3Ò3Ñ5Ô5ØÔ&ØÔ'ØÔ(ñ	ô ð ð ×*Ò*¨;Ô+CÑDÔDÐDð ×Ò¨<Ô+DÐX]ÐÑ^Ô^Ð^ð ×5Ò5°lÑCÔCÐCå!ØØ.9Ô.WØ(3Ô(KØ)Ýðñ ô ð ð øÕ!ØØ.9Ô.WØ(3Ô(KØ)Ýðñ ô ð ð øøøð Ðs   ‚BB? Â?%C$c                 óz  — d„ |D ¦   «         }|                      d„ |D ¦   «         ¦  «        }g }|                      ¦   «         D ]x\  }}|r | j        › d�}|                     |¦  «        }n8|r6t	          |¦  «        dk    rd                     | j        |g¦  «        n| j        }||v r|                     |¦  «         Œy|S )Nc                 ón   — h | ]2}d                       |                     d ¦  «        dd…         ¦  «        ’Œ3S )rH  Nr<  )r±  r  ©rþ   r  s     r£   rx  z>PreTrainedModel.retrieve_modules_from_names.<locals>.<setcomp>$  s7   € ÐFÐFÐF¸�s—x’x §	¢	¨#¡¤¨s°¨sÔ 3Ñ4Ô4ÐFÐFÐFr¥   c                 óÈ   — h | ]_}t          |¦  «        d k    ¯|d                              ¦   «         ¯/d                     |                     d¦  «        dd…         ¦  «        ’Œ`S )r   r<  rH  Nr	  )rß   Úisdigitr±  r  r  s     r£   rx  z>PreTrainedModel.retrieve_modules_from_names.<locals>.<setcomp>)  sY   € ÐbÐbÐb¨s½sÀ3¹x¼xÈ!º|¸|ÐPSÐTVÔPW×P_ÒP_ÑPaÔPa¸|ˆS�XŠX�c—i’i ‘n”n S b SÔ)Ñ*Ô*¸|¸|¸|r¥   rH  r   )ÚunionrL  r+  rg  rß   r±  rU  )	r¢   rm  Ú
add_prefixÚremove_prefixÚmodule_keysÚretrieved_modulesrJ  rC  Ú_prefixs	            r£   Úretrieve_modules_from_namesz+PreTrainedModel.retrieve_modules_from_names#  sò   € ØFÐFÀÐFÑFÔFˆð "×'Ò'ØbÐb°eÐbÑbÔbñ
ô 
ˆð Ðà ×.Ò.Ñ0Ô0ð 	1ð 	1‰LˆD�&Øð mØ!Ô3Ð6Ð6Ð6�Ø×(Ò(¨Ñ1Ô1��Øð mÝCFÀtÁ9Ä9ÈqÂ=À=�s—x’x Ô!7¸Ð >Ñ?Ô?Ð?ÐVZÔVl�à�{Ð"Ð"Ø!×(Ò(¨Ñ0Ô0Ð0øà Ð r¥   Ú	AutoModelc                 ó¢   — t          |t          ¦  «        s|j        }ddlmc m} t          ||¦  «        st          |› d�¦  «        ‚|| _        dS )aU  
        Register this class with a given auto class. This should only be used for custom models as the ones in the
        library are already mapped with an auto class.



        Args:
            auto_class (`str` or `type`, *optional*, defaults to `"AutoModel"`):
                The auto class to register this new model with.
        r   Nz is not a valid auto class.)	rs  rª   r¦   Útransformers.models.autoÚmodelsrÉ  rµ   rÍ   Ú_auto_class)rO  Ú
auto_classÚauto_modules      r£   Úregister_for_auto_classz'PreTrainedModel.register_for_auto_class:  sn   € õ ˜*¥cÑ*Ô*ð 	-Ø#Ô,ˆJà6Ð6Ð6Ð6Ð6Ð6Ð6Ð6Ð6å�{ JÑ/Ô/ð 	IÝ 
ÐGÐGÐGÑHÔHÐHà$ˆŒˆˆr¥   c           
      ó   — t          |¦  «        rdS |€| j        j        €dS | j        j        |dd…ddgf         v rÂd}t          | j        dd¦  «        }| j        j        �| j        j        | j        j        k    s8| j        j        �| j        j        | j        j        k    s|�@|| j        j        k    r0|d| j        j        › d| j        j        › d| j        j        › d	|› d
�	z  }t                               |¦  «         dS dS )zv
        Shows a one-time warning if the input_ids appear to contain padding and no attention mask was given.
        Nr<  r   zÈWe strongly recommend passing in an `attention_mask` since your input_ids may be padded. See https://huggingface.co/docs/transformers/troubleshooting#incorrect-output-when-padding-tokens-arent-masked.Úsep_token_idz5
You may ignore this warning if your `pad_token_id` (z&) is identical to the `bos_token_id` (z), `eos_token_id` (z), or the `sep_token_id` (z ), and your input is not padded.)ry   rÇ  Úpad_token_idrM  Úbos_token_idÚeos_token_idr¶  rË  )r¢   r.  rð  Úwarn_stringr/  s        r£   Ú%warn_if_padding_and_no_attention_maskz5PreTrainedModel.warn_if_padding_and_no_attention_maskP  sM  € õ �iÑ Ô ð 	ØˆFàÐ&¨D¬KÔ,DÐ,LØˆFð Œ;Ô# y°°°°R¸°G°Ô'<Ð<Ð<ðFð õ # 4¤;°ÀÑEÔEˆLà”Ô)Ð5¸$¼+Ô:RÐVZÔVaÔVnÒ:nÐ:nØ”KÔ,Ð8¸T¼[Ô=UÐY]ÔYdÔYqÒ=qÐ=qØ Ð,°ÀÄÔAYÒ1YÐ1Yàð]ÈTÌ[ÔMeð ]ð ]Ø.2¬kÔ.Fð]ð ]Ø[_Ô[fÔ[sð]ð ]à.:ð]ð ]ð ]ñ�õ ×Ò Ñ,Ô,Ð,Ð,Ð,ð- =Ð<r¥   c                 óP   — | j         rdS | j        j         rdS | j        j        rdS dS )zJ
        Returns whether the model has a tensor parallelism plan.
        TF)r=  r  rÇ  r{  r¡   s    r£   Úsupports_tp_planz PreTrainedModel.supports_tp_planu  s<   € ð Œ=ð 	Ø�4àŒ?Ô#ð 	Ø�4àŒ;Ô)ð 	Ø�4Øˆur¥   c                 ó   — | j         S )z@
        Returns the model's tensor parallelism degree.
        )r_  r¡   s    r£   rÏ  zPreTrainedModel.tp_size…  s   € ð Œ}Ðr¥   c                 óP   — | j         rdS | j        j         rdS | j        j        rdS dS rÃ   )r>  r  rÇ  rz  r¡   s    r£   Úsupports_pp_planz PreTrainedModel.supports_pp_plan�  s<   € ð Œ=ð 	Ø�4àŒ?Ô#ð 	Ø�4àŒ;Ô)ð 	Ø�4Øˆur¥   c                 óÂ   — t          | d¦  «        r| j        S t          | dd ¦  «        }|�	|t          vr t                               d|› d�¦  «         d}t          |         S )NÚ_loss_functionri  z`loss_type=zZ` was set in the config but it is unrecognized. Using the default loss: `ForCausalLMLoss`.ÚForCausalLM)rµ   r;  rM  rJ   r¶  rË  )r¢   ri  s     r£   Úloss_functionzPreTrainedModel.loss_functionš  s   € å�4Ð)Ñ*Ô*ð 	'ØÔ&Ð&å˜D +¨tÑ4Ô4ˆ	àÐ 	µÐ =Ð =Ý×Òð>˜ið >ð >ð >ñô ð ð &ˆIÝ˜IÔ&Ð&r¥   c                 ó   — || _         d S r    )r;  ©r¢   r  s     r£   r=  zPreTrainedModel.loss_function©  s   € à#ˆÔÐÐr¥   c                 ó$   — t          | dd¦  «        S )NrÃ  Fr™  r¡   s    r£   rÄ  zPreTrainedModel.use_kernels­  s   € å�t˜^¨UÑ3Ô3Ð3r¥   r  c                 óà   — t          |¦  «        rt          | dd¦  «        rd S |r|                      d¦  «         d S t          | dd¦  «        rt                               d¦  «         d| _        d S )NrÃ  FTzmDisabling kernels at runtime is a no-op as there is no 'unkernelize' routine; keeping current kernels active.)r¬   rM  rÅ  r¶  rË  rÃ  r?  s     r£   rÄ  zPreTrainedModel.use_kernels±  sŽ   € õ �‰;Œ;ð 	�7 4¨¸Ñ?Ô?ð 	ØˆFàð 	&Ø× Ò  Ñ&Ô&Ð&Ð&Ð&å�t˜^¨UÑ3Ô3ð Ý×#Ò#ð Dñô ð ð !&ˆDÔÐÐr¥   c                 ób   — | j         j        dk    rt          ddd¬¦  «        S t          ¦   «         S )aU  Build the default `CompileConfig` for `get_compiled_call`.

        Inductor + `reduce-overhead` (the `CompileConfig` defaults) target CUDA.
        torch_tpu registers its own TorchDynamo backend named `"tpu"`; route
        `device.type == "tpu"` through it with static shapes to match the
        common StaticCache + fixed-prefill usage.ÚtpuFrx  )r  Údynamicr¼  )rÖ   ru  r(   r¡   s    r£   Ú_default_compile_configz'PreTrainedModel._default_compile_configÀ  s3   € ð Œ;Ô˜uÒ$Ð$Ý ¨¸ÀIÐNÑNÔNÐNÝ‰ŒÐr¥   Úcompile_configc                 ón  — d| j         j        v r| j        S |p|                      ¦   «         }t	          | j        dd¦  «        p|                      ¦   «         }t          | d¦  «        rt	          | d|¦  «        |k    r5|| _        t          j	        | j        fi | 
                    ¦   «         ¤Ž| _        | j        S )aŒ  Return a `torch.compile`'d version of `self.__call__`. This is useful to dynamically choose between
        non-compiled/compiled `forward` during inference, especially to switch between prefill (where we don't
        want to use compiled version to avoid recomputing the graph with new shapes) and iterative decoding
        (where we want the speed-ups of compiled version with static shapes).Úllama4rF  NÚ_compiled_callÚ_last_compile_config)rÇ  Ú
model_typeÚ__call__rE  rM  rg  rµ   rJ  r®   r¹  rð  rI  )r¢   rF  Údefault_configs      r£   Úget_compiled_callz!PreTrainedModel.get_compiled_callË  s¾   € ð �t”{Ô-Ð-Ð-Ø”=Ð Ø'ÐI¨4×+GÒ+GÑ+IÔ+IˆÝ  Ô!7Ð9IÈ4ÑPÔPÐrÐTX×TpÒTpÑTrÔTrˆå˜Ð.Ñ/Ô/ð	[å�tÐ3°^ÑDÔDÈÒVÐVà(6ˆDÔ%Ý"'¤-°´Ð"ZÐ"ZÀ×AWÒAWÑAYÔAYÐ"ZÐ"ZˆDÔØÔ"Ð"r¥   c                 ó   — | j         S r    )rC  ©rO  s    r£   Úis_backend_compatiblez%PreTrainedModel.is_backend_compatibleÝ  s   € àÔ.Ð.r¥   r“   rš   r™   r˜   c                 óL  — |du}t          ¦   «         r|sdS t          ¦   «         r”t          ¦   «         s†|s„|                      ¦   «         D ],\  }}t	          j        |d¬¦  «        }t          | ||¦  «         Œ-|                      ¦   «         D ],\  }}	t	          j        |	d¬¦  «        }t          | ||¦  «         Œ-dS || j         	                    ¦   «         z
  D ]{}|  
                    |¦  «        }t          ||d¬¦  «        }
t	          j        ||
¬¦  «        }|�)t          | |||dd|                     ¦   «         |¦  «         Œjt          | ||¦  «         Œ||                      ¦   «         D ]>\  }}	t          ||d¬¦  «        }t	          j        |	|¬¦  «        }t          | ||¦  «         Œ?dS )aÛ  Move the missing keys (keys that are part of the model parameters, but were NOT found in the loaded state dicts)
        back from meta device to their device according to the `device_map` if any, else cpu. Takes care of sharding those
        missing parameters if `device_mesh` is provided, i.e. we are using TP.
        All non-persistent buffers are also moved back to the correct device (they are not part of the state_dict, but are
        not missing either).
        NrÔ   rÕ  T)Úvalid_torch_deviceF)r-   r.   rÁ   r  r®   Ú
zeros_liker’  rº  ru  r-  rv  r4   Ú
empty_likerH   Úget_local_rankÚnamed_non_persistent_buffers)r¢   rÄ  r“   rš   r˜   r¤   r  r×  r  rœ  Úparam_deviceÚbuffer_devices               r£   r  z6PreTrainedModel._move_missing_keys_from_meta_to_deviceá  sþ  € ð $¨4Ð/ˆå%Ñ'Ô'ð 	°ð 	ØˆFõ ÑÔð 	Õ%9Ñ%;Ô%;ð 	ÀLð 	Ø"×3Ò3Ñ5Ô5ð =ð =‘
��UÝÔ(¨°uÐ=Ñ=Ô=�Ý*¨4°°eÑ<Ô<Ð<Ð<Ø#×1Ò1Ñ3Ô3ð =ð =‘��VÝÔ(¨¸Ð>Ñ>Ô>�Ý*¨4°°eÑ<Ô<Ð<Ð<ØˆFð
   $Ô"<×"AÒ"AÑ"CÔ"CÑCð 	=ð 	=ˆCØ×0Ò0°Ñ5Ô5ˆEÝ% j°#È$ÐOÑOÔOˆLÝÔ$ U°<Ð@Ñ@Ô@ˆEàÐ&Ý+Ø˜% ¨¨T°5¸+×:TÒ:TÑ:VÔ:VÐXcñô ð ð õ
 +¨4°°eÑ<Ô<Ð<Ð<à×<Ò<Ñ>Ô>ð 	9ð 	9‰KˆC�Ý& z°3È4ÐPÑPÔPˆMÝÔ$ V°MÐBÑBÔBˆEÝ& t¨S°%Ñ8Ô8Ð8Ð8ð	9ð 	9r¥   c                 ó8  — t          ¦   «         rYt          ¦   «         sK|                      ¦   «         D ]/}	 |                      |¦  «        }d|_        Œ # t
          $ r Y Œ,w xY wd| _        t          ¦   «         r�|sŽddl}t          d„ |                      d¬¦  «         	                    ¦   «         D ¦   «         ¦  «        }|j
                             |d¬¦  «        5  |                      ¦   «          ddd¦  «         dS # 1 swxY w Y   dS |                      ¦   «          dS )a1  
        Initialize the missing keys (keys that are part of the model parameters, but were NOT found in the loaded state dicts), according to
        `_initialize_weights`. Indeed, since the corresponding weights are missing from the state dict, they will not be replaced and need to
        be initialized correctly (i.e. weight initialization distribution).

        Also marks non-missing params/buffers with `_is_hf_initialized` and propagates this flag to modules,
        so that `_initialize_weights` can skip fully-initialized modules entirely.
        Tr   Nc                 ó4   — h | ]}t          |d d¦  «        °|’ŒS ©rr  Fr™  )rþ   r"  s     r£   rx  z;PreTrainedModel._initialize_missing_keys.<locals>.<setcomp>/  s,   € ÐtÐtÐt�qÍGÐTUÐWkÐmrÑLsÔLsÐt�ÐtÐtÐtr¥   )Ú	keep_varsræ  )r.   rÁ   rã   rv  rr  ÚAttributeErrorr-   r´  r¯   rÞ   r·  ré  r¦  )r¢   r¤   r  Úparam_or_bufferr´  Únot_initialized_parameterss         r£   r  z(PreTrainedModel._initialize_missing_keys  s•  € õ ÑÔð 	+Õ%9Ñ%;Ô%;ð 	+ð —’Ñ(Ô(ð ð �ðØ&*×&BÒ&BÀ3Ñ&GÔ&G�OØ9=�OÔ6Ð6øÝ%ð ð ð Ø�Dðøøøà&*ˆDÔ#õ &Ñ'Ô'ð 
	&°ð 
	&ØÐÐÐõ *.ØtÐt˜DŸOšO°d˜OÑ;Ô;×BÒBÑDÔDÐtÑtÔtñ*ô *Ð&ð ”×2Ò2Ð3MÐ]^Ð2Ñ_Ô_ð *ð *Ø×'Ò'Ñ)Ô)Ð)ð*ð *ð *ñ *ô *ð *ð *ð *ð *ð *ð *ð *øøøð *ð *ð *ð *ð *ð *ð ×#Ò#Ñ%Ô%Ð%Ð%Ð%s#   ´AÁ
AÁAÃC9Ã9C=Ä C=c                 óê  ‡‡— t          d„ |                      ¦   «         D ¦   «         ¦  «        }|rdhnt          ¦   «         }t          d„ |                      ¦   «         D ¦   «         ¦  «        }|r|                     d¦  «         | j        pt          ¦   «         }| j        pt          ¦   «         |z  }d\  ŠŠt          |¦  «        dk    r1t          j        d 	                    d„ |D ¦   «         ¦  «        ¦  «        Št          |¦  «        dk    r1t          j        d 	                    d	„ |D ¦   «         ¦  «        ¦  «        Š‰�ˆfd„|j
        D ¦   «         |_
        ‰�ˆfd„|j        D ¦   «         |_        d
S d
S )z±Adjust the `missing_keys` and `unexpected_keys` based on current model's exception rules, to avoid
        raising unneeded warnings/errors. This is performed in-place.
        c              3   óF   K  — | ]\  }}|                      d ¦  «        V — ŒdS )zrotary_emb.inv_freqN©r)  ©rþ   rœ  r\  s      r£   r   zFPreTrainedModel._adjust_missing_and_unexpected_keys.<locals>.<genexpr>=  s4   è è € Ð"pÐ"pÉiÈfÐVW 6§?¢?Ð3HÑ#IÔ#IÐ"pÐ"pÐ"pÐ"pÐ"pÐ"pr¥   zrotary_emb\.inv_freqc              3   óF   K  — | ]\  }}|                      d ¦  «        V — ŒdS )Úposition_idsNrc  rd  s      r£   r   zFPreTrainedModel._adjust_missing_and_unexpected_keys.<locals>.<genexpr>@  s3   è è € Ð&mÐ&mÉ9È6ÐST v§¢°~Ñ'FÔ'FÐ&mÐ&mÐ&mÐ&mÐ&mÐ&mr¥   z(^|\.)position_ids$©NNr   rY  c              3   ó"   K  — | ]
}d |› d�V — ŒdS ©rX  rú  Nr±   ©rþ   Úpatterns     r£   r   zFPreTrainedModel._adjust_missing_and_unexpected_keys.<locals>.<genexpr>H  s*   è è € Ð6gÐ6gÈ7°¸G°°°Ð6gÐ6gÐ6gÐ6gÐ6gÐ6gr¥   c              3   ó"   K  — | ]
}d |› d�V — ŒdS ri  r±   rj  s     r£   r   zFPreTrainedModel._adjust_missing_and_unexpected_keys.<locals>.<genexpr>J  s*   è è € Ð9mÐ9mÈg¸/¸w¸/¸/¸/Ð9mÐ9mÐ9mÐ9mÐ9mÐ9mr¥   Nc                 ó>   •— h | ]}‰                      |¦  «        ­|’ŒS r    ©rq  )rþ   r  Úignore_missing_regexs     €r£   rx  zFPreTrainedModel._adjust_missing_and_unexpected_keys.<locals>.<setcomp>N  s5   ø€ ð )ð )ð )ØÐ<P×<WÒ<WÐX[Ñ<\Ô<\Ð<d�Ð<dÐ<dÐ<dr¥   c                 ó>   •— h | ]}‰                      |¦  «        ­|’ŒS r    rn  )rþ   r  Úignore_unexpected_regexs     €r£   rx  zFPreTrainedModel._adjust_missing_and_unexpected_keys.<locals>.<setcomp>T  s5   ø€ ð ,ð ,ð ,ØÐ?V×?]Ò?]Ð^aÑ?bÔ?bÐ?j�Ð?jÐ?jÐ?jr¥   )rw  rº  re  rW  r6  r7  rß   rp  r¹  r±  rÄ  r   )	r¢   rû  Úhas_inv_freq_buffersÚadditional_unexpected_patternsÚhas_position_ids_buffersÚmissing_patternsÚunexpected_patternsro  rq  s	          @@r£   r  z3PreTrainedModel._adjust_missing_and_unexpected_keys6  sÃ  øø€ õ  #Ð"pÐ"pÐ[_×[mÒ[mÑ[oÔ[oÐ"pÑ"pÔ"pÑpÔpÐØFZÐ)eÐ*AÐ)BÐ)BÕ`cÑ`eÔ`eÐ&å#&Ð&mÐ&mÐX\×XjÒXjÑXlÔXlÐ&mÑ&mÔ&mÑ#mÔ#mÐ Ø#ð 	GØ*×.Ò.Ð/EÑFÔFÐFàÔ?ÐHÅ3Á5Ä5ÐØ#ÔFÐOÍ#É%Ì%ÐSqÑqÐØ8BÑ5ÐÐ5ÝÐÑ Ô  1Ò$Ð$Ý#%¤:¨c¯hªhÐ6gÐ6gÐVfÐ6gÑ6gÔ6gÑ.gÔ.gÑ#hÔ#hÐ ÝÐ"Ñ#Ô# aÒ'Ð'Ý&(¤j°·²Ð9mÐ9mÐYlÐ9mÑ9mÔ9mÑ1mÔ1mÑ&nÔ&nÐ#ð  Ð+ð)ð )ð )ð )Ø+Ô8ð)ñ )ô )ˆLÔ%ð
 #Ð.ð,ð ,ð ,ð ,Ø+Ô;ð,ñ ,ô ,ˆLÔ(Ð(Ð(ð /Ð.r¥   c                 óü   ‡ — t          ‰ di ¦  «                             ¦   «         D ](}‰                      |¦  «        }t          |dd¦  «         Œ)‰                      ¦   «         rˆ fd„|j        D ¦   «         |_        dS dS )aE  Adds the `_is_hf_initialized` flag on parameters that will be tied, in order to avoid initializing them
        later as they will be tied (overwritten) anyway.
        This is very important as most embeddings are tied, and they are huge params (vocabularies are often 256k), so
        running inits on them is very costly.ru  rr  Tc                 ón   •— h | ]1}|‰j         v s$t          ‰                     |¦  «        d d¦  «        °/|’Œ2S r\  )ru  rM  rv  )rþ   r  r¢   s     €r£   rx  zCPreTrainedModel.mark_tied_weights_as_initialized.<locals>.<setcomp>m  sU   ø€ ð )ð )ð )àØ˜$Ô4Ð4Ð4Ý˜t×;Ò;¸CÑ@Ô@ÐBVÐX]Ñ^Ô^ð 5ð à4Ð4Ð4r¥   N)rM  r-  rÈ  r�  r–  rÄ  )r¢   rû  Ú
tied_paramr×  s   `   r£   r  z0PreTrainedModel.mark_tied_weights_as_initializedX  sª   ø€ õ
 " $Ð(?ÀÑDÔD×IÒIÑKÔKð 	7ð 	7ˆJØ×&Ò& zÑ2Ô2ˆEÝ�EÐ/°Ñ6Ô6Ð6Ð6ð ×ÒÑ Ô ð 	ð)ð )ð )ð )à'Ô4ð)ñ )ô )ˆLÔ%Ð%Ð%ð	ð 	r¥   r­  c                 óš  — 	 |                       |¦  «        S # t          $ r Y nw xY w	 |                      |¦  «        S # t          $ r Y nw xY wt          | |¦  «        \  }}|dk    rTt	          |j        dt          j        j        j	        ¦  «        t          j        j        j	        ur| 	                    ¦   «         S t          d|› d�¦  «        ‚)ai  
        Return the parameter or buffer given by `target` if it exists, otherwise throw an error. This combines
        `get_parameter()` and `get_buffer()` in a single handy function. If the target is an `_extra_state` attribute,
        it will return the extra state provided by the module. Note that it only work if `target` is a leaf of the model.
        Ú_extra_stateÚget_extra_stater¾  z2` is neither a parameter, buffer, nor extra state.)
rÈ  r^  Ú
get_bufferrU   rM  r  r®   r   r'  r|  )r¢   r­  rC  rŠ  s       r£   rv  z'PreTrainedModel.get_parameter_or_buffert  sí   € ð	Ø×%Ò% fÑ-Ô-Ð-øÝð 	ð 	ð 	ØˆDð	øøøð	Ø—?’? 6Ñ*Ô*Ð*øÝð 	ð 	ð 	ØˆDð	øøøå1°$¸Ñ?Ô?Ñˆ�
à˜.Ò(Ð(Ý˜Ô(Ð*;½U¼X¼_Ô=\Ñ]Ô]Ý”8”?Ô2ð3ð 3ð ×)Ò)Ñ+Ô+Ð+åÐ[ Ð[Ð[Ð[Ñ\Ô\Ð\s   ‚ —
$£$¨= ½
A
Á	A
rš  r¨  c              #   óÎ   K  — |                       ||¬¦  «        D ]J\  }}d|v r|                     dd¦  «        nd|f\  }}|                      |¦  «        }||j        v r||fV — ŒKdS )zŸSimilar to `named_buffers`, but only yield non-persistent ones. It is handy as it's not perfectly straightforward
        to know if they are persistent or not)rš  r¨  rH  r   r¦  N)rº  r–  rÊ  Ú_non_persistent_buffers_set)r¢   rš  r¨  rJ  rÕ   r�  Úbuf_names          r£   rW  z,PreTrainedModel.named_non_persistent_buffersŒ  s–   è è € ð
 !×.Ò.°wÐQaÐ.ÑbÔbð 	#ð 	#‰LˆD�&ð 7:¸T°k°k˜tŸ{š{¨3°Ñ2Ô2Ð2ÈÈDÀzÑˆF�HØ×'Ò'¨Ñ/Ô/ˆFØ˜6Ô=Ð=Ð=Ø˜F�lÐ"Ð"Ð"øð	#ð 	#r¥   c                 óœ   •— | j         |k    }t          ¦   «                              |¦  «        }| j        r|r|                      d¦  «         |S r8  )ÚtrainingrJ  ÚtrainrÄ  rÅ  )r¢   r¼  Úchanged_modeÚoutr  s       €r£   rƒ  zPreTrainedModel.train™  sO   ø€ Ø”}¨Ò,ˆÝ‰gŒg�mŠm˜DÑ!Ô!ˆàÔð 	' ð 	'Ø× Ò  Ñ&Ô&Ð&Øˆ
r¥   c                 ó,   — |                       d¦  «        S )NF)rƒ  r¡   s    r£   rí  zPreTrainedModel.eval¡  s   € Ø�zŠz˜%Ñ Ô Ð r¥   c                 ó   — | j         duS )z—Return whether the current model is custom code, i.e. code loaded from the hub, or class that we just registered
        via `register_for_auto_class`.N)r*  rP  s    r£   rš  zPreTrainedModel.is_remote_code¤  s   € ð Œ dÐ*Ð*r¥   c                 ó`   — |                       ¦   «         p| j                             d¦  «         S )zŽReturn whether the current model is custom code, i.e. either code loaded from the hub, or defined in any user-specific
        module/session.ztransformers.)rš  r§   r  rP  s    r£   r–  zPreTrainedModel.is_custom_codeª  s.   € ð ×!Ò!Ñ#Ô#ÐU¨3¬>×+DÒ+DÀ_Ñ+UÔ+UÐ'UÐUr¥   r    )r±   N©Fr  r8  )NNT)NFT)TNFr@  NNTT)Trg  )r&  )TT)¥r¦   r§   r¨   r©   r)  ru  r    r«   r)   r*  r*  r+  rª   r,  r¬   r-  r¯   r/  r1  r2  re  r3  r4  r5  rE  r­   r6  r7  r8  r9  r:  r;  r<  r=  r_  r>  r  r?  r@  rA  rB  rC  rD  r°   r®   ÚcompilerÚallow_in_graphr}   rF  r   rH  rK  rZ  rŒ  r�  r’  r”  Úsetterr£  r‚  r«  Úclassmethodr½  r   r'  r  re  r½   r   rØ  rã  rë  rî  rð  r^  rb  r÷  r   r$  rí  r.  r0  r5  rE  rI  r\  ra  rf  rh  Úno_gradr•  rŸ  rµ  Úguard_torch_init_functionsr¦  r~  rº  rË  r   rì  rè  rï  rz  rð  rø  r   r  r  r$  r&  r�  r¦  r   r1  r;  r?  r¾   ÚPathLikeri  r   r_   rC  r¡  rÑ  r   r±  r´  r–   r¸  r»  r^   rÅ  r‡   r   ræ  r  rŒ   rz   rë  rì  r%  r-  r4  r6  rÏ  r9  r=  rÄ  r(   rE  rN  rQ  rS   r  r  r  r  rv  r   rW  rƒ  rí  rš  r–  Ú__classcell__©r  s   @r£   rˆ   rˆ   ¯  s»  ø€ € € € € € ðð ð( 37€L�$Ð'Ô(¨4Ñ/Ð6Ð6Ñ6Ø6FÐ˜TÐ"2Ô3ÐFÐFÑFØ€KØÐ�sÐÐÑØ€L�$ÐÐÑØ#'€J��S”	˜DÑ Ð'Ð'Ñ'ð '€O�SÐ&Ð&Ñ&ð )/Ð�c˜D œI‘oÐ.Ð.Ñ.ð 6:Ð�s˜3”x $ s¤)Ñ+¨dÑ2Ð9Ð9Ñ9Ø?CÐ  S¤¨D°¬IÑ!5¸Ñ!<ÐCÐCÑCð
 :>Ð˜3˜sœ8 d¨3¤iÑ/°$Ñ6Ð=Ð=Ñ=Ø@DÐ  # c¤(¨T°#¬YÑ"6¸Ñ"=ÐDÐDÑDð *.Ð˜˜S #˜XœÐ-Ð-Ñ-àCGÐ# S¨¤X°°S´	Ñ%9¸DÑ%@ÐGÐGÑGàFJÐ&¨¨C¬°4¸´9Ñ(<¸tÑ(CÐJÐJÑJà;?Ð˜S œX¨¨S¬	Ñ1°DÑ8Ð?Ð?Ñ?ð !€N�DÐ Ð Ñ Ø!&Ð˜$Ð&Ð&Ñ&Ø %Ð˜Ð%Ð%Ñ%à:>Ð% t¨C¤y°4Ñ'7Ð>Ð>Ñ>ð  $€Hˆd�3˜�8ŒnÐ#Ð#Ñ#à€Hð ,0€Hˆd�3˜˜c 3˜hœÐ'Ô(Ð/Ð/Ñ/ð  $€Hˆd�3˜�8ŒnÐ#Ð#Ñ#ð "&€J��S˜#�X”Ð%Ð%Ñ%ð -2Ð# TÐ1Ð1Ñ1Ø#(Ð˜DÐ(Ð(Ñ(ð ).Ð Ð-Ð-Ñ-à'+Ð˜ ™Ð+Ð+Ñ+àØ
„^Ô"ð). D¨¨nÐ)<Ô$=ð ).ð ).ð ).ñ #Ô"ñ „Xð).ðV ð9˜d 3¨¬Ð#4Ô5ð 9ð 9ð 9ñ „Xð9ð/ð /ð /ð /ð /ð0-MÐ/ð -Mð -Mð -Mð -Mð -Mð -Mð^G>ð G>ð G>ðR ð˜˜c 3˜hœð ð ð ñ „Xðð ð˜4  S œ>ð ð ð ñ „Xðð ð˜˜c 5¨¨c¨¤?Ð2Ô3ð ð ð ñ „Xðð „^ð!˜D  c œN¨TÑ1ð !ð !ð !ñ „^ð!ðF „^ð˜D  e¨C°¨H¤oÐ!5Ô6¸Ñ=ð ð ð ñ „^ðð
:ð 
:ð 
:ð 
:ð;ð ;ð ;ð, 4¨¤9¨s¡?ð ,°tð ,ð ,ð ,ð ,ð@ ðGð Gñ „[ðGðR ð;˜BœIð ;ð ;ð ;ñ „Xð;ð ð$˜Tð $ð $ð $ñ „[ð$ðX FHØ-1ð@ð @àð@ð %-ð@ð !)ð	@ð
 !  x° }Ô!5°sÐ!:Ô;ð@ð #(¨¨h¸¨mÔ(<¸cÐ(AÔ"Bð@ð !$ d¡
ð@ð @ð @ð @ðDIð I¸3ð IÈtð IÐ`dð Ið Ið Ið IðVð °ð Àð ð ð ð ð<	¨$ð 	ð 	ð 	ð 	ðð °Tð Àdð ð ð ð ð8 glðq.ð q.Ø#&¨¡:ðq.Ø>Bðq.Ø_cðq.à	ðq.ð q.ð q.ð q.ðf1ÈsÐUYÉzð 1Ð^að 1ð 1ð 1ð 1ð$$ð $$À3ÈÁ:ð $$Ð^bð $$Ðorð $$ð $$ð $$ð $$ðL"ÀCÈ$ÁJð "ÐSVð "ð "ð "ð "ð0 ð=¨Tð =ð =ð =ñ „[ð=ð8 ð°ð ð ð ñ „[ðð.d8ð d8¸3À¹:ð d8ÐZ^ð d8ð d8ð d8ð d8ðL&¨D°°c¸D±j°Ô,Að &ð &ð &ð &ð:WÀÀtÁð :Wð :Wð :Wð :Wðx*ð *ð *ðX)ð )ð )ðð  C¨$¡Jð ð ð ð ð@%ð %¨S°4©Zð %ð %ð %ð %ð4ð ð ð.%ð %ð %ð" €U„]�_„_ð=?ð =?ñ „_ð=?ð~)ð )¸$ð )ð )ð )ð )ð2 €U„]�_„_Ø$€TÔ$Ñ&Ô&ðHð Hñ 'Ô&ñ „_ðHð8p%ð p%¸Dð p%ÈTð p%ð p%ð p%ð p%ðdY8ð Y8¨¨C¬°4©ð Y8ÐSWð Y8ð Y8ð Y8ð Y8ðv
Mð 
Mð 
Mð &*Ø)-Ø"ð	9ð 9à˜d™
ð9ð   $™Jð9ð ð	9ð
 
Œð9ð 9ð 9ð 9ðv%+ð %+ð %+ð %+ðT &*Ø)-Ø"ð]ð ]àœð]ð ˜d™
ð]ð   $™Jð	]ð
 ð]ð 
Œð]ð ]ð ]ð ]ðD &*Ø Ø"ðAð Aà”YðAð ˜d™
ðAð ð	Að
 ðAð 
ŒðAð Að Að AðFð ð ð@ !ð@ð @ð ð@ð @ð @ð @ð,dð dð dð
dð dð dð
Àcð 
ð 
ð 
ð 
ð
¨¬¸¸b¼lÔ8KÑ)Kð 
ð 
ð 
ð 
ð
2ð 
2ð 
2ð(.ð (.ð (.ð (.ðT :>Ðgqð ð °$ð Ð\dð ð ð ð ð,/ð /ð /ð( ðn¨4ð nð nð nñ „Xðnð !%Ø"&Ø!Ø$*Ø"Ø#'Ø!%Ø%)ð|ð |à˜bœkÑ)ð|ð ð|ð ˜4‘Kð	|ð
 ð|ð ˜c™	ð|ð �t‘ð|ð �T‰z˜DÑ ð|ð ð|ð #ð|ð |ð |ð |ð|	 €Uˆ>Ô%Ñ&Ô&ð4ð 4ð 4ð 4ñ 'Ô&ð4ðð ð ð ð$ €Uˆ5Œ8Œ?ÔÑ Ô ð-ð -ð -ð -ñ !Ô ð-ð2 €Uˆ5Œ8Œ?ÔÑÔð:+ð :+ð :+ð :+ñ Ôð:+ðx'ð 'ð 'ð 'ð 'ð(ð (ð (ð (ð (ð ðØ”KðØ/3ðØIMðØbfÐimÑbmðð ð ñ „[ðð> U¤[ð °Tð ð ð ð ð (&ð (&¸,ÈÑ:Mð (&Ð\ið (&ð (&ð (&ð (&ðT ð
 ?CØ.2Ø(-Ø$Ø!&Ø#'ØØ'+Ø!ØAEØ$(ðVð Vð VØÐ-Ô.ðVà'*¨R¬[Ñ'8¸4Ñ'?ðVð ! 3Ñ&¨¬Ñ4°tÑ;ð	Vð
 ˜œÑ$ tÑ+ðVð "&ðVð ðVð ðVð �T‰z˜DÑ ðVð ðVð  ™ðVð ðVð ˜C ¨¨S°#¨X¬Ñ!6Ð6Ô7¸$Ñ>ðVð ˜T‘kðVð  
%ð!Vð Vð Vñ „[ðVðp ð +/ðg0ð g0Ø ðg0à˜4‘Kðg0ð ˜sœ) dÑ*ðg0ð )ð	g0ð
 ˜C”y 4Ñ'ðg0ð 
Ð  $Ð&Ô	'ðg0ð g0ð g0ñ „\ðg0ðR ð$Ø/ð$Ø?Pð$à	ð$ð $ð $ñ „\ð$ðL!ð !ð !ð !ð. ð%ð %ð %ñ „[ð%ð*#-ð #-ð #-ðJ ðð ñ „Xðð ðð ñ „Xðð ð
ð 
ñ „Xð
ð ð'ð 'ñ „Xð'ð Ôð$ð $ñ Ôð$ð ð4˜Tð 4ð 4ð 4ñ „Xð4ð Ôð& ð &¨$ð &ð &ð &ñ Ôð&ð	¨ð 	ð 	ð 	ð 	ð#°ÀÑ0Dð #Èð #ð #ð #ð #ð$ ð/ð /ñ „[ð/ð/9à˜3”ið/9ð ˜4‘Kð/9ð -ð	/9ð
 " DÑ(ð/9ð 
ð/9ð /9ð /9ð /9ðb"&°Tð "&¸dð "&ð "&ð "&ð "&ðH Ð@Qð  ÐVZð  ð  ð  ð  ðDð ð ð8]¨cð ]ð ]ð ]ð ]ð2 >Bð#ð #Øð#Ø6:ð#à	�%˜˜Uœ\Ð)Ô*Ô	+ð#ð #ð #ð #ðð ˜$ð ð ð ð ð ð ð!ð !ð !ð ð+˜tð +ð +ð +ñ „[ð+ð
 ðV˜tð Vð Vð Vñ „[ðVð Vð Vð Vð Vr¥   r&  z
model file)ÚobjectÚobject_classÚobject_filesÚ	recursivec                 ó   — d S r    r±   ©ri  r–  s     r£   rf  rf  ¸  s   € ØVYÐVYr¥   c                 ó   — d S r    r±   r˜  s     r£   rf  rf  ¼  s   € ØJMÈ#r¥   c                 ó–   — t          ¦   «         ri }|r||d<   t          | fi |¤ŽS t          | d¦  «        rt          | j        ¦  «        S | S )a¡  
    Recursively unwraps a model from potential containers (as used in distributed training).

    Args:
        model (`torch.nn.Module`): The model to unwrap.
        recursive (`bool`, *optional*, defaults to `False`):
            Whether to recursively extract all cases of `module.module` from `model` as well as unwrap child sublayers
            recursively, not just the top-level distributed containers.
    r–  rC  )rd   r€   rµ   rf  rC  )ri  r–  r¯  s      r£   rf  rf  À  sh   € õ Ñ Ô ð 
ØˆØð 	,Ø"+ˆF�;ÑÝ*¨5Ð;Ð;°FÐ;Ð;Ð;õ �5˜(Ñ#Ô#ð 	Ý ¤Ñ-Ô-Ð-àˆLr¥   rÖ   c                 óH   — | dk    rdS t          j        | ¦  «        j        dvS )z�Check if the device is an accelerator. We need to function, as device_map can be "disk" as well, which is not
    a proper `torch.device`.
    rO  F)r  rÔ   )r®   rÖ   ru  rÕ  s    r£   Úis_accelerator_devicerœ  Ù  s,   € ð �ÒÐØˆuåŒ|˜FÑ#Ô#Ô(°Ð?Ð?r¥   Úaccelerator_device_mapc                 ó  — t          d„ ¦  «        }| j                             ¦   «         }t          ¦   «         r| j        ng }|                     ¦   «         D ]°\  }}||v rŒ
|                      |¦  «        }|�|                     | ||¦  «        }	n|                     ¦   «         }	| 	                    ¦   «         |	z  }
t          |¦  «        dk    r)t          ||d¬¦  «        du}|
|rt          ¦   «         ndz  }
||xx         |
z  cc<   Œ±|S )zÔ
    This utility function calculates the total bytes count needed to load the model on each device.
    This is useful for caching_allocator_warmup as we want to know how much cache we need to pre-allocate.
    c                  ó   — dS )Nr   r±   r±   r¥   r£   r  z&get_total_byte_count.<locals>.<lambda>ë  s   € ¨1€ r¥   Nr   T)Ú	is_weightr   )r   ru  r-  r·   r�  r,  rv  Úparam_element_sizer@  r  rß   rD   rº   )ri  r�  r˜   Útotal_byte_countÚtied_param_namesr�  rŠ  rÖ   r×  Ú
dtype_sizeÚparam_byte_countÚis_part_of_plans               r£   Úget_total_byte_countr§  ã  s3  € õ # 9 9Ñ-Ô-ÐØÔ2×7Ò7Ñ9Ô9ÐÝ@ÑBÔBÐJˆeŒmˆmÈ€Gà4×:Ò:Ñ<Ô<ð 5ð 5Ñˆ
�FàÐ)Ð)Ð)Øà×-Ò-¨jÑ9Ô9ˆàÐ#Ø%×8Ò8¸À
ÈEÑRÔRˆJˆJà×+Ò+Ñ-Ô-ˆJà Ÿ;š;™=œ=¨:Ñ5Ðåˆw‰<Œ<˜!ÒÐÝ4°ZÀÐTXÐYÑYÔYÐaeÐeˆOØÈÐ!^Õ!BÑ!DÔ!DÐ!DÐ]^Ñ^Ðà˜Ð Ð Ô Ð$4Ñ4Ð Ð Ñ Ð ØÐr¥   r  c                 óÈ  — d„ |                      ¦   «         D ¦   «         }|sdS t          | ||¦  «        }|                      ¦   «         D �]\  }}|j        dv r¿t          t          |j        ¦  «        }|j        �|j        n|                     ¦   «         }|                     |¦  «        \  }	}
|                     |¦  «        | 	                    |¦  «        z
  }||z
  |k    r||z
  }n||z
  dk    r|dz   |	k    rd}n|dz   }nd}t          ||
dz
  ¦  «        }n|j        dk    rŒÚ|j        d	k    rŒæt	          j        t          |d
z  ¦  «        t          j        |d¬¦  «        }�ŒdS )aH  This function warm-ups the caching allocator based on the size of the model tensors that will reside on each
    device. It allows to have one large call to Malloc, instead of recursively calling it later when loading
    the model, which is actually the loading speed bottleneck.
    Calling this function allows to cut the model loading time by a very large margin.

    A few facts related to loading speed (taking into account the use of this function):
    - When loading a model the first time, it is usually slower than the subsequent times, because the OS is very likely
    to cache the different state dicts (if enough resources/RAM are available)
    - Trying to force the OS to cache the files in advance (by e.g. accessing a small portion of them) is really hard,
    and not a good idea in general as this is low level OS optimizations that depend on resource usage anyway
    - As of 18/03/2025, loading a Llama 70B model with TP takes ~1 min without file cache, and ~13s with full file cache.
    The baseline, i.e. only loading the tensor shards on device and adjusting dtype (i.e. copying them) is ~5s with full cache.
    These numbers are reported for TP on 4 H100 GPUs.
    - It is useless to pre-allocate more than the model size in this function (i.e. using an `allocation_factor` > 1) as
    cudaMalloc is not a bottleneck at all anymore
    - Loading speed bottleneck is now almost only tensor copy (i.e. changing the dtype) and moving the tensors to the devices.
    However, we cannot really improve on those aspects obviously, as the data needs to be moved/copied in the end.
    c                 ó\   — i | ])\  }}t          |¦  «        ¯|t          j        |¦  «        “Œ*S r±   )rœ  r®   rÖ   )rþ   r×  rÖ   s      r£   r#  z,caching_allocator_warmup.<locals>.<dictcomp>  sG   € ð ð ð Ù(5¨¨vÕXmÐntÑXuÔXuðØ�uŒ|˜FÑ#Ô#ðð ð r¥   N)rÑ  Úxpug      ØAr   r   g333333ÓAr  Úneuronrü   F)r–   rÖ   rŒ  )r,  r§  ru  rM  r®   r‘  Úcurrent_deviceÚmem_get_infoÚmemory_reservedÚmemory_allocatedrã  r1  r½   rÞ  )ri  r  r˜   r�  r¢  rÖ   Ú
byte_countÚaccelerator_moduler‘  Úfree_device_memoryÚtotal_device_memoryÚunused_memoryr\  s                r£   r
  r
    sÂ  € ð(ð Ø9L×9RÒ9RÑ9TÔ9Tðñ ô Ðð "ð Øˆå+¨EÐ3IÈ<ÑXÔXÐð /×4Ò4Ñ6Ô6ð ,gñ ,gÑˆ�
ØŒ;˜/Ð)Ð)Ý!(­°´Ñ!<Ô!<ÐØ$*¤LÐ$<�F”L�LÐBT×BcÒBcÑBeÔBeˆEØ6H×6UÒ6UÐV[Ñ6\Ô6\Ñ3ÐÐ 3Ø.×>Ò>¸uÑEÔEÐHZ×HkÒHkÐlqÑHrÔHrÑrˆMð ˜MÑ)¨MÒ9Ð9Ø'¨-Ñ7�
�
à˜mÑ+¨mÒ;Ð;ð ! 1Ñ$Ð'9Ò9Ð9Ø!"�J�Jð "/°Ñ!2�J�Jð �
õ ˜ZÐ)<¸}Ñ)LÑMÔMˆJˆJØŒ[˜EÒ!Ð!ð
 ØŒ[˜HÒ$Ð$àåŒK�˜J¨!™OÑ,Ô,µE´MÈ&Ð`eÐfÑfÔfˆ‰ðY,gð ,gr¥   c                   óJ   ‡ — e Zd ZdZeeeeeeeeee	dœ
Z
dededefˆ fd„Zˆ xZS )ÚAttentionInterfacea_  
    Dict-like object keeping track of allowed attention functions. You can easily add a new attention function
    with a call to `register()`. If a model needs to locally overwrite an existing attention function, say `sdpa`,
    it needs to declare a new instance of this class inside the `modeling_<model>.py`, and declare it on that instance.
    )
Úflash_attention_4Úflash_attention_3rõ  r
  r  zpaged|flash_attention_4zpaged|flash_attention_3zpaged|flash_attention_2z
paged|sdpazpaged|eagerr®  rx  rž   c                 ó¼   •— |€t                                d¦  «         n|dk    r|| vrt          d|› d�¦  «        ‚t          ¦   «                              ||¦  «        S )zcReturn the requested `attn_implementation`. Also strictly check its validity, and raise if invalid.Na	  You tried to access the `AttentionInterface` with a `config._attn_implementation` set to `None`. This is expected if you use an Attention Module as a standalone Module. If this is not the case, something went wrong with the dispatch of `config._attn_implementation`r  r¾  zP` is not a valid attention implementation registered in the `AttentionInterface`)r¶  rË  ÚKeyErrorrJ  rÀ   )r¢   r®  rx  r  s      €r£   Úget_interfacez AttentionInterface.get_interfaceg  s€   ø€ àÐ&Ý×ÒðKñô ð ð ð
 ! GÒ+Ð+Ð0CÈ4Ð0OÐ0OÝØyÐ'ÐyÐyÐyñô ð õ ‰wŒw�{Š{Ð.°Ñ8Ô8Ð8r¥   )r¦   r§   r¨   r©   r9   r;   rA   r:   rB   r7   Ú_global_mappingrª   r   r»  r‘  r’  s   @r£   r¶  r¶  Q  sˆ   ø€ € € € € ðð ð 5Ø4Ø4Ø0Ø&Ø#:Ø#:Ø#:Ø2Ø4ðð €Oð9°ð 9¸xð 9ÈHð 9ð 9ð 9ð 9ð 9ð 9ð 9ð 9ð 9ð 9r¥   r¶  r  c                   ó^   — e Zd ZdZedej        fd„¦   «         Zedej        fd„¦   «         ZdS )ÚPreTrainedAudioTokenizerBasea¬  
    Class that additionally defines the behavior of any `audio_tokenizer` to be added.
    Characteristic for any of them:
        1. Encode raw audio into discrete audio codebooks (with x channels)
        2. Decode from discrete audio codebooks back to raw audio
    It is possible that they can decode in different ways given a different representation
    but they are forced to support 2. nonetheless, e.g. see `DAC`.
    Úinput_valuesc                 ó   — dS )z�
        Encode raw audio retrieved from a respective `FeatureExtractor` into discrete audio codebooks (with x channels)
        Nr±   )r¢   r¿  r®  r¯  s       r£   Úencodez#PreTrainedAudioTokenizerBase.encode„  ó   € € € r¥   Úaudio_codesc                 ó   — dS )z6Decode from discrete audio codebooks back to raw audioNr±   )r¢   rÃ  r®  r¯  s       r£   Údecodez#PreTrainedAudioTokenizerBase.decodeŠ  rÂ  r¥   N)	r¦   r§   r¨   r©   r   r®   r   rÁ  rÅ  r±   r¥   r£   r¾  r¾  z  sw   € € € € € ðð ð ð 5¤<ð ð ð ñ „^ðð
 ðE %¤,ð Eð Eð Eñ „^ðEð Eð Er¥   r¾  r    )rÔ   TN)NNNr‰  (  rd  r€  r/  rL  rz  r¾   rp  r
  rš  Úabcr   r   Úcollections.abcr   r   Ú
contextlibr   Údataclassesr   r	   r
   r   Ú	itertoolsr   Ú	threadingr   Útypingr   r   r   r   r   Úzipfiler   r®   Úhuggingface_hubr   r   Ú	packagingr   Úsafetensorsr   Úsafetensors.torchr   r*  r   ry  r   r   Útorch.distributionsr   Útorch.utils.checkpointr   r¦  r   rµ  Úconfiguration_utilsr    Úconversion_mappingr!   Úcore_model_loadingr"   r#   r$   r%   r¶   r&   Údynamic_module_utilsr'   Ú
generationr(   r)   Úintegrationsr*   r+   r,   r-   r.   Úintegrations.accelerater/   r0   r1   r2   r3   r4   r5   r¹  r6   Úintegrations.eager_pagedr7   Úintegrations.finegrained_fp8r8   Úintegrations.flash_attentionr9   Úintegrations.flash_pagedr:   Úintegrations.flex_attentionr;   rÀ  r<   r=   r>   Úintegrations.moer?   Úintegrations.peftr@   Úintegrations.sdpa_attentionrA   Úintegrations.sdpa_pagedrB   Úintegrations.tensor_parallelrC   rD   rE   rF   rG   rH   rI   Úloss.loss_utilsrJ   Úmodeling_flash_attention_utilsrK   rL   rM   rN   Úmodeling_rope_utilsrO   Úmonkey_patchingrP   rQ   Úpytorch_utilsrR   Ú
quantizersrS   Úquantizers.autorT   Úquantizers.quantizers_utilsrU   Úsafetensors_conversionrV   ÚutilsrW   rX   rY   rZ   r[   r\   r]   r^   r_   r`   ra   rb   rc   rd   re   rf   rg   rh   ri   rj   rk   Úutils.genericrl   rm   rn   Ú	utils.hubro   rp   rq   rr   Úutils.import_utilsrs   rt   ru   rv   rw   rx   ry   Úutils.loading_reportrz   r{   Úutils.output_capturingr|   r}   Úutils.quantization_configr~   Úaccelerate.hooksr   Úaccelerate.utilsr€   Úkernels.layer.moder�   Ú_typingr‚   Úis_availabler´   Ú!smdistributed.modelparallel.torchÚmodelparallelrn  Úsmdistributed.modelparallelrƒ   ÚSMP_VERSIONrç  rm  Ú
get_loggerr¦   r¶  r¿   rÀ   Úupperr„   r†   r‡   rÄ   rÈ   rŒ   r¬   r·   r½   rº   rÁ   rÅ   rÉ   r–   rª   rÒ   rÚ   rå   Úuint8Úint8Úint16Úuint16rÞ  rß  Úint32Úuint32rà   Úfloat64Úint64Úuint64Úfloat8_e4m3fnÚfloat8_e5m2r0  r  r�  rÖ   r­   r:  rB  r'  r¯   rR  re  r  rb  rh  r‰  r’  r—  ru  rÆ  rÐ  rÒ  r  rˆ   rC  r©   rL  rf  rœ  r§  r
  r¶  r  r«   r¾  r±   r¥   r£   ú<module>r     sC  ðð Ð Ð Ð Ð Ø €€€Ø Ð Ð Ð Ø €€€Ø €€€Ø 	€	€	€	Ø 	€	€	€	Ø 
€
€
€
Ø €€€Ø Ð Ð Ð Ð Ð Ø #Ð #Ð #Ð #Ð #Ð #Ø .Ð .Ð .Ð .Ð .Ð .Ð .Ð .Ø %Ð %Ð %Ð %Ð %Ð %Ø (Ð (Ð (Ð (Ð (Ð (Ð (Ð (Ø $Ð $Ð $Ð $Ð $Ð $Ð $Ð $Ø Ð Ð Ð Ð Ð Ø Ð Ð Ð Ð Ð Ø HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HØ Ð Ð Ð Ð Ð à €€€Ø OÐ OÐ OÐ OÐ OÐ OÐ OÐ OØ Ð Ð Ð Ð Ð Ø !Ð !Ð !Ð !Ð !Ð !Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6Ø 9Ð 9Ð 9Ð 9Ð 9Ð 9Ø Ð Ð Ð Ð Ð Ð Ð Ø +Ð +Ð +Ð +Ð +Ð +Ø -Ð -Ð -Ð -Ð -Ð -à $Ð $Ð $Ð $Ð $Ð $Ø 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø <Ð <Ð <Ð <Ð <Ð <ðð ð ð ð ð ð ð ð ð ð ð ð +Ð *Ð *Ð *Ð *Ð *Ø 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð FÐ EÐ EÐ EÐ EÐ EØ CÐ CÐ CÐ CÐ CÐ CØ CÐ CÐ CÐ CÐ CÐ CØ AÐ AÐ AÐ AÐ AÐ AØ =Ð =Ð =Ð =Ð =Ð =Ø ?Ð ?Ð ?Ð ?Ð ?Ð ?Ø QÐ QÐ QÐ QÐ QÐ QÐ QÐ QÐ QÐ QØ 3Ð 3Ð 3Ð 3Ð 3Ð 3Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2Ø ?Ð ?Ð ?Ð ?Ð ?Ð ?Ø AÐ AÐ AÐ AÐ AÐ Aðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð *Ð )Ð )Ð )Ð )Ð )ðð ð ð ð ð ð ð ð ð ð ð ð 5Ð 4Ð 4Ð 4Ð 4Ð 4Ø BÐ BÐ BÐ BÐ BÐ BÐ BÐ BØ ,Ð ,Ð ,Ð ,Ð ,Ð ,Ø #Ð #Ð #Ð #Ð #Ð #Ø -Ð -Ð -Ð -Ð -Ð -Ø =Ð =Ð =Ð =Ð =Ð =Ø 3Ð 3Ð 3Ð 3Ð 3Ð 3ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð. jÐ iÐ iÐ iÐ iÐ iÐ iÐ iÐ iÐ iØ dÐ dÐ dÐ dÐ dÐ dÐ dÐ dÐ dÐ dÐ dÐ dðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð KÐ JÐ JÐ JÐ JÐ JÐ JÐ JØ HÐ HÐ HÐ HÐ HÐ HÐ HÐ HØ 9Ð 9Ð 9Ð 9Ð 9Ð 9ð ÐÑÔð =Ø3Ð3Ð3Ð3Ð3Ð3Ø<Ð<Ð<Ð<Ð<Ð<àð (Ø'Ð'Ð'Ð'Ð'Ð'à'Ð'Ð'Ð'Ð'Ð'ð  %Ô0×=Ò=Ñ?Ô?Ð àÐÑÔð &Ø3Ð3Ð3Ð3Ð3Ð3Ð3Ð3Ð3ØFÐFÐFÐFÐFÐFà - ¤¨kÑ :Ô :¸m¸g¼mÈFÑ>SÔ>SÒ SÐÐà %Ðð 
ˆÔ	˜HÑ	%Ô	%€àŒz�~Š~˜n¨cÑ2Ô2×8Ò8Ñ:Ô:€Ø”J—N’NÐ#6¸Ñ<Ô<×BÒBÑDÔDÐ Ø%˜gÐ&CÐK\Ð]Ñ]Ô]Ð Ø€ØÐ ð €�$ÐÑÔð-ð -ð -ð -ð -ñ -ô -ñ Ôð-ð6¨4ð ð ð ð ð.¨3ð .ð .ð .ð .ð`ð `ð `ð ðð ñ „ðð ð#ð #ñ „ð#ð ð0ð 0˜Uœ[ð 0¸CÀ$¹Jð 0ð 0ð 0ñ „ð0ð.ð ð ð1ð 1ð 1ð  ŒJØ
Œ+Ø
Œ*ØŒ;ØŒ<ØŒ=ØŒNØŒ;ØŒ<ØŒ=ØŒ=ØŒ;ØŒ<ØÔ"ØÔ ðð Ð ð&Ð-ð °$ð ð ð ð ð6 (-ØØ $ð	/kð /kØ˜2œ;Ñ&ð/kà˜œÑ$ð/kð ð/kð ˜‘+ð	/kð
 
ˆ#ˆuŒ|Ð
Ôð/kð /kð /kð /kðd�U”\ð  cð ð ð ð ð "¤)ð °°S´	ð ð ð ð ð,˜D  S¤œNð ,¸¸SÀ%Ä,Ð=NÔ8Oð ,ÐTYÐZ^Ð_bÐcfÔ_gÔZhÐjnÐorÔjsÐZsÔTtð ,ð ,ð ,ð ,ð>%Ø�#�c”(Œ^ð%Ø)-¨c°5´<Ð.?Ô)@ð%à
ˆ4��C”Œ>˜4  C¤œ>Ð)Ô*ð%ð %ð %ñ %ð*MØ�S˜%œ,Ð&Ô'ðMØ0AðMà	ˆ#ˆuŒ|Ð
ÔðMð Mð Mñ Mð`(Ð&7ð (ÀSð (ÐRWÔR^ð (ð (ð (ñ (ðð ˜sð ¨S°4©Zð À3ð ð ð ñ ð 26Ø-1Ø"ðV.ð V.Ø#&¨¬Ñ#4°tÑ#;ðV.à�4‰ZðV.ð �T‰zðV.ð ˜D‘[ð	V.ð
 �t‘ðV.ð ðV.ð %(¨$¡JðV.ð $ dÑ*ðV.ñ �t‘ðV.ð ˆ4�Œ9�tÑ˜T D™[Ð(Ô)ðV.ð V.ð V.ñ V.ð@	 (,ðPð PØ�”Ñ˜tÑ# dÑ*ðPà˜3”i $Ñ&ðPð ðPð ˜T‘kð	Pð
 �t‘ðPð ðPð  Ñ$ðPð Ð˜Uœ[Ð(Ô)ðPð Pð Pñ Pðfjð jð jð jð jñ jô jñ jðZj*ð j*ð j*ð j*ð j*ñ j*ô j*ñ j*ðZ;Vð ;Vð ;Vð ;Vð ;V�b”iÑ!5Ñ7GÈÐYiñ ;Vô ;Vñ ;VðDx (˜i©Õ(CÑDÔD�Õ ÙÕÕ&Ð2Ù*9Õ*EÕ*M×*TÓ*TØ [¸|ð +Uñ +ô +�OÕÕ'ð
 
Ø YÐ Y™Ð Y°DÐ YÁ_Ð YÐ YÐ Yñ 
„Ù Yð 
Ø MÐ M˜œ	Ð M¨dÐ M¸r¼yÐ MÐ MÐ Mñ 
„Ù Mðð ˜œ	ð ¨dð ¸r¼yð ð ð ñ ð2@ #¨¡)¨e¬lÑ":ð @¸tð @ð @ð @ñ @ð ^bðð ÙðØ48ðØHSÐVZÑHZðð ð ñ ðDIg¡Oð IgÈ$ð IgÐ^iÐlpÑ^pð Igð Igð Igñ IgðX"9ð "9ð "9ð "9ð "9Ð)ñ "9ô "9ñ "9ðL /AÑ.@Ñ.BÔ.BÑ Ñ+Ñ BÐ BÑ BðEð Eð Eð Eð E¡?ñ Eô Eñ Eð Eð Er¥   