Ë
    S^(hgs ã            %       ót  — U d dl Z d dlZd dlZd dlZd dlZd dlZd dlZd dlZd dl	Z	d dl
Z
d dlZd dlZd dlZd dlZd dl mZ d dlmZ d dlmZ d dlmZ d dlmZ d dlmZmZ d dlmZ d d	lmZmZmZm Z m!Z!m"Z"m#Z#m$Z$m%Z%m&Z& d d
l'm(Z( d dl)Z)d dl*Z)d dl+m,Z, d dl-m.Z. d dl)m/Z/m0Z0 d dl1m2Z2 d dl3m4Z4m5Z5 d dl6m7Z7 d dl8m9Z9  e9«       rd dl:m;Z; ddl<m=Z= ddl>m?Z? ddl@mAZA ddlBmCZCmDZDmEZE ddlFmGZGmHZHmIZI ddlJmKZKmLZL ddlMmNZNmOZO ddlPmQZQ ddlRmSZS ddlTmUZU ddlVmWZWmXZX ddlYmZZZ dd l[m\Z\m]Z]m^Z^m_Z_m`Z`maZambZb dd!lcmdZdmeZe dd"lfmgZg dd#lhmiZi dd$ljmkZkmlZlmmZmmnZnmoZompZpmqZqmrZrmsZsmtZtmuZumvZvmwZwmxZxmyZymzZzm{Z{m|Z|m}Z}m~Z~mZm€Z€m�Z�m‚Z‚mƒZƒm„Z„m…Z…m†Z†m‡Z‡mˆZˆm‰Z‰mŠZŠm‹Z‹mŒZŒm�Z�mŽZŽ dd%l�m�Z�m‘Z‘ dd&l’m“Z“m”Z”m•Z•m–Z– dd'l—m˜Z˜m™Z™ e
�j4                  �j7                  d(d)«      �j9                  «       Z�e
�j4                  �j7                  d*d)«      �j9                  «       Zž e~«       rid d+lŸm Z m¡Z¡ d d,l¢m£Z£ d d-l¤m¥Z¥m¦Z¦m§Z§m¨Z¨m©Z©mªZªm«Z«  e.�jX                  e�jZ                  j]                  d.«      «      Z®e® e.�jX                  d/«      k\  rd d0l¯m°Z°  e…«       rd d1l±m²Z² d d2l³m´Zµ d d3l³m¶Z·  eO«       rd dl¸Z¸ eŒ�jr                  eº«      Z»d4a¼d5a½d5a¾e)�j~                  �j�                  «       ZÁd6„ ZÂd7„ ZÃ e”«       r7d dlÄmÅc m)ZÆ d d8lÇmÈZÉ  e.�jX                  eÉ«       e.�jX                  d9«      k\  ZÊnd5ZÊ eƒ«       rdd:ljmËZË  e%d;d<¬=«      ZÌe0�jš                  �jœ                  e0�jš                  �jž                  e0�jš                  �j                   e0�jš                  �j¢                  e0�jš                  �j¤                  e0�jš                  �j¦                  e0�jš                  �j¨                  e0�jš                  �jª                  e0�jš                  �j¬                  e0�jš                  �j®                  e0�jš                  �j°                  e0�jš                  �j²                  e0�jš                  �j´                  e0�jš                  �j¶                  d>œZÜed?„ «       ZÝed@„ «       ZÞedA„ «       ZßdB„ ZàdCe&e0�jÂ                  dDf   fdE„ZâdCe&e0�jÂ                  dDf   fdF„ZãdCe&e0�jÂ                  dDf   fdG„ZädH„ ZådI„ Zæd²dJ„Zçe)�jÐ                  e)�jÒ                  e)�jÔ                  e)�jÖ                  e)�jØ                  e)�jÚ                  e)�jÜ                  e)�jÞ                  e)�jà                  e)�jâ                  e)�jä                  dKœZó e‡dL«      re)�jä                  eódM<    e‡dN«      r0e)�jè                  eódO<   e)�jê                  eódP<   e)�jì                  eódQ<    e‡dL«      r e)�jä                  eódM<   e)�jî                  eódR<   	 	 	 d³dSe&eøe
�jò                  f   dTeèdUe!e&eøe)�jô                  f      dVeèfdW„ZûdX„ ZüdYe)j^                  dZeýfd[„Zþd´d\e0�jÂ                  fd]„Zÿd^e e"eø      d_eeøe)j^                  f   dZe#e e"eø      e eø   f   fd`„�Z d^e e"eø      d_eeøe)j^                  f   dZe#e e"eø      e"eø   f   fda„�Z	 	 dµdbd<dceødde)j^                  dee!e�j                     dfe!ee   dZe&eèe!e)�j                     f   fdg„�Zdbd<dceødYe)j^                  fdh„�Z e)�j                  «       	 	 	 	 	 	 	 	 	 	 d¶dbd<d_edieødje eø   dkeeøeøf   dle!e   dme!eø   dne!e   doe!eø   dpe!e   dfe!ee   dqeèdee!e�j                     dre!e eø      dse!dt   dZe#e!e   e!e   f   f du„«       �Zd·dveødwe!eø   dZeøfdx„�Zdye!e&eøe
�jò                  f      dzeødwe!eø   d{e!eø   d|eèd}eèd~eèdeød€eèd�e!eeøeøf      d‚eèdƒe!e&eøeèf      d„�e	d…eød†e!eø   dZe#e!e eø      e!e   f   f d‡„�Z
dˆe!e&eøe)�j                  ef      d‰e!e eø      dŠe?d‹e!e   d_e!e   dVeèdZe#e?e!e)�j                     e!e)�j                     f   fdŒ„�Zdbd<dle!e&eøef      d�e!e   dfe!ee   dˆe!e)�j                     dee!e�j                     dZefdŽ„�Zdbd<d�e eø   d�e eø   d‘eèdfe!ee   dledZe#e eø   e eø   f   fd’„�Zdbd<d_e!e   d‰e!e eø      d“eèd”eeøeøf   dTeèdVeèdZe#e eø   e e#eýeýf      f   fd•„�Z G d–„ d—e«      �Z G d˜„ dD«      �Z G d™„ d<e0�jÂ                  �eeEexeG«      �Z ez�e�j$                  «      �e�_        �e�j$                  �j&                  �>�e�j$                  �j&                  �j)                  dbdšd›¬œ«      �e�j$                  �_         G d�„ dže0�jÂ                  «      �Z G dŸ„ d e0�jÂ                  «      �Z G d¡„ d¢e0�jÂ                  «      �Ze G d£„ d¤ew«      «       �Z G d¥„ d¦e0�jÂ                  «      �Z G d§„ d¨e0�jÂ                  «      �Zd¸dbe0�jÂ                  d©eèdZe0�jÂ                  fdª„�Zd«„ �Zd¹db�ed¬efd­„�Zd®„ �Z G d¯„ d°e«      �Z �e«       �Z �e�e!d±<   y)ºé    N)Údefaultdict)ÚMutableMapping)Úcontextmanager)Ú	dataclass)ÚEnum)ÚpartialÚwraps)ÚThread)
ÚAnyÚCallableÚDictÚListÚOptionalÚSetÚTupleÚTypeÚTypeVarÚUnion)Ú
is_zipfile)Ú"split_torch_state_dict_into_shards)Úversion)ÚTensorÚnn)Úconstraints)ÚCrossEntropyLossÚIdentity)Ú
checkpoint)Úis_torchao_available)ÚInt4WeightOnlyConfigé   )Úget_activation)ÚPretrainedConfig)Úcustom_object_save)ÚCompileConfigÚGenerationConfigÚGenerationMixin)ÚPeftAdapterMixinÚdeepspeed_configÚis_deepspeed_zero3_enabled)Úfind_tied_parametersÚinit_empty_weights)Ú!_load_state_dict_into_zero3_modelÚis_deepspeed_available)Úflash_attention_forward)Úflex_attention_forward)Úsdpa_attention_forward)ÚSUPPORTED_TP_STYLESÚshard_and_distribute_module)ÚLOSS_MAPPING)ÚConv1DÚapply_chunking_to_forwardÚ find_pruneable_heads_and_indicesÚid_tensor_storageÚprune_conv1d_layerÚprune_layerÚprune_linear_layer)ÚAutoHfQuantizerÚHfQuantizer)Úget_module_from_name)Úauto_conversion)$ÚADAPTER_SAFE_WEIGHTS_NAMEÚADAPTER_WEIGHTS_NAMEÚCONFIG_NAMEÚDUMMY_INPUTSÚFLAX_WEIGHTS_NAMEÚSAFE_WEIGHTS_INDEX_NAMEÚSAFE_WEIGHTS_NAMEÚTF2_WEIGHTS_NAMEÚTF_WEIGHTS_NAMEÚWEIGHTS_INDEX_NAMEÚWEIGHTS_NAMEÚContextManagersÚModelOutputÚPushToHubMixinÚcached_fileÚ	copy_funcÚdownload_urlÚextract_commit_hashÚhas_fileÚis_accelerate_availableÚis_bitsandbytes_availableÚis_flash_attn_2_availableÚis_offline_modeÚis_optimum_availableÚis_peft_availableÚis_remote_urlÚis_safetensors_availableÚis_torch_flex_attn_availableÚis_torch_greater_or_equalÚis_torch_mlu_availableÚis_torch_npu_availableÚis_torch_sdpa_availableÚis_torch_xla_availableÚloggingÚreplace_return_docstringsÚ	strtobool)Úcreate_and_tag_model_cardÚget_checkpoint_shard_files)ÚENV_VARS_TRUE_VALUESÚis_sagemaker_mp_enabledÚis_torch_fx_proxyÚis_torchdynamo_compiling)ÚBitsAndBytesConfigÚQuantizationMethodÚXLA_USE_BF16Ú0ÚXLA_DOWNCAST_BF16)Údispatch_modelÚinfer_auto_device_map)Úadd_hook_to_module)Ú$check_tied_parameters_on_same_deviceÚextract_model_from_parallelÚget_balanced_memoryÚget_max_memoryÚload_offloaded_weightsÚoffload_weightÚsave_offload_indexÚ
accelerateú0.31)Úget_state_dict_from_offload)Ú	safe_open)Ú	load_file)Ú	save_fileTFc                  ó6  — t         j                  j                  «       xrz t         j                  j                  «       xrZ t	        t
        j                  j                  dd«      «      dk(  xr, t	        t
        j                  j                  dd«      «      dk(  S )NÚACCELERATE_USE_FSDPÚFalser    ÚFSDP_CPU_RAM_EFFICIENT_LOADING)ÚtorchÚdistributedÚis_availableÚis_initializedrb   ÚosÚenvironÚget© ó    úY/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/modeling_utils.pyÚis_fsdp_enabledrŒ   ©   sz   € ä×Ñ×&Ñ&Ó(ò 	VÜ×Ñ×,Ñ,Ó.ò	Vä”b—j‘j—n‘nÐ%:¸GÓDÓEÈÑJò	Vô ”b—j‘j—n‘nÐ%EÀwÓOÓPÐTUÑUð	rŠ   c                  óÚ   — t         j                  j                  «       xrL t         j                  j                  «       xr, t	        t
        j                  j                  dd«      «      dk(  S )NÚ
LOCAL_RANKéÿÿÿÿr   )r‚   rƒ   r„   r…   Úintr†   r‡   rˆ   r‰   rŠ   r‹   Úis_local_dist_rank_0r‘   ²   sQ   € ä×Ñ×&Ñ&Ó(ò 	7Ü×Ñ×,Ñ,Ó.ò	7ä”—
‘
—‘˜|¨RÓ0Ó1°QÑ6ðrŠ   ©Ú__version__z1.10)Úfind_adapter_config_fileÚSpecificPreTrainedModelTypeÚPreTrainedModel)Úbound)Úuniform_Únormal_Útrunc_normal_Ú	constant_Úxavier_uniform_Úxavier_normal_Úkaiming_uniform_Úkaiming_normal_ÚuniformÚnormalÚxavier_uniformÚxavier_normalÚkaiming_uniformÚkaiming_normalc               #   óÄ  K  — t         } da d„ }t        j                  «       D ]*  \  }}t        t        j
                  j                  ||«       Œ, 	 d–— | a t        j                  «       D ]*  \  }}t        t        j
                  j                  ||«       Œ, y# | a t        j                  «       D ]*  \  }}t        t        j
                  j                  ||«       Œ, w xY w­w)ze
    Context manager to globally disable weight initialization to speed up loading large models.
    Fc                   ó   — y ©Nr‰   )ÚargsÚkwargss     r‹   Ú
_skip_initz#no_init_weights.<locals>._skip_initä   s   € ØrŠ   N)Ú_init_weightsÚTORCH_INIT_FUNCTIONSÚitemsÚsetattrr‚   r   Úinit)Úold_init_weightsr«   ÚnameÚ	init_funcs       r‹   Úno_init_weightsr´   Ú   s¿   è ø€ ô %Ðà€Mòô 0×5Ñ5Ó7ò 1‰ˆˆiÜ”—‘—‘˜t ZÕ0ð1ð4Ûà(ˆä3×9Ñ9Ó;ò 	4‰OˆD�)Ü”E—H‘H—M‘M 4¨Õ3ñ	4øð )ˆä3×9Ñ9Ó;ò 	4‰OˆD�)Ü”E—H‘H—M‘M 4¨Õ3ñ	4üs    ‚AC ÁB ÁAC ÂACÃC c               #   ó,   K  — da 	 d –— da y # da w xY w­w©NTF)Ú_is_quantizedr‰   rŠ   r‹   Úset_quantized_stater¸   ô   s   è ø€ ð €MðÛà‰ø˜‰üó   ‚† Š�‘c               #   ó,   K  — da 	 d –— da y # da w xY w­wr¶   )Ú_is_ds_init_calledr‰   rŠ   r‹   Úset_zero3_stater¼     s"   è ø€ ð Ðð#Ûà"Ñø˜UÑür¹   c                 ó.   ‡ — t        ‰ «      ˆ fd„«       }|S )zï
    Decorator to restore the default torch dtype
    at the end of the function. Serves
    as a backup in case calling the function raises
    an error after the function has changed the default dtype but before it could restore it.
    c                  óœ   •— t        j                  «       }	  ‰| i |¤Žt        j                  |«       S # t        j                  |«       w xY wr¨   )r‚   Úget_default_dtypeÚset_default_dtype)r©   rª   Ú	old_dtypeÚfuncs      €r‹   Ú_wrapperz-restore_default_torch_dtype.<locals>._wrapper  s@   ø€ ä×+Ñ+Ó-ˆ	ð	/Ù˜Ð( Ñ(ä×#Ñ# IÕ.øŒE×#Ñ# IÕ.ús	   —4 ´A)r	   )rÂ   rÃ   s   ` r‹   Úrestore_default_torch_dtyperÄ     s"   ø€ ô ˆ4ƒ[ó/ó ð/ð €OrŠ   Ú	parameterÚModuleUtilsMixinc                 ó  — 	 t        | j                  «       «      j                  S # t        $ r] dt        j
                  dt        t        t        t        f      fd„}| j                  |¬«      }t        |«      }|d   j                  cY S w xY w)NÚmoduleÚreturnc                 óš   — | j                   j                  «       D ��cg c]  \  }}t        j                  |«      sŒ||f‘Œ! }}}|S c c}}w r¨   ©Ú__dict__r®   r‚   Ú	is_tensor©rÈ   ÚkÚvÚtupless       r‹   Úfind_tensor_attributesz4get_parameter_device.<locals>.find_tensor_attributes$  óA   € Ø)/¯©×)>Ñ)>Ó)@×W¡  AÄEÇOÁOÐTUÕDV�q˜!’fÐWˆFÑWØˆMùó Xó
   žA¼A©Úget_members_fnr    )ÚnextÚ
parametersÚdeviceÚStopIterationr   ÚModuler   r   Ústrr   Ú_named_members©rÅ   rÒ   ÚgenÚfirst_tuples       r‹   Úget_parameter_devicerá     s„   € ð%Ü�I×(Ñ(Ó*Ó+×2Ñ2Ð2øÜò 	%ð	¬2¯9©9ð 	¼¼eÄCÌÀKÑ>PÑ9Qó 	ð ×&Ñ&Ð6LÐ&ÓMˆÜ˜3“iˆØ˜1‰~×$Ñ$Ò$ð	%úó   ‚"% ¥A#BÂ
Bc                 ó  — 	 t        | j                  «       «      j                  S # t        $ r] dt        j
                  dt        t        t        t        f      fd„}| j                  |¬«      }t        |«      }|d   j                  cY S w xY w)z`
    Returns the first parameter dtype (can be non-floating) or asserts if none were found.
    rÈ   rÉ   c                 óš   — | j                   j                  «       D ��cg c]  \  }}t        j                  |«      sŒ||f‘Œ! }}}|S c c}}w r¨   rË   rÎ   s       r‹   rÒ   z9get_first_parameter_dtype.<locals>.find_tensor_attributes6  rÓ   rÔ   rÕ   r    )r×   rØ   ÚdtyperÚ   r   rÛ   r   r   rÜ   r   rÝ   rÞ   s       r‹   Úget_first_parameter_dtyperæ   -  s„   € ð$Ü�I×(Ñ(Ó*Ó+×1Ñ1Ð1øÜò 	$ð	¬2¯9©9ð 	¼¼eÄCÌÀKÑ>PÑ9Qó 	ð ×&Ñ&Ð6LÐ&ÓMˆÜ˜3“iˆØ˜1‰~×#Ñ#Ò#ð	$úrâ   c                 óF  — d}| j                  «       D ]È  }|j                  }|j                  «       sŒ t        t        v rt        «       rt        j                  c S t        t        v rht        «       r^|j                  t        j                  k(  rt        j                  c S |j                  t        j                  k(  rt        j                  c S |j                  c S  |�|S dt        j                  dt        t        t         t"        f      fd„}| j%                  |¬«      }d}|D ](  }|}|d   j                  «       sŒ|d   j                  c S  |�|d   j                  S | j'                  «       D ],  }|j                  }|j                  «       sŒ |j                  c S  |S )zz
    Returns the first found floating dtype in parameters if there is one, otherwise returns the last dtype it found.
    NrÈ   rÉ   c                 óš   — | j                   j                  «       D ��cg c]  \  }}t        j                  |«      sŒ||f‘Œ! }}}|S c c}}w r¨   rË   rÎ   s       r‹   rÒ   z3get_parameter_dtype.<locals>.find_tensor_attributesY  sA   € Ø%+§_¡_×%:Ñ%:Ó%<×S™T˜Q ÄÇÁÐPQÕ@R�1�a’&ÐSˆÑSØˆùó TrÔ   rÕ   r    )rØ   rå   Úis_floating_pointrk   re   r_   r‚   Úbfloat16rm   ÚfloatÚdoubleÚfloat32r   rÛ   r   r   rÜ   r   rÝ   Úbuffers)rÅ   Ú
last_dtypeÚtrÒ   rß   Ú
last_tupleÚtuples          r‹   Úget_parameter_dtyperó   ?  su  € ð €JØ×!Ñ!Ó#ò ˆØ—W‘Wˆ
Ø×ÑÕ ô
 Ô3Ñ3Ô8NÔ8PÜ—~‘~Ò%Ü Ô$8Ñ8Ô=SÔ=UØ—7‘7œeŸk™kÒ)Ü Ÿ>™>Ò)Ø—7‘7œeŸl™lÒ*Ü Ÿ=™=Ò(Ø—7‘7ŠNðð  ÐàÐð¤r§y¡yð ´T¼%ÄÄVÀÑ:LÑ5Mó ð ×
"Ñ
"Ð2HÐ
"Ó
I€CØ€JØò "ˆØˆ
Ø�‰8×%Ñ%Õ'Ø˜‘8—>‘>Ò!ð"ð
 Ðà˜!‰}×"Ñ"Ð"ð ×ÑÓ ò ˆØ—W‘Wˆ
Ø×ÑÕ Ø—7‘7ŠNðð ÐrŠ   c                 ó~   — | j                  «       D ]   }|j                  «       sŒ|j                  c S  t        d«      ‚)z_
    Returns the first found floating dtype in `state_dict` or asserts if none were found.
    z5couldn't find any floating point dtypes in state_dict)Úvaluesré   rå   Ú
ValueError©Ú
state_dictrð   s     r‹   Úget_state_dict_float_dtyperù   p  s?   € ð ×ÑÓ ò ˆØ×ÑÕ Ø—7‘7ŠNðô ÐLÓ
MÐMrŠ   c                 ó®   — | j                  «       D ]   }|j                  «       sŒ|j                  c S  t        | j                  «       «      j                  S )zt
    Returns the first found floating dtype in `state_dict` if there is one, otherwise returns the first dtype.
    )rõ   ré   rå   r×   r÷   s     r‹   Úget_state_dict_dtyperû   {  sM   € ð ×ÑÓ ò /ˆØ×ÑÕ Ø—7‘7ŠNð/ô �J×%Ñ%Ó'Ó(×.Ñ.Ð.rŠ   c                 óp  — t         j                  j                  |t        «      }t         j                  j                  |t        «      }t         j                  j                  |«      }t         j                  j                  |«      }|sJ|r
t        «       s>t        «       rt        t        fnt        f}t        ddj                  |«      › d|› d�«      ‚d}	|r-|r't        «       rd}	nt        j                  d|› d�«       n|sd}	|	r|n|}
t        |
d	d
¬«      5 }t        j                  |«      }ddd«       t        t        d   j                  «       «      «      }|d   j!                  «       }| j#                  «       j!                  «       }|D �cg c]	  }||vsŒ|‘Œ }}|D �cg c]	  }||vsŒ|‘Œ }}|r´t%        |«      dkD  st%        |«      dkD  r˜d| j&                  j(                  › �}t%        |«      dkD  r,dj                  |D �cg c]  }d|› d�‘Œ
 c}«      }|d|› d�z  }t%        |«      dkD  r,dj                  |D �cg c]  }d|› d�‘Œ
 c}«      }|d|› d�z  }t+        |«      ‚|	rt,        nt/        t0        j                  dd¬«      }|D ]P  } |t         j                  j                  ||«      «      }| j3                  |d¬«       ~t5        j6                  «        ŒR t0        j8                  j:                  j<                  j?                  ||«      S # 1 sw Y   �ŒëxY wc c}w c c}w c c}w c c}w )aê  
    This is the same as
    [`torch.nn.Module.load_state_dict`](https://pytorch.org/docs/stable/generated/torch.nn.Module.html?highlight=load_state_dict#torch.nn.Module.load_state_dict)
    but for a sharded checkpoint.

    This load is performed efficiently: each checkpoint shard is loaded one by one in RAM and deleted after being
    loaded in the model.

    Args:
        model (`torch.nn.Module`): The model in which to load the checkpoint.
        folder (`str` or `os.PathLike`): A path to a folder containing the sharded checkpoint.
        strict (`bool`, *optional*, defaults to `True`):
            Whether to strictly enforce that the keys in the model state dict match the keys in the sharded checkpoint.
        prefer_safe (`bool`, *optional*, defaults to `False`):
            If both safetensors and PyTorch save files are present in checkpoint and `prefer_safe` is True, the
            safetensors files will be loaded. Otherwise, PyTorch files are always loaded when possible.

    Returns:
        `NamedTuple`: A named tuple with `missing_keys` and `unexpected_keys` fields
            - `missing_keys` is a list of str containing the missing keys
            - `unexpected_keys` is a list of str containing the unexpected keys
    zCan't find a checkpoint index (ú or z) in ú.FTz"Cannot load sharded checkpoint at z+ safely since safetensors is not installed!Úrúutf-8©ÚencodingNÚ
weight_mapr   ú#Error(s) in loading state_dict for ú,ú"z
Missing key(s): Úcpu©Úmap_locationÚweights_only)Ústrict) r†   ÚpathÚjoinrH   rD   ÚisfilerY   rö   ÚloggerÚwarningÚopenÚjsonÚloadÚlistÚsetrõ   Úkeysrø   ÚlenÚ	__class__Ú__name__ÚRuntimeErrorÚsafe_load_filer   r‚   Úload_state_dictÚgcÚcollectr   ÚmodulesrÈ   Ú_IncompatibleKeys)ÚmodelÚfolderr  Úprefer_safeÚ
index_fileÚsafe_index_fileÚindex_presentÚsafe_index_presentÚ	filenamesÚ	load_safeÚ
load_indexÚfÚindexÚshard_filesÚloaded_keysÚ
model_keysÚkeyÚmissing_keysÚunexpected_keysÚerror_messagerÏ   Ústr_missing_keysÚstr_unexpected_keysÚloaderÚ
shard_filerø   s                             r‹   Úload_sharded_checkpointr8  ˆ  sõ  € ô0 —‘—‘˜fÔ&8Ó9€JÜ—g‘g—l‘l 6Ô+BÓC€Oä—G‘G—N‘N :Ó.€MÜŸ™Ÿ™¨Ó8ÐáÑ"4Ô9QÔ9Sä=UÔ=WÔÔ!8Ñ9Ô^pÐ]rð 	ô Ð:¸6¿;¹;ÀyÓ;QÐ:RÐRWÐX^ÐW_Ð_`ÐaÓbÐbà€IÙÙÜ'Ô)Ø ‘	ä—‘Ø8¸¸Ð@kÐlõñ ØˆIá$-‘°:€Jä	ˆj˜#¨Ô	0ð °AÜ—	‘	˜!“ˆ÷ô ”s˜5 Ñ.×5Ñ5Ó7Ó8Ó9€Kð ˜Ñ%×*Ñ*Ó,€KØ×!Ñ!Ó#×(Ñ(Ó*€JØ#-ÖH˜C°¸KÒ1G’CÐH€LÐHØ&1ÖK˜s°SÀ
Ò5J’sÐK€OÐKÙ”3�|Ó$ qÒ(¬C°Ó,@À1Ò,DØ=¸e¿o¹o×>VÑ>VÐ=WÐXˆÜˆ|Ó˜qÒ Ø"Ÿx™x¸<Ö(H°a¨1¨Q¨C¨qªÒ(HÓIÐØÐ1Ð2BÐ1CÀ1ÐEÑEˆMÜˆÓ !Ò#Ø"%§(¡(¸oÖ+N¸¨a°¨s°!ªHÒ+NÓ"OÐØÐ1Ð2EÐ1FÀaÐHÑHˆMÜ˜=Ó)Ð)á(�^¬g´e·j±jÈuÐcgÔ.h€Fà!ò ˆ
ÙœBŸG™GŸL™L¨°Ó<Ó=ˆ
Ø×Ñ˜j°ÐÔ7ð Ü
�
‰
�ðô �8‰8×Ñ×"Ñ"×4Ñ4°\À?ÓSÐS÷?ñ üò IùÚKùò )Iùò ,Os0   ÄLÆ	L$ÆL$Æ$	L)Æ.L)ÈL.ÉL3ÌL!)ÚBOOLÚU8ÚI8ÚI16ÚF16ÚBF16ÚI32ÚF32ÚF64ÚI64ÚF8_E4M3ú2.1.0rC  z2.3.0ÚU16ÚU32ÚU64ÚF8_E5M2Úcheckpoint_fileÚis_quantizedr	  r
  c           	      ó   — | j                  d«      rÿt        «       rõt        | d¬«      5 }|j                  «       }|�"|j	                  d«      dvrt        d| › d�«      ‚i }|j                  «       D ]“  }|j                  |«      j                  «       }|t        v r
t        |   }	nt        d	|› �«      ‚|d
k(  r9t        j                  |j                  |«      j                  «       |	d
¬«      ||<   Œ€|j                  |«      ||<   Œ• |cddd«       S 	 |€dt        «       r?t        j                   j#                  «       r!t        j                   j%                  «       dkD  st'        «       rt)        «       s|sd
}nd}i }
t+        | t,        «      rM|d
k7  rHt/        j0                  t        j2                  «      t/        j0                  d«      k\  rt5        | «      rddi}
t        j6                  | f||dœ|
¤ŽS # 1 sw Y   ŒèxY w# t8        $ rx}	 t;        | «      5 }|j=                  d«      dk(  rt        d«      ‚t        d| › d�«      |‚# 1 sw Y   nxY wn%# t>        t        f$ r t        d| › d| › d�«      ‚w xY wY d}~yd}~ww xY w)zg
    Reads a `safetensor` or a `.bin` checkpoint file. We load the checkpoint on "cpu" by default.
    ú.safetensorsÚpt©Ú	frameworkNÚformat)rM  ÚtfÚflaxÚmlxz"The safetensors archive passed at zf does not contain the valid metadata. Make sure you save your model with the `save_pretrained` method.z)Cannot load safetensors of unknown dtype Úmeta)Úsizerå   rÙ   r   r  rD  ÚmmapTr  é   r   z¬You seem to have cloned a repository without having git-lfs installed. Please install git-lfs and run `git lfs install` followed by `git lfs pull` in the folder you cloned.zUnable to locate the file z_ which is necessary to load this pretrained model. Make sure you have saved the model properly.z9Unable to load weights from pytorch checkpoint file for 'z' at 'zZ'. If you tried to load a PyTorch model from a TF 2.0 checkpoint, please set from_tf=True.) ÚendswithrY   r{   Úmetadatarˆ   ÚOSErrorr  Ú	get_sliceÚ	get_dtypeÚstr_to_torch_dtyperö   r‚   ÚemptyÚ	get_shapeÚ
get_tensorr)   rƒ   r…   Úget_rankrŒ   r‘   Ú
isinstancerÜ   r   Úparser“   r   r  Ú	Exceptionr  ÚreadÚUnicodeDecodeError)rI  rJ  r	  r
  r+  rY  rø   rÏ   Úk_dtyperå   Ú
extra_argsÚes               r‹   r  r  ÷  sµ  € ð ×Ñ Ô/Ô4LÔ4NÜ�°$Ô7ð 	¸1Ø—z‘z“|ˆHàÐ#¨¯©°XÓ(>ÐFaÑ(aÜØ8¸Ð8Ið JMð Móð ð ˆJØ—V‘V“Xò 	4�ØŸ+™+ a›.×2Ñ2Ó4�ØÔ0Ñ0Ü.¨wÑ7‘Eä$Ð'PÐQXÐPYÐ%ZÓ[Ð[Ø 6Ò)Ü$)§K¡K°Q·[±[À³^×5MÑ5MÓ5OÐW\ÐekÔ$l�J˜q’Mà$%§L¡L°£O�J˜q’Mð	4ð ÷'	ñ 	ð*/ØÐô /Ô0Ü×)Ñ)×8Ñ8Ô:Ü×)Ñ)×2Ñ2Ó4°qÒ8ä#Ô%Ô.BÔ.DÙ"Ø%‘à$�Øˆ
ô �¬Ô,Ø Ò&Ü—‘œe×/Ñ/Ó0´G·M±MÀ'Ó4JÒJÜ˜?Ô+à  $˜ˆJÜ�z‰zØð
à%Ø%ñ
ð ñ	
ð 	
÷W	ð 	ûôb ò ð	Ü�oÓ&ð ¨!Ø—6‘6˜!“9 	Ò)Ü!ð&óð ô %Ø4°_Ð4Eð FNð Nóð ð÷ð úð øô #¤JÐ/ò 	ÜØKÈOÐK\ð ]Ø&Ð'ð (jðjóð ð	úôûðúsI   ©CG0ÄCG< Ç0G9Ç<	I=ÈIÈ0IÉI
	ÉIÉI8É"I0É0I8É8I=c                 ó  — t        |«      }i }| j                  «       D ]d  \  }}|dk(  rt        |j                  «       «      }n"|j                  «       D �ch c]	  }|› d|› �’Œ }}|j                  |«      rd|_        Œ`|||<   Œf |S c c}w )z†
    Sets the `_is_hf_initialized` flag in all submodules of a given model when all its weights are in the loaded state
    dict.
    Ú rþ   T)r  Únamed_modulesrø   ÚissubsetÚ_is_hf_initialized)r!  Ústate_dict_keysÚnot_initialized_submodulesÚmodule_namerÈ   Úmodule_keysrÏ   s          r‹   Úset_initialized_submodulesrs  H  s¢   € ô
 ˜/Ó*€OØ!#ÐØ$×2Ñ2Ó4ò 	=Ñˆ�VØ˜"Òä˜f×/Ñ/Ó1Ó2‰Kà9?×9JÑ9JÓ9LÖM°A˜k˜]¨!¨A¨3Ò/ÐMˆKÐMØ×Ñ Ô0Ø(,ˆFÕ%à6<Ð& {Ò3ð	=ð &Ð%ùò Ns   ÁBÚtensorrÉ   c                 ó°   — | j                  «       r5| j                  d«      d   j                  «       | j                  «       z   }|S | j                  «       }|S )Nr�   )ÚnelementÚviewÚdata_ptrÚelement_size)rt  Ústops     r‹   Ú_end_ptrr{  \  sO   € à‡�ÔØ�{‰{˜2‹˜rÑ"×+Ñ+Ó-°×0CÑ0CÓ0EÑEˆð €Kð �‰Ó ˆØ€KrŠ   rÈ   c                 óœ  — g }t        | dd «      �3| j                  D �cg c]  }|r|› d|› �n|‘Œ }}|j                  |«       t        | dd «      �3| j                  D �cg c]  }|r|› d|› �n|‘Œ }}|j                  |«       | j	                  «       D ],  \  }}|r|› d|› �n|}|j                  t        ||¬«      «       Œ. |S c c}w c c}w )NÚ_tied_weights_keysrþ   Ú_dynamic_tied_weights_keys)Úprefix)Úgetattrr}  Úextendr~  Únamed_childrenÚ_get_tied_weight_keys)rÈ   r  Útied_weight_keysrÏ   Únamesr²   Ú	submoduleÚlocal_prefixs           r‹   rƒ  rƒ  e  sô   € ØÐÜˆvÐ+¨TÓ2Ð>Ø;A×;TÑ;TÖU°a¡F�F�8˜1˜Q˜C‘°Ñ1ÐUˆÐUØ×Ñ Ô&ÜˆvÐ3°TÓ:ÐFØ;A×;\Ñ;\Ö]°a¡F�F�8˜1˜Q˜C‘°Ñ1Ð]ˆÐ]Ø×Ñ Ô&Ø!×0Ñ0Ó2ò W‰ˆˆiÙ-3˜&˜  4 &Ñ)¸ˆØ×ÑÔ 5°iÈÔ UÕVðWð Ðùò Vùò ^s   žCÁC	Útensorsrø   c                 ó0  — g }| D ]Â  }t        |«      dk  r|j                  |«       Œ#g }|D ]2  }||   }|j                  |j                  «       t        |«      |f«       Œ4 |j	                  «        |d   \  }}}	|j                  |	h«       |dd  D ]4  \  }
}}|
|k\  r|j                  |h«       n|d   j                  |«       |}Œ6 ŒÄ g }g }|D ]A  } t        | «      dk(  r |j                  | j                  «       «       Œ1|j                  | «       ŒC ||fS )Né   r   r    r�   )r  Úappendrx  r{  ÚsortÚaddÚpop)rˆ  rø   Úfiltered_tensorsÚsharedÚareasr²   rt  Ú_Ú	last_stopÚ	last_nameÚstartrz  Údisjoint_tensorsÚshared_tensorss                 r‹   Ú_find_disjointr˜  s  sA  € ØÐØò ˆÜˆv‹;˜Š?Ø×#Ñ# FÔ+ØàˆØò 	FˆDØ Ñ%ˆFØ�L‰L˜&Ÿ/™/Ó+¬X°fÓ-=¸tÐDÕEð	Fð 	�
‰
Œà"'¨¡(Ñˆˆ9�iØ×Ñ  Ô,Ø!& q r ò 	ÑˆE�4˜Ø˜	Ò!Ø ×'Ñ'¨¨Õ/à  Ñ$×(Ñ(¨Ô.Ø‰Iñ	ðð& ÐØ€NØ#ò +ˆÜˆw‹<˜1ÒØ×#Ñ# G§K¡K£MÕ2à×!Ñ! 'Õ*ð	+ð
 Ð+Ð+Ð+rŠ   c                 ó^  — g }g }| D ]¡  }t        |«      dk  rŒt        j                  t        «      }|D ]A  }||   }|j                  |j                  «       t        |«      f}||   j                  |«       ŒC t        |«      dk(  r|j                  |«       Œ‘|j                  |«       Œ£ ||fS )NrŠ  r    )	r  Úcollectionsr   r  rÙ   rx  r{  r�  r‹  )	rˆ  rø   r—  Ú	identicalr�  r‘  r²   rt  Úareas	            r‹   Ú_find_identicalr�  ’  s´   € Ø€NØ€IØò *ˆÜˆv‹;˜Š?Øä×'Ñ'¬Ó,ˆØò 	"ˆDØ Ñ%ˆFØ—M‘M 6§?¡?Ó#4´h¸vÓ6FÐGˆDØ�$‰K�O‰O˜DÕ!ð	"ô ˆu‹:˜Š?Ø×Ñ˜VÕ$à×!Ñ! &Õ)ð*ð ˜9Ð$Ð$rŠ   r!  Ú
param_nameÚempty_paramÚkeep_in_fp32_regexÚhf_quantizerc                 ó   — 	 | j                  |«      }t        t        d«      }d }|xr |j                  t        j                  k(  }	|j                  j                  rK|	sI|�"|j                  |«      rt        j                  }n%|�| j                  j                  }n|j                  }|d uxr |j                  «       |fS # t        $ r5}|�,|j                  j                  t        j
                  k(  rY d }~y|‚d }~ww xY w)N)TNÚfloat8_e4m3fn)Úget_parameter_or_bufferrd  Úquantization_configÚquant_methodrj   ÚHQQÚhasattrr‚   rå   r£  ré   Úsearchrí   ÚconfigÚ_pre_quantization_dtypeÚis_contiguous)
r!  rž  rŸ  r   r¡  Ú	old_paramri  Úis_torch_e4m3fn_availableÚcasting_dtypeÚis_param_float8_e4m3fns
             r‹   Ú_infer_parameter_dtyper±  ¥  sï   € ðØ×1Ñ1°*Ó=ˆ	ô !(¬¨Ó ?Ðð €MØ6Òc¸;×;LÑ;LÔPU×PcÑPcÑ;cÐØ×Ñ×*Ò*Ñ3IàÐ)Ð.@×.GÑ.GÈ
Ô.SÜ!ŸM™M‰MàÐ%Ø!ŸL™L×@Ñ@‰Mà%ŸO™OˆMØ˜DÐ Ò> Y×%<Ñ%<Ó%>ÀÐMÐMøô' ò ØÐ#¨×(HÑ(H×(UÑ(UÔYk×YoÑYoÒ(oÜàˆGûð	ús   ‚B? Â?	C=Ã)C8Ã6C8Ã8C=c                 óN   — t        | |«      \  }}|j                  ||idd¬«       y)zKCast a single parameter `param_name` into the `model`, with value `tensor`.FT)r  ÚassignN)r=   r  )r!  rž  rt  rÈ   Ú
param_types        r‹   Ú_load_parameter_into_modelrµ  Ä  s-   € ä-¨e°ZÓ@Ñ€FˆJà
×Ñ˜J¨Ð/¸ÀdÐÕKrŠ   r7  Úexpected_keysÚreverse_renaming_mappingÚ
device_mapÚdisk_offload_folderÚdisk_offload_indexÚcpu_offload_folderÚcpu_offload_indexÚis_safetensorsr2  Údevice_meshú(torch.distributed.device_mesh.DeviceMeshc                 ó  — d}|�_|j                  dd«      �M|d   dt        j                  d«      fvr1t        |d   t        j                  «      r|d   j                  n|d   }|�Kdj                  t        |j                  «       d¬«      D �cg c]  }t        j                  |«      ‘Œ c}«      }|
du}|xr6 |
j                  j                  t        j                  t        j                  fv }|j                  d«      xr | }d}|rt!        |d|¬	«      }|j#                  «       D �]@  \  }}||vrŒ|r||   }|j%                  |«      }n|j'                  |«      }t)        | ||||
«      \  }}|�-t+        | |||||t-        t.        j0                  d
   «      |«       Œw|d   }|�|j'                  |«      }|r|j3                  «       }|€d}n9t        j4                  |«      }|st7        |› d�«      ‚||j9                  «          }|dk(  r|rŒçt;        ||||«      }Œö|dk(  r|	�t;        ||||	«      }	�Œ|r#|
j<                  r|
j?                  | |||||¬«      s6tA        «       rtC        «       rdnd}tE        | ||j'                  |«      «       �Œh|
jG                  | |||||«       tA        «       stI        «       s�Œ”tK        | |«      \  }}tM        ||«      } d}!tA        «       rtC        «       sd}!i }"tO        |d«      r(|jP                  jR                  jT                  dk(  rd|"d<    tW        | «      | jX                  j'                  |!«      fi |"¤| jZ                  ¤Ž} t]        ||| «       �ŒC |�|j_                  ddd«       ||	fS c c}w )a¤  Load parameters from `meta_state_dict` into the model. The parameters of the `meta_state_dict` are on the meta
    device in order to easily infer the shapes and dtypes that they will have. Then proper parameters are then loaded
    from `shard_file`, which is the actual state dict file on disk.
    This function takes care of correctly casting dtypes, devices, and sharding tensors in case of tensor parallelism.
    r  Nrk  ú|T)ÚreverserL  rM  )rO  rÙ   ÚRANK.z doesn't have any device set.Údisk)Úparam_devicer¸  rT  ÚweightÚ
Int8ParamsFÚrequires_grad)0rˆ   r‚   rÙ   rb  r,  r  Úsortedr  ÚreÚescaper¥  r¦  rj   r§  ÚBITS_AND_BYTESrX  r{   r®   r[  Útor±  r2   r�   r†   r‡   Ú
contiguousr©  rö   Úgrouprv   Ú requires_parameters_quantizationÚcheck_quantized_paramrŒ   r‘   rµ  Úcreate_quantized_paramr)   r=   r€  r¨  rÆ  r  r  ÚtypeÚdatarÌ   r¯   Ú__exit__)#r!  rø   r7  r¶  r·  r¸  r¹  rº  r»  r¼  r¡  r½  r   r2  r¾  Útensor_devicerÏ   Údevice_map_regexrJ  Úis_hqq_or_bnbÚis_meta_state_dictÚfile_pointerrž  rŸ  Úserialized_param_nameÚparamÚto_contiguousr¯  rÅ  Úmodule_layerrÈ   r´  ÚvalueÚparam_toÚ
val_kwargss#                                      r‹   Ú _load_state_dict_into_meta_modelrâ  Ë  s¼  € ð. €MØÐ *§.¡.°°TÓ":Ð"FØ�b‰> %¬¯©°eÓ)<Ð!=Ñ=Ü4>¸zÈ"¹~ÌuÏ|É|Ô4\˜J r™N×0Ò0ÐblÐmoÑbpˆMØÐØŸ8™8¼6À*Ç/Á/ÓBSÐ]aÔ;bÖ$c°a¤R§Y¡Y¨q¥\Ò$cÓdÐà tÐ+€LØ ò  \×%EÑ%E×%RÑ%RÜ×ÑÜ×)Ñ)ðWð &€Mð $×,Ñ,¨^Ó<ÒRÀ]ÐARÐØ€LÙÜ  °tÀMÔRˆà#-×#3Ñ#3Ó#5ó X7Ñˆ
�KØ˜]Ñ*Øñ à$<¸ZÑ$HÐ!Ø ×*Ñ*Ð+@ÓA‰Eà—N‘N =Ó1ˆEä'=ØØØØØó(
Ñ$ˆ�}ð Ð"Ü'ØØØØØØÜ”B—J‘J˜vÑ&Ó'Øõ	ð ˜#‘JˆEØÐ(ØŸ™ Ó/�ÙØ×(Ñ(Ó*�àÐ!Ø$‘ä!Ÿy™yÐ)9¸:ÓF�Ù#Ü$¨
 |Ð3PÐ%QÓRÐRà#-¨l×.@Ñ.@Ó.BÑ#C�Là˜vÒ%Ú%Ü)7¸¸zÐK^Ð`rÓ)sÑ&Ø Ò&Ð+<Ð+HÜ$2°5¸*ÐFXÐZkÓ$lÒ!á Ø$×EÒEà$×:Ñ:ØØØ"Ø"Ø%1Ø#-ð ;ô ô #Ô$Ü,@Ô,B¡5È�Lä*¨5°*¸e¿h¹hÀ|Ó>TÖUð ×3Ñ3Ø˜5 *¨l¸JÈôô #Ô$Ô(BÖ(DÜ)=¸eÀZÓ)PÑ&�F˜JÜ# F¨JÓ7�EØ$�HÜ&Ô(Ô1EÔ1GØ#)˜Ø!#�JÜ˜v xÔ0°V·]±]×5LÑ5L×5UÑ5UÐYeÒ5eØ6;˜
 ?Ñ3Ø'œD ›K¨¯
©
¯©°hÓ(?Ñ`À:Ð`ÐQV×Q_ÑQ_Ñ`�EÜ˜F J°Ö6ðqX7ðt ÐØ×Ñ˜d D¨$Ô/àÐ0Ð0Ð0ùòS %ds   ÂNÚweights_nameÚvariantc                 óH   — |�| j                  dd«      \  }}|› d|› d|› �} | S )Nrþ   r    )Úrsplit)rã  rä  r  r²   s       r‹   Ú_add_variantrç  S  s:   € ØÐØ!×(Ñ(¨¨aÓ0‰
ˆˆdØ˜˜q  	¨¨4¨&Ð1ˆØÐrŠ   Úpretrained_model_name_or_pathÚ	subfolderÚ	gguf_fileÚfrom_tfÚ	from_flaxÚuse_safetensorsÚ	cache_dirÚforce_downloadÚproxiesÚlocal_files_onlyÚtokenÚ
user_agentÚrevisionÚcommit_hashc                 ó�  — d}| ��„|�€�t        | «      } t        j                  j                  | «      }|�rÃ|rot        j                  j	                  t        j                  j                  | |t        dz   «      «      r*t        j                  j                  | |t        dz   «      }�nª|rit        j                  j	                  t        j                  j                  | |t        «      «      r't        j                  j                  | |t        «      }�n?|rit        j                  j	                  t        j                  j                  | |t        «      «      r't        j                  j                  | |t        «      }�nÔ|dur}t        j                  j	                  t        j                  j                  | |t        t        |«      «      «      r1t        j                  j                  | |t        t        |«      «      }�nS|durt        j                  j	                  t        j                  j                  | |t        t        |«      «      «      r3t        j                  j                  | |t        t        |«      «      }d}�nÐ|s}t        j                  j	                  t        j                  j                  | |t        t        |«      «      «      r1t        j                  j                  | |t        t        |«      «      }�nQ|st        j                  j	                  t        j                  j                  | |t        t        |«      «      «      r3t        j                  j                  | |t        t        |«      «      }d}�nÐ|s§t        j                  j	                  t        j                  j                  | |t        dz   «      «      sBt        j                  j	                  t        j                  j                  | |t        «      «      r t        dt        t        |«      › d| › d�«      ‚|sbt        j                  j	                  t        j                  j                  | |t        «      «      r t        dt        t        |«      › d| › d�«      ‚|r t        dt        t        |«      › d| › d	�«      ‚t        dt        t        |«      › d
t        t        |«      › d
t        › d
t        dz   › dt        › d| › d	�«      ‚t        j                  j	                  t        j                  j                  || «      «      r| }d}�nt        j                  j	                  t        j                  j                  || dz   «      «      r;|st        d| dz   › d�«      ‚t        j                  j                  || dz   «      }d}�nšt!        | «      r| }t#        | «      }�n€|rt        }n.|rt        }n%|durt        t        |«      }nt        t        |«      }	 |||	|
||||dd|dœ}t%        | |fi |¤Ž}|€ž|t        t        |«      k(  r‹t%        | t        t        |«      fi |¤Ž}|�d}nk|rL|dk(  rt'        | fi |¤Ž\  }}}||d<   |€Mt        | › dt        t        |«      › dt        t        |«      › d�«      ‚t        t        |«      }t%        | |fi |¤Ž}|€2|t        t        |«      k(  rt%        | t        t        |«      fi |¤Ž}|�d}|
�sVt)        «       �sK|�g|t        t        fv �r:|rt        nt        }||	|||
dœ}|||
||dd|dœ|¥}t+        | |fi |¤Ž�s	t-        t&        | fddi|¥d¬«      j/                  «        nâ||	|||
dœ}t+        | t        fi |¤Žrt        | › dt        t        |«      › d�«      ‚t+        | t        fi |¤Žrt        | › dt        t        |«      › d�«      ‚|�3t+        | t        fi |¤Žr"t        | › dt        t        |«      › d|› d�«      ‚t        | › dt        t        |«      › d
t        t        |«      › d
t        › d
t        › dt        › d	�«      ‚|rt2        j5                  d› �«       |}n[t2        j5                  d› d› �«       n?|r=t        j                  j	                  |«      r|}n|||	|
||||dd|dœ}t%        | |fi |¤Ž}d}|rt7        | |||	|
|||||¬«      \  }}||fS | �gnd}||fS # t        $ r ‚ t0        $ r>}t        d| › d| › dt        t        |«      › d
t        › d
t        › dt        › d	�«      |‚d}~ww xY w) z¿Get all the checkpoint filenames based on `pretrained_model_name_or_path`, and optional metadata if the
    checkpoints are sharded.
    This function will download the data if necesary.
    FNú.indexTzError no file named z found in directory zf but there is a file for TensorFlow weights. Use `from_tf=True` to load this model from those weights.zb but there is a file for Flax weights. Use `from_flax=True` to load this model from those weights.rþ   z, rý   z$We found a TensorFlow checkpoint at z:, please set from_tf to True to load from this checkpoint.)rî  rï  rð  rñ  rò  ró  rô  ré  Ú _raise_exceptions_for_gated_repoÚ%_raise_exceptions_for_missing_entriesÚ_commit_hashÚmainrô  z& does not appear to have a file named z¢ and thus cannot be loaded with `safetensors`. Please make sure that the model has been saved with `safe_serialization=True` or do not set `use_safetensors=True`.)rô  rð  rò  rî  rñ  )rî  rï  rñ  ró  ré  rø  rù  rú  Úignore_errors_during_conversionzThread-auto_conversion)Útargetr©   rª   r²   z) but there is a file without the variant z;. Use `variant=None` to load this model from those weights.zCan't load the model for 'zœ'. If you were trying to load it from 'https://huggingface.co/models', make sure you don't have a local directory with the same name. Otherwise, make sure 'z=' is the correct path to a directory containing a file named zloading weights file z from cache at )	rî  rï  rð  rñ  rò  ró  rô  ré  rú  )rÜ   r†   r  Úisdirr  r  rG   rF   rC   rç  rE   rD   rI   rH   ÚEnvironmentErrorrö   rX   rO   rM   r>   rU   rQ   r
   r•  rd  r  Úinford   )rè  ré  rä  rê  rë  rì  rí  rî  rï  rð  rñ  rò  ró  rô  rõ  Ú
is_shardedÚis_localÚarchive_fileÚfilenameÚresolved_archive_fileÚcached_file_kwargsÚsafe_weights_nameÚhas_file_kwargsri  Úsharded_metadataÚcheckpoint_filess                             r‹   Ú_get_resolved_checkpoint_filesr  Z  sâ	  € ð* €Jà$Ñ0°YÑ5FÜ(+Ð,IÓ(JÐ%Ü—7‘7—=‘=Ð!>Ó?ˆÚÙœ2Ÿ7™7Ÿ>™>Ü—‘—‘Ð:¸IÄÐYaÑGaÓbôô  "Ÿw™wŸ|™|Ð,IÈ9ÔVeÐhpÑVpÓq’ÙœRŸW™WŸ^™^¬B¯G©G¯L©LÐ9VÐXaÔcsÓ,tÔuä!Ÿw™wŸ|™|Ð,IÈ9ÔVfÓg’ÙœrŸw™wŸ~™~Ü—‘—‘Ð:¸IÔGXÓYô ô  "Ÿw™wŸ|™|Ð,IÈ9ÔVgÓh’Ø ¨Ñ-´"·'±'·.±.Ü—‘—‘Ð:¸IÄ|ÔTeÐgnÓGoÓpô3ô  "Ÿw™wŸ|™|Ø1°9¼lÔK\Ð^eÓ>fó ’ð !¨Ñ-´"·'±'·.±.Ü—‘—‘Ð:¸IÄ|ÔTkÐmtÓGuÓvô3ô  "Ÿw™wŸ|™|Ø1°9¼lÔKbÐdkÓ>ló �ð "’
Ù$¬¯©¯©Ü—‘—‘Ð:¸IÄ|ÔT`ÐbiÓGjÓkô*ô  "Ÿw™wŸ|™|Ø1°9¼lÌ<ÐY`Ó>aó ’ñ %¬¯©¯©Ü—‘—‘Ð:¸IÄ|ÔTfÐhoÓGpÓqô*ô  "Ÿw™wŸ|™|Ø1°9¼lÔK]Ð_fÓ>gó �ð "’
á$Ü—‘—‘œrŸw™wŸ|™|Ð,IÈ9ÔVeÐhpÑVpÓqÔrÜ—7‘7—>‘>¤"§'¡'§,¡,Ð/LÈiÔYiÓ"jÔkä&Ø*¬<¼ÀgÓ+NÐ*Oð PØ5Ð6ð 7MðMóð ñ
 %¬¯©¯©Ü—‘—‘Ð:¸IÔGXÓYô*ô 'Ø*¬<¼ÀgÓ+NÐ*Oð PØ5Ð6ð 7>ð>óð ñ
 !Ü&Ø*¬<Ô8IÈ7Ó+SÐ*Tð UØ5Ð6°að9óð ô
 'Ø*¬<¼ÀgÓ+NÐ*OÈrÔR^Ô_pÐryÓRzÐQ{ð |Ü(Ð)¨¬O¸hÑ,FÐ+GÀtÔL]ÐK^ð _Ø5Ð6°að9óð ô
 �W‰W�^‰^œBŸG™GŸL™L¨Ð4QÓRÔSØ8ˆLØŠHÜ�W‰W�^‰^œBŸG™GŸL™L¨Ð4QÐT\Ñ4\Ó]Ô^ÙÜ Ø:Ð;XÐ[cÑ;cÐ:dð eDð Dóð ô Ÿ7™7Ÿ<™<¨	Ð3PÐS[Ñ3[Ó\ˆLØŠHÜÐ8Ô9Ø4ˆHÜ$0Ð1NÓ$OÒ!ñ Ü+‘ÙÜ,‘Ø ¨Ñ-Ü'Ô(9¸7ÓC‘ä'¬°gÓ>�ðHð "+Ø&4Ø&Ø(8Ø"Ø",Ø (Ø!*Ø8=Ø=BØ$/ñ&Ð"ô )4Ð4QÐS[Ñ(rÐ_qÑ(rÐ%ð )Ð0°XÄÔN_ÐahÓAiÒ5iä,7Ø5Ü$Ô%<¸gÓFñ-ð -ñ-Ð)ð
 -Ð8Ø%)™
Ù(Ø# vÒ-ÜJYØ =ñKØASñKÑGÐ1°8¸Zð :BÐ*¨:Ñ6Ø0Ð8Ü"2Ø#@Ð"Að B$Ü$0Ô1BÀGÓ$LÐ#MÈTÔR^Ô_vÐxó  SAð  RBð Brð!ró#ð ô $0´¸gÓ#F˜Ü0;Ø9¸8ñ1ØGYñ1Ð-ð )Ð0°XÄÌlÐ\cÓAdÒ5dä,7Ø5Ü$Ô%7¸ÓAñ-ð -ñ-Ð)ð
 -Ð8Ø%)˜
Ú'´Õ0AØ,Ð8Ø#¬Ô6HÐ'IÒIñ LVÕ0GÔ[lÐ-à,4Ø+2Ø).Ø-6Ø4Dñ/˜Oð .7Ø2@Ø4DØ.8Ø-6ØDIØINØ0;ñ
2ð #2ð
2Ð.ô $,Ð,IÐK\Ñ#pÐ`oÓ#pÜ &Ü+:Ø*GÐ)IØ,MÈtÐ+jÐWiÐ+jØ)Aô	!"÷
 #(¡%¥'ð
 )1Ø'.Ø%*Ø)2Ø0@ñ+˜ô $Ð$AÔCSÑgÐWfÒgÜ"2Ø#@Ð"Að B$Ü$0´¸wÓ$GÐ#Hð I]ð!]ó#ð ô
 &Ð&CÔEVÑjÐZiÒjÜ"2Ø#@Ð"Að B$Ü$0´¸wÓ$GÐ#Hð I[ð![ó#ð ð
 %Ð0´XØ9¼<ñ6ØKZò6ô #3Ø#@Ð"Að B$Ü$0´¸wÓ$GÐ#Hð I$Ø$+ 9Ð,gð!ió#ð ô #3Ø#@Ð"Að B$Ü$0´¸wÓ$GÐ#HÈÌ<ÔXiÐkrÓKsÐJtð u$Ü$4Ð#5°R¼Ð7HÈÔM^ÐL_Ð_`ð!bó#ð ñ( Ü�K‰KÐ/°¨~Ð>Ô?Ø$0Ñ!ä�K‰KÐ/°¨z¸ÐI^ÐH_Ð`Õaá	ä�7‰7�>‰>˜)Ô$Ø$-Ñ!ð
 'Ø"0Ø"Ø$4ØØ(Ø$Ø&Ø49Ø9>Ø +ñ"Ðô %0Ð0MÈyÑ$oÐ\nÑ$oÐ!ð ÐÙÜ-GØ)Ø!ØØ)ØØ-ØØ!ØØØ$ô.
Ñ*ÐÐ*ð  Ð-Ð-Ð-ð 7TÐ6_Ð1Ñ2ÐeiÐàÐ-Ð-Ð-øô} $ò ð Üò ä&Ø0Ð1NÐ0Oð P9à9VÐ8Wð X:Ü:FÄ|ÐU\Ó:]Ð9^ð _Ü(Ð)¨¬OÐ+<¸DÔARÐ@SÐSTð	Vóð ðûðús   Ø	Ic5 ã5eä9e å eÚtorch_dtyper
  rª  r	  c                 ó  — d}|du}|��(t        |t        «      rí|dk(  rŒt        |d«      r3|j                  �'|j                  }t        j                  d|› d�«       �nÂ|r
d|v r|d   }n*|�t        |«      }nt        |d   d|¬	«      }t        |«      }t        j                  d
«       �nut        t        |«      �rdt        t        |«      }||_        |j                  j                  «       D ]  }	t        ||	«      }
||
_        Œ �nt        |t        j                  «      r:||_        |j                  j                  «       D ]  }	t        ||	«      }
||
_        Œ nÅt        |t        «      r§|j                  «       D ]G  \  }}t        ||«      sŒt        ||«      }t        |t        «      s|nt        t        |«      }||_        ŒI |j                  d«      }t        |t        «      s|nt        t        |«      }||_        |€t        j                   }nt#        d|› �«      ‚| j%                  |«      }nMt        j&                  «       }||_        |j                  j                  «       D ]  }t        ||«      }||_        Œ |||fS )aô  Find the correct `torch_dtype` to use based on provided arguments. Also update the `config` based on the
    inferred dtype. We do the following:
    1. If torch_dtype is not None, we use that dtype
    2. If torch_dtype is "auto", we auto-detect dtype from the loaded state_dict, by checking its first
        weights entry that is of a floating type - we assume all floating dtype weights are of the same dtype
    we also may have config.torch_dtype available, but we won't rely on it till v5
    NÚautor  zWill use torch_dtype=z$ as defined in model's config objectrå   r   rT  r  z�Since the `torch_dtype` attribute can't be found in model's config object, will use torch_dtype={torch_dtype} as derived from model's weightsrk  z´`torch_dtype` can be one of: `torch.dtype`, `'auto'`, a string of a valid `torch.dtype` or a `dict` with valid `torch_dtype` for each sub-config in composite configs, but received )rb  rÜ   r¨  r  r  r   rû   r  r‚   r€  Úsub_configsr  rå   Údictr®   rˆ   rí   rö   Ú_set_default_torch_dtyper¿   )Úclsr  r
  rª  r	  rø   r
  Ú
dtype_origr  Úsub_config_keyÚ
sub_configr0  Ú
curr_dtyperß  Údefault_dtypes                  r‹   Ú_get_torch_dtyper  “  s�  € ð  €JØ!¨Ð-€JàÑÜ�k¤3Ô'Ø˜fÒ$Ü˜6 =Ô1°f×6HÑ6HÐ6TØ"(×"4Ñ"4�KÜ—K‘KÐ"7¸°}ÐDhÐ iÖjá! gÐ1AÑ&AØ&6°wÑ&?™Ø#Ð/Ü&:¸:Ó&F™ä%4Ø,¨QÑ/¸fÐS_ô&˜
ô ';¸:Ó&F˜Ü—K‘Kð]öô œ Õ,Ü%¤e¨[Ó9�Ø%0�Ô"Ø&,×&8Ñ&8×&=Ñ&=Ó&?ò 9�NÜ!(¨°Ó!@�JØ-8�JÕ*ò9ô ˜¤U§[¡[Ô1Ø!,ˆFÔØ"(×"4Ñ"4×"9Ñ"9Ó";ò 5�Ü$ V¨^Ó<�
Ø)4�
Õ&ñ5ô ˜¤TÔ*Ø#.×#4Ñ#4Ó#6ò 3‘��ZÜ˜6 3Õ'Ü# F¨CÓ0�EÜ3=¸jÌ#Ô3N¡ÔT[Ô\aÐcmÓTn�JØ(2�EÕ%ð	3ð &Ÿ/™/¨"Ó-ˆKÜ-7¸ÄSÔ-I™+ÌwÔW\Ð^iÓOjˆKØ!,ˆFÔØÐ"Ü#Ÿm™m‘äðJØJUÈðXóð ð
 ×1Ñ1°+Ó>‰
ô ×/Ñ/Ó1ˆØ*ˆÔØ×%Ñ%×*Ñ*Ó,ò 	.ˆCÜ˜F CÓ(ˆEØ -ˆEÕð	.ð �; 
Ð*Ð*rŠ   Ú
max_memoryc           	      ó
  — t        |t        «      �rRi }|�!|j                  |j                  | |«      «       |�S|j                  | j	                  «       D ��ci c](  \  }}|j                  |«      sŒ|t        j                  “Œ* c}}«       |}	|�|j                  |	«      }	| j                  |«      }
d|
i}dt        j                  t        «      j                  v r||d<   n#t        |«      dkD  rt        j!                  d«       |dk7  rt#        | f|	|dk(  |dœ|¤Ž}nt%        |«      }|�|j'                  |«      }||d<   t        | fd	|	i|¤Ž}|�|j)                  |¬
«       |S |�t+        | «      }t-        ||«       |S c c}}w )zÂCompute the final `device_map` to use if we passed a value in ['auto', 'balanced', 'balanced_low_0', 'sequential'].
    Otherwise, we check for any device inconsistencies in the device_map.
    Úno_split_module_classesÚspecial_dtypesr   z¦This model has some weights that should be kept in higher precision, you need to upgrade `accelerate` to properly deal with them (`pip install --upgrade accelerate`).Ú
sequentialÚbalanced_low_0)rå   Úlow_zeror  r  rå   )r¸  )rb  rÜ   ÚupdateÚget_special_dtypes_updateÚnamed_parametersr©  r‚   rí   Úadjust_target_dtypeÚ_get_no_split_modulesÚinspectÚ	signaturero   rØ   r  r  r  rs   rt   Úadjust_max_memoryÚvalidate_environmentr*   rq   )r!  r¸  r  r¡  r  r   r  r²   r’  Útarget_dtypeÚno_split_modulesÚdevice_map_kwargsÚtied_paramss                r‹   Ú_get_device_mapr-  ã  sº  € ô �*œcÕ"ØˆØÐ#Ø×!Ñ! ,×"HÑ"HÈÐP[Ó"\Ô]ØÐ)Ø×!Ñ!Ø49×4JÑ4JÓ4L×p©¨¨qÐPb×PiÑPiÐjnÕPo�”u—}‘}Ñ$Ópôð #ˆàÐ#Ø'×;Ñ;¸LÓIˆLà ×6Ñ6°zÓBÐØ6Ð8HÐIÐàœw×0Ñ0Ô1FÓG×RÑRÑRØ2@ÐÐ.Ò/Ü�Ó  1Ò$Ü�N‰Nð`ôð
 ˜Ò%Ü,Øðà"Ø$Ð(8Ñ8Ø%ñ	ð
 $ñ‰Jô (¨
Ó3ˆJØÐ#Ø%×7Ñ7¸
ÓCˆJØ*4Ð˜,Ñ'ä*¨5ÑZ¸ÐZÐHYÑZˆ
àÐ#Ø×-Ñ-¸Ð-ÔDð Ðð 
Ð	Ü*¨5Ó1ˆä,¨[¸*ÔEàÐùóW qs   ÁE?
Á1E?
Úoriginal_checkpoint_keysÚcheckpoint_keysÚ'loading_base_model_from_task_state_dictc                 óŒ  — |j                   }t        |j                  «       j                  «       «      }|�|j	                  |||«      }t        t        |«      t        |«      z
  «      }	t        |«      t        |«      z
  }
|r5|D �cg c]  }|j                  |› d�«      rŒ|‘Œ }}|
j                  |«       |j                  «       D ��ch c]  \  }}|’Œ	 }}}t        |
|z
  «      }
t        d„ |D «       «      }|r|
D �cg c]	  }d|vsŒ|‘Œ }
}t        |«      }|D ]Q  }|	D �cg c]	  }||v sŒ|‘Œ }}t        |«      dkD  sŒ&t        |«      t        |«      k  sŒ>|	D �cg c]	  }||vsŒ|‘Œ }	}ŒS |�&|j                  ||	|«      }	|j                  ||
|«      }
| j                  �7| j                  D ](  }|	D �cg c]  }t!        j"                  ||«      �Œ|‘Œ }	}Œ* | j$                  �7| j$                  D ](  }|
D �cg c]  }t!        j"                  ||«      �Œ|‘Œ }
}Œ* |	|
fS c c}w c c}}w c c}w c c}w c c}w c c}w c c}w )zæFind missing keys (keys that are part of the model parameters but were NOT found in the loaded state dict keys) and unexpected keys
    (keys found in the loaded state dict keys, but that are NOT part of the model parameters)
    rþ   c              3   ó>   K  — | ]  }|j                  d «      –— Œ y­w)úrotary_emb.inv_freqN)rX  )Ú.0Úbuffers     r‹   ú	<genexpr>z4_find_missing_and_unexpected_keys.<locals>.<genexpr>D  s   è ø€ ÒbÈ&˜vŸ™Ð/D×EÑbùs   ‚r3  r   )Úbase_model_prefixr  rø   r  Úupdate_expected_keysrÉ  r  Ú
startswithr   Únamed_buffersÚanyr*   r  Úupdate_missing_keysÚupdate_unexpected_keysÚ_keys_to_ignore_on_load_missingrÊ  r©  Ú"_keys_to_ignore_on_load_unexpected)r  r!  r.  r/  r0  r¡  r¸  r  r¶  r1  r2  rÏ   Útask_specific_keysÚnr’  Úmodel_buffersÚhas_inv_freq_buffersr,  rÏ  Úmissing_in_groupÚpatterns                        r‹   Ú!_find_missing_and_unexpected_keysrF  "  s{  € ð ×$Ñ$€Fô ˜×)Ñ)Ó+×0Ñ0Ó2Ó3€MØÐØ$×9Ñ9¸%ÀÐP_Ó`ˆô œ#˜mÓ,¬s°?Ó/CÑCÓD€LÜ˜/Ó*¬S°Ó-?Ñ?€Oá.Ø)AÖd AÈÏÉÐY_ÐX`Ð`aÐVbÕIcšaÐdÐÐdØ×ÑÐ1Ô2ð $)×#6Ñ#6Ó#8×9™4˜1˜a’QÐ9€MÑ9Ü˜_¨}Ñ<Ó=€Oô ÑbÐTaÔbÓbÐÙØ&5ÖX Ð9NÐVWÒ9Wš1ÐXˆÐXä& uÓ-€KØò RˆØ'3ÖB !°q¸E²zšAÐBÐÐBÜÐÓ  1Ó$¬Ð-=Ó)>ÄÀUÃÓ)KØ'3ÖQ !°qÐ@PÒ7PšAÐQˆLÑQðRð
 ÐØ#×7Ñ7¸¸|ÈVÓTˆØ&×=Ñ=¸eÀ_ÐV\Ó]ˆð ×*Ñ*Ð6Ø×:Ñ:ò 	VˆGØ'3ÖU !´r·y±yÀÈ!Ó7LÑ7TšAÐUˆLÑUð	Vð ×-Ñ-Ð9Ø×=Ñ=ò 	\ˆGØ*9Ö[ Q¼R¿Y¹YÀwÐPQÓ=RÑ=ZšqÐ[ˆOÑ[ð	\ð ˜Ð(Ð(ùòC eùó
 :ùò Yùò CùâQùò Vùò \sN   ÂH"Â H"Ã
H'Ã?	H-Ä	H-Ä#	H2Ä-H2Å	H7Å(H7Æ7H<ÇH<Ç:IÈIÚignore_mismatched_sizesÚkeys_to_rename_mappingc                 óP  — |sg g fS |�dg}| j                  «       }g }g }	|D ]ù  }
|
dk7  rt        |
|d|¬«      }|j                  «       D ��ci c]  \  }}||v sŒ||   |“Œ }}}|j                  «       D ]¥  }||v sŒ||   j                  ||   j                  k7  sŒ(||   j                  d   dk(  r+||   j                  «       dz  ||   j                  «       k(  rŒh|j                  |«       |	j                  ||   j                  ||   j                  f«       Œ§ Œû ||	fS c c}}w )a   
    Find potential shape mismatch between the different state dicts and the model parameters, but only if `ignore_mismatched_sizes`
    is True. Otherwise, return immediately and any shape mismatch that may exist will be raised later on. This avoids checking
    every parameter in advance, as shape mismatch are extremely rare in practice. If we want to ignore them however, we do
    need to check in advance as we need to know which parameters we need to move back from meta to cpu, and initialize
    correctly. Indeed, as our model initialization takes place at the module level, and not the weight level, in the
    case of a sharded checkpoint we cannot correctly initialize the weights according to `model._init_weights()` if we perform
    this check on each state dict at loading time (after the first loaded checkpoint, there are no way to initialize only the
    mismatched weights if any, without overwriting the previously loaded weights as well because all the module will be
    initialized, not only the weights that are mismatched).
    rk  rT  ©rJ  r	  r
  r�   r    rŠ  )rø   r  r®   r  ÚshapeÚnumelr‹  )r!  rø   r
  rG  rH  rJ  r
  Úmodel_state_dictÚmismatched_keysÚmismatched_shapesr7  rÏ   rÐ   Únew_state_dictr0  s                  r‹   Ú_find_mismatched_keysrQ  ^  sp  € ñ. #Ø�2ˆvˆàÐØ˜4Ðà×'Ñ'Ó)ÐØ€OØÐØ&ò gˆ
à˜ÒÜ(Ø¨ÀFÐYeôˆJð
 DN×CSÑCSÓCU×u¹4¸1¸aÐYZÐ^tÒYtÐ0°Ñ3°QÑ6ÐuˆÑuà!×&Ñ&Ó(ò 		gˆCØÐ&Ò&¨>¸#Ñ+>×+DÑ+DÐHXÐY\ÑH]×HcÑHcÓ+cð # 3Ñ'×-Ñ-¨bÑ1°QÒ6Ø& sÑ+×1Ñ1Ó3°aÑ7Ð;KÈCÑ;P×;VÑ;VÓ;XÓXà#×*Ñ*¨3Ô/Ø%×,Ñ,¨n¸SÑ.A×.GÑ.GÐIYÐZ]ÑI^×IdÑIdÐ-eÕfñ		gðgð* Ð-Ð-Ð-ùó vs   ÁD"ÁD"c                   ó"   — e Zd ZU ded<   ded<   y)ÚPipelineParallelr   Úinputsr    ÚoutputsN)r  Ú
__module__Ú__qualname__Ú__annotations__r‰   rŠ   r‹   rS  rS  –  s   … ØƒIØ„JrŠ   rS  c                   ó¶  — e Zd ZdZed„ «       Zed„ «       Zd„ Zd„ Ze	de
j                  fd„«       Ze	de
j                  fd„«       Zd	edefd
„Zedd„«       Z	 ddedee   de
j                  de
j$                  def
d„Z	 ddee   dededefd„Zd„ Zd dededefd„Zdeeee
j                  ef   f   defd„Z	 d!deeee
j                  ef   f   dedefd„Zy)"rÆ   zH
    A few utilities for `torch.nn.Modules`, to be used as a mixin.
    c                 óÆ   — 	 dd l }|j                  t        j                  «       «      }|j                  «       }|j                  | _        y # t        $ r t        d«      ‚w xY w)Nr   úFYou need to install psutil (pip install psutil) to use memory tracing.)ÚpsutilÚImportErrorÚProcessr†   ÚgetpidÚmemory_infoÚrssÚmem_rss_pre_forward)rÈ   r©   rª   r\  ÚprocessÚmems         r‹   Ú_hook_rss_memory_pre_forwardz-ModuleUtilsMixin._hook_rss_memory_pre_forward   s]   € ð	hÛð —.‘.¤§¡£Ó-ˆØ×!Ñ!Ó#ˆØ%(§W¡WˆÔ"Øøô ò 	hÜÐfÓgÐgð	hús   ‚A ÁA c                 óL  — 	 dd l }|j                  t        j                  «       «      }|j                  «       }|j                  | _        | j                  | j                  z
  }|t        | d«      r| j                  z   | _
        y dz   | _
        y # t        $ r t        d«      ‚w xY w)Nr   r[  Úmem_rss_diff)r\  r]  r^  r†   r_  r`  ra  Úmem_rss_post_forwardrb  r¨  rg  )rÈ   r©   rª   r\  rc  rd  rg  s          r‹   Ú_hook_rss_memory_post_forwardz.ModuleUtilsMixin._hook_rss_memory_post_forward¬  s¢   € ð	hÛð —.‘.¤§¡£Ó-ˆØ×!Ñ!Ó#ˆØ&)§g¡gˆÔ#Ø×2Ñ2°V×5OÑ5OÑOˆØ*ÄWÈVÐUcÔEd¨f×.AÑ.AÑlˆÔØð klÑlˆÔØøô ò 	hÜÐfÓgÐgð	hús   ‚B ÂB#c                 óº   — | j                  «       D ]8  }|j                  | j                  «       |j                  | j                  «       Œ: | j                  «        y)a%  
        Add a memory hook before and after each sub-module forward pass to record increase in memory consumption.

        Increase in memory consumption is stored in a `mem_rss_diff` attribute for each module and can be reset to zero
        with `model.reset_memory_hooks_state()`.
        N)r  Úregister_forward_pre_hookre  Úregister_forward_hookri  Úreset_memory_hooks_state©ÚselfrÈ   s     r‹   Úadd_memory_hooksz!ModuleUtilsMixin.add_memory_hooksº  sQ   € ð —l‘l“nò 	MˆFØ×,Ñ,¨T×-NÑ-NÔOØ×(Ñ(¨×)KÑ)KÕLð	Mð 	×%Ñ%Õ'rŠ   c                 óX   — | j                  «       D ]  }d|_        d|_        d|_        Œ y)z€
        Reset the `mem_rss_diff` attribute of each module (see [`~modeling_utils.ModuleUtilsMixin.add_memory_hooks`]).
        r   N)r  rg  rh  rb  rn  s     r‹   rm  z)ModuleUtilsMixin.reset_memory_hooks_stateÆ  s1   € ð —l‘l“nò 	+ˆFØ"#ˆFÔØ*+ˆFÔ'Ø)*ˆFÕ&ñ	+rŠ   rÉ   c                 ó   — t        | «      S )z�
        `torch.device`: The device on which the module is (assuming that all the module parameters are on the same
        device).
        )rá   ©ro  s    r‹   rÙ   zModuleUtilsMixin.deviceÏ  s   € ô $ DÓ)Ð)rŠ   c                 ó   — t        | «      S )zw
        `torch.dtype`: The dtype of the module (assuming that all the module parameters have the same dtype).
        )ró   rs  s    r‹   rå   zModuleUtilsMixin.dtype×  s   € ô
 # 4Ó(Ð(rŠ   Úencoder_attention_maskc                 ó   — |j                  «       dk(  r|dd…ddd…dd…f   }|j                  «       dk(  r|dd…dddd…f   }j                  | j                  ¬«      }d|z
  t        j                  | j                  «      j
                  z  }|S )zè
        Invert an attention mask (e.g., switches 0. and 1.).

        Args:
            encoder_attention_mask (`torch.Tensor`): An attention mask.

        Returns:
            `torch.Tensor`: The inverted attention mask.
        é   NrŠ  ©rå   ç      ð?)ÚdimrÍ  rå   r‚   ÚfinfoÚmin)ro  ru  Úencoder_extended_attention_masks      r‹   Úinvert_attention_maskz&ModuleUtilsMixin.invert_attention_maskÞ  s�   € ð "×%Ñ%Ó'¨1Ò,Ø.DÂQÈÊaÒQRÀ]Ñ.SÐ+Ø!×%Ñ%Ó'¨1Ò,Ø.DÂQÈÈdÒTUÐEUÑ.VÐ+ð +J×*LÑ*LÐSW×S]ÑS]Ð*LÓ*^Ð'Ø+.Ð1PÑ+PÔTY×T_ÑT_Ð`d×`jÑ`jÓTk×ToÑToÑ*oÐ'à.Ð.rŠ   Nc                 ó@  — |�t        j                  dt        «       n|j                  }| \  }}t	        j
                  ||¬«      }|d d d d …f   j                  ||d«      |d d d …d f   k  }|j                  |j                  «      }|j                  d   |j                  d   k  r[|j                  d   |j                  d   z
  }t	        j                  t	        j                  |||f||j                  ¬«      |gd¬«      }|d d …d d d …d d …f   |d d …d d d d …f   z  }|S )NúNThe `device` argument is deprecated and will be removed in v5 of Transformers.)rÙ   r    ©rÙ   rå   r�   ©Úaxis)ÚwarningsÚwarnÚFutureWarningrÙ   r‚   ÚarangeÚrepeatrÍ  rå   rK  ÚcatÚones)	Úinput_shapeÚattention_maskrÙ   Ú
batch_sizeÚ
seq_lengthÚseq_idsÚcausal_maskÚprefix_seq_lenÚextended_attention_masks	            r‹   Ú*create_extended_attention_mask_for_decoderz;ModuleUtilsMixin.create_extended_attention_mask_for_decoderö  s4  € àÐÜ�M‰MØ`Ôboõð $×*Ñ*ˆFØ!,Ñˆ
�JÜ—,‘,˜z°&Ô9ˆØ˜d Dª!˜mÑ,×3Ñ3°JÀ
ÈAÓNÐRYÐZ^Ò`aÐcgÐZgÑRhÑhˆð "—n‘n ^×%9Ñ%9Ó:ˆà×Ñ˜QÑ .×"6Ñ"6°qÑ"9Ò9Ø+×1Ñ1°!Ñ4°{×7HÑ7HÈÑ7KÑKˆNÜŸ)™)ä—J‘J 
¨J¸ÐGÐPVÐ^i×^oÑ^oÔpØðð ôˆKð #.ªa°²qº!¨mÑ"<¸~ÊaÐQUÐW[Ò]^ÐN^Ñ?_Ñ"_ÐØ&Ð&rŠ   rŒ  r‹  rÙ   rå   c                 ó6  — |€| j                   }|j                  «       dk(  r| j                  j                  s|�t	        j
                  dt        «       |j                  «       dk(  r|dd…ddd…dd…f   }nk|j                  «       dk(  r<| j                  j                  rt        j                  |||«      }n*|dd…dddd…f   }nt        d|› d|j                  › d�«      ‚|j                  |¬«      }d	|z
  t        j                  |«      j                  z  }|S )
aâ  
        Makes broadcastable attention and causal masks so that future and masked tokens are ignored.

        Arguments:
            attention_mask (`torch.Tensor`):
                Mask with ones indicating tokens to attend to, zeros for tokens to ignore.
            input_shape (`Tuple[int]`):
                The shape of the input to the model.

        Returns:
            `torch.Tensor` The extended attention mask, with a the same dtype as `attention_mask.dtype`.
        NrŠ  r€  rw  z!Wrong shape for input_ids (shape z) or attention_mask (shape ú)rx  ry  )rå   rz  rª  Ú
is_decoderr„  r…  r†  rÆ   r“  rö   rK  rÍ  r‚   r{  r|  )ro  rŒ  r‹  rÙ   rå   r’  s         r‹   Úget_extended_attention_maskz,ModuleUtilsMixin.get_extended_attention_mask  s!  € ð ˆ=Ø—J‘JˆEà×"Ñ"Ó$¨Ò)¨d¯k©k×.DÒ.DàÐ!Ü—‘ØdÔfsôð
 ×ÑÓ 1Ò$Ø&4²Q¸ºaÂ°]Ñ&CÑ#Ø×ÑÓ! QÒ&ð �{‰{×%Ò%Ü*:×*eÑ*eØ °ó+Ñ'ð +9º¸DÀ$ÊÐ9IÑ*JÑ'äØ3°K°=Ð@[Ð\j×\pÑ\pÐ[qÐqrÐsóð ð #:×"<Ñ"<À5Ð"<Ó"IÐØ#&Ð)@Ñ#@ÄEÇKÁKÐPUÓDV×DZÑDZÑ"ZÐØ&Ð&rŠ   Ú	head_maskÚnum_hidden_layersÚis_attention_chunkedc                 óh   — |�)| j                  ||«      }|du r|j                  d«      }|S dg|z  }|S )aÊ  
        Prepare the head mask if needed.

        Args:
            head_mask (`torch.Tensor` with shape `[num_heads]` or `[num_hidden_layers x num_heads]`, *optional*):
                The mask indicating if we should keep the heads or not (1.0 for keep, 0.0 for discard).
            num_hidden_layers (`int`):
                The number of hidden layers in the model.
            is_attention_chunked (`bool`, *optional*, defaults to `False`):
                Whether or not the attentions scores are computed by chunks or not.

        Returns:
            `torch.Tensor` with shape `[num_hidden_layers x batch x num_heads x seq_length x seq_length]` or list with
            `[None]` for each layer.
        NTr�   )Ú_convert_head_mask_to_5dÚ	unsqueeze)ro  r˜  r™  rš  s       r‹   Úget_head_maskzModuleUtilsMixin.get_head_maskF  sR   € ð$ Ð Ø×5Ñ5°iÐARÓSˆIØ# tÑ+Ø%×/Ñ/°Ó3�	ð Ðð ˜Ð!2Ñ2ˆIàÐrŠ   c                 óæ  — |j                  «       dk(  rT|j                  d«      j                  d«      j                  d«      j                  d«      }|j                  |dddd«      }nB|j                  «       dk(  r/|j                  d«      j                  d«      j                  d«      }|j                  «       dk(  sJ d|j                  «       › �«       ‚|j                  | j                  ¬«      }|S )zD-> [num_hidden_layers x batch x num_heads x seq_length x seq_length]r    r   r�   rŠ  é   zhead_mask.dim != 5, instead rx  )rz  r�  ÚexpandrÍ  rå   )ro  r˜  r™  s      r‹   rœ  z)ModuleUtilsMixin._convert_head_mask_to_5da  sÑ   € à�=‰=‹?˜aÒØ!×+Ñ+¨AÓ.×8Ñ8¸Ó;×EÑEÀbÓI×SÑSÐTVÓWˆIØ!×(Ñ(Ð):¸BÀÀBÈÓK‰IØ�]‰]‹_ Ò!Ø!×+Ñ+¨AÓ.×8Ñ8¸Ó<×FÑFÀrÓJˆIØ�}‰}‹ !Ò#ÐUÐ'CÀIÇMÁMÃOÐCTÐ%UÓUÐ#Ø—L‘L t§z¡z�LÓ2ˆ	ØÐrŠ   Úonly_trainableÚexclude_embeddingsc                 ó
  — |rh| j                  «       D ��cg c]%  \  }}t        |t        j                  «      sŒ!|› d�‘Œ' }}}| j	                  «       D ��cg c]  \  }}||vsŒ|‘Œ }}}nt        | j                  «       «      }g }t        | dd«      }	|	rt        «       rddl	}
nt        d«      ‚|D ]º  }|j                  s|rŒ|	rˆt        |
j                  j                  «      rht        |d«      r|j                  «       }n%t        |d«      r|j                  j                   }nd	}|j#                  |j%                  «       d
z  |z  «       Œœ|j#                  |j%                  «       «       Œ¼ t'        |«      S c c}}w c c}}w )aé  
        Get number of (optionally, trainable or non-embeddings) parameters in the module.

        Args:
            only_trainable (`bool`, *optional*, defaults to `False`):
                Whether or not to return only the number of trainable parameters

            exclude_embeddings (`bool`, *optional*, defaults to `False`):
                Whether or not to return only the number of non-embeddings parameters

        Returns:
            `int`: The number of parameters.
        ú.weightÚis_loaded_in_4bitFr   NzÊbitsandbytes is not installed but it seems that the model has been loaded in 4bit precision, something went wrong make sure to install bitsandbytes with `pip install bitsandbytes`. You also need a GPU. ry  Úquant_storager    rŠ  )rl  rb  r   Ú	Embeddingr"  r  rØ   r€  rS   Úbitsandbytesrö   rÈ  Ú
Params4bitr¨  ry  r§  Úitemsizer‹  rL  Úsum)ro  r¢  r£  r²   Úmodule_typeÚembedding_param_namesrÅ   Útotal_parametersÚtotal_numelr¦  ÚbnbrÜ  Ú	num_bytess                r‹   Únum_parameterszModuleUtilsMixin.num_parametersl  sx  € ñ à:>×:LÑ:LÓ:N÷%Ù%6 T¨;ÔR\Ð]hÔjl×jvÑjvÕRw�4�&˜Ò ð%Ð!ñ %ð 26×1FÑ1FÓ1H÷ Ù-˜d IÈDÐXmÒLm’	ð Ðò  ô  $ D§O¡OÓ$5Ó6ÐàˆÜ# DÐ*=¸uÓEÐáÜ(Ô*Ü*ä ðpóð ð
 &ò 	6ˆEØ×"Ò"ª.ñ %¬°E¸3¿6¹6×;LÑ;LÔ)MÜ˜u nÔ5Ø$)×$6Ñ$6Ó$8™	Ü  ¨Ô8Ø$)×$7Ñ$7×$@Ñ$@™	à$%˜	Ø×&Ñ& u§{¡{£}°qÑ'8¸9Ñ'DÕEà×&Ñ& u§{¡{£}Õ5ð	6ô �;ÓÐùóI%ùó s   –"E9¹E9ÁE?Á#E?Ú
input_dictc                 óä   — t        | d«      si | _        | j                  |v r|| j                     j                  «       S d| j                  vr$t        j                  d«       d| j                  d<   y)zÞ
        Helper function to estimate the total number of tokens from the model inputs.

        Args:
            inputs (`dict`): The model inputs.

        Returns:
            `int`: The total number of tokens.
        Úwarnings_issuedÚestimate_tokenszdCould not estimate the number of tokens of the input, floating-point operations will not be computedTr   )r¨  r¶  Úmain_input_namerL  r  r  )ro  r´  s     r‹   r·  z ModuleUtilsMixin.estimate_tokens¢  sr   € ô �tÐ.Ô/Ø#%ˆDÔ Ø×Ñ :Ñ-Ø˜d×2Ñ2Ñ3×9Ñ9Ó;Ð;Ø d×&:Ñ&:Ñ:Ü�N‰NØvôð 7;ˆD× Ñ Ð!2Ñ3ØrŠ   c                 óP   — d| j                  |«      z  | j                  |¬«      z  S )aÞ  
        Get number of (optionally, non-embeddings) floating-point operations for the forward and backward passes of a
        batch with this transformer model. Default approximation neglects the quadratic dependency on the number of
        tokens (valid if `12 * d_model << sequence_length`) as laid out in [this
        paper](https://arxiv.org/pdf/2001.08361.pdf) section 2.1. Should be overridden for transformers with parameter
        re-use e.g. Albert or Universal Transformers, or if doing long-range modeling with very high sequence lengths.

        Args:
            batch_size (`int`):
                The batch size for the forward pass.

            sequence_length (`int`):
                The number of tokens in each line of the batch.

            exclude_embeddings (`bool`, *optional*, defaults to `True`):
                Whether or not to count embedding and softmax operations.

        Returns:
            `int`: The number of floating-point operations.
        é   )r£  )r·  r³  )ro  r´  r£  s      r‹   Úfloating_point_opsz#ModuleUtilsMixin.floating_point_ops·  s.   € ð0 �4×'Ñ'¨
Ó3Ñ3°d×6IÑ6IÐ]oÐ6IÓ6pÑpÐprŠ   r¨   ©NN©F©FF©T)r  rV  rW  Ú__doc__Ústaticmethodre  ri  rp  rm  Úpropertyr‚   rÙ   rå   r   r~  r“  r   r�   rë   r—  r   Úboolrž  rœ  r³  r   rÜ   r   r   r·  r»  r‰   rŠ   r‹   rÆ   rÆ   ›  sŸ  „ ñð ñ	ó ð	ð ñó ðò
(ò+ð ð*˜Ÿ™ò *ó ð*ð ð)�u—{‘{ò )ó ð)ð/¸Fð /Àvó /ð0 ò'ó ð'ð8 rvñ2'Ø$ð2'Ø38¸±:ð2'ØGLÇ|Á|ð2'Øch×cnÑcnð2'à	ó2'ðj afñØ! &Ñ)ðØ>AðØY]ðà	óò6	ñ4 ¨Tð 4 Ètð 4 Ð`có 4 ðl¨$¨s°E¸%¿,¹,ÈÐ:KÑ4LÐ/LÑ*Mð ÐRUó ð, [_ñqØ˜s E¨%¯,©,¸Ð*;Ñ$<Ð<Ñ=ðqØSWðqà	ôqrŠ   c            !       óv
  ‡ — e Zd ZdZdZdZdZdZdZdZ	dZ
dZdZdZdZdZdZdZdZdZdZdZdZdZdZdZdZdZedeeej@                  f   fd„«       Z!edefd„«       Z"d	e#fˆ fd
„Z$d„ Z%d„ Z&d„ Z'de(e)e   ef   ddfd„Z*e+e,d„ «       «       Z-e+	 	 	 	 d’de.de/ej`                     de/e(eeee1f   f      de.fd„«       Z2e+dej`                  dej`                  fd„«       Z3ede4jj                  fd„«       Z6e+de.fd„«       Z7e+	 	 	 	 d“de/ej`                     de/e(eeee1f   f      de.de.de#f
d„«       Z8e+d”de.de#fd„«       Z9e+d”de.de#fd„«       Z:d„ Z;d „ Z<de4jj                  fd!„Z=d"e4jj                  fd#„Z>de4jj                  fd$„Z?d%„ Z@d&„ ZAd'„ ZBeCd(e4jj                  d)e4jj                  d*ed+efd,„«       ZDd-„ ZEdefd.„ZF	 	 	 d•d/e/e1   d0e/e1   d1e.de4jŽ                  fd2„ZHd–d3„ZI	 	 	 d•d4e4jŽ                  d/e/e1   d0e/e1   d1e.de4jŽ                  f
d5„ZJ	 	 	 d—d6e4j–                  d/e/e1   d7e/e.   d1e.de4j–                  f
d8„ZLd9„ ZM	 d”d:„ZNd;„ ZOd<„ ZPd=e1fd>„ZQde(e4jŽ                  eRe4jŽ                     f   fd?„ZSd@„ ZTdAee1e)e1   f   fdB„ZUd˜dC„ZVdeWfdDe.dEeXfdF„ZYdG„ ZZede.fdH„«       Z[ddej¸                  ddIddddf	dJe(ee]j¼                  f   dKe.dLe/e_   dMeXdNe.dOe(e1ef   dPe.dQe/e   dRe/e(ee.f      dSe.fdT„Z` eaebjÆ                  «      ˆ fdU„«       Zcd™dV„Zd eaejh                  jj                  jÊ                  «      ˆ fdW„«       Ze eaejh                  jj                  jÌ                  «      ˆ fdX„«       Zfˆ fdY„Zgˆ fdZ„Zhe+d[e.d\e.fd]„«       Zie+e,ddddddd^ddd_œ	d`ejek   dae/e(ee]j¼                  f      d	e/e(e#ee]j¼                  f      dbe/e(ee]j¼                  f      dce.dde.dee.dRe/e(ee.f      dfedge/e.   dhe.dekfdi„«       «       ZleCdjedeRee.f   fdk„«       Zm	 	 	 dšdle)e   dme/eeef      dne.doe.fdp„ZneCdeRee.f   fdq„«       Zodr„ Zpe+	 	 	 	 	 	 	 	 	 	 	 d›dsd dLe/e   dte/e)e      dae/e   dce.due/e   de/e   dve/e   dwe/e.   de/ej`                     dxe/eq   dye/erjæ                     dze/d{   dme/eeef      dhe.fd|„«       Zte+d}„ «       Zue+d~„ «       Zvdœd„Zwe+d�d€„«       Zxdžd�„Zyd‚„ Zzdƒ„ Z{ed„„ «       Z|ed…„ «       Z}ed†„ «       Z~e~jþ                  d‡„ «       Z~dˆe€fd‰„Z�e+dŠ„ «       Z‚d‹e)e   dŒe)e   de/ej`                     dxe/eq   dd f
d�„ZƒdŽe)e   dce.d[e.dd fd�„Z„d�efd‘„Z…ˆ xZ†S )Ÿr–   aà  
    Base class for all models.

    [`PreTrainedModel`] takes care of storing the configuration of the models and handles methods for loading,
    downloading and saving models as well as a few methods common to all models to:

        - resize the input embeddings,
        - prune heads in the self-attention heads.

    Class attributes (overridden by derived classes):

        - **config_class** ([`PretrainedConfig`]) -- A subclass of [`PretrainedConfig`] to use as configuration class
          for this model architecture.
        - **load_tf_weights** (`Callable`) -- A python *method* for loading a TensorFlow checkpoint in a PyTorch model,
          taking as arguments:

            - **model** ([`PreTrainedModel`]) -- An instance of the model on which to load the TensorFlow checkpoint.
            - **config** ([`PreTrainedConfig`]) -- An instance of the configuration associated to the model.
            - **path** (`str`) -- A path to the TensorFlow checkpoint.

        - **base_model_prefix** (`str`) -- A string indicating the attribute associated to the base model in derived
          classes of the same architecture adding modules on top of the base model.
        - **is_parallelizable** (`bool`) -- A flag indicating whether this model supports model parallelization.
        - **main_input_name** (`str`) -- The name of the principal input to the model (often `input_ids` for NLP
          models, `pixel_values` for vision models and `input_values` for speech models).
    Nrk  Ú	input_idsFrÉ   c                 ó8   — dt        j                  t        «      iS )z^
        `Dict[str, torch.Tensor]`: Dummy inputs to do a forward pass in the network.
        rÅ  )r‚   rt  rB   rs  s    r‹   Údummy_inputszPreTrainedModel.dummy_inputs3  s   € ð
 œUŸ\™\¬,Ó7Ð8Ð8rŠ   c                  ó   — y)z@
        :str: Identifies that this is a PyTorch model.
        rM  r‰   rs  s    r‹   rO  zPreTrainedModel.framework:  s   € ð
 rŠ   rª  c                 óh  •— t         ‰| �  «        t        |t        «      s:t	        d| j
                  j                  › d| j
                  j                  › d�«      ‚t        |dd«      s@t        |d«      r|j                  nt        j                  «       }| j                  ||d¬«      }|| _        | j
                  j                  }|t        vrYdd	j                  t        «      › d
�}t!        j"                  || j
                  j                  «      }t%        |«      dkD  r|d   }nd }|| _        |j(                  | _        i | _        | j-                  «       rt/        j0                  |«      nd | _        t5        j4                  | j
                  j6                  «      | _        | j8                  xs g | _        y )NzParameter config in `zt(config)` should be an instance of class `PretrainedConfig`. To create a model from a pretrained model use `model = z(.from_pretrained(PRETRAINED_MODEL_NAME)`Ú_attn_implementation_autosetFr  )r  Úcheck_device_mapú(rÁ  r•  r   )ÚsuperÚ__init__rb  r"   rö   r  r  r€  r¨  r  r‚   r¿   Ú_autoset_attn_implementationrª  r3   r  rÊ  Úfindallr  Ú	loss_typeÚname_or_pathr¶  Úcan_generater%   Úfrom_model_configÚgeneration_configÚcopyÚ_keep_in_fp32_modulesÚ_no_split_modules)ro  rª  rT  rª   rå   rÑ  Úloss_groupsr  s          €r‹   rÎ  zPreTrainedModel.__init__A  sx  ø€ Ü‰ÑÔÜ˜&Ô"2Ô3ÜØ'¨¯©×(?Ñ(?Ð'@ð Aà ŸN™N×3Ñ3Ð4Ð4\ð^óð ô
 �vÐ=¸uÔEä*1°&¸-Ô*H�F×&Ò&Ìe×NeÑNeÓNgˆEØ×6Ñ6°vÈ5ÐchÐ6ÓiˆFØˆŒð —N‘N×+Ñ+ˆ	ØœLÑ(Ø˜cŸh™h¤|Ó4Ð5°QÐ7ˆKÜŸ
™
 ;°·±×0GÑ0GÓHˆIÜ�9‹~ Ò!Ø% a™L‘	à �	Ø"ˆŒà"×/Ñ/ˆÔØ!ˆÔØOS×O`ÑO`ÔObÔ!1×!CÑ!CÀFÔ!KÐhlˆÔô &*§Y¡Y¨t¯~©~×/SÑ/SÓ%TˆÔ"à!%×!7Ñ!7Ò!=¸2ˆÕrŠ   c           
      ó   — | j                  «        | j                  «        | j                  �¿| j                  «       D ��ch c]  \  }}t	        |«      dkD  sŒ|’Œ }}}t        «       }|D ]F  }|j                  |j                  d«      D �cg c]  }|j                  «       rŒ|dvsŒ|‘Œ c}«       ŒH | j                  D ]*  }||vsŒt        |› d| j                  j                  › �«      ‚ | j                  j                  �$| j                  j                  j                  «       nd| _        | j                  j                   �$| j                  j                   j                  «       ni | _        | j%                  «       D ]e  \  }}t'        |dd«      x}sŒ| j"                  j                  |j                  «       j)                  «       D ��	ci c]  \  }}	|› d|› �|	“Œ c}	}«       Œg | j"                  �Lt+        d«      r@| j"                  j)                  «       D ]"  \  }}	|	t,        vsŒt        d|	› d	t,        › �«      ‚ yyyc c}}w c c}w c c}	}w )
zÅ
        A method executed at the end of each Transformer model initialization, to execute code that needs the model's
        modules properly initialized (such as weight initialization).
        Nr   rþ   )rÆ  ÚbiaszV was specified in the `_keep_in_fp32_modules` list, but is not part of the modules in Ú_tp_planz2.3z"Unsupported tensor parallel style z. Supported styles are )Úinit_weightsÚ._backward_compatibility_gradient_checkpointingr×  r"  r  r  r   ÚsplitÚ	isnumericrö   r  r  rª  Úbase_model_pp_planrÖ  Ú_pp_planÚbase_model_tp_planrÜ  r‚  r€  r®   r[   r1   )
ro  r²   r’  Úall_parametersÚunique_module_namesrÜ  rÈ   ÚplanrÏ   rÐ   s
             r‹   Ú	post_initzPreTrainedModel.post_initd  s5  € ð
 	×ÑÔØ×;Ñ;Ô=ð ×%Ñ%Ð1Ø26×2GÑ2GÓ2I×[¡w t¨QÌSÐQUËYÐYZË]šdÐ[ˆNÑ[Ü"%£%Ðà'ò �Ø#×*Ñ*Ø&+§k¡k°#Ó&6Ör˜d¸d¿n¹nÕ>NÐSWÐ_qÒSq’TÒrõðð
 ×4Ñ4ò �ØÐ!4Ò4Ü$Ø!˜(ð #Ø ŸN™N×3Ñ3Ð4ð6óð ðð BFÇÁ×A_ÑA_ÐAk˜Ÿ™×6Ñ6×;Ñ;Ô=ÐquˆŒØAEÇÁ×A_ÑA_ÐAk˜Ÿ™×6Ñ6×;Ñ;Ô=ÐqsˆŒØ ×/Ñ/Ó1ò 	Y‰LˆD�&Ü˜v z°4Ó8Ð8ˆtÑ8Ø—‘×$Ñ$À4Ç9Á9Ã;×CTÑCTÓCV×%W¹4¸1¸a¨¨¨a°¨s m°QÑ&6Ó%WÕXð	Yð �=‰=Ð$Ô)BÀ5Ô)IØŸ™×+Ñ+Ó-ò ‘��1ØÔ/Ò/Ü$Ø<¸Q¸CÐ?VÔWjÐVkÐlóð ñð *JÐ$ùó- \ùò
 sùó &Xs$   Á H?ÁH?ÂI
Â!I
Â&I
Ç
I
c                 óX   — t        | dd«      }|€t        d«      ‚|j                  | «      S )zŽ
        Potentially dequantize the model in case it has been quantized by a quantization method that support
        dequantization.
        r¡  Nz?You need to first quantize your model in order to dequantize it)r€  rö   Ú
dequantize)ro  r¡  s     r‹   ré  zPreTrainedModel.dequantize‹  s5   € ô
 ˜t ^°TÓ:ˆàÐÜÐ^Ó_Ð_à×&Ñ& tÓ,Ð,rŠ   c                 óš   — | j                   r?t        | j                  dd«      r'| j                  «        t	        | j                  d«       y y y )NÚgradient_checkpointingF)Úsupports_gradient_checkpointingr€  rª  Úgradient_checkpointing_enableÚdelattrrs  s    r‹   rÞ  z>PreTrainedModel._backward_compatibility_gradient_checkpointing—  s@   € Ø×/Ò/´G¸D¿K¹KÐIaÐchÔ4iØ×.Ñ.Ô0ä�D—K‘KÐ!9Õ:ð 5jÐ/rŠ   Útagsc                 ó²   — t        |t        «      r|g}| j                  €g | _        |D ],  }|| j                  vsŒ| j                  j                  |«       Œ. y)a\  
        Add custom tags into the model that gets pushed to the Hugging Face Hub. Will
        not overwrite existing tags in the model.

        Args:
            tags (`Union[List[str], str]`):
                The desired tags to inject in the model

        Examples:

        ```python
        from transformers import AutoModel

        model = AutoModel.from_pretrained("google-bert/bert-base-cased")

        model.add_model_tags(["custom", "custom-bert"])

        # Push the model to your namespace with the name "my-custom-bert".
        model.push_to_hub("my-custom-bert")
        ```
        N)rb  rÜ   Ú
model_tagsr‹  )ro  rï  Útags      r‹   Úadd_model_tagszPreTrainedModel.add_model_tags�  sS   € ô, �dœCÔ Ø�6ˆDà�?‰?Ð"Ø ˆDŒOàò 	,ˆCØ˜$Ÿ/™/Ò)Ø—‘×&Ñ& sÕ+ñ	,rŠ   c                 óì  — |j                  d|j                  «      }t        |t        «      rt	        t
        |«      }|j                  dd«      }d}|�| j                  |«      }t        j                  |«      }|j                  �|j                  }nd}|j                  d|«      |_
        t	        |dd«      s| j                  ||d|¬«      }t        «       rqt        skt        set        j!                  d«       t"        j$                  j'                  t)        «       ¬	«      t+        «       g}t-        |«      5   | |fi |¤Ž}ddd«       n	 | |fi |¤Ž}|�t        j.                  |«       S # 1 sw Y   Œ"xY w)
zö
        All context managers that the model should be initialized under go here.

        Args:
            torch_dtype (`torch.dtype`, *optional*):
                Override the default `torch.dtype` and load the model under this dtype.
        r  Úuse_flash_attention_2FNÚattn_implementationrÊ  )rõ  rË  r  ú@Detected DeepSpeed ZeRO-3: activating zero.init() for this model©Úconfig_dict_or_path)rŽ  r  rb  rÜ   r€  r‚   r  rÖ  ÚdeepcopyÚ_attn_implementation_internalÚ_attn_implementationrÏ  r)   r·   r»   r  r   Ú	deepspeedÚzeroÚInitr(   r¼   rJ   rÀ   )	r  rª  rª   r  rõ  r  rö  Úinit_contextsr!  s	            r‹   Ú_from_configzPreTrainedModel._from_config½  sm  € ð —j‘j °×0BÑ0BÓCˆÜ�k¤3Ô'Ü!¤%¨Ó5ˆKà &§
¡
Ð+BÀEÓ JÐð ˆ
ØÐ"Ø×5Ñ5°kÓBˆJä—‘˜vÓ&ˆà×/Ñ/Ð;ð #)×"FÑ"FÑà"&Ðà&,§j¡jÐ1FÐH[Ó&\ˆÔ#Ü�vÐ=¸uÔEØ×5Ñ5ØØ&;Ø!&Ø'ð	 6ó ˆFô &Ô'µÕFXÜ�K‰KÐZÔ[ô 'Ÿ^™^×0Ñ0ÔEUÓEWÐ0ÓXÔZiÓZkÐlˆMÜ  Ó/ñ .Ù˜FÑ- fÑ-�÷.ð .ñ ˜Ñ) &Ñ)ˆEð Ð!Ü×#Ñ# JÔ/àˆ÷.ð .ús   Ä5
E*Å*E3Trõ  r  r¸  rË  c                 óŠ  — d}t        |d«      rÑ|j                  �Å|j                  dk7  r|rt        d|j                  › d�«      ‚t	        |j                  t
        «      su|j                  dgt        j                  «       z   vrQd|j                  › d�}| j                  r|d	z  }| j                  r|d
z  }| j                  r|dz  }t        |dz   «      ‚|j                  }|j                  j                  «       D ]<  }t        ||«      }	t	        |t
        «      s|n|j                  |d«      }
|	€Œ6|
|	_        Œ> |rt        j!                  d«       d|_        |j                  dk(  r| j#                  |||d|¬«       �n>|dk(  r| j%                  |d¬«      }�n$|dv rãt'        «       sÙ| j)                  ||€dnd¬«      }t*        j,                  j.                  �å|j                  dk(  rÖt*        j0                  j3                  «       dkD  rµt-        j4                  t*        j6                  «      t-        j4                  d«      k  r|t        j!                  d«       t*        j8                  j0                  j;                  d«       n=|t        j                  «       v r||_        nt	        |t
        «      rd|_        nd|_        d|_        |S )az  
        Automatically checks and dispatches to a default attention implementation. In order of priority:
            1. An implementation specified in `config._attn_implementation` (due for example to the argument attn_implementation="sdpa" in from_pretrained).
            2. DEPRECATED: if use_flash_attention_2 is set to `True` and `flash_attn` is available, flash attention. (`LlamaFlashAttention` for example)
            3. SDPA implementation, if available and supported by the model type. (`LlamaSdpaAttention` for example)
            4. The default model's implementation otherwise (`LlamaAttention` for example) .
        Nrû  Úflash_attention_2zBoth attn_implementation="z¹" and `use_flash_attention_2=True` were used when loading the model, which are not compatible. We recommend to just use `attn_implementation="flash_attention_2"` when loading the model.Úeagerz Specified `attn_implementation="zt"` is not supported. The only possible arguments are `attn_implementation="eager"` (manual attention implementation)zT, `"attn_implementation=flash_attention_2"` (implementation using flash attention 2)zf, `"attn_implementation=sdpa"` (implementation using torch.nn.functional.scaled_dot_product_attention)zV, `"attn_implementation=flex_attention"` (implementation using torch's flex_attention)rþ   z¯The model was loaded with use_flash_attention_2=True, which is deprecated and may be removed in a future release. Please use `attn_implementation="flash_attention_2"` instead.F)r  r¸  Úhard_check_onlyrË  Úflex_attentionT)r  )NÚsdpar  r    z2.4.1z¦Using the `SDPA` attention implementation on multi-gpu setup with ROCM may lead to performance issues due to the FA backend. Disabling it to use alternative backends.)r¨  rû  rü  rö   rb  r  ÚALL_ATTENTION_FUNCTIONSÚ
valid_keysÚ_supports_flash_attn_2Ú_supports_sdpaÚ_supports_flex_attnr  r  r€  rˆ   r  Úwarning_onceÚ_check_and_enable_flash_attn_2Ú_check_and_enable_flex_attnr_   Ú_check_and_enable_sdpar‚   r   ÚhipÚcudaÚdevice_countrc  r“   ÚbackendsÚenable_flash_sdprÊ  )r  rª  rõ  r  r¸  rË  Úrequested_attn_implementationÚmessager0  r  Úcurr_attn_implementations              r‹   rÏ  z,PreTrainedModel._autoset_attn_implementationø  sÚ  € ð& )-Ð%Ü�6Ð:Ô;À×@dÑ@dÐ@pØ×*Ñ*Ð.AÒAÑF[Ü Ø0°×1LÑ1LÐ0Mð Nrð róð ô ˜v×:Ñ:¼DÔAØ×/Ñ/¸°yÔCZ×CeÑCeÓCgÑ7gÑgà<¸V×=XÑ=XÐ<Yð  ZNð  O�Ø×-Ò-ØÐuÑu�GØ×%Ò%Øð   Hñ  H�GØ×*Ò*ØØqñ�Gô ! ¨3¡Ó/Ð/ð -3×,PÑ,PÐ)ð ×%Ñ%×*Ñ*Ó,ò 		TˆCÜ  ¨Ó-ˆJô "Ð"?ÄÔFñ .à2×6Ñ6°s¸DÓAð %ð Ñ%Ø;S�
Õ8ð		Tñ !Ü×Ñð Bôð +>ˆFÔ'à×&Ñ&Ð*=Ò=Ø×.Ñ.ØØ'Ø%Ø %Ø!1ð /ö ð +Ð.>Ò>Ø×4Ñ4°VÈTÐ4ÓRŠFØ*¨nÑ<ÔE[ÔE]à×/Ñ/ØØ)FÐ)N¡ÐTXð 0ó ˆFô —‘×!Ñ!Ð-Ø×/Ñ/°6Ò9Ü—J‘J×+Ñ+Ó-°Ò1Ü—M‘M¤%×"3Ñ"3Ó4´w·}±}ÀWÓ7MÒMä×#Ñ#ð }ôô —‘×#Ñ#×4Ñ4°UÕ;Ø*Ô.E×.PÑ.PÓ.RÑRØ*GˆFÕ'ÜÐ5´tÔ<Ø*.ˆFÕ'à*1ˆFÔ'à.2ˆÔ+ØˆrŠ   rå   c                 óô   — |j                   st        d| j                  › d|› d�«      ‚t        j	                  d| j                  › d|› d�«       t        j                  «       }t        j                  |«       |S )a�  
        Change the default dtype and return the previous one. This is needed when wanting to instantiate the model
        under specific dtype.

        Args:
            dtype (`torch.dtype`):
                a floating dtype to set to.

        Returns:
            `torch.dtype`: the original `dtype` that can be used to restore `torch.set_default_dtype(dtype)` if it was
            modified. If it wasn't, returns `None`.

        Note `set_default_dtype` currently only works with floating-point types and asserts if for example,
        `torch.int64` is passed. So if a non-float `dtype` is passed this functions will throw an exception.
        zCan't instantiate z model under dtype=z' since it is not a floating point dtypezInstantiating z model under default dtype rþ   )ré   rö   r  r  r   r‚   r¿   rÀ   )r  rå   r  s      r‹   r  z(PreTrainedModel._set_default_torch_dtypea  sw   € ð" ×&Ò&ÜØ$ S§\¡\ NÐ2EÀeÀWÐLsÐtóð ô 	�‰�n S§\¡\ NÐ2MÈeÈWÐTUÐVÔWÜ×,Ñ,Ó.ˆ
Ü×Ñ Ô&ØÐrŠ   c                 ó0   — t        | | j                  | «      S )z@
        `torch.nn.Module`: The main body of the model.
        )r€  r7  rs  s    r‹   Ú
base_modelzPreTrainedModel.base_model|  s   € ô
 �t˜T×3Ñ3°TÓ:Ð:rŠ   c                 ó$  — dt        | j                  «      v ry| j                  D ]/  }t        |d«      sŒdt        |«      vsŒ|j                  «       sŒ/ y dt        | j                  «      vr#t
        j                  | j                  › d�«       yy)aª  
        Returns whether this model can generate sequences with `.generate()` from the `GenerationMixin`.

        Under the hood, on classes where this function returns True, some generation-specific changes are triggered:
        for instance, the model instance will have a populated `generation_config` attribute.

        Returns:
            `bool`: Whether this model can generate sequences with `.generate()`.
        r&   TrÓ  r–   u:   has generative capabilities, as `prepare_inputs_for_generation` is explicitly overwritten. However, it doesn't directly inherit from `GenerationMixin`. From ðŸ‘‰v4.50ðŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
  - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.F)rÜ   Ú	__bases__r¨  rÓ  Úprepare_inputs_for_generationr  r  r  )r  Úbases     r‹   rÓ  zPreTrainedModel.can_generateƒ  s�   € ð ¤ C§M¡MÓ 2Ñ2Øà—M‘Mò 	ˆDÜ˜4 Ô0ØØ ¬¨D«	Ò1°d×6GÑ6GÕ6IÙð		ð ¤C¨×(IÑ(IÓ$JÑJÜ×ÑØ—<‘<�.ð 	! ð 	 ôð àrŠ   r  c                 óô  — | j                   s%t        | j                  › d|j                  › d�«      ‚t	        «       �sed}d}t
        j                  j                  d«      €:t        «       r |sd|_	        t        j                  d«       |S t        |› d|› �«      ‚t        j                  t
        j                  j                  d«      «      }t         j                  j"                  rg|t        j                  d	«      k  rt        |› d
|› d|› �«      ‚t         j"                  j%                  «       st        |› d�«      ‚t        |› d|› �«      ‚t         j                  j&                  r;|t        j                  d«      k  rt        |› d|› d|› �«      ‚t        |› d|› �«      ‚t)        | dd«      }	|	rt        d«      ‚|€t        j+                  d«       nJ|�H|t         j,                  t         j.                  fvr&t        j+                  d| j                  › d|› d�«       |rŒ|€Št!        j0                  d«      j2                  j4                  dvr_t         j"                  j%                  «       rt        j+                  d«       nnt7        «       rt        j+                  d«       nNt        d«      ‚|rA|�?t9        |t:        «      r/d|j=                  «       v sd|j=                  «       v rt        d«      ‚|sd|_	        |S )a9  
        Checks the availability of Flash Attention 2 and compatibility with the current model.

        If all checks pass and `hard_check_only` is False, the method will set the config attribute `attn_implementation` to "flash_attention_2" so that the model can initialize the correct attention module.
        z’ does not support Flash Attention 2.0 yet. Please request to add support where the model is hosted, on its model hub page: https://huggingface.co/zk/discussions/new or in the Transformers GitHub repo: https://github.com/huggingface/transformers/issues/newzVFlashAttention2 has been toggled on, but it cannot be used due to the following error:z�Please refer to the documentation of https://huggingface.co/docs/transformers/perf_infer_gpu_one#flashattention-2 to install Flash Attention 2.Ú
flash_attnr  z+Detect using FlashAttention2 on Ascend NPU.z3 the package flash_attn seems to be not installed. rD  zY you need flash_attn package version to be greater or equal than 2.1.0. Detected version z. z\ Flash Attention 2 is not available on CPU. Please make sure torch can access a CUDA device.z% Flash Attention 2 is not available. z2.0.4z„ you need flash_attn package version to be greater or equal than 2.0.4. Make sure to have that version installed - detected version Úuse_bettertransformerFz™Flash Attention 2 and BetterTransformer API are not compatible. Please make sure to disable BetterTransformers by doing model.reverse_bettertransformer()zwYou are attempting to use Flash Attention 2.0 without specifying a torch dtype. This might lead to unexpected behaviourzcFlash Attention 2.0 only supports torch.float16 and torch.bfloat16 dtypes, but the current dype in z is aG  . You should run training or inference using Automatic Mixed-Precision via the `with torch.autocast(device_type='torch_device'):` decorator, or load the model with the `torch_dtype` argument. Example: `model = AutoModel.from_pretrained("openai/whisper-tiny", attn_implementation="flash_attention_2", torch_dtype=torch.float16)`r   )r  Úmluz«You are attempting to use Flash Attention 2.0 with a model not initialized on GPU. Make sure to move the model to GPU after initializing it on CPU with `model.to('cuda')`.zªYou are attempting to use Flash Attention 2.0 with a model not initialized on MLU. Make sure to move the model to MLU after initializing it on CPU with `model.to('mlu')`.a-  You are attempting to use Flash Attention 2.0 with a model not initialized on GPU and with no GPU available. This is not supported yet. Please make sure to have access to a GPU and either initialise the model on a GPU by passing a device_map or initialising the model on CPU and then moving it to GPU.r  rÄ  zÞYou are attempting to use Flash Attention 2.0 with a model dispatched on CPU or disk. This is not supported. Please make sure to initialise the model on a GPU by passing a device_map that contains only GPU devices as keys.)r
  rö   r  Ú_name_or_pathrT   Ú	importlibÚutilÚ	find_specr]   rü  r  r   r]  r   rc  rY  r‚   r  r„   r  r€  r  Úfloat16rê   r^  rÙ   rÓ  r\   rb  r  rõ   )
r  rª  r  r¸  rË  r  ÚprefaceÚinstall_messageÚflash_attention_versionÚ_is_bettertransformers
             r‹   r  z.PreTrainedModel._check_and_enable_flash_attn_2ª  sI  € ð ×)Ò)ÜØ—<‘<�.ð !WØW]×WkÑWkÐVlð mnðnóð ô )Õ*ØnˆGð pˆOä�~‰~×'Ñ'¨Ó5Ð=ä)Ô+Ù*Ø6I˜Ô3ä—K‘KÐ MÔNØ!�Mä%¨¨	Ð1dÐetÐduÐ&vÓwÐwä&-§m¡m´I×4FÑ4F×4NÑ4NÈ|Ó4\Ó&]Ð#Ü�}‰}×!Ò!Ø*¬W¯]©]¸7Ó-CÒCÜ%Ø"˜)Ð#|ð  ~Uð  }Vð  VXð  Yhð  Xið  jóð ô Ÿ™×0Ñ0Ô2Ü$Ø"˜)Ð#ð  Aóð ô &¨¨	Ð1VÐWfÐVgÐ&hÓiÐiÜ—‘×"Ò"Ø*¬W¯]©]¸7Ó-CÒCÜ%Ø"˜)ð  $hð  i@ð  hAð  ACð  DSð  CTð  Uóð ô &¨¨	Ð1VÐWfÐVgÐ&hÓiÐiä '¨Ð-DÀeÓ LÐá Üð lóð ð ÐÜ×Ñð Jõð Ð$¨¼U¿]¹]ÌEÏNÉNÐ<[Ñ)[Ü×Ñð(Ø(+¯© ~°T¸+¸ð GNðNôñ  
Ð 2´u·{±{À1³~×7LÑ7L×7QÑ7QÐYhÑ7hÜ�z‰z×&Ñ&Ô(Ü×#Ñ#ðMõô (Ô)Ü×#Ñ#ðLõô
 !ðRóð ñ ØÐ&Ü˜:¤tÔ,Ø˜*×+Ñ+Ó-Ñ-°¸:×;LÑ;LÓ;NÑ1Näðpóð ñ Ø*=ˆFÔ'ØˆrŠ   c                 óà   — |r9| j                   st        | j                  › d�«      ‚t        «       st	        d«      ‚t        «       r| j                   s|S t        | dd«      }|r|S |sd|_        |S )a	  
        Checks the availability of SDPA for a given model.

        If all checks pass and `hard_check_only` is False, the method will set the config attribute `_attn_implementation` to "sdpa" so that the model can initialize the correct attention module.
        aâ   does not support an attention implementation through torch.nn.functional.scaled_dot_product_attention yet. Please request the support for this architecture: https://github.com/huggingface/transformers/issues/28005. If you believe this error is a bug, please open an issue in Transformers GitHub repository and load your model with the argument `attn_implementation="eager"` meanwhile. Example: `model = AutoModel.from_pretrained("openai/whisper-tiny", attn_implementation="eager")`zSPyTorch SDPA requirements in Transformers are not met. Please install torch>=2.1.1.r"  Fr  )r  rö   r  r^   r]  r€  rü  )r  rª  r  r,  s       r‹   r  z&PreTrainedModel._check_and_enable_sdpa	  s‡   € ñ Ø×%Ò%Ü Ø—|‘|�nð %Sð Sóð ô
 +Ô,Ü!Øióð ô 'Ô(°×0BÒ0BØˆMä '¨Ð-DÀeÓ LÐÙ ØˆMáØ*0ˆFÔ'ØˆrŠ   c                 ó¾   — |r9| j                   st        | j                  › d�«      ‚t        «       st	        d«      ‚t        «       r| j                   s|S |sd|_        |S )a  
        Checks the availability of Flex Attention for a given model.

        If all checks pass and `hard_check_only` is False, the method will set the config attribute `_attn_implementation` to "flex_attention" so that the model can initialize the correct attention module.
        aÄ   does not support an attention implementation through torch's flex_attention. Please request the support for this architecture: https://github.com/huggingface/transformers/issues/34809. If you believe this error is a bug, please open an issue in Transformers GitHub repository and load your model with the argument `attn_implementation="eager"` meanwhile. Example: `model = AutoModel.from_pretrained("openai/whisper-tiny", attn_implementation="eager")`z]PyTorch Flex Attention requirements in Transformers are not met. Please install torch>=2.5.0.r  )r  rö   r  rZ   r]  rü  )r  rª  r  s      r‹   r  z+PreTrainedModel._check_and_enable_flex_attn3	  so   € ñ Ø×*Ò*Ü Ø—|‘|�nð %xð xóð ô 0Ô1Ü!Øsóð ô ,Ô-°S×5LÒ5LØˆMáØ*:ˆFÔ'àˆrŠ   c                 óR   — d„ }| j                  «       j                  |«      | _        y)zŸ
        Enables the gradients for the input embeddings. This is useful for fine-tuning adapter weights while keeping
        the model weights fixed.
        c                 ó&   — |j                  d«       y ©NT)Úrequires_grad_)rÈ   ÚinputÚoutputs      r‹   Úmake_inputs_require_gradszMPreTrainedModel.enable_input_require_grads.<locals>.make_inputs_require_gradsV	  s   € Ø×!Ñ! $Õ'rŠ   N)Úget_input_embeddingsrl  Ú_require_grads_hook)ro  r5  s     r‹   Úenable_input_require_gradsz*PreTrainedModel.enable_input_require_gradsP	  s&   € ò	(ð $(×#<Ñ#<Ó#>×#TÑ#TÐUnÓ#oˆÕ rŠ   c                 ó8   — | j                   j                  «        y)z4
        Removes the `_require_grads_hook`.
        N)r7  Úremovers  s    r‹   Údisable_input_require_gradsz+PreTrainedModel.disable_input_require_grads[	  s   € ð 	× Ñ ×'Ñ'Õ)rŠ   c                 ód   — t        | | j                  | «      }|| ur|j                  «       S t        ‚)z–
        Returns the model's input embeddings.

        Returns:
            `nn.Module`: A torch module mapping vocabulary to hidden states.
        )r€  r7  r6  ÚNotImplementedError)ro  r  s     r‹   r6  z$PreTrainedModel.get_input_embeddingsa	  s5   € ô ˜T 4×#9Ñ#9¸4Ó@ˆ
Ø˜TÑ!Ø×2Ñ2Ó4Ð4ä%Ð%rŠ   rß  c                 óh   — t        | | j                  | «      }|| ur|j                  |«       yt        ‚)z�
        Set model's input embeddings.

        Args:
            value (`nn.Module`): A module mapping vocabulary to hidden states.
        N)r€  r7  Úset_input_embeddingsr=  )ro  rß  r  s      r‹   r?  z$PreTrainedModel.set_input_embeddingsn	  s4   € ô ˜T 4×#9Ñ#9¸4Ó@ˆ
Ø˜TÑ!Ø×+Ñ+¨EÕ2ä%Ð%rŠ   c                  ó   — y)z—
        Returns the model's output embeddings.

        Returns:
            `nn.Module`: A torch module mapping hidden states to vocabulary.
        Nr‰   rs  s    r‹   Úget_output_embeddingsz%PreTrainedModel.get_output_embeddings{	  s   € ð rŠ   c                  ó   — y)a]  
        Initialize the weights. This method should be overridden by derived class and is
        the only initialization method that will be called when loading a checkpoint
        using `from_pretrained`. Any attempt to initialize outside of this function
        will be useless as the torch.nn.init function are all replaced with skip.
        Nr‰   rn  s     r‹   r¬   zPreTrainedModel._init_weights„	  s   € ð 	rŠ   c                 óP   — t        |dd«      ry| j                  |«       d|_        y)zM
        Initialize the weights if they are not already initialized.
        rn  FNT)r€  r¬   rn  rn  s     r‹   Ú_initialize_weightsz#PreTrainedModel._initialize_weights�	  s*   € ô �6Ð/°Ô7ØØ×Ñ˜6Ô"Ø$(ˆÕ!rŠ   c                 ó@  — t        | j                  j                  d¬«      dd«      r2| j                  «       }|� | j	                  || j                  «       «       t        | j                  dd«      r|t        | j                  dd«      ret        | | j                  «      rt        | | j                  «      } | j                  | j                  | j                  | j                  d«      }|| _        | j                  «       D ]  }t        |d	«      sŒ|j                  «        Œ! y)
zç
        Tie the weights between the input embeddings and the output embeddings.

        If the `torchscript` flag is set in the configuration, can't handle parameter sharing so we are cloning the
        weights instead.
        T©ÚdecoderÚtie_word_embeddingsNÚis_encoder_decoderFÚtie_encoder_decoderÚencoderÚ_tie_weights)r€  rª  Úget_text_configrA  Ú_tie_or_clone_weightsr6  r¨  r7  Ú_tie_encoder_decoder_weightsrK  rG  r~  r  rL  )ro  Úoutput_embeddingsÚtied_weightsrÈ   s       r‹   Útie_weightszPreTrainedModel.tie_weights–	  só   € ô �4—;‘;×.Ñ.°tÐ.Ó<Ð>SÐUYÔZØ $× :Ñ :Ó <ÐØ Ð,Ø×*Ñ*Ð+<¸d×>WÑ>WÓ>YÔZä�4—;‘;Ð 4°eÔ<ÄÈÏÉÐVkÐmrÔAsÜ�t˜T×3Ñ3Ô4Ü˜t T×%;Ñ%;Ó<�Ø×<Ñ<Ø—‘˜dŸl™l¨D×,BÑ,BÀIóˆLð /;ˆDÔ+à—l‘l“nò 	&ˆFÜ�v˜~Õ.Ø×#Ñ#Õ%ñ	&rŠ   rK  rG  r7  Úbase_encoder_namec                 óŽ  ‡‡— g }g Š|j                   | j                   k7  r/t        j                  |j                   › d| j                   › d�«       	 	 	 ddt        j                  dt        j                  dt
        dt
        dt        t
           f
ˆˆfd	„Š ‰|| |||«       t        |«      dkD  rt        j                  d
|› �«       ‰S )Nú and zZ are not equal. In this case make sure that all encoder weights are correctly initialized.r   Údecoder_pointerÚencoder_pointerrq  rS  Úuninitialized_encoder_weightsc                 ó(  •— t        | t        j                  «      rt        |t        j                  «      sJ | › d|› d�«       ‚t        | d«      rwt        |d«      sJ ‚| j                  |_        ‰j                  |› |› d�«       t        | d«      r5t        |d«      sJ ‚‰j                  |› |› d�«       | j                  |_        y |j                  }| j                  }	t        |	«      dkD  �r!t        |«      dkD  sJ d|› d	| › �«       ‚|j                  «       D �
ch c]
  }
|d
z   |
z   ’Œ }}
d}|	j                  «       D ]¿  \  }}|j                  «       rQt        t        |«      |z   «      }|}t        |	|   t        ||   «      «      s6t        |«      t        |	«      k7  r|dz  }Œg||vrŒl|dkD  rt        d«      ‚|x}} ‰|	|   ||   |d
z   |z   |||dz   |› d|› �|› d|› �¬«       |j!                  |d
z   |z   «       ŒÁ |t#        |«      z  }y y c c}
w )NrU  z have to be of type nn.ModulerÆ  r¥  rÛ  z.biasr   zEncoder module z does not match decoder module ú/r    iô  zžMax depth of recursive function `tie_encoder_to_decoder` reached. It seems that there is a circular dependency between two or more `nn.Modules` of your model.rþ   )ÚdepthÚtotal_encoder_nameÚtotal_decoder_name)rb  r   rÛ   r¨  rÆ  r‹  rÛ  Ú_modulesr  r  r®   ÚisdigitrÜ   r�   rÓ  rö   r:  r  )rV  rW  rq  rS  rX  r[  r]  r\  Úencoder_modulesÚdecoder_modulesÚsub_nameÚall_encoder_weightsÚencoder_layer_posr²   rÈ   Úencoder_nameÚdecoder_nameÚ"tie_encoder_to_decoder_recursivelyrQ  s                    €€r‹   rg  zXPreTrainedModel._tie_encoder_decoder_weights.<locals>.tie_encoder_to_decoder_recursively½	  s”  ø€ ô ˜o¬r¯y©yÔ9¼jÈÔZ\×ZcÑZcÔ>dð Ø"Ð# 5¨Ð(9Ð9VÐWóÐdô �¨Ô1Ü˜°Ô9Ð9Ð9Ø)8×)?Ñ)?�Ô&Ø×#Ñ#Ð'8Ð&9Ð:LÐ9MÈWÐ$UÔVÜ˜?¨FÔ3Ü" ?°FÔ;Ð;Ð;Ø ×'Ñ'Ð+<Ð*=Ð>PÐ=QÐQVÐ(WÔXØ+:×+?Ñ+?�OÔ(Øà-×6Ñ6ˆOØ-×6Ñ6ˆOÜ�?Ó# aÓ'Ü˜?Ó+¨aÒ/ð Ø% oÐ%6Ð6UÐVeÐUfÐgóÐ/ð Ud×ThÑThÓTjÖ&kÈ {°SÑ'8¸8Ó'CÐ&kÐ#Ð&kØ$%Ð!Ø$3×$9Ñ$9Ó$;ò Q‘L�D˜&Ø—|‘|”~Ü'*¬3¨t«9Ð7HÑ+HÓ'I˜Ø'+˜Ü)¨/¸,Ñ*GÌÈoÐ^jÑNkÓIlÔmÔruØ+ósä  Ó1òs2ð .°Ñ2Ð-Ø$Ø _Ñ4Ø Ø šÜ(ðeóð ð
 7;Ð:˜ |Ù6Ø'¨Ñ5Ø'¨Ñ5Ø# cÑ)¨DÑ0Ø)Ø5Ø# a™iØ.@Ð-AÀÀ<À.Ð+QØ.@Ð-AÀÀ<À.Ð+Qõ	ð (×.Ñ.¨{¸SÑ/@À<Ñ/OÕPð?QðB .´Ð6IÓ1JÑJÑ-ðQ (ùò
 'ls   ÄHz;The following encoder weights were not tied to the decoder )r   rk  rk  )	r  r  r   r   rÛ   rÜ   r   r  r  )rK  rG  r7  rS  rX  rg  rQ  s        @@r‹   rO  z,PreTrainedModel._tie_encoder_decoder_weights±	  sò   ù€ ð 46Ð%Ø"$ˆØ×Ñ × 1Ñ 1Ò1Ü�K‰KØ×$Ñ$Ð% U¨7×+<Ñ+<Ð*=ð >6ð 6ôð Ø!Ø!ñA	KÜŸY™YðA	KäŸY™YðA	Kô ðA	Kô  #ð	A	Kô
 ,0´©9öA	KñH 	+Ø�WÐ/Ð1BÐDaô	
ô Ð,Ó-°Ò1Ü�N‰NØMÐNkÐMlÐmôð ÐrŠ   c                 ó  — | j                   j                  r3t        j                  |j                  j                  «       «      |_        n|j                  |_        t        |dd«      �xt        j                  j                  |j                  j                  d|j                  j                  d   |j                  j                  d   z
  fdd«      |j                  _
        t        |d«      rt        |d«      r|j                  |_        yyy)zPTie or clone module weights depending of whether we are using TorchScript or notrÛ  Nr   ÚconstantÚout_featuresÚnum_embeddings)rª  Útorchscriptr   Ú	ParameterrÆ  Úcloner€  Ú
functionalÚpadrÛ  rÔ  rK  r¨  rk  rj  )ro  rP  Úinput_embeddingss      r‹   rN  z%PreTrainedModel._tie_or_clone_weights
  sé   € à�;‰;×"Ò"Ü')§|¡|Ð4D×4KÑ4K×4QÑ4QÓ4SÓ'TÐÕ$à'7×'>Ñ'>ÐÔ$äÐ$ f¨dÓ3Ð?Ü*,¯-©-×*;Ñ*;Ø!×&Ñ&×+Ñ+àØ%×,Ñ,×2Ñ2°1Ñ5Ð8I×8NÑ8N×8TÑ8TÐUVÑ8WÑWðð Øó+Ð×"Ñ"Ô'ô Ð$ nÔ5¼'ÐBRÐTdÔ:eØ-=×-LÑ-LÐÕ*ð ;fÐ5rŠ   c                 ó¨  — t        «       }| g}t        |«      dkD  r­|j                  d«      }|j                  j                  |vrut        |t        «      rI|j                  €%t        |j                  j                  › d|› d�«      ‚|t        |j                  «      z  }|t        |j                  «       «      z  }t        |«      dkD  rŒ­t        |«      S )a™  
        Get the modules of the model that should not be spit when using device_map. We iterate through the modules to
        get the underlying `_no_split_modules`.

        Args:
            device_map (`str`):
                The device map value. Options are ["auto", "balanced", "balanced_low_0", "sequential"]

        Returns:
            `List[str]`: List of modules that should not be split
        r   r�   z does not support `device_map='z_'`. To implement support, the model class needs to implement the `_no_split_modules` attribute.)r  r  rŽ  r  r  rb  r–   rØ  rö   r  Úchildren)ro  r¸  rØ  Úmodules_to_checkrÈ   s        r‹   r$  z%PreTrainedModel._get_no_split_modules
  sß   € ô  ›EÐØ ˜6ÐÜÐ"Ó# aÒ'Ø%×)Ñ)¨"Ó-ˆFà×Ñ×(Ñ(Ð0AÑAÜ˜f¤oÔ6Ø×/Ñ/Ð7Ü(Ø%×/Ñ/×8Ñ8Ð9Ð9XÐYcÐXdð eZð Zóð ð
 ->ÄÀF×D\ÑD\Ó@]Ñ,]Ð)Ø ¤D¨¯©Ó):Ó$;Ñ;Ð ô Ð"Ó# aÓ'ô Ð%Ó&Ð&rŠ   Únew_num_tokensÚpad_to_multiple_ofÚmean_resizingc                 óÚ  — | j                  |||«      }|€|€|S t        | d«      xr | j                  du}t        «       rP|sNt        j
                  j                  |j                  d¬«      5  |j                  j                  d   }ddd«       n|j                  j                  d   }| j                  j                  «       _        || _        | j                  «        |S # 1 sw Y   ŒAxY w)a$	  
        Resizes input token embeddings matrix of the model if `new_num_tokens != config.vocab_size`.

        Takes care of tying weights embeddings afterwards if the model class has a `tie_weights()` method.

        Arguments:
            new_num_tokens (`int`, *optional*):
                The new number of tokens in the embedding matrix. Increasing the size will add newly initialized
                vectors at the end. Reducing the size will remove vectors from the end. If not provided or `None`, just
                returns a pointer to the input tokens `torch.nn.Embedding` module of the model without doing anything.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the embedding matrix to a multiple of the provided value.If `new_num_tokens` is set to
                `None` will just pad the embedding to a multiple of `pad_to_multiple_of`.

                This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
                `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128. For more
                details about this, or help on choosing the correct value for resizing, refer to this guide:
                https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tc
            mean_resizing (`bool`):
                Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
                covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

                Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
                where the generated tokens' probabilities won't be affected by the added embeddings because initializing the new embeddings with the
                old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
                Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html

        Return:
            `torch.nn.Embedding`: Pointer to the input tokens Embeddings Module of the model.
        Nr¡  ©Úmodifier_rankr   )Ú_resize_token_embeddingsr¨  r¡  r)   rý  rþ  ÚGatheredParametersrÆ  rK  rª  rM  Ú
vocab_sizerR  )ro  ru  rv  rw  Úmodel_embedsrJ  r}  s          r‹   Úresize_token_embeddingsz'PreTrainedModel.resize_token_embeddings<
  sì   € ðH ×4Ñ4°^ÐEWÐYfÓgˆØÐ!Ð&8Ð&@ØÐô ˜t ^Ó4ÒV¸×9JÑ9JÐRVÐ9VˆÜ%Ô'±Ü—‘×2Ñ2°<×3FÑ3FÐVZÐ2Ó[ñ :Ø)×0Ñ0×6Ñ6°qÑ9�
÷:ð :ð &×,Ñ,×2Ñ2°1Ñ5ˆJð 4>ˆ�‰×#Ñ#Ó%Ô0Ø$ˆŒð 	×ÑÔàÐ÷:ð :ús   Á-C!Ã!C*c                 ó<  — | j                  «       }| j                  ||||«      }t        |d«      r|j                  }t	        ||«       |j
                  j                  }|j                  |«       | j                  |«       t        | d«      xr | j                  d u}|�st        «       rP|sNt        j                  j                  |j
                  d ¬«      5  |j
                  j                  d   }d d d «       n|j
                  j                  d   }| j                  «       �ß| j                   j#                  d¬«      j$                  s¹| j                  «       }	t'        |	t(        j*                  j,                  «      r| j                  |	||¬«      }
n| j/                  |	||¬«      }
t        |	d«      r|	j                  }t	        |
|«       |	j
                  j                  }|
j                  |«       | j1                  |
«       | j                  «       S # 1 sw Y   �Œ	xY w)NÚ_hf_hookr¡  ry  r   TrF  )rw  )r6  Ú_get_resized_embeddingsr¨  r�  rp   rÆ  rÈ  r2  r?  r¡  r)   rý  rþ  r|  rK  rA  rª  rM  rH  rb  r‚   r   r¨  Ú_get_resized_lm_headÚset_output_embeddings)ro  ru  rv  rw  Úold_embeddingsÚnew_embeddingsÚhookÚold_embeddings_requires_gradrJ  Úold_lm_headÚnew_lm_headÚold_lm_head_requires_grads               r‹   r{  z(PreTrainedModel._resize_token_embeddingsu
  sí  € Ø×2Ñ2Ó4ˆØ×5Ñ5Ø˜NÐ,>Àó
ˆô �> :Ô.Ø!×*Ñ*ˆDÜ˜~¨tÔ4Ø'5×'<Ñ'<×'JÑ'JÐ$Ø×%Ñ%Ð&BÔCØ×!Ñ! .Ô1Ü˜t ^Ó4ÒV¸×9JÑ9JÐRVÐ9Vˆð Ð)Ü)Ô+±LÜ—^‘^×6Ñ6°~×7LÑ7LÐ\`Ð6Óañ DØ%3×%:Ñ%:×%@Ñ%@ÀÑ%C�N÷Dð Dð "0×!6Ñ!6×!<Ñ!<¸QÑ!?�ð ×&Ñ&Ó(Ð4Ø—K‘K×/Ñ/¸Ð/Ó=×QÒQà×4Ñ4Ó6ˆKÜ˜+¤u§x¡x×'9Ñ'9Ô:Ø"×:Ñ:¸;ÈÐfsÐ:Ót‘à"×7Ñ7¸À^ÐcpÐ7Óq�Ü�{ JÔ/Ø"×+Ñ+�Ü" ;°Ô5Ø(3×(:Ñ(:×(HÑ(HÐ%Ø×&Ñ&Ð'@ÔAØ×&Ñ& {Ô3à×(Ñ(Ó*Ð*÷-Dñ Dús   ÃHÈHr…  c           	      óê  — |�It        |t        «      st        d|› d�«      ‚|€|j                  j                  d   }||z   dz
  |z  |z  }nt
        j                  d|› d�«       |€|S t        | d«      xr | j                  du}t        «       rT|sRt        j                  j                  |j                  d¬	«      5  |j                  j                  «       \  }}ddd«       n|j                  j                  «       \  }}|k(  rt        «       s|S t        |t        j                  «      s:t!        d
t#        |«      › dt        j                  › dt        j                  › d�«      ‚t        j                  ||j                  j$                  |j                  j&                  ¬«      }||kD  r|s| j)                  |«       n�||kD  rˆ|r†t
        j+                  d«       ||z
  }	t        «       rM|sKt        j                  j                  |j                  gd¬	«      5  | j-                  |||||	«       ddd«       n| j-                  |||||	«       t/        ||«      }
t        «       r�|s|j                  |j                  g}t        j                  j                  |d¬	«      5  |j                  j0                  d|
…dd…f   |j                  j0                  d|
…dd…f<   ddd«       n<|j                  j0                  d|
…dd…f   |j                  j0                  d|
…dd…f<   t        «       r¤|s¢|j                  |j                  g}t        j                  j                  |d¬	«      5  |j                  |_        |j                  j0                  j                  d   |_        |j4                  �|dz
  |j4                  k  rd|_        ddd«       |S |j                  j0                  |j                  _        |j                  j0                  j                  d   |_        |j4                  �|dz
  |j4                  k  rd|_        |S # 1 sw Y   �ŒKxY w# 1 sw Y   �ŒxY w# 1 sw Y   �ŒFxY w# 1 sw Y   |S xY w)aÂ	  
        Build a resized Embedding Module from a provided token Embedding Module. Increasing the size will add newly
        initialized vectors at the end. Reducing the size will remove vectors from the end

        Args:
            old_embeddings (`torch.nn.Embedding`):
                Old embeddings to be resized.
            new_num_tokens (`int`, *optional*):
                New number of tokens in the embedding matrix.

                Increasing the size will add newly initialized vectors at the end. Reducing the size will remove
                vectors from the end. If not provided or `None`, just returns a pointer to the input tokens
                `torch.nn.Embedding` module of the model without doing anything.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the embedding matrix to a multiple of the provided value. If `new_num_tokens` is set to
                `None` will just pad the embedding to a multiple of `pad_to_multiple_of`.

                This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
                `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128. For more
                details about this, or help on choosing the correct value for resizing, refer to this guide:
                https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tc
            mean_resizing (`bool`):
                Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
                covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

                Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
                where the generated tokens' probabilities will not be affected by the added embeddings because initializing the new embeddings with the
                old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
                Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html


        Return:
            `torch.nn.Embedding`: Pointer to the resized Embedding Module or the old Embedding Module if
            `new_num_tokens` is `None`
        Nz5Asking to pad the embedding matrix to a multiple of `z@`, which is not and integer. Please make sure to pass an integerr   r    z�You are resizing the embedding layer without providing a `pad_to_multiple_of` parameter. This means that the new embedding dimension will be a.  . This might induce some performance reduction as *Tensor Cores* will not be available. For more details about this, or help on choosing the correct value for resizing, refer to this guide: https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tcr¡  ry  zOld embeddings are of type ú, which is not an instance of zj. You should either use a different resize function or make sure that `old_embeddings` are an instance of rþ   r�  zýThe new embeddings will be initialized from a multivariate normal distribution that has old embeddings' mean and covariance. As described in this article: https://nlp.stanford.edu/~johnhew/vocab-expansion.html. To disable this, use `mean_resizing=False`)rb  r�   rö   rÆ  rK  r  r   r¨  r¡  r)   rý  rþ  r|  rU  r   r¨  Ú	TypeErrorrÓ  rÙ   rå   r¬   r  Ú(_init_added_embeddings_weights_with_meanr|  rÔ  rk  Úpadding_idx)ro  r…  ru  rv  rw  rJ  Úold_num_tokensÚold_embedding_dimr†  Úadded_num_tokensrA  Úparamss               r‹   r‚  z'PreTrainedModel._get_resized_embeddings�
  s  € ðV Ð)ÜÐ0´#Ô6Ü ØKÐL^ÐK_ð  ``ð  aóð ð Ð%Ø!/×!6Ñ!6×!<Ñ!<¸QÑ!?�Ø-Ð0BÑBÀQÑFÐK]Ñ]ÐasÑs‰Nä�K‰Kð&Ø&4Ð%5ð 6DðDôð Ð!Ø!Ð!ä˜t ^Ó4ÒV¸×9JÑ9JÐRVÐ9VˆÜ%Ô'±Ü—‘×2Ñ2°>×3HÑ3HÐX\Ð2Ó]ñ QØ4B×4IÑ4I×4NÑ4NÓ4PÑ1�Ð 1÷Qð Qð 1?×0EÑ0E×0JÑ0JÓ0LÑ-ˆNÐ-à˜^Ò+Ô4NÔ4PØ!Ð!ä˜.¬"¯,©,Ô7ÜØ-¬d°>Ó.BÐ-CÐCaÔbd×bnÑbnÐaoð pä—L‘L�> ð$óð ô Ÿ™ØØØ!×(Ñ(×/Ñ/Ø ×'Ñ'×-Ñ-ô	
ˆð ˜NÒ*±=à×Ñ˜~Õ.à˜nÒ,±ô ×Ñð=ôð  .°Ñ>ÐÜ)Ô+±LÜ—^‘^×6Ñ6¸×8MÑ8MÐ7NÐ^bÐ6Ócñ Ø×AÑAØ&¨Ð8IÈ>Ð[kô÷ð ð
 ×=Ñ=Ø" NÐ4EÀ~ÐWgôô � Ó/ˆä%Ô'±Ø$×+Ñ+¨^×-BÑ-BÐCˆFÜ—‘×2Ñ2°6ÈÐ2ÓKñ VØ4B×4IÑ4I×4NÑ4NÈrÐPQÈrÒSTÈuÑ4U�×%Ñ%×*Ñ*¨2¨A¨2ªq¨5Ñ1÷Vð Vð 1?×0EÑ0E×0JÑ0JÈ2ÈAÈ2ÊqÈ5Ñ0QˆN×!Ñ!×&Ñ& r¨ rª1 uÑ-ô
 &Ô'±Ø$×+Ñ+¨^×-BÑ-BÐCˆFÜ—‘×2Ñ2°6ÈÐ2ÓKñ 6Ø(6×(=Ñ(=�Ô%Ø0>×0EÑ0E×0JÑ0J×0PÑ0PÐQRÑ0S�Ô-ð "×-Ñ-Ð9¸~ÐPQÑ?QÐUc×UoÑUoÒ>oØ15�NÔ.÷6ð Ðð *8×)>Ñ)>×)CÑ)CˆN×!Ñ!Ô&Ø,:×,AÑ,A×,FÑ,F×,LÑ,LÈQÑ,OˆNÔ)Ø×)Ñ)Ð5¸>ÈAÑ;MÐQ_×QkÑQkÒ:kØ-1�Ô*àÐ÷iQñ Qú÷Xñ ú÷ Vñ Vú÷6ð Ðús1   Â<QÈQÊ=QÍ$AQ(ÑQÑQÑQ%Ñ(Q2r‰  Ú
transposedc           	      ó¬  — |€|S t        | d«      xr | j                  du}t        «       r~|s|t        j                  j                  |j                  d¬«      5  |s|j                  j                  «       n'|j                  j                  «       j                  «       \  }}ddd«       nG|s|j                  j                  «       n'|j                  j                  «       j                  «       \  }}|k(  rt        «       s|S t        |t        j                  «      s:t        dt        |«      › dt        j                  › dt        j                  › d�«      ‚|s|fn|f}|j                  du}	t        j                  ||	|j                  j                  |j                  j                   dœŽ}
||kD  r|s| j#                  |
«       nÍ||kD  rÈ|rÆt$        j'                  d	«       ||z
  }t        «       rw|su|j                  g}|	r||j                  gz  }t        j                  j                  |d¬«      5  | j)                  ||
||||«       |	r| j+                  ||
|«       ddd«       n+| j)                  ||
||||«       |	r| j+                  ||
|«       t-        ||«      }t        «       rq|so|j                  |j                  |
j                  |
j                  g}t        j                  j                  |d
¬«      5  | j/                  |
||||	«       ddd«       |
S | j/                  |
||||	«       |
S # 1 sw Y   �ŒJxY w# 1 sw Y   Œ´xY w# 1 sw Y   |
S xY w)a¦  
        Build a resized Linear Module from a provided old Linear Module. Increasing the size will add newly initialized
        vectors at the end. Reducing the size will remove vectors from the end

        Args:
            old_lm_head (`torch.nn.Linear`):
                Old lm head liner layer to be resized.
            new_num_tokens (`int`, *optional*):
                New number of tokens in the linear matrix.

                Increasing the size will add newly initialized vectors at the end. Reducing the size will remove
                vectors from the end. If not provided or `None`, just returns a pointer to the input tokens
                `torch.nn.Linear` module of the model without doing anything. transposed (`bool`, *optional*, defaults
                to `False`): Whether `old_lm_head` is transposed or not. If True `old_lm_head.size()` is `lm_head_dim,
                vocab_size` else `vocab_size, lm_head_dim`.
            mean_resizing (`bool`):
                Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
                covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

                Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
                where the generated tokens' probabilities will not be affected by the added embeddings because initializing the new embeddings with the
                old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
                Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html

        Return:
            `torch.nn.Linear`: Pointer to the resized Linear Module or the old Linear Module if `new_num_tokens` is
            `None`
        Nr¡  ry  z#Old language model head is of type r�  zg. You should either use a different resize function or make sure that `old_lm_head` are an instance of rþ   )rÛ  rÙ   rå   a  The new lm_head weights will be initialized from a multivariate normal distribution that has old embeddings' mean and covariance. As described in this article: https://nlp.stanford.edu/~johnhew/vocab-expansion.html. To disable this, use `mean_resizing=False`r   )r¨  r¡  r)   rý  rþ  r|  rÆ  rU  rð   rb  r   ÚLinearrŽ  rÓ  rÛ  rÙ   rå   r¬   r  r  Ú%_init_added_lm_head_weights_with_meanÚ"_init_added_lm_head_bias_with_meanr|  Ú!_copy_lm_head_original_to_resized)ro  r‰  ru  r•  rw  rJ  r‘  Úold_lm_head_dimÚnew_lm_head_shapeÚhas_new_lm_head_biasrŠ  r“  r”  Únum_tokens_to_copys                 r‹   rƒ  z$PreTrainedModel._get_resized_lm_head3  si  € ðF Ð!ØÐä˜t ^Ó4ÒV¸×9JÑ9JÐRVÐ9VˆÜ%Ô'±Ü—‘×2Ñ2°;×3EÑ3EÐUYÐ2ÓZñ á5?�K×&Ñ&×+Ñ+Ô-À[×EWÑEW×EYÑEYÓE[×E`ÑE`ÓEbñ 0� ÷ð ñ 2<�×"Ñ"×'Ñ'Ô)À×ASÑAS×AUÑAUÓAW×A\ÑA\ÓA^ñ ,ˆN˜Oð ˜^Ò+Ô4NÔ4PØÐä˜+¤r§y¡yÔ1ÜØ5´d¸;Ó6GÐ5HÐHfÔgi×gpÑgpÐfqð rä—I‘I�;˜að!óð ñ FP˜_¨nÑ=ÐVdÐfuÐUvÐØ*×/Ñ/°tÐ;Ðô —i‘iØØ%Ø×%Ñ%×,Ñ,Ø×$Ñ$×*Ñ*ò	
ˆð ˜NÒ*±=à×Ñ˜{Õ+à˜nÒ,±ô ×Ñð=ôð  .°Ñ>ÐÜ)Ô+±LØ%×,Ñ,Ð-�Ù'Ø˜{×/Ñ/Ð0Ñ0�FÜ—^‘^×6Ñ6°vÈTÐ6ÓRñ lØ×>Ñ>Ø# [°/À>ÐScÐeoôñ ,Ø×?Ñ?ÀÈ[ÐZjÔk÷lð lð ×:Ñ:Ø ¨o¸~ÐO_Ðakôñ (Ø×;Ñ;¸KÈÐVfÔgä  °Ó@Ðä%Ô'±Ø!×(Ñ(¨+×*:Ñ*:¸K×<NÑ<NÐP[×P`ÑP`ÐaˆFÜ—‘×2Ñ2°6ÈÐ2ÓKñ Ø×6Ñ6Ø Ð.@À*ÐNbô÷ð Ðð	 ×2Ñ2Ø˜[Ð*<¸jÐJ^ôð Ð÷añ ú÷jlð lú÷$ð Ðús%   ÁAL0È2,L=Ë9M	Ì0L:Ì=MÍ	Mc                 óð  — |j                   j                  j                  t        j                  «      }t        j
                  |d¬«      }||z
  }|j                  |z  |z  }	d}
t        j                  j                  |
|	z  «      j                  «       }|r…t        j                  j                  j                  ||
|	z  ¬«      }|j                  |f¬«      j                  |j                   j                  «      |j                   j                  d|z  d …d d …f<   y |d d d …f   j!                  |d«      j                  |j                   j                  «      |j                   j                  d|z  d …d d …f<   y )Nr   r‚  ç•Ö&è.>)Úcovariance_matrix)Úsample_shaper�   r    )rÆ  rÔ  rÍ  r‚   rí   ÚmeanÚTr   Úpositive_definiteÚcheckÚallÚdistributionsÚmultivariate_normalÚMultivariateNormalÚsamplerå   rˆ  )ro  r…  r†  r’  r‘  r“  Úold_embeddings_weightÚmean_embeddingsÚold_centered_embeddingsÚ
covarianceÚepsilonÚis_covariance_psdÚdistributions                r‹   r�  z8PreTrainedModel._init_added_embeddings_weights_with_mean­  sc  € ð !/× 5Ñ 5× :Ñ :× =Ñ =¼e¿m¹mÓ LÐÜŸ*™*Ð%:ÀÔCˆØ"7¸/Ñ"IÐØ,×.Ñ.Ð1HÑHÈ>ÑYˆ
ð ˆÜ'×9Ñ9×?Ñ?ÀÈ*Ñ@TÓU×YÑYÓ[ÐÙä ×.Ñ.×BÑB×UÑUØ°7¸ZÑ3Gð Vó ˆLð FR×EXÑEXØ.Ð0ð FYó Fç‰b�×&Ñ&×,Ñ,Ó-ð ×!Ñ!×&Ñ& rÐ,<Ñ'<Ñ'>ÂÐ'AÒBð   ¢a Ñ(×/Ñ/Ð0@À!ÓD×GÑGÈ×H]ÑH]×HcÑHcÓdð ×!Ñ!×&Ñ& rÐ,<Ñ'<Ñ'>ÂÐ'AÒBrŠ   c                 ó°  — |r^|j                   j                  j                  |j                   _        |j                   j                  j                  |j                   _        | j                  |||||«       |r_|j                   j                  j                  |j                   _        |j                   j                  j                  |j                   _        y y r¨   )rÆ  rÔ  r¤  r�  )ro  r‰  rŠ  r›  r‘  r“  r•  s          r‹   r˜  z5PreTrainedModel._init_added_lm_head_weights_with_meanÆ  s­   € ñ à&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÔ#Ø&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÔ#ð 	×5Ñ5Ø˜ o°~ÐGWô	
ñ à&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÔ#Ø&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÕ#ð rŠ   c                 óh  — t        j                  |j                  j                  dt         j                  ¬«      }t        j
                  |j                  j                  d¬«      j                  t         j                  «      }|j                  j                  d|z  d  j                  |d|z  ¬«       y )Nr   )rƒ  rå   r‚  r�   r   )r£  Ústd)r‚   r£  rÛ  rÔ  rí   rµ  rÍ  r™   )ro  r‰  rŠ  r“  Ú	bias_meanÚbias_stds         r‹   r™  z2PreTrainedModel._init_added_lm_head_bias_with_meanÞ  sƒ   € Ü—J‘J˜{×/Ñ/×4Ñ4¸1ÄEÇMÁMÔRˆ	Ü—9‘9˜[×-Ñ-×2Ñ2¸Ô;×>Ñ>¼u¿}¹}ÓMˆØ×Ñ×Ñ˜bÐ#3Ñ3Ð5Ð6×>Ñ>ÀIÐSWÐZbÑSbÐ>ÕcrŠ   c                 ó`  — |s=|j                   j                  d |…d d …f   |j                   j                  d |…d d …f<   n<|j                   j                  d d …d |…f   |j                   j                  d d …d |…f<   |r1|j                  j                  d | |j                  j                  d | y y r¨   )rÆ  rÔ  rÛ  )ro  rŠ  r‰  rž  r•  r�  s         r‹   rš  z1PreTrainedModel._copy_lm_head_original_to_resizedã  sÀ   € ñ Ø>I×>PÑ>P×>UÑ>UÐViÐWiÐViÒklÐVlÑ>mˆK×Ñ×#Ñ#Ð$7Ð%7Ð$7ºÐ$:Ò;à>I×>PÑ>P×>UÑ>UÒVWÐYlÐZlÐYlÐVlÑ>mˆK×Ñ×#Ñ#¢AÐ':Ð(:Ð':Ð$:Ñ;ñ  Ø9D×9IÑ9I×9NÑ9NÐObÐPbÐ9cˆK×Ñ×!Ñ!Ð"5Ð#5Ñ6ð  rŠ   Únew_num_position_embeddingsc           	      ó|   — t        d| j                  › d| j                  › d| j                  j                  › d�«      ‚)Nz4`resize_position_embeddings` is not implemented for úB`. To implement it, you should overwrite this method in the class ú in `modeling_ú.py`©r=  r  rV  )ro  r¹  s     r‹   Úresize_position_embeddingsz*PreTrainedModel.resize_position_embeddingsð  sH   € Ü!ØBÀ4Ç>Á>ÐBRð S2Ø26·.±.Ð1AÀÐPT×P^ÑP^×PiÑPiÐOjÐjnðpó
ð 	
rŠ   c           	      ó|   — t        d| j                  › d| j                  › d| j                  j                  › d�«      ‚)Nz1`get_position_embeddings` is not implemented for r»  r¼  r½  r¾  rs  s    r‹   Úget_position_embeddingsz'PreTrainedModel.get_position_embeddingsö  sH   € Ü!Ø?ÀÇÁÐ?Oð P2Ø26·.±.Ð1AÀÐPT×P^ÑP^×PiÑPiÐOjÐjnðpó
ð 	
rŠ   c                 óÞ   — | j                   j                  r%| j                  | j                   j                  «       t        r,| j	                  | j
                  «       | j                  «        yy)z±
        If needed prunes and maybe initializes weights. If using a custom `PreTrainedModel`, you need to implement any
        initialization logic in `_init_weights`.
        N)rª  Úpruned_headsÚprune_headsr¬   ÚapplyrD  rR  rs  s    r‹   rÝ  zPreTrainedModel.init_weightsü  sQ   € ð �;‰;×#Ò#Ø×Ñ˜TŸ[™[×5Ñ5Ô6åà�J‰J�t×/Ñ/Ô0ð ×ÑÕð rŠ   Úheads_to_prunec                 ó$  — |j                  «       D ]b  \  }}t        | j                  j                  j	                  |g «      «      t        |«      z  }t        |«      | j                  j                  |<   Œd | j                  j                  |«       y)a   
        Prunes heads of the base model.

        Arguments:
            heads_to_prune (`Dict[int, List[int]]`):
                Dictionary with keys being selected layer indices (`int`) and associated values being the list of heads
                to prune in said layer (list of `int`). For instance {1: [0, 2], 2: [2, 3]} will prune heads 0 and 2 on
                layer 1 and heads 2 and 3 on layer 2.
        N)r®   r  rª  rÃ  rˆ   r  r  Ú_prune_heads)ro  rÆ  ÚlayerÚheadsÚunion_headss        r‹   rÄ  zPreTrainedModel.prune_heads  sy   € ð +×0Ñ0Ó2ò 	@‰LˆE�5Ü˜dŸk™k×6Ñ6×:Ñ:¸5À"ÓEÓFÌÈUËÑSˆKÜ.2°;Ó.?ˆD�K‰K×$Ñ$ UÒ+ð	@ð 	�‰×$Ñ$ ^Õ4rŠ   c                 óÔ  — | j                   s"t        | j                  j                  › d�«      ‚|€ddi}t	        j
                  t        fi |¤Ž}dt        j                  | j                  «      j                  v }|s| j                  d|¬«       n;| j                  t        | j                  d¬«      «       t        j                  d«       t        | d	d
«      r| j                  «        yy)az  
        Activates gradient checkpointing for the current model.

        Note that in other frameworks this feature can be referred to as "activation checkpointing" or "checkpoint
        activations".

        We pass the `__call__` method of the modules instead of `forward` because `__call__` attaches all the hooks of
        the module. https://discuss.pytorch.org/t/any-different-between-model-input-and-model-forward-input/3690/2

        Args:
            gradient_checkpointing_kwargs (dict, *optional*):
                Additional keyword arguments passed along to the `torch.utils.checkpoint.checkpoint` function.
        z) does not support gradient checkpointing.NÚuse_reentrantTrß  )ÚenableÚgradient_checkpointing_func©rß  áV  You are using an old version of the checkpointing format that is deprecated (We will also silently ignore `gradient_checkpointing_kwargs` in case you passed it).Please update to the new format on your modeling file. To use the new format, you need to completely remove the definition of the method `_set_gradient_checkpointing` in your model.Ú_hf_peft_config_loadedF)rì  rö   r  r  Ú	functoolsr   r   r%  r&  Ú_set_gradient_checkpointingrØ   rÅ  r  r  r€  r8  )ro  Úgradient_checkpointing_kwargsrÏ  Ú_is_using_old_formats       r‹   rí  z-PreTrainedModel.gradient_checkpointing_enable  sÛ   € ð ×3Ò3Ü §¡× 7Ñ 7Ð8Ð8aÐbÓcÐcà(Ð0Ø-<¸dÐ,CÐ)ä&/×&7Ñ&7¼
Ñ&dÐFcÑ&dÐ#ð  '¬'×*;Ñ*;¸D×<\Ñ<\Ó*]×*hÑ*hÐhÐá#Ø×,Ñ,°DÐVqÐ,Õrà�J‰J”w˜t×?Ñ?ÀtÔLÔMÜ�N‰NðHôô
 �4Ð1°5Ô9ð
 ×+Ñ+Õ-ð :rŠ   rÎ  rÏ  c                 óì   — d}t        | d«      r|| _        || _        d}| j                  «       D ]  }t        |d«      sŒ||_        ||_        d}Œ! |s"t	        | j
                  j                  › d�«      ‚y )NFrë  TzÂ is not compatible with gradient checkpointing. Make sure all the architecture support it by setting a boolean attribute `gradient_checkpointing` to modules of the model that uses checkpointing.)r¨  Ú_gradient_checkpointing_funcrë  r  rö   r  r  )ro  rÎ  rÏ  Úis_gradient_checkpointing_setrÈ   s        r‹   rÔ  z+PreTrainedModel._set_gradient_checkpointingH  s”   € Ø(-Ð%ô �4Ð1Ô2Ø0KˆDÔ-Ø*0ˆDÔ'Ø,0Ð)à—l‘l“nò 	5ˆFÜ�vÐ7Õ8Ø6Q�Ô3Ø06�Ô-Ø04Ñ-ð		5ñ -ÜØ—>‘>×*Ñ*Ð+ð ,]ð ]óð ð -rŠ   c                 óN  — | j                   r{dt        j                  | j                  «      j                  v }|s| j                  d¬«       n;t
        j                  d«       | j                  t        | j                  d¬«      «       t        | dd«      r| j                  «        yy)zÕ
        Deactivates gradient checkpointing for the current model.

        Note that in other frameworks this feature can be referred to as "activation checkpointing" or "checkpoint
        activations".
        rß  F)rÎ  rÑ  rÐ  rÒ  N)rì  r%  r&  rÔ  rØ   r  r  rÅ  r   r€  r;  )ro  rÖ  s     r‹   Úgradient_checkpointing_disablez.PreTrainedModel.gradient_checkpointing_disable^  s�   € ð ×/Ò/ð $+¬g×.?Ñ.?À×@`Ñ@`Ó.a×.lÑ.lÐ#lÐ Ù'Ø×0Ñ0¸Ð0Õ>ä—‘ðLôð —
‘
œ7 4×#CÑ#CÈ5ÔQÔRä�4Ð1°5Ô9Ø×,Ñ,Õ.ð :rŠ   c                 óB   — t        d„ | j                  «       D «       «      S )zÞ
        Whether gradient checkpointing is activated for this model or not.

        Note that in other frameworks this feature can be referred to as "activation checkpointing" or "checkpoint
        activations".
        c              3   óP   K  — | ]  }t        |d «      xr |j                  –— Œ  y­w)rë  N)r¨  rë  )r4  Úms     r‹   r6  z<PreTrainedModel.is_gradient_checkpointing.<locals>.<genexpr>}  s'   è ø€ ÒmÐYZ”7˜1Ð6Ó7ÒT¸A×<TÑ<TÓTÑmùs   ‚$&)r;  r  rs  s    r‹   Úis_gradient_checkpointingz)PreTrainedModel.is_gradient_checkpointingu  s   € ô ÑmÐ^b×^jÑ^jÓ^lÔmÓmÐmrŠ   Ú5GBÚsave_directoryÚis_main_processrø   Úsave_functionÚpush_to_hubÚmax_shard_sizeÚsafe_serializationrä  rò  Úsave_peft_formatc           	      óú  ‡I‡J— |j                  dd«      }|j                  dd«      }|�)t        j                  dt        «       |	�t	        d«      ‚|}	|	�|	|d<   t        | dd«      }t        | d	d«      }|duxr$ t        |t        «      xr |j                  |¬
«      }|�'|s%|s#t	        d|j                  j                  › d�«      ‚d|v r&t        j                  d«       |j                  d«      }|rt        «       st        d«      ‚t        j                  j                  |«      rt         j#                  d|› d�«       yt        j$                  |d¬«       |rr|j                  dd«      }|j                  d|j'                  t        j                  j(                  «      d   «      } | j*                  |fi |¤Ž}| j-                  |«      }t/        | «      }t1        |«      }t3        |«      j'                  d«      d   |j4                  _        |j8                  j:                  g|j4                  _        d|j4                  _        | j@                  �tC        | || j4                  ¬«       |�r–|s·|j4                  jE                  «       }| jG                  «       rrtI        |«      dkD  rdt        j                  d|› d�tJ        «       |jM                  «       D ]3  \  }}tO        |jP                  ||«       tO        |j4                  |d«       Œ5 |j4                  jS                  |«       | jG                  «       r|jP                  jS                  |«       |r°t         jU                  d«       |jW                  |¬«      }|
r9t         jU                  d«       i }|jM                  «       D ]  \  }}||d |› �<   Œ |}| jY                  «       }tI        |«      dkD  rt	        d!«      ‚|d   }| jZ                  |   }|jS                  |«       i }|€Øt]        | d"«      r¼tI        t_        | j`                  jc                  «       «      «      dkD  r�d#| j`                  jc                  «       v sd$| j`                  jc                  «       v rUt        j                  d%«       |je                  «       D ]-  \  ŠI}‰Id&k(  rŒ|jg                  «       } | D ]  }||‰Id|› �z   <   Œ Œ/ |jg                  «       }th        r4tj        jl                  jn                  jp                  D ]  \  }!}" |!|«      }Œ | jr                  �'| jr                  D ]  }#|#|ju                  «       v sŒ||#= Œ | jw                  |«      }|�rity        jz                  t|        «      }$|jM                  «       D ]Z  \  ŠI}%t        |%t~        j€                  «      r|$tƒ        |%«         j…                  ‰I«       Œ>|$t‡        |%«         j…                  ‰I«       Œ\ t]        | d"«      rNt‰        | «      }&|&r>|&d   ŠJ|$jM                  «       D �'�(ci c]  \  }'}(t‹        ˆJfd'„|(D «       «      sŒ|'|(“Œ })}'}(n5i })n2|$jM                  «       D �'�(ci c]  \  }'}(tI        |(«      dkD  sŒ|'|(“Œ })}'}(t�        | «      }*g }+t_        «       },|)jc                  «       D ]X  }(|*€Œd}-t�        |(«      D ]C  ŠIt‹        ˆIfd(„|*D «       «      }.|.sŒ‰I|v sŒ|-dz  }-|-tI        |(«      k  sŒ3|,j‘                  ‰I«       ŒE ŒZ t“        |)jc                  «       |«      \  }/}0|0D ]  ŠI|‰I   j•                  «       |‰I<   Œ t—        |/|«      \  }/}1|1D ]N  }2|2j™                  |,«      }3|3D ]  ŠI|‰I= Œ |2j›                  |,«      }4tI        |4«      dkD  sŒ>|+j…                  |4«       ŒP |/r|+j…                  t_        |/«      «       tI        |+«      dkD  rt�        d)|+› d*�«      ‚|s|rtž        nt         }5t£        |5|«      }5n|rt¤        nt¦        }5|5j©                  d+d,«      j©                  d-d.«      }6t«        ||6|¬/«      }7d}8|7j¬                  r|7j®                  |7j°                  d0œ}8t        j²                  |«      D ]ô  }9t        j                  jµ                  ||9«      }:|5j©                  d+d&«      j©                  d-d&«      };|9j©                  d+d&«      j©                  d-d&«      }<t·        j¸                  d1«      }=|9j»                  |;«      sŒŽt        j                  j                  |:«      sŒ®|9|7j¼                  ju                  «       vsŒË|sŒÎ|=j¿                  |<«      €Œàt        jÀ                  |:«       Œö |7j¼                  jM                  «       }>|rtÃ        jÄ                  |>d2¬3«      }>|>D �]  \  }?}@i }A|@D ]  }%||%   jÇ                  «       A|%<   ||%= Œ |rŠtÈ        tË        jÌ                  d4«      k  rt        d5tÈ        › d6�«      ‚tÎ        jÑ                  Ad&«      }B|AD ])  }CBjÓ                  |C«      d&k7  rŒ|C   }tÕ        ||CB«      }BŒ+ B}A~Bt×        jØ                  «        |r/tÛ        At        j                  jµ                  ||?«      d7d8i¬9«       Œæ |At        j                  jµ                  ||?«      «       �Œ ~|8€9t        j                  jµ                  ||5«      }Dt         jU                  d:|D› �«       n­|rtÜ        ntÞ        }Et        j                  jµ                  |t£        |E|«      «      }Etá        |Ed;d<¬=«      5 }Ftã        jä                  |8d>d¬?«      d@z   }G|Fjç                  |G«       ddd«       t         jU                  dA|› dBtI        |7j¼                  «      › dCE› d�«       |r_té        | jê                  |	|¬D«      }H|Hjí                  t        j                  jµ                  |dE«      «       | jï                  |||	¬F«       yyc c}(}'w c c}(}'w # 1 sw Y   Œ©xY w)Gaƒ  
        Save a model and its configuration file to a directory, so that it can be re-loaded using the
        [`~PreTrainedModel.from_pretrained`] class method.

        Arguments:
            save_directory (`str` or `os.PathLike`):
                Directory to which to save. Will be created if it doesn't exist.
            is_main_process (`bool`, *optional*, defaults to `True`):
                Whether the process calling this is the main process or not. Useful when in distributed training like
                TPUs and need to call this function on all processes. In this case, set `is_main_process=True` only on
                the main process to avoid race conditions.
            state_dict (nested dictionary of `torch.Tensor`):
                The state dictionary of the model to save. Will default to `self.state_dict()`, but can be used to only
                save parts of the model or if special precautions need to be taken when recovering the state dictionary
                of a model (like when using model parallelism).
            save_function (`Callable`):
                The function to use to save the state dictionary. Useful on distributed training like TPUs when one
                need to replace `torch.save` by another method.
            push_to_hub (`bool`, *optional*, defaults to `False`):
                Whether or not to push your model to the Hugging Face model hub after saving it. You can specify the
                repository you want to push to with `repo_id` (will default to the name of `save_directory` in your
                namespace).
            max_shard_size (`int` or `str`, *optional*, defaults to `"5GB"`):
                The maximum size for a checkpoint before being sharded. Checkpoints shard will then be each of size
                lower than this size. If expressed as a string, needs to be digits followed by a unit (like `"5MB"`).
                We default it to 5GB in order for models to be able to run easily on free-tier google colab instances
                without CPU OOM issues.

                <Tip warning={true}>

                If a single weight of the model is bigger than `max_shard_size`, it will be in its own checkpoint shard
                which will be bigger than `max_shard_size`.

                </Tip>

            safe_serialization (`bool`, *optional*, defaults to `True`):
                Whether to save the model using `safetensors` or the traditional PyTorch way (that uses `pickle`).
            variant (`str`, *optional*):
                If specified, weights are saved in the format pytorch_model.<variant>.bin.
            token (`str` or `bool`, *optional*):
                The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
                the token generated when running `huggingface-cli login` (stored in `~/.huggingface`).
            save_peft_format (`bool`, *optional*, defaults to `True`):
                For backward compatibility with PEFT library, in case adapter weights are attached to the model, all
                keys of the state dict of adapters needs to be pre-pended with `base_model.model`. Advanced users can
                disable this behaviours by setting `save_peft_format` to `False`.
            kwargs (`Dict[str, Any]`, *optional*):
                Additional key word arguments passed along to the [`~utils.PushToHubMixin.push_to_hub`] method.
        Úuse_auth_tokenNÚignore_metadata_errorsFúrThe `use_auth_token` argument is deprecated and will be removed in v5 of Transformers. Please use `token` instead.úV`token` and `use_auth_token` are both specified. Please set only the argument `token`.rò  rÒ  r¡  )ræ  zThe model is quantized with z˜ and is not serializable - check out the warnings from the logger on the traceback to understand the reason why the quantized model is not serializable.Úsave_configze`save_config` is deprecated and will be removed in v5 of Transformers. Use `is_main_process` instead.zR`safe_serialization` requires the `safetensors library: `pip install safetensors`.zProvided path (z#) should be a directory, not a fileT©Úexist_okÚcommit_messageÚrepo_idr�   rþ   r    ©rª  r   zHMoving the following attributes in the config to the generation config: zƒ. You are seeing this warning because you've set generation parameters in the model config, as opposed to in the generation config.zhDetected adapters on the model, saving the model in the PEFT format, only adapter weights will be saved.)rø   z„To match the expected format of the PEFT library, all keys of the state dict of adapters will be pre-pended with `base_model.model`.zbase_model.model.zßMultiple active adapters detected, saving multiple active adapters is not supported yet. You can save adapters separately one by one by iteratively calling `model.set_adapter(adapter_name)` then `model.save_pretrained(...)`Úhf_device_mapr  rÄ  z|Attempting to save a model with offloaded modules. Ensure that unallocated cpu memory exceeds the `shard_size` (5GB default)rk  c              3   ó&   •K  — | ]  }|‰v –— Œ
 y ­wr¨   r‰   )r4  r²   Ú
tied_namess     €r‹   r6  z2PreTrainedModel.save_pretrained.<locals>.<genexpr>n  s   øè ø€ ÒHnÐ`dÈÐQ[ÔI[ÑHnùs   ƒc              3   óJ   •K  — | ]  }t        j                  |‰«      –— Œ y ­wr¨   )rÊ  r©  )r4  Úpatr²   s     €r‹   r6  z2PreTrainedModel.save_pretrained.<locals>.<genexpr>  s   øè ø€ Ò-aÀs¬b¯i©i¸¸T×.BÑ-aùs   ƒ #z8The weights trying to be saved contained shared tensors z… that are mismatching the transformers base configuration. Try saving using `safe_serialization=False` or remove this tensor sharing.z.binz{suffix}.binrL  z{suffix}.safetensors)Úfilename_patternrå  )rY  r  z(.*?)-\d{5}-of-\d{5}zSaving checkpoint shards©Údescry   zxYou need accelerate version to be greater or equal than 0.31 to save models with offloaded parameters. Detected version z<. Please upgrade accelerate with `pip install -U accelerate`rP  rM  )rY  zModel weights saved in Úwr   r  rŠ  )ÚindentÚ	sort_keysú
z:The model is bigger than the maximum size per checkpoint (z) and is going to be split in z^ checkpoint shards. You can find where each parameters has been saved in the index located at )rò  rê  z	README.md)rð  rò  )xrŽ  r„  r…  r†  rö   r€  rb  r<   Úis_serializabler¥  r¦  rY   r]  r†   r  r  r  ÚerrorÚmakedirsrß  ÚsepÚ_create_repoÚ_get_files_timestampsÚunwrap_modelró   rÜ   rª  r  r  r  ÚarchitecturesrÊ  Ú_auto_classr#   Ú&_get_non_default_generation_parametersrÓ  r  ÚUserWarningr®   r¯   rÕ  Úsave_pretrainedr   Úget_adapter_state_dictÚactive_adaptersÚpeft_configr¨  r  ró  rõ   rl  rø   ÚIS_SAGEMAKER_MP_POST_1_10ÚsmpÚstateÚmodule_managerÚtranslate_functionsÚ_keys_to_ignore_on_saver  Ú_fix_state_dict_keys_on_saverš  r   r  r‚   r   r7   r‹  Úidr*   r;  rƒ  rÉ  r�  r˜  rn  r�  ÚintersectionÚ
differencer  rE   rI   rç  r?   r@   Úreplacer   r  rY  Útensor_to_filenameÚlistdirr  rÊ  Úcompiler9  Úfilename_to_tensorsÚ	fullmatchr:  r`   ÚtqdmrÎ  Úaccelerate_versionr   rc  r  Úfromkeysrˆ   rz   r  r  Úsafe_save_filerD   rH   r  r  ÚdumpsÚwriterc   rñ  ÚsaveÚ_upload_modified_files)Kro  rá  râ  rø   rã  rä  rå  ræ  rä  rò  rç  rª   ré  rê  rÒ  r¡  Úquantization_serializablerð  rñ  Úfiles_timestampsÚmodel_to_saverå   Úmisplaced_generation_parametersrž  Úparam_valueÚpeft_state_dictr0  rß  Úactive_adapterÚcurrent_peft_configÚ
module_maprÈ   Úmodule_state_dictÚ	smp_to_hfr’  Ú
ignore_keyÚptrsrt  r,  Úptrr…  Úshared_ptrsr}  Úerror_namesÚto_delete_namesÚfoundÚmatches_patternÚshared_namesÚdisjoint_namesÚidentical_namesÚinamesÚknownÚunknownrã  rø  Ústate_dict_splitr,  r  Úfull_filenameÚweights_no_suffixÚfilename_no_suffixÚregr  r7  rˆ  ÚshardÚshard_state_dictrq  Úpath_to_weightsÚsave_index_filer+  ÚcontentÚ
model_cardr²   rõ  sK                                                                            @@r‹   r
  zPreTrainedModel.save_pretrained  s  ù€ ð~  Ÿ™Ð$4°dÓ;ˆØ!'§¡Ð,DÀeÓ!LÐàÐ%Ü�M‰Mð EÜôð Ð Ü Ølóð ð #ˆEàÐØ#ˆF�7‰Oä!(¨Ð/GÈÓ!OÐä˜t ^°TÓ:ˆà Ð$ò TÜ˜<¬Ó5òTà×,Ñ,Ð@RÐ,ÓSð 	"ð Ð#Ñ,BÑKdÜØ.¨|×/OÑ/O×/\Ñ/\Ð.]ð ^uð uóð ð
 ˜FÑ"Ü�M‰MØwôð %Ÿj™j¨Ó7ˆOÙÔ&>Ô&@ÜÐrÓsÐsä�7‰7�>‰>˜.Ô)Ü�L‰L˜?¨>Ð*:Ð:]Ð^Ô_Øä
�‰�N¨TÕ2áØ#ŸZ™ZÐ(8¸$Ó?ˆNØ—j‘j ¨N×,@Ñ,@ÄÇÁÇÁÓ,MÈbÑ,QÓRˆGØ'�d×'Ñ'¨Ñ:°6Ñ:ˆGØ#×9Ñ9¸.ÓIÐô % TÓ*ˆô $ MÓ2ˆÜ+.¨u«:×+;Ñ+;¸CÓ+@ÀÑ+Cˆ×ÑÔ(ð /<×.EÑ.E×.NÑ.NÐ-Oˆ×ÑÔ*ð =Bˆ×ÑÔ9ð ×ÑÐ'Ü˜t ^¸D¿K¹KÕHò Ù)à2?×2FÑ2F×2mÑ2mÓ2oÐ/Ø×$Ñ$Ô&¬3Ð/NÓ+OÐRSÒ+SÜ—M‘MØbØ:Ð;ð <mðmô $ô	ð 4S×3XÑ3XÓ3Zò HÑ/˜
 KÜ × ?Ñ ?ÀÈ[ÔYÜ × 4Ñ 4°jÀ$ÕGðHð ×$Ñ$×4Ñ4°^ÔDØ× Ñ Ô"Ø×/Ñ/×?Ñ?ÀÔOá%Ü—‘Ø~ôð +×AÑAÈZÐAÓX�
á#Ü—K‘Kð _ôð ')�OØ&0×&6Ñ&6Ó&8ò K™
˜˜UØEJ˜Ð*;¸C¸5Ð(AÒBðKà!0�Jà!%×!5Ñ!5Ó!7�ä�~Ó&¨Ò*Ü$ðuóð ð "0°Ñ!2�à&*×&6Ñ&6°~Ñ&FÐ#Ø#×3Ñ3°NÔCð ˆ
ð Ðô ˜˜oÔ.Üœ˜D×.Ñ.×5Ñ5Ó7Ó8Ó9¸AÒ=Ø˜d×0Ñ0×7Ñ7Ó9Ñ9¸VÀt×GYÑGY×G`ÑG`ÓGbÑ=bä—‘ð Sôð %2×$?Ñ$?Ó$Aò >‘L�D˜&Ø˜r’zØ Ø(.×(9Ñ(9Ó(;Ð%à0ò >˜Ø7=˜
 4¨A¨c¨U¨)Ñ#3Ò4ñ>ð>ð '×1Ñ1Ó3ˆJõ %Ü #§	¡	× 8Ñ 8× LÑ Lò 3‘�	˜1Ù& zÓ2‘
ð3ð ×'Ñ'Ð3Ø"×:Ñ:ò /�
Ø §¡Ó!2Ò2Ø" :Ñ.ð/ð ×6Ñ6°zÓBˆ
âô ×*Ñ*¬4Ó0ˆDØ *× 0Ñ 0Ó 2ò 2‘��fô ˜f¤e§l¡lÔ3ØÔ*¨6Ó2Ñ3×:Ñ:¸4Õ@ð œ˜F›Ñ$×+Ñ+¨DÕ1ð2ô �t˜_Ô-ä2°4Ó8�ÙØ!,¨Q¡�Jà59·Z±Z³\÷#Ù'1 s¨EÄSÓHnÐhmÔHnÕEn˜˜U™
ð#�Kò #ð #%‘Kà<@¿J¹J»L×[©j¨c°5ÌCÐPUËJÐYZËN˜s E™zÐ[�Ñ[ô "7°tÓ!<ÐØˆKÜ!›eˆOØ$×+Ñ+Ó-ò 
:�ð &Ñ1Ø�EÜ & u£ò :˜Ü*-Ó-aÐN`Ô-aÓ*a˜Ú*¨t°zÒ/AØ! Q™J˜EØ$¤s¨5£zÓ1Ø /× 3Ñ 3°DÕ 9ñ:ð
:ô ,:¸+×:LÑ:LÓ:NÐPZÓ+[Ñ(ˆL˜.ð 'ò <�Ø#-¨dÑ#3×#9Ñ#9Ó#;�
˜4Ò ð<ô -<¸LÈ*Ó,UÑ)ˆL˜/à)ò 0�Ø×+Ñ+¨OÓ<�Ø!ò )�DØ" 4Ñ(ð)à ×+Ñ+¨OÓ<�Ü�w“< !Ó#Ø×&Ñ& wÕ/ð0ñ Ø×"Ñ"¤3 |Ó#4Ô5ä�;Ó !Ò#Ü"ØNÈ{Èmð  \að  bóð ñ
 &Ù0BÕ,ÌˆLÜ'¨°gÓ>‰Lá8JÕ4ÔPdˆLà'×/Ñ/°¸ÓG×OÑOÐP^Ð`vÓwÐÜ=ØÐ)9È.ô
Ðð ˆØ×&Ò&à,×5Ñ5Ø.×AÑAñˆEô Ÿ
™
 >Ó2ò 	)ˆHÜŸG™GŸL™L¨¸ÓBˆMð !-× 4Ñ 4°V¸RÓ @× HÑ HÈÐY[Ó \Ðð "*×!1Ñ!1°&¸"Ó!=×!EÑ!EÀnÐVXÓ!YÐÜ—*‘*Ð4Ó5ˆCð ×#Ñ#Ð$5Õ6Ü—G‘G—N‘N =Õ1ØÐ$4×$HÑ$H×$MÑ$MÓ$OÒOÚ#Ø—M‘MÐ"4Ó5ÑAä—	‘	˜-Õ(ð#	)ð& /×BÑB×HÑHÓJÐÙÜ")§,¡,Ð/BÐIcÔ"dÐØ#6ó "	OÑˆJ˜ØˆEØ!ò '�Ø *¨6Ñ 2× =Ñ =Ó ?��f‘à˜vÑ&ð'ñ Ü%¬¯©°fÓ(=Ò=Ü%ð Sô  Tfð  Sgð gUð Vóð ô
 $(§=¡=°¸Ó#;Ð Ø#(ò j�Kà'×+Ñ+¨KÓ8¸BÒ>Ø Ø'¨Ñ4�Fä'BÀ6È;ÐXhÓ'iÑ$ðjð )�Ø$Ü—
‘
”á!ô ˜u¤b§g¡g§l¡l°>À:Ó&NÐZbÐdhÐYiÖjá˜e¤R§W¡W§\¡\°.À*Ó%MÖNðE"	OðH àˆ=Ü Ÿg™gŸl™l¨>¸<ÓHˆOÜ�K‰KÐ1°/Ð1BÐCÕDá9KÕ5ÔQcˆOÜ Ÿg™gŸl™l¨>¼<ÈÐY`Ó;aÓbˆOä�o s°WÔ=ð !ÀÜŸ*™* U°1ÀÔEÈÑL�Ø—‘˜Ô ÷!ô �K‰KØLÈ^ÐL\ð ]ÜÐ 0× DÑ DÓEÐFð G$Ø$3Ð#4°Að7ôñ ä2Ø˜Ÿ™°ÐNdôˆJð
 �O‰OœBŸG™GŸL™L¨¸ÓEÔFà×'Ñ'ØØØ Ø-Øð (õ ð ùók#ùó \÷L!ð !ús$   Øo%Ø2o%Ùo+Ù(o+ì-o1ï1o:c                 óè   •— | j                   �| j                   ng }|j                  dg «      }t        |t        «      r|g}|D ]  }||vsŒ|j	                  |«       Œ |r||d<   t        ‰| �  |i |¤ŽS )Nrï  )rñ  rˆ   rb  rÜ   r‹  rÍ  rä  )ro  r©   rª   rï  Útags_kwargsrò  r  s         €r‹   rä  zPreTrainedModel.push_to_hub  s}   ø€ à"&§/¡/Ð"=ˆt�ŠÀ2ˆà—j‘j ¨Ó,ˆÜ�k¤3Ô'Ø&˜-ˆKàò 	!ˆCØ˜$ŠØ—‘˜CÕ ð	!ñ Ø!ˆF�6‰NÜ‰wÑ" DÐ3¨FÑ3Ð3rŠ   c                 ó@  — t        | j                  «       D �cg c]#  }|j                  «       |j                  «       z  ‘Œ% c}«      }|rKt        | j	                  «       D �cg c]#  }|j                  «       |j                  «       z  ‘Œ% c}«      }||z   }|S c c}w c c}w )a  
        Get the memory footprint of a model. This will return the memory footprint of the current model in bytes.
        Useful to benchmark the memory footprint of the current model and design some tests. Solution inspired from the
        PyTorch discussions: https://discuss.pytorch.org/t/gpu-memory-that-model-uses/56822/2

        Arguments:
            return_buffers (`bool`, *optional*, defaults to `True`):
                Whether to return the size of the buffer tensors in the computation of the memory footprint. Buffers
                are tensors that do not require gradients and not registered as parameters. E.g. mean and std in batch
                norm layers. Please see: https://discuss.pytorch.org/t/what-pytorch-means-by-buffers/120266/2
        )r¬  rØ   rv  ry  rî   )ro  Úreturn_buffersrÜ  rd  ÚbufÚmem_bufss         r‹   Úget_memory_footprintz$PreTrainedModel.get_memory_footprint#  s~   € ô ÈÏÉÓHYÖZ¸u�5—>‘>Ó# e×&8Ñ&8Ó&:Ó:ÒZÓ[ˆÙÜÈ4Ï<É<Ë>ÖZÀC˜CŸL™L›N¨S×-=Ñ-=Ó-?Ó?ÒZÓ[ˆHØ˜‘.ˆCØˆ
ùò	 [ùâZs   ˜(BÁ (Bc                 ó¤  •— t        | dd «      t        j                  k(  rt        d«      ‚t        | dd «      t        j                  k(  rzt        | dd«      rt        d«      ‚t        j                  t        j                  j                  d«      «      t        j                  d«      k  rt        d| j                  › d	�«      ‚y t        ‰| �,  |i |¤ŽS )
NÚquantization_methodz2`.cuda` is not supported for HQQ-quantized models.Úis_loaded_in_8bitFzœCalling `cuda()` is not supported for `8-bit` quantized models.  Please use the model as it is, since the model has already been set to the correct devices.r©  ú0.43.2z‚Calling `cuda()` is not supported for `4-bit` quantized models with the installed version of bitsandbytes. The current device is `úL`. If you intended to move the model, please install bitsandbytes >= 0.43.2.)r€  rj   r§  rö   rÌ  r   rc  r%  rY  rÙ   rÍ  r  )ro  r©   rª   r  s      €r‹   r  zPreTrainedModel.cuda5  sÒ   ø€ ä�4Ð.°Ó5Ô9K×9OÑ9OÒOÜÐQÓRÐRä�4Ð.°Ó5Ô9K×9ZÑ9ZÒZÜ�tÐ0°%Ô8Ü ðsóð ô —‘œy×1Ñ1×9Ñ9¸.ÓIÓJÌWÏ]É]Ð[cÓMdÒdÜ ð.Ø.2¯k©k¨]ð  ;GðHóð ð eô ‘7‘< Ð0¨Ñ0Ð0rŠ   c                 ó¾  •— d|v }|s%|D ]   }t        |t        j                  «      sŒd} n t        | dd «      t        j
                  k(  rt        d«      ‚|r)t        | dd «      t        j                  k(  rt        d«      ‚t        | dd «      t        j                  k(  r†|rt        d«      ‚t        | dd«      rt        d	«      ‚t        j                  t        j                  j                  d
«      «      t        j                  d«      k  rDt        d| j                  › d�«      ‚t        | dd «      t        j                  k(  r|rt        d«      ‚t        ‰| �@  |i |¤ŽS )Nrå   TrR  z0`.to` is not supported for HQQ-quantized models.zBCasting a Quark quantized model to a new `dtype` is not supported.z³You cannot cast a bitsandbytes model in a new `dtype`. Make sure to load the model using `from_pretrained` using the desired `dtype` by passing the correct `torch_dtype` argument.rS  Fzµ`.to` is not supported for `8-bit` bitsandbytes models. Please use the model as it is, since the model has already been set to the correct devices and casted to the correct `dtype`.r©  rT  z€Calling `to()` is not supported for `4-bit` quantized models with the installed version of bitsandbytes. The current device is `rU  z«You cannot cast a GPTQ model in a new `dtype`. Make sure to load the model using `from_pretrained` using the desired `dtype` by passing the correct `torch_dtype` argument.)rb  r‚   rå   r€  rj   r§  rö   ÚQUARKrÌ  r   rc  r%  rY  rÙ   ÚGPTQrÍ  rÍ  )ro  r©   rª   Údtype_present_in_argsÚargr  s        €r‹   rÍ  zPreTrainedModel.toH  sv  ø€ ð !(¨6Ð 1Ðá$Øò �Ü˜c¤5§;¡;Õ/Ø,0Ð)Ùðô
 �4Ð.°Ó5Ô9K×9OÑ9OÒOÜÐOÓPÐPá ¤W¨TÐ3HÈ$Ó%OÔSe×SkÑSkÒ%kÜÐaÓbÐbô �4Ð.°Ó5Ô9K×9ZÑ9ZÒZÙ$Ü ðVóð ô
 �tÐ0°%Ô8Ü ðlóð ô —‘œy×1Ñ1×9Ñ9¸.ÓIÓJÌWÏ]É]Ð[cÓMdÒdÜ ð.Ø.2¯k©k¨]ð  ;GðHóð ô �TÐ0°$Ó7Ô;M×;RÑ;RÒRÙ$Ü ðNóð ô ‰w‰z˜4Ð* 6Ñ*Ð*rŠ   c                 óL   •— t        | dd«      rt        d«      ‚t        ‰| �  |Ž S )NrJ  FzŽ`.half()` is not supported for quantized model. Please use the model as it is, since the model has already been casted to the correct `dtype`.)r€  rö   rÍ  Úhalf©ro  r©   r  s     €r‹   r\  zPreTrainedModel.halft  s3   ø€ ä�4˜¨Ô/ÜðIóð ô
 ‘7‘< Ð&Ð&rŠ   c                 óL   •— t        | dd«      rt        d«      ‚t        ‰| �  |Ž S )NrJ  Fz�`.float()` is not supported for quantized model. Please use the model as it is, since the model has already been casted to the correct `dtype`.)r€  rö   rÍ  rë   r]  s     €r‹   rë   zPreTrainedModel.float~  s3   ø€ ä�4˜¨Ô/ÜðIóð ô
 ‘7‘= $Ð'Ð'rŠ   rJ  r»   c                 ó^  — t        «       rŽt        «       g}|sZ|sXt        j                  d«       |j	                  t
        j                  j                  t        «       ¬«      t        «       g«       |S |r#|j	                  t        «       t        «       g«       |S t        «       t        «       g}|S )Nr÷  rø  )r)   r´   r  r   r�  rý  rþ  rÿ  r(   r¼   r+   r¸   )r  rJ  r»   r   s       r‹   Úget_init_contextz PreTrainedModel.get_init_contextˆ  sœ   € ä%Ô'Ü,Ó.Ð/ˆMáÑ(:Ü—‘Ð^Ô_Ø×$Ñ$¤i§n¡n×&9Ñ&9ÔN^ÓN`Ð&9Ó&aÔcrÓctÐ%uÔvð Ðñ Ø×$Ñ$Ô&8Ó&:Ô<OÓ<QÐ%RÔSð Ðô -Ó.Ô0BÓ0DÐEˆMàÐrŠ   rû  )	rª  rî  rG  rï  rñ  rò  rô  rí  r
  r  rè  rî  rG  rï  rñ  rô  rí  r
  c       	         óè  — |j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  d	d«      }|j                  d
d«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      }|j                  dd«      } |j                  di «      }!|j                  dd«      }"|j                  dd«      }#|j                  dd«      }$|j                  dd«      }%|j                  dd«      }&|j                  dd«      }'|j                  d d«      }(|j                  d!d«      }(|j                  d"d«      }(|j                  d#d$«      }(|j                  d%d«      }(|�|€|%�t        d&«      ‚|&�|&d'k7  rt        d(|&› d)�«      ‚|&�|�t        d*«      ‚|d'k(  r/|&€-t        t        j                  j                  d+d,«      «      rd'}&d}d})|&��t        d-«      st        d.«      ‚t        j                  j                  «       j                  }*t        j                  j                  «       sï	 t        t        j                  d/   «      }+t        t        j                  d+   «      },|*d0k(  r]t        j                  j                  d1|+|,d2¬3«       t        j                  j!                  t        t        j                  d4   «      «       nT|*d5k(  rOt        t        j                  j                  d6d,«      «      rd7nd8}-t        j                  j                  |-|+|,¬9«       |*d5k(  rdnt        j                  j%                  «       }/t        j&                  |*|/«      }0|/�G|/d,kD  rBd,dl}1t+        t        j,                  d;«      |1_        t+        t        j,                  d;«      |1_        |0}t        j                  j3                  «       },t        j                  j5                  |0j                  |,f«      })|�)t7        j8                  d<t:        «       |�t        d=«      ‚|}|�|!�	d>|!vr||!d><   |	€t=        «       sd}	|%�t?        «       st        d?«      ‚|€EtA        |tB        «      s(tE        |tF        |||||||ddd¬@«      }2tI        |2|«      }ntK        |dd«      }tM        «       ry|!j                  dAd«      }3|3€tO        |f|||||dBœ|!¤Ž}3|3�St        jP                  jS                  |3«      r4t+        |3dCdD¬E«      5 }4|}3tU        jV                  |4«      dF   }ddd«       nd}3tA        |t        j&                  «      rd|i}nQtA        |tX        «      r|dGvr	 dt        j&                  |«      i}n$tA        |t        «      r|d,k  rt        dI«      ‚d|i}|�*t]        «       rt        dJ«      ‚t?        «       st        dK«      ‚|s|r�|�t        dL«      ‚|j_                  «       D �5�6ci c].  \  }5}6|5ta        jb                  td        «      jf                  v sŒ,|5|6“Œ0 }7}5}6i |7¥||dMœ¥}7te        jh                  d€|7d$dNœ|¤Ž\  }}tj        jm                  dO«       ||z   }8dPdQ|dRœ}9|�||9dS<   to        «       r|stj        jq                  dT«       d$}tA        |tB        «      sH|�|n|}: | jr                  jt                  |:f|d$|||||||%||dUœ|¤Ž\  }};d|;v rD|;j                  d«       n2tw        jx                  |«      }|j                  dVd«      }<|<�|<|_=        |};t}        |d«      }=|=r!t        j€                  |j‚                  «      sd}=|=s|�Q|=r&t        j„                  |j‚                  |«      |_A        n||_A        t        j†                  |j‚                  |=¬W«      }>nd}>|>�¦|>j‰                  |||||
¬X«       |>j‹                  |«      }|>j�                  |«      }|>j�                  |«      }t}        |>j‚                  j�                  dY«      r$|>j‚                  j�                  j’                  |9dZ<   n|>j‚                  j�                  |9dZ<   |%�|>�t        d[«      ‚|%r3|�1tA        |t”        «      rd\|j—                  «       v sd\|v rt[        d]«      ‚t™        ||| |%|||	||||||9||¬^«      \  }?}@|@du}A|>du}B|duxs |%du}Ct=        «       rãCráAsß|?d,   j›                  d_«      rËt�        |?d,   d`¬a«      5 }4|4jŸ                  «       }Dddd«       D€nŸDj                  db«      d`k(  rnŠDj                  db«      dck(  rd$}tj        jq                  dd«       n^Dj                  db«      dek(  rd$}tj        jq                  df«       n2Dj                  db«      dgk(  rnt        dhDj                  db«      › �«      ‚||z   }8|8rT|%r=didjlPmQ}E t        j&                  dk«      5   | |«      }Fddd«        E|?d,   d$F¬l«      dm   }t¥        | ||?|@||
«      \  }}}G||_S        | j©                  Btª        «      }Htw        jx                  |«      }tK        |dnd«      s| j­                  ||#||¬o«      }t¯        H«      5   | |g|¢­i |;¤Ž}Iddd«       Ij±                  «        |)�=Ij²                  s1|j´                  €%|j·                  «       j´                  €t¹        dp«      ‚Ijº                  }d}J|Ij¼                  �`|t        j¾                  k(  stK        |>dqd«      r@tÁ        jÂ                  drjÅ                  Ij¼                  D �Kcg c]  }Kds|K› dt�‘Œ
 c}K«      «      }J|>�<|>jÇ                  I||Ij¼                  |¬u«       |�|nt        jÈ                  «       |_e        |�tÍ        I|||>|J«      }|r| jÏ                  I||?«      \  }I}LnU|r| jÑ                  I|?«      }In@|8r>G�t        jÒ                  G«       | jÕ                  I||?||@|||||>J|)|'|
¬v«      \  }I}M}N}O}P}QIj±                  «        |Ij×                  «        |IjÙ                  «       rF|$�Dtj        jq                  dw«       IjÚ                  ji                  |$jÝ                  «       «      |I_m        n8IjÙ                  «       r(|�&	 tß        jt                  |f|||||||||dxœ	|¤ŽI_m        |��|)�€||P|dzœ}Rd{ta        jb                  tâ        «      jf                  v rIjä                  Rd{<   d|ta        jb                  tâ        «      jf                  v r.|>�,|>j‚                  j�                  tæ        jè                  k(  rd$Rd|<   |>�`|>j‚                  j�                  tæ        jê                  k(  r9tA        |t”        «      r)d5|j—                  «       v sd\|j—                  «       v rd$Rd<   tí        «       st]        «       stã        Ifi R¤Ž |>�|>jï                  I|¬}«       |>|I_x        |3�Ijó                  |3|"||!¬~«       |r|8rMNOQdœ}LI|LfS |rd}LILfS IS # t"        $ r}.t        d:«      |.‚d}.~.ww xY w# 1 sw Y   �	ŒxY w# tZ        $ r t        dH|› d)�«      ‚w xY wc c}6}5w # 1 sw Y   �Œ_xY w# 1 sw Y   �Œ”xY w# 1 sw Y   �ŒxY wc c}Kw # tà        $ r tj        jq                  dy«       Y �Œûw xY w)�aË;  
        Instantiate a pretrained pytorch model from a pre-trained model configuration.

        The model is set in evaluation mode by default using `model.eval()` (Dropout modules are deactivated). To train
        the model, you should first set it back in training mode with `model.train()`.

        The warning *Weights from XXX not initialized from pretrained model* means that the weights of XXX do not come
        pretrained with the rest of the model. It is up to you to train those weights with a downstream fine-tuning
        task.

        The warning *Weights from XXX not used in YYY* means that the layer XXX is not used by YYY, therefore those
        weights are discarded.

        Parameters:
            pretrained_model_name_or_path (`str` or `os.PathLike`, *optional*):
                Can be either:

                    - A string, the *model id* of a pretrained model hosted inside a model repo on huggingface.co.
                    - A path to a *directory* containing model weights saved using
                      [`~PreTrainedModel.save_pretrained`], e.g., `./my_model_directory/`.
                    - A path or url to a *tensorflow index checkpoint file* (e.g, `./tf_model/model.ckpt.index`). In
                      this case, `from_tf` should be set to `True` and a configuration object should be provided as
                      `config` argument. This loading path is slower than converting the TensorFlow checkpoint in a
                      PyTorch model using the provided conversion scripts and loading the PyTorch model afterwards.
                    - A path or url to a model folder containing a *flax checkpoint file* in *.msgpack* format (e.g,
                      `./flax_model/` containing `flax_model.msgpack`). In this case, `from_flax` should be set to
                      `True`.
                    - `None` if you are both providing the configuration and state dictionary (resp. with keyword
                      arguments `config` and `state_dict`).
            model_args (sequence of positional arguments, *optional*):
                All remaining positional arguments will be passed to the underlying model's `__init__` method.
            config (`Union[PretrainedConfig, str, os.PathLike]`, *optional*):
                Can be either:

                    - an instance of a class derived from [`PretrainedConfig`],
                    - a string or path valid as input to [`~PretrainedConfig.from_pretrained`].

                Configuration for the model to use instead of an automatically loaded configuration. Configuration can
                be automatically loaded when:

                    - The model is a model provided by the library (loaded with the *model id* string of a pretrained
                      model).
                    - The model was saved using [`~PreTrainedModel.save_pretrained`] and is reloaded by supplying the
                      save directory.
                    - The model is loaded by supplying a local directory as `pretrained_model_name_or_path` and a
                      configuration JSON file named *config.json* is found in the directory.
            state_dict (`Dict[str, torch.Tensor]`, *optional*):
                A state dictionary to use instead of a state dictionary loaded from saved weights file.

                This option can be used if you want to create a model from a pretrained configuration but load your own
                weights. In this case though, you should check if using [`~PreTrainedModel.save_pretrained`] and
                [`~PreTrainedModel.from_pretrained`] is not a simpler option.
            cache_dir (`Union[str, os.PathLike]`, *optional*):
                Path to a directory in which a downloaded pretrained model configuration should be cached if the
                standard cache should not be used.
            from_tf (`bool`, *optional*, defaults to `False`):
                Load the model weights from a TensorFlow checkpoint save file (see docstring of
                `pretrained_model_name_or_path` argument).
            from_flax (`bool`, *optional*, defaults to `False`):
                Load the model weights from a Flax checkpoint save file (see docstring of
                `pretrained_model_name_or_path` argument).
            ignore_mismatched_sizes (`bool`, *optional*, defaults to `False`):
                Whether or not to raise an error if some of the weights from the checkpoint do not have the same size
                as the weights of the model (if for instance, you are instantiating a model with 10 labels from a
                checkpoint with 3 labels).
            force_download (`bool`, *optional*, defaults to `False`):
                Whether or not to force the (re-)download of the model weights and configuration files, overriding the
                cached versions if they exist.
            resume_download:
                Deprecated and ignored. All downloads are now resumed by default when possible.
                Will be removed in v5 of Transformers.
            proxies (`Dict[str, str]`, *optional*):
                A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
                'http://hostname': 'foo.bar:4012'}`. The proxies are used on each request.
            output_loading_info(`bool`, *optional*, defaults to `False`):
                Whether ot not to also return a dictionary containing missing keys, unexpected keys and error messages.
            local_files_only(`bool`, *optional*, defaults to `False`):
                Whether or not to only look at local files (i.e., do not try to download the model).
            token (`str` or `bool`, *optional*):
                The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
                the token generated when running `huggingface-cli login` (stored in `~/.huggingface`).
            revision (`str`, *optional*, defaults to `"main"`):
                The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a
                git-based system for storing models and other artifacts on huggingface.co, so `revision` can be any
                identifier allowed by git.

                <Tip>

                To test a pull request you made on the Hub, you can pass `revision="refs/pr/<pr_number>"`.

                </Tip>
            attn_implementation (`str`, *optional*):
                The attention implementation to use in the model (if relevant). Can be any of `"eager"` (manual implementation of the attention), `"sdpa"` (using [`F.scaled_dot_product_attention`](https://pytorch.org/docs/master/generated/torch.nn.functional.scaled_dot_product_attention.html)), or `"flash_attention_2"` (using [Dao-AILab/flash-attention](https://github.com/Dao-AILab/flash-attention)). By default, if available, SDPA will be used for torch>=2.1.1. The default is otherwise the manual `"eager"` implementation.

            > Parameters for big model inference

            torch_dtype (`str` or `torch.dtype`, *optional*):
                Override the default `torch.dtype` and load the model under a specific `dtype`. The different options
                are:

                1. `torch.float16` or `torch.bfloat16` or `torch.float`: load in a specified
                  `dtype`, ignoring the model's `config.torch_dtype` if one exists. If not specified
                  - the model will get loaded in `torch.float` (fp32).

                2. `"auto"` - A `torch_dtype` entry in the `config.json` file of the model will be
                  attempted to be used. If this entry isn't found then next check the `dtype` of the first weight in
                  the checkpoint that's of a floating point type and use that as `dtype`. This will load the model
                  using the `dtype` it was saved in at the end of the training. It can't be used as an indicator of how
                  the model was trained. Since it could be trained in one of half precision dtypes, but saved in fp32.

                3. A string that is a valid `torch.dtype`. E.g. "float32" loads the model in `torch.float32`, "float16" loads in `torch.float16` etc.

                <Tip>

                For some models the `dtype` they were trained in is unknown - you may try to check the model's paper or
                reach out to the authors and ask them to add this information to the model's card and to insert the
                `torch_dtype` entry in `config.json` on the hub.

                </Tip>

            device_map (`str` or `Dict[str, Union[int, str, torch.device]]` or `int` or `torch.device`, *optional*):
                A map that specifies where each submodule should go. It doesn't need to be refined to each
                parameter/buffer name, once a given module name is inside, every submodule of it will be sent to the
                same device. If we only pass the device (*e.g.*, `"cpu"`, `"cuda:1"`, `"mps"`, or a GPU ordinal rank
                like `1`) on which the model will be allocated, the device map will map the entire model to this
                device. Passing `device_map = 0` means put the whole model on GPU 0.

                To have Accelerate compute the most optimized `device_map` automatically, set `device_map="auto"`. For
                more information about each option see [designing a device
                map](https://hf.co/docs/accelerate/main/en/usage_guides/big_modeling#designing-a-device-map).
            max_memory (`Dict`, *optional*):
                A dictionary device identifier to maximum memory if using `device_map`. Will default to the maximum memory available for each
                GPU and the available CPU RAM if unset.
            tp_plan (`str`, *optional*):
                A torch tensor parallel plan, see [here](https://pytorch.org/tutorials/intermediate/TP_tutorial.html). Currently, it only accepts
                `tp_plan="auto"` to use predefined plan based on the model. Note that if you use it, you should launch your script accordingly with
                `torchrun [args] script.py`. This will be much faster than using a `device_map`, but has limitations.
            offload_folder (`str` or `os.PathLike`, *optional*):
                If the `device_map` contains any value `"disk"`, the folder where we will offload weights.
            offload_state_dict (`bool`, *optional*):
                If `True`, will temporarily offload the CPU state dict to the hard drive to avoid getting out of CPU
                RAM if the weight of the CPU state dict + the biggest shard of the checkpoint does not fit. Defaults to
                `True` when there is some disk offload.
            offload_buffers (`bool`, *optional*):
                Whether or not to offload the buffers with the model parameters.
            quantization_config (`Union[QuantizationConfigMixin,Dict]`, *optional*):
                A dictionary of configuration parameters or a QuantizationConfigMixin object for quantization (e.g
                bitsandbytes, gptq). There may be other quantization-related kwargs, including `load_in_4bit` and
                `load_in_8bit`, which are parsed by QuantizationConfigParser. Supported only for bitsandbytes
                quantizations and not preferred. consider inserting all such arguments into quantization_config
                instead.
            subfolder (`str`, *optional*, defaults to `""`):
                In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can
                specify the folder name here.
            variant (`str`, *optional*):
                If specified load weights from `variant` filename, *e.g.* pytorch_model.<variant>.bin. `variant` is
                ignored when using `from_tf` or `from_flax`.
            use_safetensors (`bool`, *optional*, defaults to `None`):
                Whether or not to use `safetensors` checkpoints. Defaults to `None`. If not specified and `safetensors`
                is not installed, it will be set to `False`.
            weights_only (`bool`, *optional*, defaults to `True`):
                Indicates whether unpickler should be restricted to loading only tensors, primitive types,
                dictionaries and any types added via torch.serialization.add_safe_globals().
                When set to False, we can load wrapper tensor subclass weights.
            key_mapping (`Dict[str, str], *optional*):
                A potential mapping of the weight names if using a model on the Hub which is compatible to a Transformers
                architecture, but was not converted accordingly.
            kwargs (remaining dictionary of keyword arguments, *optional*):
                Can be used to update the configuration object (after it being loaded) and initiate the model (e.g.,
                `output_attentions=True`). Behaves differently depending on whether a `config` is provided or
                automatically loaded:

                    - If a configuration is provided with `config`, `**kwargs` will be directly passed to the
                      underlying model's `__init__` method (we assume all relevant updates to the configuration have
                      already been done)
                    - If a configuration is not provided, `kwargs` will be first passed to the configuration class
                      initialization function ([`~PretrainedConfig.from_pretrained`]). Each key of `kwargs` that
                      corresponds to a configuration attribute will be used to override said attribute with the
                      supplied `kwargs` value. Remaining keys that do not correspond to any configuration attribute
                      will be passed to the underlying model's `__init__` function.

        <Tip>

        Activate the special ["offline-mode"](https://huggingface.co/transformers/installation.html#offline-mode) to
        use this method in a firewalled environment.

        </Tip>

        Examples:

        ```python
        >>> from transformers import BertConfig, BertModel

        >>> # Download model and configuration from huggingface.co and cache.
        >>> model = BertModel.from_pretrained("google-bert/bert-base-uncased")
        >>> # Model was saved using *save_pretrained('./test/saved_model/')* (for example purposes, not runnable).
        >>> model = BertModel.from_pretrained("./test/saved_model/")
        >>> # Update configuration during loading.
        >>> model = BertModel.from_pretrained("google-bert/bert-base-uncased", output_attentions=True)
        >>> assert model.config.output_attentions == True
        >>> # Loading from a TF checkpoint file instead of a PyTorch model (slower, for example purposes, not runnable).
        >>> config = BertConfig.from_json_file("./tf_model/my_tf_model_config.json")
        >>> model = BertModel.from_pretrained("./tf_model/my_tf_checkpoint.ckpt.index", from_tf=True, config=config)
        >>> # Loading from a Flax checkpoint file instead of a PyTorch model (slower)
        >>> model = BertModel.from_pretrained("google-bert/bert-base-uncased", from_flax=True)
        ```
        rø   Nrë  Frì  rð  Úoutput_loading_inforé  Ú_from_pipelineÚ
_from_autor  r¸  r  Úoffload_folderÚoffload_state_dictÚoffload_buffersÚload_in_8bitÚload_in_4bitr¥  ré  rk  rú  rä  Úadapter_kwargsÚadapter_nameÚdefaultrõ  rÕ  rê  Útp_planÚkey_mappingÚresume_downloadÚtrust_remote_codeÚmirrorÚ
_fast_initTÚlow_cpu_mem_usagezq`state_dict` cannot be passed together with a model name or a `gguf_file`. Use one of the two loading strategies.r  z-tp_plan supports 'auto' only for now but got rþ   zY`tp_plan` and `device_map` are mutually exclusive. Choose either one for parallelization.Ú
WORLD_SIZEr   z2.5z3tensor parallel is only supported for `torch>=2.5`.rÃ  r  Úncclzenv://)ÚrankÚ
world_sizeÚinit_methodrŽ   r  ÚCCL_WORKER_COUNTÚcclÚgloo)rv  rw  z‹We tried to initialize torch.distributed for you, but it failed, makesure you init torch distributed in your script to use `tp_plan='auto'`rû  rë  rì  rò  zIaccelerate is required when loading a GGUF file `pip install accelerate`.)
rî  rï  rð  rñ  rò  rô  ré  rø  rù  Ú'_raise_exceptions_for_connection_errorsÚ_adapter_model_path)rî  rï  rð  rñ  rú  rÿ   r   r  Úbase_model_name_or_path)r  Úbalancedr  r  zœWhen passing device_map as a string, the value needs to be a device name (e.g. cpu, cuda:0) or 'auto', 'balanced', 'balanced_low_0', 'sequential' but found znYou can't pass device_map as a negative int. If you want to put the model on the cpu, pass device_map = 'cpu' z?DeepSpeed Zero-3 is not compatible with passing a `device_map`.ziUsing a `device_map` or `tp_plan` requires `accelerate`. You can install it with `pip install accelerate`zwYou can't pass `load_in_4bit`or `load_in_8bit` as a kwarg when passing `quantization_config` argument at the same time.)ri  rh  )Úconfig_dictÚreturn_unused_kwargszÀThe `load_in_4bit` and `load_in_8bit` arguments are deprecated and will be removed in the future versions. Please, pass a `BitsAndBytesConfig` object in `quantization_config` argument instead.r!  Úpytorch)Ú	file_typerO  Úfrom_auto_classÚusing_pipelinez+Offline mode: forcing local_files_only=True)rî  r�  rï  rð  rñ  rò  rô  ré  rê  rd  rc  rö  )Úpre_quantized)r  rë  rì  r¸  r
  rß  ÚquantzÂYou cannot combine Quantization and loading a model from a GGUF file, try again by making sure you did not passed a `quantization_config` or that you did not load a quantized model from the Hub.rÄ  zxOne or more modules is configured to be mapped to disk. Disk offload is not supported for models loaded from GGUF files.)rè  ré  rä  rê  rë  rì  rí  rî  rï  rð  rñ  rò  ró  rô  rõ  rL  rM  rN  rP  rQ  zAA TensorFlow safetensors file is being loaded in a PyTorch model.rR  z;A Flax safetensors file is being loaded in a PyTorch model.rS  zTIncompatible safetensors file. File metadata is not ['pt', 'tf', 'flax', 'mlx'] but r    )Úload_gguf_checkpointrT  )Úreturn_tensorsÚmodel_to_loadrˆ  rÊ  )rõ  r  r¸  z0This model does not have a tensor parallel plan.Úuse_keep_in_fp32_modulesrÁ  z((^|\.)z($|\.)))r!  r¸  Úkeep_in_fp32_modulesrª  )rG  r	  r¸  r¹  rf  rå   r¡  r   r¾  rn  r
  z\The user-defined `generation_config` will be used to override the default generation config.)	rî  rï  rð  rñ  rò  rô  ré  rd  rc  zZGeneration config file not found, using a generation config created from the model config.)r¸  Úoffload_dirÚoffload_indexrg  Ú	skip_keysÚforce_hooksrò  )rk  rò  rj  )r1  r2  rN  Ú
error_msgsr‰   )zrŽ  rö   r�   r†   r‡   rˆ   r[   rÿ  r‚   Ú_CÚ_get_acceleratorrÓ  rƒ   r…   Úinit_process_groupr  Ú
set_devicerd  Úcurrent_devicerÙ   Úsysr  ÚdevnullÚstdoutÚstderrÚget_world_sizeÚinit_device_meshr„  r…  r†  rY   rR   rb  r"   rM   rA   rP   r€  rW   r”   r  r  r  r  rÜ   r  r)   r®   r%  r&  ri   rØ   Ú	from_dictr  r  rU   r   Úconfig_classÚfrom_pretrainedrÖ  rú  rü  r¨  r;   Úsupports_quant_methodr¥  Úmerge_quantization_configsÚfrom_configr(  Úupdate_torch_dtypeÚupdate_device_mapÚupdate_tp_planr¦  rß  r  rõ   r  rX  r{   rY  Úmodeling_gguf_pytorch_utilsrˆ  r  rÒ  r`  r»   rÏ  rJ   rR  Úsupports_tp_planrã  rM  r=  rª  r×  r(  rÊ  r  r  Úpreprocess_modelr¿   r«  r-  Ú_load_from_tfÚ_load_from_flaxrÀ   Ú_load_pretrained_modelÚevalrÓ  rÕ  Úto_dictr%   rZ  rn   Ú_skip_keys_device_placementrj   r§  Ú
FBGEMM_FP8rŒ   Úpostprocess_modelr¡  Úload_adapter)Sr  rè  rª  rî  rG  rï  rñ  rò  rô  rí  r
  Ú
model_argsrª   rø   rë  rì  rð  rb  ré  Úfrom_pipeliner„  r  r¸  r  re  rf  rg  rh  ri  r¥  ré  rõ  rä  rj  rk  rõ  rÕ  rê  rm  rn  r’  r¾  Údevice_typerv  rw  Úcpu_backendri  r,  Ú	tp_devicer—  Úresolved_config_filer}  r+  rÏ   rÐ   r€  Úfrom_ptró  Úconfig_pathÚmodel_kwargsÚkwarg_attn_impr†  r¡  r
  r	  r  rJ  Úis_from_filerY  rˆ  Údummy_modelr  Úmodel_init_contextr!  r   rÈ   Úloading_infor1  r2  rN  rŽ  r‘  r+  sS                                                                                      r‹   rŸ  zPreTrainedModel.from_pretrained—  sÄ  € ð@ —Z‘Z ¨dÓ3ˆ
Ø—*‘*˜Y¨Ó.ˆØ—J‘J˜{¨EÓ2ˆ	Ø—*‘*˜Y¨Ó-ˆØ$Ÿj™jÐ)>ÀÓFÐØŸ™Ð$4°dÓ;ˆØŸ
™
Ð#3°TÓ:ˆØ Ÿ*™* \°5Ó9ˆØ—j‘j °Ó5ˆØ—Z‘Z ¨dÓ3ˆ
Ø—Z‘Z ¨dÓ3ˆ
ØŸ™Ð$4°dÓ;ˆØ#ŸZ™ZÐ(<¸eÓDÐØ Ÿ*™*Ð%6¸Ó>ˆØ—z‘z .°%Ó8ˆØ—z‘z .°%Ó8ˆØ$Ÿj™jÐ)>ÀÓEÐØ—J‘J˜{¨BÓ/ˆ	Ø—j‘j °Ó6ˆØ—*‘*˜Y¨Ó-ˆØŸ™Ð$4°bÓ9ˆØ—z‘z .°)Ó<ˆØ &§
¡
Ð+BÀEÓ JÐØ"ŸJ™JÐ':¸DÓAÐØ—J‘J˜{¨DÓ1ˆ	Ø—*‘*˜Y¨Ó-ˆØ—j‘j °Ó5ˆà�J‰JÐ(¨$Ó/ˆØ�J‰JÐ*¨DÓ1ˆØ�J‰J�x Ó&ˆØ�J‰J�| TÓ*ˆØ�J‰JÐ*¨DÓ1ˆàÐ!Ð'DÐ'PÐT]ÐTiÜð Dóð ð Ð 7¨fÒ#4äÐLÈWÈIÐUVÐWÓXÐXØÐ :Ð#9ÜØkóð ð
 ˜Ò G O¼¼B¿J¹J¿N¹NÈ<ÐYZÓ<[Ô8\ØˆGØˆJð ˆØÑÜ,¨UÔ3Ü&Ð'\Ó]Ð]ô  Ÿ(™(×3Ñ3Ó5×:Ñ:ˆKä×$Ñ$×3Ñ3Ô5ðÜœrŸz™z¨&Ñ1Ó2�DÜ!$¤R§Z¡Z°Ñ%=Ó!>�JØ" fÒ,Ü×)Ñ)×<Ñ<Ø"¨¸*ÐRZð =ô ô Ÿ
™
×-Ñ-¬c´"·*±*¸\Ñ2JÓ.KÕLØ$¨Ò-Ü/2´2·:±:·>±>ÐBTÐVWÓ3XÔ/Y¡eÐ_e˜Ü×)Ñ)×<Ñ<¸[ÈtÐ`jÐ<Ôkð (¨5Ò0‘D´e·j±j×6OÑ6OÓ6QˆEÜŸ™ [°%Ó8ˆIàÐ  U¨Q¢YÛä!¤"§*¡*¨cÓ2�”
Ü!¤"§*¡*¨cÓ2�”
à"ˆJä×*Ñ*×9Ñ9Ó;ˆJÜ×+Ñ+×<Ñ<¸Y¿^¹^ÈjÈ]Ó[ˆKàÐ%Ü�M‰Mð EÜôð Ð Ü Ølóð ð #ˆEàÐ Ð!;ÀÈ~Ñ@]Ø&+ˆN˜7Ñ#àÐ"Ô+CÔ+EØ#ˆOàÐ Ô)@Ô)BÜÐhÓiÐiàÐÜ˜fÔ&6Ô7ä'2Ø1ÜØ'Ø#1Ø#Ø%5ØØ%Ø'Ø5:Ø:?Ø<Aô(Ð$ô 2Ð2FÈÓT‘ä% f¨n¸dÓC�äÔØ"0×"4Ñ"4Ð5JÈDÓ"QÐà"Ð*Ü&>Ø1ð'à'Ø#1Ø#Ø%5Ø!,ñ'ð %ñ'Ð#ð #Ð.´2·7±7·>±>ÐBUÔ3VÜÐ-¨s¸WÔEð \ÈØ*GÐ'Ü48·I±I¸a³LÐAZÑ4[Ð1÷\ð \ð #'Ðô �j¤%§,¡,Ô/Ø˜jÐ)‰JÜ˜
¤CÔ(¨ZÐ?sÑ-sðØ ¤%§,¡,¨zÓ":Ð;‘
ô ˜
¤CÔ(Ø˜AŠ~Ü ð Eóð ð ! *Ð-�
àÐ!Ü)Ô+Ü Ð!bÓcÐcÜ*Ô,Ü Øóð ñ
 ™<Ø"Ð.Ü ðGóð ð -3¯L©L«N×t¡D A q¸aÄ7×CTÑCTÔUgÓCh×CsÑCsÒ>s˜1˜a™4ÐtˆKÑtØe˜[Ðe¸,ÐXdÒeˆKÜ*<×*FÑ*Fð +Ø'¸dñ+ØFLñ+Ñ'Ð ô �N‰Nðhôð
  Ñ*Ð+ˆà#*¸ÐWfÑgˆ
ØÐ$Ø+8ˆJÐ'Ñ(äÔÑ%5Ü�K‰KÐEÔFØ#Ðô ˜&Ô"2Ô3Ø$*Ð$6™&Ð<YˆKØ#C 3×#3Ñ#3×#CÑ#CØð$à#Ø%)Ø-ØØ!1ØØ!Ø#Ø#Ø*Ø,ñ$ð ñ$Ñ ˆF�Lð ˜lÑ*Ø× Ñ  Õ-ô —]‘] 6Ó*ˆFà#ŸZ™ZÐ(=¸tÓDˆNØÐ)Ø.<�Ô+à!ˆLä Ð(=Ó>ˆÙ¤×!FÑ!FÀv×GaÑGaÔ!bØ!ˆMáÐ/Ð;ÙÜ-<×-WÑ-WØ×.Ñ.Ð0Có.�Õ*ð .A�Ô*ä*×6Ñ6Ø×*Ñ*Ø+ô‰Lð
  ˆLàÐ#Ø×-Ñ-Ø'ØØ#Ø%Ø)ð .ô ð '×9Ñ9¸+ÓFˆKØ%×7Ñ7¸
ÓCˆJØ!×0Ñ0°Ó8ˆFô �|×7Ñ7×DÑDÀgÔNØ&2×&FÑ&F×&SÑ&S×&YÑ&Y�
˜7Ò#à&2×&FÑ&F×&SÑ&S�
˜7Ñ#àÐ  \Ð%=Üð Uóð ñ
 ØÐ&Ü˜Z¬Ô.°6¸Z×=NÑ=NÓ=PÑ3PÐU[Ð_iÑUiäð*óð ô
 .LØ*GØØØØØØ+ØØ)ØØ-ØØ!ØØ#ô.
Ñ*ÐÐ*ð$ &¨TÐ1ˆ
Ø#¨4Ð/ˆØ4¸DÐ@ÒYÀIÐUYÐDYˆô %Ô&ÙÙØ  Ñ#×,Ñ,¨^Ô<äÐ+¨AÑ.¸$Ô?ð (À1ØŸ:™:›<�÷(ð ÐàØ—‘˜hÓ'¨4Ò/ØØ—‘˜hÓ'¨4Ò/Ø�Ü—‘Ð_Õ`Ø—‘˜hÓ'¨6Ò1Ø �	Ü—‘ÐYÕZØ—‘˜hÓ'¨5Ò0àä ØjÐks×kwÑkwð  yAó  lBð  kCð  Dóð ð  Ñ*Ð+ˆáÙÝMô —\‘\ &Ó)ñ .Ù"% f£+�K÷.á1Ð2BÀ1Ñ2EÐVZÐjuÔvØñ�
ô
 /?Ø�[Ð"2°FÐ<LÈjÐZfó/Ñ+ˆF�K ð <ˆÔð !×1Ñ1°,Ô@RÓSÐä—‘˜vÓ&ˆÜ�vÐ=¸uÔEØ×5Ñ5ØÐ.CÐQ\Ðisð 6ó ˆFô Ð/Ó0ñ 	=á˜Ð< Ò<¨|Ñ<ˆE÷	=ð
 	×ÑÔð Ð"¨5×+AÒ+AØ×(Ñ(Ð0°V×5KÑ5KÓ5M×5`Ñ5`Ð5hÜ)Ð*\Ó]Ð]ð —‘ˆð "Ðð ×&Ñ&Ð2Øœ5Ÿ=™=Ò(¬G°LÐB\Ð^cÔ,dô "$§¡Ø—‘À5×C^ÑC^Ö_¸˜W V H¨GÒ4Ò_Ó`ó"Ðð Ð#Ø×)Ñ)Ø¨
È×IdÑIdÐmsð *ô ð =HÐ<S©[ÔY^×YpÑYpÓYrˆFÔ*ð Ð!Ü(¨°
¸JÈÐVaÐcuÓvˆJñ Ø"%×"3Ñ"3°E¸6ÐCSÓ"TÑˆE‘<ÙØ×'Ñ'¨Ð/?Ó@‰EÙàÐ%Ü×'Ñ'¨
Ô3ð ×*Ñ*ØØØ Ø-Ø(?Ø!1Ø%Ø$2Ø#5Ø!Ø)Ø#5Ø'Ø'Ø)ð +ó ñØØØØØØð( 	×ÑÔð 	�
‰
Œð ×ÑÔÐ$5Ð$AÜ�K‰KÐvÔwØ&+×&=Ñ&=×&GÑ&GÐHY×HaÑHaÓHcÓ&dˆEÕ#Ø×ÑÔ!Ð&CÐ&OðÜ*:×*JÑ*JØ1ð+à'Ø#1Ø#Ø%5ØØ%Ø'Ø.Ø#0ñ+ð ñ+�Ô'ð* Ñ! kÑ&9à(Ø-Ø!.Ø#2ñ	!Ðð œg×/Ñ/´Ó?×JÑJÑJØ16×1RÑ1RÐ! +Ñ.ð ¤×!2Ñ!2´>Ó!B×!MÑ!MÑMØ Ð,Ø ×4Ñ4×AÑAÔEW×E[ÑE[Ò[à37Ð! -Ñ0àÐ(Ø ×4Ñ4×AÑAÔEW×EbÑEbÒbÜ˜z¬4Ô0Ø˜j×/Ñ/Ó1Ñ1°V¸z×?PÑ?PÓ?RÑ5Rà7;Ð!Ð"3Ñ4ä"Ô$Ô-GÔ-IÜ˜uÑ:Ð(9Ò:àÐ#Ø×*Ñ*¨5¸Ð*Ô@Ø!-ˆEÔàÐ*Ø×ÑØ#Ø)ØØ-ð	 ô ñ Ùà$0Ø'6Ø'6Ø",ñ	 �ð ˜,Ð&Ð&ñ Ø#�Ø˜,Ð&Ð&àˆøôg !ò Ü*ðaóð ðûðú÷X\ñ \ûô  ò Ü ðTØT^ÐS_Ð_`ðbóð ðüó< u÷P(ñ (ú÷<.ñ .ú÷,	=ñ 	=üò2 `øôZ ò Ü—‘Øpôò ð	úsy   ÌC.{ ×{4Ø'| Ú3-|Û!|æ5|#ê	|0ì+|=ï1}
õ%} û	{1û {,û,{1û4{>ü|ü#|-ü0|:ü=}ý}1ý0}1r0  c                 ó  — | j                  d«      r| j                  dd«      dfS | j                  d«      r| j                  dd«      dfS t        t        j                  j
                  d«      rN| j                  d«      r| j                  dd«      dfS | j                  d	«      r| j                  d	d
«      dfS | dfS | j                  d«      r| j                  dd«      dfS | j                  d
«      r| j                  d
d	«      dfS | dfS )zaReplace legacy parameter names with their modern equivalents. E.g. beta -> bias, gamma -> weight.úLayerNorm.betazLayerNorm.biasTúLayerNorm.gammazLayerNorm.weightÚweight_normÚweight_gz!parametrizations.weight.original0Úweight_vz!parametrizations.weight.original1F)rX  r  r¨  r   ÚutilsÚparametrizations©r0  s    r‹   Ú_fix_state_dict_key_on_loadz+PreTrainedModel._fix_state_dict_key_on_load—  s  € ð
 �<‰<Ð(Ô)Ø—;‘;Ð/Ð1AÓBÀDÐHÐHØ�<‰<Ð)Ô*Ø—;‘;Ð0Ð2DÓEÀtÐKÐKô
 ”2—8‘8×,Ñ,¨mÔ<Ø�|‰|˜JÔ'Ø—{‘{ :Ð/RÓSÐUYÐYÐYØ�|‰|˜JÔ'Ø—{‘{ :Ð/RÓSÐUYÐYÐYð �EˆzÐð �|‰|Ð?Ô@Ø—{‘{Ð#FÈ
ÓSÐUYÐYÐYØ�|‰|Ð?Ô@Ø—{‘{Ð#FÈ
ÓSÐUYÐYÐYà�EˆzÐrŠ   r/  rn  r0  Ú'loading_task_model_from_base_state_dictc                 ó‚  — | j                   }|› d�}i }i }|D ]Å  }	| j                  |	«      \  }
}|�;|j                  «       D ](  \  }}t        j                  |||
«      \  }
}|dkD  sŒ&d} n |rdj                  ||
g«      }
n"|r |
j                  |«      sŒ~|
t        |«      d }
|
||	<   |sŒ”|	j                  d«      r|	|
f|d<   Œ­|	j                  d«      sŒ¿|	|
f|d<   ŒÇ |r]d| j                  j                  › d�}|d	z  }|j                  «       D ]  \  }}
|d
|› d|
› d�z  }Œ |dz  }t        j                  |«       |S )a&  
        Compute a mapping between the serialized keys on disk `checkpoint_keys`, and the keys that the model
        that we are loading expects. This is the single entry point for key renaming that will be used during
        loading.
        Log if any parameters have been renamed.
        rþ   Nr   TrÂ  rÁ  zA pretrained model of type `z` zrcontains parameters that have been renamed internally (a few are listed below but more are present in the model):
z* `z` -> `z`
znIf you are using a model from the Hub, consider submitting a PR to adjust these weights and help future users.)r7  rÉ  r®   rÊ  Úsubnr  r9  r  rX  r  r  rõ   r  Ú	info_once)ro  r/  rn  r0  rÊ  r  Ú_prefixÚrenamed_keysÚkey_renaming_mappingr0  Únew_keyÚhas_changedrE  ÚreplacementÚ	n_replaceÚwarning_msgÚold_keys                    r‹   Ú_get_key_renaming_mappingz)PreTrainedModel._get_key_renaming_mapping±  sª  € ð ×'Ñ'ˆØ�H˜A�,ˆàˆØ!ÐØ"ò 	DˆCà#'×#CÑ#CÀCÓ#HÑ ˆG�[ð Ð&Ø,7×,=Ñ,=Ó,?ò Ñ(�G˜[Ü)+¯©°¸+ÀwÓ)OÑ&�G˜Yà  1“}Ø&*˜Ùðñ 7ØŸ(™( F¨GÐ#4Ó5‘ñ 9Ø×)Ñ)¨'Ô2ØØ!¤# g£, .Ð1�à(/Ð  Ñ%ò Ø—<‘<Ð 1Ô2Ø7:¸G°n�LÐ!2Ò3Ø—\‘\Ð"2Õ3Ø69¸7°^�LÐ!1Ò2ð=	Dñ@ Ø8¸¿¹×9PÑ9PÐ8QÐQSÐTˆKØð  Qñ  QˆKØ$0×$7Ñ$7Ó$9ò AÑ �˜Ø  W I¨V°G°9¸CÐ@Ñ@‘ðAàð  Lñ  LˆKÜ×Ñ˜[Ô)à#Ð#rŠ   c                 ó
   — | dfS )zÆ
        Similar to `_fix_state_dict_key_on_load` allows to define hook for state dict key renaming on model save.
        Do nothing by default, but can be overridden in particular models.
        Fr‰   rÈ  s    r‹   Ú_fix_state_dict_key_on_savez+PreTrainedModel._fix_state_dict_key_on_saveí  s   € ð �EˆzÐrŠ   c                 óz   — |j                  «       D ��ci c]  \  }}| j                  |«      d   |“Œ c}}S c c}}w )zÅ
        Similar to `_fix_state_dict_keys_on_load` allows to define hook for state dict key renaming on model save.
        Apply `_fix_state_dict_key_on_save` to all keys in `state_dict`.
        r   )r®   rÙ  )ro  rø   r0  rß  s       r‹   r  z,PreTrainedModel._fix_state_dict_keys_on_saveõ  s<   € ð
 S]×RbÑRbÓRd×eÁJÀCÈ�×0Ñ0°Ó5°aÑ8¸%Ñ?ÓeÐeùÓes   ”7r!  r
  r	  r¹  rf  r¡  r   r¾  r¿  c                 óz  ‡D‡E‡F— |d u}|xr' |j                   j                  t        j                  k(  }|xr6 |j                   j                  t        j                  t        j                  fv }|�|d   }nD|�t        |j                  «       «      }n(t        t        |d   d|¬«      j                  «       «      }|j                  ŠE‰E› d�}t        ‰E«      dkD  rt        ˆEfd„|D «       «      nd}t        ‰E«      dkD  rt        |‰E«      nd}| xr |}|xr | }|j                  ||||«      }t        |j                  «       «      }t        | ||||||«      \  }}t        |||||||«      \  }}|j!                  «       D �� ci c]  \  }} | |vsŒ|| “Œ }}} t        |j                  «       «      }|j#                  ||z   ||
|«       |j%                  |||«       |�X|j'                  «       D ]E  \  }!}"|j)                  |!«      sŒ|"j*                  j-                  t.        j0                  «      |"_        ŒG |}#|�rt3        |‰E«      }#|j!                  «       D �� ci c]  \  }} || t        |«      d  “Œ }}} t        |j                  «       «      }|�B|j!                  «       D �� ci c]'  \  }} |j5                  |«      r|t        |«      d  n|| “Œ) }}} |j7                  «       j                  «       D �$cg c]  }$|$j5                  |«      rŒ|$‘Œ c}$ŠFt        |#j7                  «       j                  «       «      ŠDt        ˆDˆFfd„|D «       «      rt9        d	«      ‚|j!                  «       D �� ci c]  \  }} | |“Œ
 }%}} d}&d }'g }(|��¨d
|j                  «       v �r•|	€d}	|�t;        j<                  |d¬«       |d uxr |d   j?                  d«      }&|€|&st9        d«      ‚|&�rJtA        ||«      })|
�tC        |
«      jE                  dd«      nd}*|€tF        jI                  ||d   «      }+nÐt:        jJ                  jL                  jO                  |d   jQ                  t:        jJ                  jL                  «      d d «      },|d   j!                  «       D �� ci c]  \  }} ||v r||   | “Œ }+}} |+j!                  «       D �� ci c]&  \  }} |t:        jJ                  jO                  |,| «      “Œ( }+}} tS        ||+«      }(|+j!                  «       D �!�-ci c]  \  }!}-|)|!   d
k(  r
|!|-|%|!   |*dœ“Œ }'}!}-ni }'d }.d }/|	rtU        jV                  «       }.i }/|�&t        |«      dkD  rtY        jZ                  |d¬«      }n|�dg}t        |#j7                  «       j                  «       «      }0|�|j]                  |#|0|«      }0|� |stA        ||0«      }1t_        |#|1|€dnd¬«       g }2|D �][  }3|3|(v rŒ	d}4|3j?                  d«      r|sta        «       r|rd}4n |�ž|�œ|j                   j                  t        jb                  k(  ru|j                   jd                  dv s$tg        |j                   jd                  th        «      r9t/        jj                  |j                  «       D �5cg c]	  }5|5dvsŒ|5‘Œ c}5d   «      }4|3dk7  rt        |3||4|¬«      }j!                  «       D �� ci c]  \  }} ||v sŒ||   | “Œ }}} ta        «       r|s|2tm        |#|«      z  }2n3to        «       rtq        «       s|rts        |#||3|0|%|||'|.|/||&|||¬«      \  }'}/~�Œ^ |'�¸t        |'«      dkD  rª|r˜| j                  ŠE|&sb|'D ]]  }6tu        jv                  t:        jJ                  jO                  ||6› d �«      t:        jJ                  jO                  |‰E› d|6› d �«      «       Œ_ |'j!                  «       D �7�8ci c]  \  }7}8‰E› d|7› �|8“Œ }'}7}8|&sty        |'|«       d }'|	r"t{        |#|/|.«       tu        j|                  |.«       |�|j                  |#|‰E«      }|�ét        |j                  «       «      d   }9|j�                  «       D ](  }:|:jj                  |9k7  sŒ|:j-                  |9«      |:_        Œ* |r�|j'                  «       D �!�"ci c]  \  }!}"|!j5                  ‰E«      rŒ|!|"“Œ };}!}"|;j!                  «       D ]H  \  }!}"tƒ        ||!|"|«      \  }<}=t…        ||"j-                  |9«      |"|!|=|<t:        j†                  d!   |«       ŒJ t        |2«      dkD  r?d"jO                  |2«      }>d#|>v r|>d$z  }>t‰        d%|jŠ                  jŒ                  › d&|>› �«      ‚t        |«      dkD  r»|jŽ                  j�                  €g n|jŽ                  j�                  }?|jŠ                  jŒ                  |?v rt’        j”                  nt’        j–                  }@ |@d'|› d(|jŠ                  jŒ                  › d)|› d*|jŠ                  jŒ                  › d+|jŠ                  jŒ                  › d,�«       n-t’        j—                  d-|jŠ                  jŒ                  › d.�«       t        |«      dkD  r4t’        j•                  d/|jŠ                  jŒ                  › d0|› d1|› d2�«       nUt        |«      dk(  rGt’        j—                  d3|jŠ                  jŒ                  › d4|› d5|jŠ                  jŒ                  › d6�«       t        |«      dkD  rpd7jO                  t™        ||«      D �7�A�Bcg c]  \  }7\  }A}Bd8|7› d9|A› d:|B› d;�‘Œ c}B}A}7«      }Ct’        j•                  d/|jŠ                  jŒ                  › d0|› d<|C› d2�«       |||||'|2fS c c} }w c c} }w c c} }w c c}$w c c} }w c c} }w c c} }w c c}-}!w c c}5w c c} }w c c}8}7w c c}"}!w c c}B}A}7w )=NÚall_checkpoint_keysr   rT  r  rþ   c              3   ó@   •K  — | ]  }|j                  ‰«      –— Œ y ­wr¨   )r9  )r4  Úsr  s     €r‹   r6  z9PreTrainedModel._load_pretrained_model.<locals>.<genexpr>$  s   øè ø€ ÒW¸ §¡¨V× 4ÑWùs   ƒFc              3   ó2   •K  — | ]  }|‰v xr |‰v–— Œ y ­wr¨   r‰   )r4  r0  Úbase_model_expected_keysÚtask_specific_expected_keyss     €€r‹   r6  z9PreTrainedModel._load_pretrained_model.<locals>.<genexpr>i  s*   øè ø€ ò Ø_b�Ð2Ð2ÒZ°sÐBZÐ7ZÓZñùs   ƒzjThe state dictionary of the model you are trying to load is corrupted. Are you sure it was properly saved?rÄ  Trî  rL  zàThe current `device_map` had weights offloaded to the disk. Please provide an `offload_folder` for them. Alternatively, make sure you have `safetensors` installed if the model you are using offers the weights in this format.ztorch.rk  rí   r�   r  )Úsafetensors_fileÚweight_namerå   r    zLoading checkpoint shardsrù  rŠ  é   )Úfactorr  )Úint4_weight_onlyÚ	autoquant©r  rÄ  rJ  )
r¸  r¹  rº  r»  r¼  r¡  r½  r   r2  r¾  z.datrÃ  z
	zsize mismatchz_
	You may consider adding `ignore_mismatched_sizes=True` in the model `from_pretrained` method.r  z:
	z(Some weights of the model checkpoint at z! were not used when initializing z: z,
- This IS expected if you are initializing zß from the checkpoint of a model trained on another task or with another architecture (e.g. initializing a BertForSequenceClassification model from a BertForPreTraining model).
- This IS NOT expected if you are initializing z¨ from the checkpoint of a model that you expect to be exactly identical (initializing a BertForSequenceClassification model from a BertForSequenceClassification model).z9All model checkpoint weights were used when initializing z.
zSome weights of z3 were not initialized from the model checkpoint at z and are newly initialized: zo
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.zAll the weights of z/ were initialized from the model checkpoint at zf.
If your task is similar to the task the model of the checkpoint was trained on, you can already use z* for predictions without further training.rþ  z- z: found shape z in the checkpoint and z in the model instantiatedz= and are newly initialized because the shapes did not match:
)Mr¥  r¦  rj   r§  rÌ  r  r  r  r7  r  r;  r¨  r×  rõ   rF  rQ  r®   Ú#_move_missing_keys_from_meta_to_cpuÚ_initialize_missing_keysr"  r©  rÔ  rÍ  r‚   rí   r€  r9  rø   rö   r†   r  rX  Úexpand_device_maprÜ   r  r  r   r  r  r  rß  Úget_disk_only_shard_filesÚtempfileÚmkdtempr`   r  r8  Úcaching_allocator_warmupr)   ÚTORCHAOÚ
quant_typerb  r   rÙ   r,   rŒ   r‘   râ  ÚshutilÚmoverw   ru   ÚrmtreeÚ!update_missing_keys_after_loadingrî   r±  r2   r‡   r  r  r  rª  r  r  r  r   Úzip)Gr  r!  rø   r
  rè  rG  r	  r¸  r¹  rf  rå   r¡  r   r¾  rn  r
  rJ  Úis_hqqrØ  r.  rÎ  Úhas_prefix_moduleÚexpects_prefix_modulerÊ  r0  rÐ  r/  r1  r2  rN  rO  rÏ   rÐ   r²   rÜ  rŠ  rÞ  Úreverse_key_renaming_mappingÚis_offloaded_safetensorsrº  Údisk_only_shard_filesÚparam_device_mapÚ	str_dtyper  r"  Úfiler»  r¼  r¶  Úexpanded_device_mapr‘  r7  r	  Údrã  r0  rß  r¶  r5  Úparameters_to_initializerÝ  r¯  Ú	error_msgÚarchsÚwarnerÚshape1Úshape2Úmismatched_warningrà  r  rá  sG                                                                       @@@r‹   r«  z&PreTrainedModel._load_pretrained_modelü  sÔ  ú€ ð( $¨4Ð/ˆØÒi ,×"BÑ"B×"OÑ"OÔSe×SiÑSiÑ"iˆØ$ò 
¨×)IÑ)I×)VÑ)VÜ×"Ñ"Ü×-Ñ-ð[
ð *
ˆð Ð'Ø'7Ð8MÑ'NÑ$ØÐ#Ü'+¨J¯O©OÓ,=Ó'>Ñ$ä'+ÜÐ 0°Ñ 3À&ÐWcÔd×iÑiÓkó(Ð$ð
 ×(Ñ(ˆØ�H˜A�,ˆÜ[^Ð_eÓ[fÐijÒ[jœCÓWÐ>VÔWÔWÐpuÐÜ:=¸f»+Èº/¤¨¨vÔ 6ÈuÐØ6GÐ2GÒ2aÐLaÐ/Ø2CÒ2aÐLaÐHaÐ/ð  %×>Ñ>Ø$ØØ3Ø3ó	 
Ðô Ð3×:Ñ:Ó<Ó=ˆô )JØØØ$ØØ3ØØó)
Ñ%ˆ�oô .CØØØØ#Ø ØØó.
Ñ*ˆÐ*ð 2F×1KÑ1KÓ1M×j©¨¨AÐQRÐZiÒQi  1¡ÐjÐÑjÜÐ3×:Ñ:Ó<Ó=ˆð 	×1Ñ1°,ÀÑ2PÐRaÐchÐjvÔwð 	×&Ñ& Ð8OÐQ]Ô^ð Ð)Ø$×5Ñ5Ó7ò >‘��eØ%×,Ñ,¨TÕ2à!&§¡§¡¬u¯}©}Ó!=�E•Jð>ð ˆâ2Ü# E¨6Ó2ˆMð FZ×E_ÑE_ÓEa×#b¹T¸QÀ A q¬¨W«¨Ð'8Ñ$8Ð#bÐ Ñ#bÜ"Ð#7×#>Ñ#>Ó#@ÓAˆOàÐ%Ø_i×_oÑ_oÓ_q×rÑW[ÐWXÐZ[°1·<±<ÀÔ3H˜a¤ G£ Ñ/ÈaÐQRÑRÐr�
Ñrà6;×6FÑ6FÓ6H×6MÑ6MÓ6OÖ*m°ÐWX×WcÑWcÐdkÕWlª1Ò*mÐ'Ü'+¨M×,DÑ,DÓ,F×,KÑ,KÓ,MÓ'NÐ$Üô Øfuôô ô !ð&óð ð :N×9SÑ9SÓ9U×'V±°°A¨¨1©Ð'VÐ$Ñ'Và#(Ð à!ÐØ "ÐàÑ! f°
×0AÑ0AÓ0CÒ&CØ!Ð)Ø%)Ð"Ø"Ð.Ü—‘Ð/¸$Õ?Ø'7¸tÐ'CÒ'tÐHXÐYZÑH[×HdÑHdÐesÓHtÐ$Ø"Ð*Ñ3KÜ ð:óð ò
 (Ü#4°ZÀÓ#QÐ Ø@EÐ@QœC ›J×.Ñ.¨x¸Ô<ÐW`�	Ø#Ð+Ü!%§¡¨Ð@PÐQRÑ@SÓ!T‘JäŸW™WŸ[™[×-Ñ-Ð.>¸qÑ.A×.GÑ.GÌÏÉÏÉÓ.TÐUXÐVXÐ.YÓZ�Fð %5°\Ñ$B×$HÑ$HÓ$J÷"á ˜A˜qØÐ 4Ñ4ð -¨QÑ/°Ñ2ð"�Jñ "ð
 JT×IYÑIYÓI[×!\ÁÀÀA !¤R§W¡W§\¡\°&¸!Ó%<Ñ"<Ð!\�JÑ!\ä,EÀjÐR\Ó,]Ð)ð '1×&6Ñ&6Ó&8÷&ñ #˜˜dØ'¨Ñ-°Ò7ð Ø,0Ø'CÀDÑ'IØ!*ññ ð&Ð"ò &ð &(Ð"ð
 "ÐØ ÐÙÜ!)×!1Ñ!1Ó!3ÐØ "Ðð Ð'¬CÐ0@Ó,AÀAÒ,EÜ&Ÿ|™|Ð,<ÐC^Ô_ÑàÐ#Ø "˜tÐô ˜]×5Ñ5Ó7×<Ñ<Ó>Ó?ˆØÐ#Ø(×=Ñ=¸mÈ]Ð\kÓlˆMð Ð!©&Ü"3°JÀÓ"NÐÜ$ ]Ð4GÐUaÐUiÑPQÐopÕqàˆ
à*ó 7	ˆJàÐ2Ñ2Øà ˆLà×#Ñ# NÔ3Ù%Ü3Ô5¹là%‘àÐ&Ø Ð,Ø ×4Ñ4×AÑAÔEW×E_ÑE_Ò_à ×4Ñ4×?Ñ?ÐCdÑdÜ! ,×"BÑ"B×"MÑ"MÔOcÔdô  %Ÿ|™|¸
×8IÑ8IÓ8KÖ,h°1ÈqÐXgÒOgªQÒ,hÐijÑ,kÓl�ð ˜RÒÜ,Ø¨\ÈÐcoô�
ð
 BL×AQÑAQÓAS×q¹¸¸AÐWXÐ\pÒWpÐ.¨qÑ1°1Ñ4ÐqˆJÑqä)Ô+±LØÔ?ÀÈzÓZÑZ‘
ä%Ô'Ô0DÔ0FÉ|Ü8XØ!ØØØ!Ø0Ø)Ø(;Ø'9Ø'9Ø&7Ø!-Ø#;Ø'9Ø$3Ø +ô9Ñ5Ð"Ð$5ò& ðo7	ðt Ð)¬cÐ2DÓ.EÈÒ.IÙ6à×.Ñ.�Ù/Ø'9ò ˜ÜŸ™ÜŸG™GŸL™LÐ)<ÀÀÈTÐ>RÓSÜŸG™GŸL™LÐ)<ÀÀÈÈ+ÈÐVZÐ>[Ó\õðð
 Rd×QiÑQiÓQk×%lÁ:À3È¨¨°°#°Ð&7¸Ñ&>Ð%lÐ"Ñ%lÙ+Ü"Ð#5Ð7JÔKØ%)Ð"ñ ä" =Ð2CÐEWÔXÜ�M‰MÐ,Ô-àÐ#Ø'×IÑIÈ-ÐYeÐgmÓnˆLð Ð"ä˜Z×.Ñ.Ó0Ó1°!Ñ4ˆIð  Ÿ-™-›/ò 7�Ø—=‘= IÓ-Ø"(§)¡)¨IÓ"6�F•Kð7ñ 7à38×3IÑ3IÓ3K÷,Ù$/ D¨%ÐSW×SbÑSbÐciÕSj�D˜%‘Kð,Ð(ñ ,ð $<×#AÑ#AÓ#Cò ‘K�D˜%ä3IÈ%ÐQUÐW\Ð^pÓ3qÑ0�M =Ü/ØØŸ™ Ó+ØØØ%Ø%ÜŸ
™
 6Ñ*Ø#õ	ðô ˆz‹?˜QÒØŸ™ JÓ/ˆIØ )Ñ+ØØwñ�	ô Ð!DÀUÇ_Á_×E]ÑE]ÐD^Ð^cÐdmÐcnÐoÓpÐpÜˆÓ !Ò#ØŸ,™,×4Ñ4Ð<‘BÀ%Ç,Á,×B\ÑB\ˆEØ',§¡×'?Ñ'?À5Ñ'H”V—^’^ÌfÏkÉkˆFÙØ:Ð;XÐ:Yð Z!Ø!&§¡×!9Ñ!9Ð :¸"¸_Ð<Mð N!Ø!&§¡×!9Ñ!9Ð :ð ;ð —O‘O×,Ñ,Ð-ð .tðtõô �K‰KÐSÐTY×TcÑTc×TlÑTlÐSmÐmpÐqÔrÜˆ|Ó˜qÒ Ü�N‰NØ" 5§?¡?×#;Ñ#;Ð"<ð =Ø1Ð2Ð2NÈ|Ènð ]nðnõô
 �Ó! QÒ&Ü�K‰KØ% e§o¡o×&>Ñ&>Ð%?ð @Ø1Ð2ð 38Ø8=¿¹×8PÑ8PÐ7Qð Rðôô ˆÓ !Ò#Ø!%§¡ô 25°_ÐFWÓ1X÷ð á-˜Ñ-˜f fð ˜˜˜^¨F¨8Ð3JÈ6È(ÐRlÒmôó"Ðô �N‰NØ" 5§?¡?×#;Ñ#;Ð"<ð =Ø1Ð2ð 3Ø.Ð/ð 0<ð<ôð �l O°_ÐFXÐZdÐdÐdùók  kùó0 $cùó sùâ*mùó (Wùó6"ùó
 "]ùó&ùòz -iùó rùóL &mùó6,ùôlsl   Æ m0Æm0Ém6Ê ,m<Ë/nÌnÍ'nÒnÒ9+nÔnÚ	n
Ún
Ûn$Ûn$ßn*ân0â-n0ìn6c                 óè   — |d   j                  d«      r| j                  |||d   d d «      }d }||fS 	 ddlm}  |||d   dd¬«      \  }}||fS # t        $ r t
        j                  d«       ‚ w xY w)	Nr   r÷  iúÿÿÿr    )Ú$load_tf2_checkpoint_in_pytorch_modelT)Úallow_missing_keysrb  zÃLoading a TensorFlow model in PyTorch, requires both PyTorch and TensorFlow to be installed. Please see https://pytorch.org/ and https://www.tensorflow.org/install/ for installation instructions.)rX  Úload_tf_weightsÚmodeling_tf_pytorch_utilsr
  r]  r  r   )r  r!  rª  r
  r¿  r
  s         r‹   r©  zPreTrainedModel._load_from_tf`  s£   € à˜AÑ×'Ñ'¨Ô1à×'Ñ'¨¨vÐ7GÈÑ7JÈ3ÈBÐ7OÓPˆEØˆLð  �lÐ"Ð"ðÝ[á&JØÐ+¨AÑ.À4Ð]aô'Ñ#��|ð �lÐ"Ð"øô ò Ü—‘ð%ôð
 ðús   µA Á A1c                 ór   — 	 ddl m}  |||d   «      }|S # t        $ r t        j	                  d«       ‚ w xY w)Nr    )Ú%load_flax_checkpoint_in_pytorch_modelr   zËLoading a Flax model in PyTorch, requires both PyTorch and Flax to be installed. Please see https://pytorch.org/ and https://flax.readthedocs.io/en/latest/installation.html for installation instructions.)Úmodeling_flax_pytorch_utilsr  r]  r  r   )r  r!  r
  r  s       r‹   rª  zPreTrainedModel._load_from_flaxw  sK   € ð
	ÝZá9¸%ÐAQÐRSÑATÓUˆEð ˆøô ò 	Ü�L‰Lð.ôð
 ð	ús   ‚ – 6c           
      óx  — |D �ch c]%  }dj                  |j                  d«      d d «      ’Œ' }}|j                  |D �ch c]H  }t        |«      dkD  sŒ|d   j	                  «       sŒ&dj                  |j                  d«      d d «      ’ŒJ c}«      }g }| j                  «       D ]‡  \  }}|r1| j                  › d�}	|j                  |	«      r|t        |	«      d  n|}n9|r7t        |«      dkD  rdj                  | j                  |g«      n| j                  }||v sŒw|j                  |«       Œ‰ |S c c}w c c}w )Nrþ   r�   r   éþÿÿÿ)	r  rß  Úunionr  r_  rl  r7  r9  r‹  )
ro  r…  Ú
add_prefixÚremove_prefixr0  rr  Úretrieved_modulesr²   rÈ   rÎ  s
             r‹   Úretrieve_modules_from_namesz+PreTrainedModel.retrieve_modules_from_names†  s1  € Ø@EÖF¸�s—x‘x §	¡	¨#£¨s°Ð 3Õ4ÐFˆÐFð "×'Ñ'Ø6;Öb¨s¼sÀ3»xÈ!»|ÐPSÐTVÑPW×P_ÑP_ÕPaˆS�X‰X�c—i‘i “n S bÐ)Õ*Òbó
ˆð Ðà ×.Ñ.Ó0ò 	1‰LˆD�&ÙØ!×3Ñ3Ð4°AÐ6�Ø/3¯©¸wÔ/G�tœC ›L˜NÑ+ÈT‘ÙÜCFÀtÃ9ÈqÂ=�s—x‘x ×!7Ñ!7¸Ð >Ô?ÐVZ×VlÑVl�à�{Ò"Ø!×(Ñ(¨Õ0ð	1ð !Ð ùò) Gùò
 cs   …*D2Á D7ÁD7Á(%D7c                 ó�   — t        |t        «      s|j                  }ddlmc m} t        ||«      st        |› d�«      ‚|| _        y)aã  
        Register this class with a given auto class. This should only be used for custom models as the ones in the
        library are already mapped with an auto class.

        <Tip warning={true}>

        This API is experimental and may have some slight breaking changes in the next releases.

        </Tip>

        Args:
            auto_class (`str` or `type`, *optional*, defaults to `"AutoModel"`):
                The auto class to register this new model with.
        r   Nz is not a valid auto class.)	rb  rÜ   r  Útransformers.models.autoÚmodelsr  r¨  rö   r  )r  Ú
auto_classÚauto_modules      r‹   Úregister_for_auto_classz'PreTrainedModel.register_for_auto_class�  sC   € ô  ˜*¤cÔ*Ø#×,Ñ,ˆJç6Ð6ä�{ JÔ/Ü 
˜|Ð+FÐGÓHÐHà$ˆ�rŠ   c                 óÚ   — t        «       st        d«      ‚ddlm} t	        j
                  |«      t	        j
                  d«      k  rt        d|› d�«      ‚ddlm} |j                  | «      S )a(  
        Converts the model to use [PyTorch's native attention
        implementation](https://pytorch.org/docs/stable/generated/torch.nn.MultiheadAttention.html), integrated to
        Transformers through [Optimum library](https://huggingface.co/docs/optimum/bettertransformer/overview). Only a
        subset of all Transformers models are supported.

        PyTorch's attention fastpath allows to speed up inference through kernel fusions and the use of [nested
        tensors](https://pytorch.org/docs/stable/nested.html). Detailed benchmarks can be found in [this blog
        post](https://medium.com/pytorch/bettertransformer-out-of-the-box-performance-for-huggingface-transformers-3fbe27d50ab2).

        Returns:
            [`PreTrainedModel`]: The model converted to BetterTransformer.
        ú<The package `optimum` is required to use Better Transformer.r   r’   ú1.7.0úEPlease install optimum>=1.7.0 to use Better Transformer. The version ú was found.©ÚBetterTransformer)	rV   r]  Úoptimum.versionr“   r   rc  Úoptimum.bettertransformerr$  Ú	transform©ro  Úoptimum_versionr$  s      r‹   Úto_bettertransformerz$PreTrainedModel.to_bettertransformer·  sf   € ô $Ô%ÜÐ\Ó]Ð]åBä�=‰=˜Ó)¬G¯M©M¸'Ó,BÒBÜØWÐXgÐWhÐhsÐtóð õ 	@à ×*Ñ*¨4Ó0Ð0rŠ   c                 óÚ   — t        «       st        d«      ‚ddlm} t	        j
                  |«      t	        j
                  d«      k  rt        d|› d�«      ‚ddlm} |j                  | «      S )a  
        Reverts the transformation from [`~PreTrainedModel.to_bettertransformer`] so that the original modeling is
        used, for example in order to save the model.

        Returns:
            [`PreTrainedModel`]: The model converted back to the original modeling.
        r  r   r’   r   r!  r"  r#  )	rV   r]  r%  r“   r   rc  r&  r$  rÂ  r(  s      r‹   Úreverse_bettertransformerz)PreTrainedModel.reverse_bettertransformerÓ  sf   € ô $Ô%ÜÐ\Ó]Ð]åBä�=‰=˜Ó)¬G¯M©M¸'Ó,BÒBÜØWÐXgÐWhÐhsÐtóð õ 	@à ×(Ñ(¨Ó.Ð.rŠ   c           
      óh  — t        |«      s(t        j                  j                  «       s
t	        «       ry|€| j
                  j                  €y| j
                  j                  |dd…ddgf   v �rCd}| j
                  j                  �-| j
                  j                  | j
                  j                  k(  s†| j
                  j                  �-| j
                  j                  | j
                  j                  k(  sC| j
                  j                  ��| j
                  j                  | j
                  j                  k(  rb|d| j
                  j                  › d| j
                  j                  › d| j
                  j                  › d| j
                  j                  › d	�	z  }t        j                  |«       yy)
zv
        Shows a one-time warning if the input_ids appear to contain padding and no attention mask was given.
        Nr�   r   zÈWe strongly recommend passing in an `attention_mask` since your input_ids may be padded. See https://huggingface.co/docs/transformers/troubleshooting#incorrect-output-when-padding-tokens-arent-masked.z5
You may ignore this warning if your `pad_token_id` (z&) is identical to the `bos_token_id` (z), `eos_token_id` (z), or the `sep_token_id` (z ), and your input is not padded.)rg   r‚   ÚjitÚ
is_tracingrh   rª  Úpad_token_idÚbos_token_idÚeos_token_idÚsep_token_idr  r  )ro  rÅ  rŒ  Úwarn_strings       r‹   Ú%warn_if_padding_and_no_attention_maskz5PreTrainedModel.warn_if_padding_and_no_attention_maské  so  € ô ˜YÔ'¬5¯9©9×+?Ñ+?Ô+AÔE]ÔE_ØàÐ&¨D¯K©K×,DÑ,DÐ,LØð �;‰;×#Ñ# y²°R¸°G°Ñ'<Ò<ðFð ð —‘×)Ñ)Ð5¸$¿+¹+×:RÑ:RÐVZ×VaÑVa×VnÑVnÒ:nØ—K‘K×,Ñ,Ð8¸T¿[¹[×=UÑ=UÐY]×YdÑYd×YqÑYqÒ=qØ—K‘K×,Ñ,Ð8¸T¿[¹[×=UÑ=UÐY]×YdÑYd×YqÑYqÒ=qàØLÈTÏ[É[×MeÑMeÐLfð g.Ø.2¯k©k×.FÑ.FÐ-GÐGZÐ[_×[fÑ[f×[sÑ[sÐZtð u.Ø.2¯k©k×.FÑ.FÐ-GÐGgðiñ�ô ×Ñ Õ,ð) =rŠ   c                 óN   — | j                   �yt        | j                  dd«      �yy)zJ
        Returns whether the model has a tensor parallelism plan.
        NTrÜ  F)rÜ  r€  r  rs  s    r‹   r§  z PreTrainedModel.supports_tp_plan  s*   € ð
 �=‰=Ð$Øä�4—?‘? J°Ó5ÐAØØrŠ   c                 óN   — | j                   �yt        | j                  dd «      �yy)NTrâ  F)râ  r€  r  rs  s    r‹   Úsupports_pp_planz PreTrainedModel.supports_pp_plan  s(   € à�=‰=Ð$Øä�4—?‘? J°Ó5ÐAØØrŠ   c                 ó¨   — t        | d«      r| j                  S t        | dd «      }|�|t        vrt        j                  d|› d�«       d}t        |   S )NÚ_loss_functionrÑ  z`loss_type=zY` was set in the config but it is unrecognised.Using the default loss: `ForCausalLMLoss`.ÚForCausalLM)r¨  r:  r€  r3   r  r  )ro  rÑ  s     r‹   Úloss_functionzPreTrainedModel.loss_function!  se   € ä�4Ð)Ô*Ø×&Ñ&Ð&ä˜D +¨tÓ4ˆ	àÐ 	´Ñ =Ü×ÑØ˜i˜[ð )=ð >ôð &ˆIÜ˜IÑ&Ð&rŠ   c                 ó   — || _         y r¨   )r:  )ro  rß  s     r‹   r<  zPreTrainedModel.loss_function0  s
   € à#ˆÕrŠ   Úcompile_configc                 óL  — d| j                   j                  v r| j                  S t        | j                  dt        «       «      }t        | d«      rt        | d|«      |k7  r:|| _        t        j                  | j                  fi |j                  «       ¤Ž| _        | j                  S )aŒ  Return a `torch.compile`'d version of `self.__call__`. This is useful to dynamically choose between
        non-compiled/compiled `forward` during inference, especially to switch between prefill (where we don't
        want to use compiled version to avoid recomputing the graph with new shapes) and iterative decoding
        (where we want the speed-ups of compiled version with static shapes).Úllama4r>  Ú_compiled_callÚ_last_compile_config)rª  Ú
model_typeÚ__call__r€  rÕ  r$   r¨  rB  r‚   r  r­  rA  )ro  r>  Údefault_configs      r‹   Úget_compiled_callz!PreTrainedModel.get_compiled_call4  s‹   € ð �t—{‘{×-Ñ-Ñ-Ø—=‘=Ð Ü  ×!7Ñ!7Ð9IÌ=Ë?Ó[ˆä˜Ð.Ô/Ü�tÐ3°^ÓDÈÒVà(6ˆDÔ%Ü"'§-¡-°·±Ñ"ZÀ×AWÑAWÓAYÑ"ZˆDÔØ×"Ñ"Ð"rŠ   c                 ó   — | j                   S r¨   )Ú_supports_attention_backend)r  s    r‹   Úis_backend_compatiblez%PreTrainedModel.is_backend_compatibleE  s   € à×.Ñ.Ð.rŠ   r1  r2  c           	      óò  — |du}t        «       rJt        «       s@|s>| j                  «       D ]*  \  }}t        j                  ||d¬«      }t        | ||«       Œ, y| j                  «       }	|D ]Š  }|	|   }|j                  t        j                  d«      k(  sŒ+t        j                  ||d¬«      }|r"t        |dd«      s|j                  | ||i ¬«      st        | ||«       Œu|j                  | ||d|	|«       ŒŒ y)zŸMove the missing keys (keys that are part of the model parameters, but were NOT found in the loaded state dicts) back
        from meta device to cpu.
        Nr  )rå   rÙ   rT  rÐ  F)r*  rž  rø   )rŒ   r‘   r"  r‚   Ú
empty_likerµ  rø   rÙ   r€  rÑ  rÒ  )
ro  r1  r2  rå   r¡  rJ  r0  rÜ  rß  rM  s
             r‹   ré  z3PreTrainedModel._move_missing_keys_from_meta_to_cpuI  s  € ð $¨4Ð/ˆô ÔÔ%9Ô%;ÁLà"×3Ñ3Ó5ò =‘
��UÜ×(Ñ(¨°eÀEÔJ�Ü*¨4°°eÕ<ð=ð àŸ?™?Ó,ÐØò 	tˆCØ$ SÑ)ˆEà�|‰|œuŸ|™|¨FÓ3Ó3Ü×(Ñ(¨°eÀEÔJ�á$Ü Ð.PÐRWÔXØ'×=Ñ=¸dÐPUÐbeÐrtÐ=Ôuä.¨t°S¸%Õ@à ×7Ñ7¸¸eÀSÈ%ÐQaÐcrÕsñ	trŠ   r.  c           	      óÌ  — |sŠt        | |«      }t        | j                  j                  d¬«      d«      rq| j                  j                  d¬«      j                  rK| j                  «       }|�9t        |d«      r|j                  €!d|_        nt        | j                  «       «      }t        «       rŽ|sŒt        t        t        j                  j                  d„ |j!                  «       D «       «      «      «      }t"        j$                  j'                  |d¬«      5  | j)                  | j*                  «       ddd«       y| j)                  | j*                  «       y# 1 sw Y   yxY w)	aÆ  Initialize the missing keys (keys that are part of the model parameters, but were NOT found in the loaded state dicts), according to
        `_initialize_weights`. Indeed, since the corresponding weights are missing from the state dict, they will not be replaced and need to
        be initialized correctly (i.e. weight initialization distribution).
        Also take care of setting the `_is_hf_initialized` flag for keys that are not missing.
        TrF  rH  NrÛ  c              3   ó@   K  — | ]  }|j                  d ¬«      –— Œ y­w)F)ÚrecurseN)rØ   )r4  r†  s     r‹   r6  z;PreTrainedModel._initialize_missing_keys.<locals>.<genexpr>‰  s#   è ø€ ò 2Ø@I˜	×,Ñ,°UÐ,×;ñ2ùs   ‚r   ry  )rs  r¨  rª  rM  rH  rA  rÛ  rn  r  rl  r)   r  r  Ú	itertoolsÚchainÚfrom_iterablerõ   rý  rþ  r|  rÅ  rD  )ro  r.  rG  rJ  rp  rP  Únot_initialized_parameterss          r‹   rê  z(PreTrainedModel._initialize_missing_keysl  s5  € ñ 'Ü)CÀDÈ+Ó)VÐ&ô ˜Ÿ™×3Ñ3¸DÐ3ÓAÐCXÔYØ—K‘K×/Ñ/¸Ð/Ó=×QÒQà$(×$>Ñ$>Ó$@Ð!Ø$Ð0ä"Ð#4°fÔ=ÐAR×AWÑAWÐA_Ø?CÐ)Õ<ä)-¨d×.@Ñ.@Ó.BÓ)CÐ&ä%Ô'±Ü)-ÜÜ—O‘O×1Ñ1ñ 2ØMg×MnÑMnÓMpô2ó óó*Ð&ô —‘×2Ñ2Ð3MÐ]^Ð2Ó_ñ 5Ø—
‘
˜4×3Ñ3Ô4÷5ð 5ð �J‰J�t×/Ñ/Õ0÷5ð 5ús   ÄEÅE#rý  c                 ó¤   — 	 | j                  |«      S # t        $ r Y nw xY w	 | j                  |«      S # t        $ r Y nw xY wt        d|› d�«      ‚)a  
        Return the parameter or buffer given by `target` if it exists, otherwise throw an error. This combines
        `get_parameter()` and `get_buffer()` in a single handy function. Note that it only work if `target` is a
        leaf of the model.
        ú`z&` is neither a parameter nor a buffer.)Úget_parameterÚAttributeErrorÚ
get_buffer)ro  rý  s     r‹   r¤  z'PreTrainedModel.get_parameter_or_buffer“  se   € ð	Ø×%Ñ% fÓ-Ð-øÜò 	Ùð	úð	Ø—?‘? 6Ó*Ð*øÜò 	Ùð	úä˜q  Ð(NÐOÓPÐPs   ‚ “	ž£4 ´	A ¿A )FNNT)NNTFr½  )NNTr1  )NFTr¨   r¿  )NFF)FNNNNNNNNNTr¾  )Ú	AutoModel)rÉ   r–   )‡r  rV  rW  rÀ  rž  r7  r¸  rñ  r  rØ  r®  r×  r>  r?  r  r}  Úis_parallelizablerì  Ú_is_statefulr
  r  r  Ú_supports_cache_classÚ_supports_static_cacheÚ_supports_quantized_cacherÜ  râ  rH  rÂ  r   rÜ   r‚   r   rÇ  rO  r"   rÎ  rç  ré  rÞ  r   r   ró  ÚclassmethodrÄ   r  rÃ  r   rå   r�   rÏ  r  r   rÛ   r  rÓ  r  r  r  r8  r;  r6  r?  rA  r¬   rD  rR  rÁ  rO  rN  r$  r¨  r  r{  r‚  r—  rƒ  r�  r˜  r™  rš  r¿  r   rÁ  rÝ  rÄ  rí  r   r   rÔ  rÛ  rß  r$  r†   ÚPathLiker  r
  r	   rL   rä  rP  r  rÍ  r\  rë   r`  r   r•   rŸ  rÉ  r×  rÙ  r  r<   rÊ  ÚPatternr«  r©  rª  r  r  r*  r,  r5  r§  r8  r<  Úsetterr$   rF  rI  ré  rê  r¤  Ú__classcell__©r  s   @r‹   r–   r–   Ó  sÁ
  ø„ ñð6 €LØÐØ!€OØ€Jà€KØÐØ"&ÐØ Ðð '+Ð#ð *.Ð&ð #ÐàÐàÐØ&+Ð#Ø€Lð #Ðð €Nð  Ðð "ÐØ"Ðð !&Ðð €Hð €Hð
 #(Ðàð9˜d 3¨¯©Ð#4Ñ5ò 9ó ð9ð ð˜3ò ó ðð!>Ð/õ !>òF%òN
-ò;ð, 5¨¨c©°C¨Ñ#8ð ,¸Tó ,ð@ Ø ñ7ó !ó ð7ðr ð ',Ø-1Ø;?Ø!%ñfð  $ðfð ˜eŸk™kÑ*ð	fð
 ˜U 3¨¨S°#¨X©Ð#6Ñ7Ñ8ðfð òfó ðfðP ð¨U¯[©[ð ¸U¿[¹[ò ó ðð4 ð;˜BŸI™Iò ;ó ð;ð ð$˜Tò $ó ð$ðL ð .2Ø;?Ø!%Ø %ñhð ˜eŸk™kÑ*ðhð ˜U 3¨¨S°#¨X©Ð#6Ñ7Ñ8ð	hð
 ðhð ðhð 
òhó ðhðT ñ¸Tð ÐN^ò ó ðð: ñÀ$ð ÐScò ó ðò8	pò*ð& b§i¡ió &ð&¨"¯)©)ó &ð r§y¡yó òò)ò&ð6 ðWØ—‘ðWØ%'§Y¡YðWØCFðWØ[^òWó ðWòrMð('°ó 'ð> )-Ø,0Ø"ñ	7à  ™ð7ð % S™Mð7ð ð	7ð
 
�‰ó7ór&+ðV )-Ø,0Ø"ñTàŸ™ðTð ! ™ðTð % S™Mð	Tð
 ðTð 
�‰óTðr )-Ø%*Ø"ñxà—Y‘Yðxð ! ™ðxð ˜T‘Nð	xð
 ðxð 
�‰óxòtð@ ó@ò0dò
dð
Àcó 
ð
¨¨r¯|©|¸UÀ2Ç<Á<Ñ=PÐ/PÑ)Qó 
òð"5¨$¨s°D¸±I¨~Ñ*>ó 5ó"(.ðT :>Ðgqñ °$ð Ð\dó ò,/ð. ðn¨4ò nó ðnð !%Ø%)Ø"'§*¡*Ø!Ø*/Ø#'Ø!%Ø,0Ø!%ñRà˜c 2§;¡;Ð.Ñ/ðRð ðRð ˜T‘Nð	Rð
  ðRð ðRð ˜c 3˜h™ðRð !ðRð ˜#‘ðRð ˜˜c 4˜iÑ(Ñ)ðRð óRñh ˆ>×%Ñ%Ó&ó4ó 'ð4óñ$ ˆ5�8‰8�?‰?×ÑÓ ó1ó !ð1ñ$ ˆ5�8‰8�?‰?×ÑÓó)+ó ð)+ôV'ô(ð ð¨Dð Àdò ó ðð Ø ð
 GKØ7;Ø(-Ø$Ø!&Ø,0ØØ*.Ø!ò|ØÐ-Ñ.ð|à'/°°c¸2¿;¹;Ð6FÑ0GÑ'Hð|ð ˜Ð/°°b·k±kÐAÑBÑCð	|ð
 ˜E # r§{¡{Ð"2Ñ3Ñ4ð|ð "&ð|ð ð|ð ð|ð ˜˜c 4˜iÑ(Ñ)ð|ð ð|ð " $™ð|ð ð|ð 
%ò|ó !ó ð|ð| ð¨ð °°s¸D°yÑ1Aò ó ðð8 15Ø8=Ø8=ñ:$à˜c™ð:$ð ˜d 3¨ 8™nÑ-ð:$ð 26ð	:$ð
 26ó:$ðx ð¨E°#°t°)Ñ,<ò ó ðòfð ð ).Ø+/Ø%)Ø-1Ø-1Ø'+Ø.2Ø37ØLPØ04Ø!ñ!aeà ðaeð ˜T‘Nðaeð # 4¨¡9Ñ-ð	aeð
 (0°¡}ðaeð "&ðaeð # 4™.ðaeð ˜T‘Nðaeð & c™]ðaeð % T™Nðaeð ˜Ÿ™Ñ$ðaeð ˜{Ñ+ðaeð % R§Z¡ZÑ0ðaeð ÐHÑIðaeð ˜d 3¨ 8™nÑ-ðaeð  ò!aeó ðaeðF ñ#ó ð#ð, ñó ðó!ð. ò%ó ð%ó21ò8/ò,!-ðF ñ	ó ð	ð ñó ðð ñ'ó ð'ð ×Ññ$ó ð$ð#°ó #ð" ñ/ó ð/ð!tà˜3‘ið!tð ˜c™ð!tð ˜Ÿ™Ñ$ð	!tð
 ˜{Ñ+ð!tð 
ó!tðF%1à˜#‘Yð%1ð "&ð%1ð ð	%1ð
 
ó%1ðNQ¨c÷ QrŠ   rX  z
model file)ÚobjectÚobject_classÚobject_filesc                   ó‚   ‡ — e Zd ZdZdefˆ fd„Z	 ddej                  deej                     dej                  fd„Z	ˆ xZ
S )	ÚPoolerStartLogitszÒ
    Compute SQuAD start logits from sequence hidden states.

    Args:
        config ([`PretrainedConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model.
    rª  c                 ól   •— t         ‰| �  «        t        j                  |j                  d«      | _        y )Nr    )rÍ  rÎ  r   r—  Úhidden_sizeÚdense©ro  rª  r  s     €r‹   rÎ  zPoolerStartLogits.__init__´  s&   ø€ Ü‰ÑÔÜ—Y‘Y˜v×1Ñ1°1Ó5ˆ�
rŠ   Úhidden_statesÚp_maskrÉ   c                 ó¾   — | j                  |«      j                  d«      }|�:t        | «      t        j                  k(  r|d|z
  z  d|z  z
  }|S |d|z
  z  d|z  z
  }|S )aì  
        Args:
            hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
                The final hidden states of the model.
            p_mask (`torch.FloatTensor` of shape `(batch_size, seq_len)`, *optional*):
                Mask for tokens at invalid position, such as query and special symbols (PAD, SEP, CLS). 1.0 means token
                should be masked.

        Returns:
            `torch.FloatTensor`: The start logits for SQuAD.
        r�   r    éÜÿ  çêŒ 9Y>)F)rk  Úsqueezeró   r‚   r(  )ro  rm  rn  Úxs       r‹   ÚforwardzPoolerStartLogits.forward¸  sp   € ð �J‰J�}Ó%×-Ñ-¨bÓ1ˆàÐÜ" 4Ó(¬E¯M©MÒ9Ø˜˜V™Ñ$ u¨v¡~Ñ5�ð ˆð ˜˜V™Ñ$ t¨f¡}Ñ4�àˆrŠ   r¨   )r  rV  rW  rÀ  r"   rÎ  r‚   ÚFloatTensorr   rt  rb  rc  s   @r‹   rh  rh  «  sP   ø„ ñð6Ð/õ 6ð
 W[ñØ"×.Ñ.ðØ8@À×ARÑARÑ8Sðà	×	Ñ	÷rŠ   rh  c                   óÂ   ‡ — e Zd ZdZdefˆ fd„Z	 	 	 d
dej                  deej                     deej                     deej                     dej                  f
d	„Z
ˆ xZS )ÚPoolerEndLogitszü
    Compute SQuAD end logits from sequence hidden states.

    Args:
        config ([`PretrainedConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model and the `layer_norm_eps`
            to use.
    rª  c                 ób  •— t         ‰| �  «        t        j                  |j                  dz  |j                  «      | _        t        j                  «       | _        t        j                  |j                  |j                  ¬«      | _        t        j                  |j                  d«      | _
        y )NrŠ  )Úepsr    )rÍ  rÎ  r   r—  rj  Údense_0ÚTanhÚ
activationÚ	LayerNormÚlayer_norm_epsÚdense_1rl  s     €r‹   rÎ  zPoolerEndLogits.__init__Û  st   ø€ Ü‰ÑÔÜ—y‘y ×!3Ñ!3°aÑ!7¸×9KÑ9KÓLˆŒÜŸ'™'›)ˆŒÜŸ™ f×&8Ñ&8¸f×>SÑ>SÔTˆŒÜ—y‘y ×!3Ñ!3°QÓ7ˆ�rŠ   rm  Ústart_statesÚstart_positionsrn  rÉ   c                 ó  — |€	|€J d«       ‚|�R|j                   dd \  }}|dd…ddf   j                  dd|«      }|j                  d|«      }|j                  d|d«      }| j                  t	        j
                  ||gd¬«      «      }| j                  |«      }| j                  |«      }| j                  |«      j                  d«      }|�:t        | «      t        j                  k(  r|d|z
  z  d|z  z
  }|S |d|z
  z  d|z  z
  }|S )	aë  
        Args:
            hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
                The final hidden states of the model.
            start_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`, *optional*):
                The hidden states of the first tokens for the labeled span.
            start_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                The position of the first token for the labeled span.
            p_mask (`torch.FloatTensor` of shape `(batch_size, seq_len)`, *optional*):
                Mask for tokens at invalid position, such as query and special symbols (PAD, SEP, CLS). 1.0 means token
                should be masked.

        <Tip>

        One of `start_states` or `start_positions` should be not `None`. If both are set, `start_positions` overrides
        `start_states`.

        </Tip>

        Returns:
            `torch.FloatTensor`: The end logits for SQuAD.
        Nú7One of start_states, start_positions should be not Noner  r�   ©rz  r    rp  rq  )rK  r¡  Úgatherrz  r‚   r‰  r|  r}  r  rr  ró   r(  )ro  rm  r€  r�  rn  ÚslenÚhszrs  s           r‹   rt  zPoolerEndLogits.forwardâ  s'  € ð: Ð'¨?Ð+Fð 	
ØEó	
ÐFð Ð&Ø%×+Ñ+¨B¨CÐ0‰IˆD�#Ø-ªa°°t¨mÑ<×CÑCÀBÈÈCÓPˆOØ(×/Ñ/°°OÓDˆLØ'×.Ñ.¨r°4¸Ó<ˆLà�L‰LœŸ™ M°<Ð#@ÀbÔIÓJˆØ�O‰O˜AÓˆØ�N‰N˜1ÓˆØ�L‰L˜‹O×#Ñ# BÓ'ˆàÐÜ" 4Ó(¬E¯M©MÒ9Ø˜˜V™Ñ$ u¨v¡~Ñ5�ð ˆð ˜˜V™Ñ$ t¨f¡}Ñ4�àˆrŠ   ©NNN©r  rV  rW  rÀ  r"   rÎ  r‚   ru  r   Ú
LongTensorrt  rb  rc  s   @r‹   rw  rw  Ñ  s‚   ø„ ñð8Ð/õ 8ð 59Ø6:Ø.2ñ1à×(Ñ(ð1ð ˜u×0Ñ0Ñ1ð1ð " %×"2Ñ"2Ñ3ð	1ð
 ˜×*Ñ*Ñ+ð1ð 
×	Ñ	÷1rŠ   rw  c                   ó¼   ‡ — e Zd ZdZˆ fd„Z	 	 	 d	dej                  deej                     deej                     deej                     dej                  f
d„Z	ˆ xZ
S )
ÚPoolerAnswerClasszí
    Compute SQuAD 2.0 answer class from classification and start tokens hidden states.

    Args:
        config ([`PretrainedConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model.
    c                 ó  •— t         ‰| �  «        t        j                  |j                  dz  |j                  «      | _        t        j                  «       | _        t        j                  |j                  dd¬«      | _        y )NrŠ  r    F)rÛ  )	rÍ  rÎ  r   r—  rj  rz  r{  r|  r  rl  s     €r‹   rÎ  zPoolerAnswerClass.__init__  sX   ø€ Ü‰ÑÔÜ—y‘y ×!3Ñ!3°aÑ!7¸×9KÑ9KÓLˆŒÜŸ'™'›)ˆŒÜ—y‘y ×!3Ñ!3°Q¸UÔCˆ�rŠ   rm  r€  r�  Ú	cls_indexrÉ   c                 óþ  — |j                   d   }|€	|€J d«       ‚|�<|dd…ddf   j                  dd|«      }|j                  d|«      j                  d«      }|�=|dd…ddf   j                  dd|«      }|j                  d|«      j                  d«      }n|dd…ddd…f   }| j	                  t        j                  ||gd¬«      «      }| j                  |«      }| j                  |«      j                  d«      }|S )a¸  
        Args:
            hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
                The final hidden states of the model.
            start_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`, *optional*):
                The hidden states of the first tokens for the labeled span.
            start_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                The position of the first token for the labeled span.
            cls_index (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                Position of the CLS token for each sentence in the batch. If `None`, takes the last token.

        <Tip>

        One of `start_states` or `start_positions` should be not `None`. If both are set, `start_positions` overrides
        `start_states`.

        </Tip>

        Returns:
            `torch.FloatTensor`: The SQuAD 2.0 answer class.
        r�   Nrƒ  r  r„  )	rK  r¡  r…  rr  rz  r‚   r‰  r|  r  )ro  rm  r€  r�  rŽ  r‡  Úcls_token_staters  s           r‹   rt  zPoolerAnswerClass.forward%  s  € ð: ×!Ñ! "Ñ%ˆØÐ'¨?Ð+Fð 	
ØEó	
ÐFð Ð&Ø-ªa°°t¨mÑ<×CÑCÀBÈÈCÓPˆOØ(×/Ñ/°°OÓD×LÑLÈRÓPˆLàÐ Ø!¢! T¨4 -Ñ0×7Ñ7¸¸BÀÓDˆIØ+×2Ñ2°2°yÓA×IÑIÈ"ÓM‰Oà+ªA¨r²1¨HÑ5ˆOà�L‰LœŸ™ L°/Ð#BÈÔKÓLˆØ�O‰O˜AÓˆØ�L‰L˜‹O×#Ñ# BÓ'ˆàˆrŠ   rˆ  )r  rV  rW  rÀ  rÎ  r‚   ru  r   rŠ  rt  rb  rc  s   @r‹   rŒ  rŒ    s{   ø„ ñôDð 59Ø6:Ø04ñ/à×(Ñ(ð/ð ˜u×0Ñ0Ñ1ð/ð " %×"2Ñ"2Ñ3ð	/ð
 ˜E×,Ñ,Ñ-ð/ð 
×	Ñ	÷/rŠ   rŒ  c                   ó  — e Zd ZU dZdZeej                     ed<   dZ	eej                     ed<   dZ
eej                     ed<   dZeej                     ed<   dZeej                     ed<   dZeej                     ed<   y)	ÚSquadHeadOutputaÜ  
    Base class for outputs of question answering models using a [`~modeling_utils.SQuADHead`].

    Args:
        loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned if both `start_positions` and `end_positions` are provided):
            Classification loss as the sum of start token, end token (and is_impossible if provided) classification
            losses.
        start_top_log_probs (`torch.FloatTensor` of shape `(batch_size, config.start_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
            Log probabilities for the top config.start_n_top start token possibilities (beam-search).
        start_top_index (`torch.LongTensor` of shape `(batch_size, config.start_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
            Indices for the top config.start_n_top start token possibilities (beam-search).
        end_top_log_probs (`torch.FloatTensor` of shape `(batch_size, config.start_n_top * config.end_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
            Log probabilities for the top `config.start_n_top * config.end_n_top` end token possibilities
            (beam-search).
        end_top_index (`torch.LongTensor` of shape `(batch_size, config.start_n_top * config.end_n_top)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
            Indices for the top `config.start_n_top * config.end_n_top` end token possibilities (beam-search).
        cls_logits (`torch.FloatTensor` of shape `(batch_size,)`, *optional*, returned if `start_positions` or `end_positions` is not provided):
            Log probabilities for the `is_impossible` label of the answers.

    NÚlossÚstart_top_log_probsÚstart_top_indexÚend_top_log_probsÚend_top_indexÚ
cls_logits)r  rV  rW  rÀ  r“  r   r‚   ru  rX  r”  r•  rŠ  r–  r—  r˜  r‰   rŠ   r‹   r’  r’  W  s‰   … ñð* )-€Dˆ(�5×$Ñ$Ñ
%Ó,Ø7;Ð˜ %×"3Ñ"3Ñ4Ó;Ø26€O�X˜e×.Ñ.Ñ/Ó6Ø59Ð�x × 1Ñ 1Ñ2Ó9Ø04€M�8˜E×,Ñ,Ñ-Ó4Ø.2€J�˜×*Ñ*Ñ+Ô2rŠ   r’  c                   ó,  ‡ — e Zd ZdZˆ fd„Z eee¬«      	 	 	 	 	 	 ddej                  de
ej                     de
ej                     de
ej                     de
ej                     d	e
ej                     d
edeeeej                     f   fd„«       Zˆ xZS )Ú	SQuADHeadzæ
    A SQuAD head inspired by XLNet.

    Args:
        config ([`PretrainedConfig`]):
            The config used by the model, will be used to grab the `hidden_size` of the model and the `layer_norm_eps`
            to use.
    c                 óÆ   •— t         ‰| �  «        |j                  | _        |j                  | _        t	        |«      | _        t        |«      | _        t        |«      | _	        y r¨   )
rÍ  rÎ  Ústart_n_topÚ	end_n_toprh  Ústart_logitsrw  Ú
end_logitsrŒ  Úanswer_classrl  s     €r‹   rÎ  zSQuADHead.__init__€  sO   ø€ Ü‰ÑÔØ!×-Ñ-ˆÔØ×)Ñ)ˆŒä-¨fÓ5ˆÔÜ)¨&Ó1ˆŒÜ-¨fÓ5ˆÕrŠ   )Úoutput_typerž  rm  r�  Úend_positionsrŽ  Úis_impossiblern  Úreturn_dictrÉ   c                 óX  — | j                  ||¬«      }|�»|�¹||||fD ]*  }	|	€Œ|	j                  «       dkD  sŒ|	j                  d«       Œ, | j                  |||¬«      }
t	        «       } |||«      } ||
|«      }||z   dz  }|�;|�9| j                  |||¬«      }t        j                  «       } |||«      }||dz  z  }|rt        |¬	«      S |fS |j                  «       \  }}}t        j                  j                  |d¬
«      }t        j                  || j                  d¬
«      \  }}|j                  d«      j!                  dd|«      }t        j"                  |d|«      }|j                  d«      j!                  d|dd«      }|j                  d«      j%                  |«      }|�|j                  d«      nd}| j                  |||¬«      }
t        j                  j                  |
d¬
«      }t        j                  || j&                  d¬
«      \  }}|j)                  d| j                  | j&                  z  «      }|j)                  d| j                  | j&                  z  «      }t        j*                  d||«      }| j                  |||¬«      }|s|||||fS t        |||||¬«      S )a÷  
        Args:
            hidden_states (`torch.FloatTensor` of shape `(batch_size, seq_len, hidden_size)`):
                Final hidden states of the model on the sequence tokens.
            start_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                Positions of the first token for the labeled span.
            end_positions (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                Positions of the last token for the labeled span.
            cls_index (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                Position of the CLS token for each sentence in the batch. If `None`, takes the last token.
            is_impossible (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
                Whether the question has a possible answer in the paragraph or not.
            p_mask (`torch.FloatTensor` of shape `(batch_size, seq_len)`, *optional*):
                Mask for tokens at invalid position, such as query and special symbols (PAD, SEP, CLS). 1.0 means token
                should be masked.
            return_dict (`bool`, *optional*, defaults to `False`):
                Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.

        Returns:
        )rn  Nr    r�   )r�  rn  rŠ  )r�  rŽ  g      à?)r“  r„  r  )r€  rn  z
blh,bl->bh)r€  rŽ  )r”  r•  r–  r—  r˜  )rž  rz  Úsqueeze_rŸ  r   r   r   ÚBCEWithLogitsLossr’  rU  ro  Úsoftmaxr‚   Útopkrœ  r�  r¡  r…  Ú	expand_asr�  rw  Úeinsum)ro  rm  r�  r¢  rŽ  r£  rn  r¤  rž  rs  rŸ  Úloss_fctÚ
start_lossÚend_lossÚ
total_lossr˜  Úloss_fct_clsÚcls_lossÚbszr†  r‡  Ústart_log_probsr”  r•  Ústart_top_index_expr€  Úhidden_states_expandedÚend_log_probsr–  r—  s                                 r‹   rt  zSQuADHead.forward‰  sÍ  € ð> ×(Ñ(¨¸vÐ(ÓFˆàÐ&¨=Ð+Dà% }°iÀÐOò #�Ø‘= Q§U¡U£W¨q£[Ø—J‘J˜r•Nð#ð
 Ÿ™¨ÈÐ`f˜ÓgˆJä'Ó)ˆHÙ! ,°Ó@ˆJÙ 
¨MÓ:ˆHØ$ xÑ/°1Ñ4ˆJàÐ$¨Ð)Bà!×.Ñ.¨}ÈoÐirÐ.Ós�
Ü!×3Ñ3Ó5�Ù'¨
°MÓB�ð ˜h¨™nÑ,�
á7B”?¨
Ô3ÐUÈÈÐUð +×/Ñ/Ó1‰NˆC��sÜ Ÿm™m×3Ñ3°LÀbÐ3ÓIˆOä38·:±:Ø ×!1Ñ!1°rô4Ñ0Ð ð #2×";Ñ";¸BÓ"?×"FÑ"FÀrÈ2ÈsÓ"SÐÜ Ÿ<™<¨°rÐ;NÓOˆLØ'×1Ñ1°!Ó4×;Ñ;¸BÀÀbÈ"ÓMˆLà%2×%<Ñ%<¸QÓ%?×%IÑ%IØó&Ð"ð .4Ð-?�V×%Ñ% bÔ)ÀTˆFØŸ™Ð)?ÈlÐci˜ÓjˆJÜŸM™M×1Ñ1°*À!Ð1ÓDˆMä/4¯z©zØ˜tŸ~™~°1ô0Ñ,Ð˜}ð !2× 6Ñ 6°r¸4×;KÑ;KÈdÏnÉnÑ;\Ó ]ÐØ)×.Ñ.¨r°4×3CÑ3CÀdÇnÁnÑ3TÓUˆMä Ÿ<™<¨°mÀ_ÓUˆLØ×*Ñ*¨=À|Ð_hÐ*ÓiˆJáØ+¨_Ð>OÐQ^Ð`jÐkÐkä&Ø(;Ø$3Ø&7Ø"/Ø)ôð rŠ   )NNNNNF)r  rV  rW  rÀ  rÎ  ra   r’  r"   r‚   ru  r   rŠ  rÃ  r   r   rt  rb  rc  s   @r‹   rš  rš  v  sè   ø„ ñô6ñ ¨?ÐIYÔZð 7;Ø48Ø04Ø48Ø.2Ø!ñ^à×(Ñ(ð^ð " %×"2Ñ"2Ñ3ð^ð   × 0Ñ 0Ñ1ð	^ð
 ˜E×,Ñ,Ñ-ð^ð   × 0Ñ 0Ñ1ð^ð ˜×*Ñ*Ñ+ð^ð ð^ð 
ˆ  e×&7Ñ&7Ñ 8Ð8Ñ	9ò^ó [ô^rŠ   rš  c                   ó‚   ‡ — e Zd ZdZdefˆ fd„Z	 ddej                  deej                     dej                  fd„Z
ˆ xZS )	ÚSequenceSummaryaÒ  
    Compute a single vector summary of a sequence hidden states.

    Args:
        config ([`PretrainedConfig`]):
            The config used by the model. Relevant arguments in the config class of the model are (refer to the actual
            config class of your model for the default values it uses):

            - **summary_type** (`str`) -- The method to use to make this summary. Accepted values are:

                - `"last"` -- Take the last token hidden state (like XLNet)
                - `"first"` -- Take the first token hidden state (like Bert)
                - `"mean"` -- Take the mean of all tokens hidden states
                - `"cls_index"` -- Supply a Tensor of classification token position (GPT/GPT-2)
                - `"attn"` -- Not implemented now, use multi-head attention

            - **summary_use_proj** (`bool`) -- Add a projection after the vector extraction.
            - **summary_proj_to_labels** (`bool`) -- If `True`, the projection outputs to `config.num_labels` classes
              (otherwise to `config.hidden_size`).
            - **summary_activation** (`Optional[str]`) -- Set to `"tanh"` to add a tanh activation to the output,
              another string or `None` will add no activation.
            - **summary_first_dropout** (`float`) -- Optional dropout probability before the projection and activation.
            - **summary_last_dropout** (`float`)-- Optional dropout probability after the projection and activation.
    rª  c                 ó  •— t         ‰| �  «        t        |dd«      | _        | j                  dk(  rt        ‚t        «       | _        t        |d«      rq|j                  ret        |d«      r(|j                  r|j                  dkD  r|j                  }n|j                  }t        j                  |j                  |«      | _        t        |dd «      }|rt        |«      n	t        «       | _        t        «       | _        t        |d«      r3|j"                  dkD  r$t        j$                  |j"                  «      | _        t        «       | _        t        |d	«      r5|j(                  dkD  r%t        j$                  |j(                  «      | _        y y y )
NÚsummary_typeÚlastÚattnÚsummary_use_projÚsummary_proj_to_labelsr   Úsummary_activationÚsummary_first_dropoutÚsummary_last_dropout)rÍ  rÎ  r€  rº  r=  r   Úsummaryr¨  r½  r¾  Ú
num_labelsrj  r   r—  r!   r|  Úfirst_dropoutrÀ  ÚDropoutÚlast_dropoutrÁ  )ro  rª  Únum_classesÚactivation_stringr  s       €r‹   rÎ  zSequenceSummary.__init__  sC  ø€ Ü‰ÑÔä# F¨N¸FÓCˆÔØ×Ñ Ò&ô &Ð%ä“zˆŒÜ�6Ð-Ô.°6×3JÒ3JÜ�vÐ7Ô8¸V×=ZÒ=ZÐ_e×_pÑ_pÐstÒ_tØ$×/Ñ/‘à$×0Ñ0�ÜŸ9™9 V×%7Ñ%7¸ÓEˆDŒLä# FÐ,@À$ÓGÐÙIZ¤NÐ3DÔ$EÔ`hÓ`jˆŒä%›ZˆÔÜ�6Ð2Ô3¸×8TÑ8TÐWXÒ8XÜ!#§¡¨F×,HÑ,HÓ!IˆDÔä$›JˆÔÜ�6Ð1Ô2°v×7RÑ7RÐUVÒ7VÜ "§
¡
¨6×+FÑ+FÓ GˆDÕð 8WÐ2rŠ   rm  rŽ  rÉ   c                 óü  — | j                   dk(  r|dd…df   }�n| j                   dk(  r|dd…df   }�n| j                   dk(  r|j                  d¬«      }ná| j                   d	k(  r½|€At        j                  |d
dd…dd…f   |j                  d   dz
  t        j
                  ¬«      }nX|j                  d«      j                  d«      }|j                  d|j                  «       dz
  z  |j                  d«      fz   «      }|j                  d|«      j                  d«      }n| j                   dk(  rt        ‚| j                  «      }| j                  |«      }| j                  |«      }| j!                  |«      }|S )ak  
        Compute a single vector summary of a sequence hidden states.

        Args:
            hidden_states (`torch.FloatTensor` of shape `[batch_size, seq_len, hidden_size]`):
                The hidden states of the last layer.
            cls_index (`torch.LongTensor` of shape `[batch_size]` or `[batch_size, ...]` where ... are optional leading dimensions of `hidden_states`, *optional*):
                Used if `summary_type == "cls_index"` and takes the last token of the sequence as classification token.

        Returns:
            `torch.FloatTensor`: The summary of the sequence hidden states.
        r»  Nr�   Úfirstr   r£  r    r„  rŽ  .r  rx  )r�   r¼  )rº  r£  r‚   Ú	full_likerK  Úlongr�  r¡  rz  rU  r…  rr  r=  rÄ  rÂ  r|  rÆ  )ro  rm  rŽ  r4  s       r‹   rt  zSequenceSummary.forward"  sn  € ð ×Ñ Ò&Ø"¢1 b 5Ñ)ŠFØ×Ñ 'Ò)Ø"¢1 a 4Ñ(ŠFØ×Ñ &Ò(Ø"×'Ñ'¨AÐ'Ó.‰FØ×Ñ +Ò-ØÐ Ü!ŸO™OØ! # r¨ rª1 *Ñ-Ø!×'Ñ'¨Ñ+¨aÑ/ÜŸ*™*ô‘	ð &×/Ñ/°Ó3×=Ñ=¸bÓA�	Ø%×,Ñ,¨U°i·m±m³oÈÑ6IÑ-JÈm×N`ÑN`ÐacÓNdÐMfÑ-fÓg�	à"×)Ñ)¨"¨iÓ8×@Ñ@ÀÓD‰FØ×Ñ &Ò(Ü%Ð%à×#Ñ# FÓ+ˆØ—‘˜fÓ%ˆØ—‘ Ó(ˆØ×"Ñ" 6Ó*ˆàˆrŠ   r¨   r‰  rc  s   @r‹   r¸  r¸  ë  sR   ø„ ñð2HÐ/õ Hð< Y]ñ)Ø"×.Ñ.ð)Ø;CÀE×DTÑDTÑ;Uð)à	×	Ñ	÷)rŠ   r¸  Ú	recursivec                 ó²   — t        «       r+i }|rt        d«      st        d«      ‚||d<   t        | fi |¤ŽS t        | d«      rt	        | j
                  «      S | S )a¡  
    Recursively unwraps a model from potential containers (as used in distributed training).

    Args:
        model (`torch.nn.Module`): The model to unwrap.
        recursive (`bool`, *optional*, defaults to `False`):
            Whether to recursively extract all cases of `module.module` from `model` as well as unwrap child sublayers
            recursively, not just the top-level distributed containers.
    z0.29.0zsSetting `recursive=True` to `unwrap_model` requires `accelerate` v0.29.0. Please upgrade your version of acceleraterÍ  rÈ   )rR   r  rr   r¨  r  rÈ   )r!  rÍ  rª   s      r‹   r  r  N  si   € ô Ô ØˆÙÜ*¨8Ô4Ü"ð Jóð ð '0��{Ñ#Ü*¨5Ñ;°FÑ;Ð;ô �5˜(Ô#Ü §¡Ó-Ð-àˆLrŠ   c           
      óÂ   — i }| j                  «       D ]D  \  }}|j                  |D �ci c]$  }||k(  s|j                  |› d�«      s|dk(  sŒ"||“Œ& c}«       ŒF |S c c}w )zT
    Expand a device map to return the correspondence parameter name to device.
    rþ   rk  )r®   r   r9  )r¸  Úparam_namesÚnew_device_maprÈ   rÙ   Úps         r‹   rë  rë  l  st   € ð €NØ$×*Ñ*Ó,ò 
‰ˆ�Ø×ÑØ +Öi˜1¨q°Fª{¸a¿l¹lÈfÈXÐUVÈ<Ô>XÐ\bÐfhÓ\hˆQ�‰YÒiõ	
ð
ð Ðùò js   ¨#A
ÁA
r   c           
      ó’  — |j                  «       D ��ci c]   \  }}|dvsŒ|t        j                  |«      “Œ" }}}t        |«      syt        rmt        j
                  j                  «       rOt        j                  dj                  | j                  D �cg c]  }t        j                  |«      ‘Œ c}«      «      nd}t        d„ «      }|j                  «       D ]   \  }	}| j                  |	«      }t        j                  |j                   «      |j#                  «       z  }
|�Kt        j$                  dd|	«      }|
|j'                  |«      rt        j
                  j)                  «       ndz  }
||xx   |
z  cc<   Œ¢ |j                  «       D ]®  \  }}|j*                  dk(  rp|j,                  �|j,                  nt        j.                  j1                  «       }t        j.                  j3                  |«      d	   }t5        |t7        d
|z  «      «      }t        j8                  ||z  t        j:                  |d¬«      }Œ° yc c}}w c c}w )aI  This function warm-ups the caching allocator based on the size of the model tensors that will reside on each
    device. It allows to have one large call to Malloc, instead of recursively calling it later when loading
    the model, which is actually the loading speed botteneck.
    Calling this function allows to cut the model loading time by a very large margin.

    A few facts related to loading speed (taking into account the use of this function):
    - When loading a model the first time, it is usually slower than the subsequent times, because the OS is very likely
    to cache the different state dicts (if enough ressources/RAM are available)
    - Trying to force the OS to cache the files in advance (by e.g. accessing a small portion of them) is really hard,
    and not a good idea in general as this is low level OS optimizations that depend on ressource usage anyway
    - As of 18/03/2025, loading a Llama 70B model with TP takes ~1 min without file cache, and ~13s with full file cache.
    The baseline, i.e. only loading the tensor shards on device and adjusting dtype (i.e. copying them) is ~5s with full cache.
    These numbers are reported for TP on 4 H100 GPUs.
    - It is useless to pre-allocate more than the model size in this function (i.e. using an `allocation_factor` > 1) as
    cudaMalloc is not a bottleneck at all anymore
    - Loading speed bottleneck is now almost only tensor copy (i.e. changing the dtype) and moving the tensors to the devices.
    However, we cannot really improve on those aspects obviously, as the data needs to be moved/copied in the end.
    rè  NrÁ  c                   ó   — y)Nr   r‰   r‰   rŠ   r‹   ú<lambda>z*caching_allocator_warmup.<locals>.<lambda>—  s   � rŠ   z\.\d+\.z.*.r    r  r   gffffffî?F)rå   rÙ   rÈ  )r®   r‚   rÙ   r  Ú_torch_distributed_availablerƒ   r…   rÊ  r  r  rÜ  rË  r   r¤  ÚmathÚprodrK  ry  Úsubr©  r›  rÓ  r,  r  r–  Úmem_get_infor|  r�   r^  r(  )r!  r   rå  rÜ  rÙ   Úaccelerator_device_mapræ  Útp_plan_regexÚtotal_byte_countrž  Úparam_byte_countÚgeneric_nameÚ
byte_countr,  Údevice_memoryr’  s                   r‹   rï  rï  x  só  € ð* :M×9RÑ9RÓ9T÷Ù(5¨¨vÐX^ÐfuÒXuˆŒu�|‰|˜FÓ#Ñ#ðÐñ ô Ð%Ô&Øõ (¬E×,=Ñ,=×,LÑ,LÔ,Nô 	�
‰
�3—8‘8¸¿¹ÖH°œRŸY™Y t�_ÒHÓIÔJàð ô
 #¡9Ó-ÐØ4×:Ñ:Ó<ò 	5Ñˆ
�FØ×-Ñ-¨jÓ9ˆäŸ9™9 U§[¡[Ó1°E×4FÑ4FÓ4HÑHÐàÐ$ÜŸ6™6 *¨e°ZÓ@ˆLØÀ}×G[ÑG[Ð\hÔGi¤×!2Ñ!2×!AÑ!AÔ!CÐopÑpÐà˜Ó Ð$4Ñ4Ô ð	5ð /×4Ñ4Ó6ò gÑˆ�
Ø�;‰;˜&Ò Ø$*§L¡LÐ$<�F—L’LÄ%Ç*Á*×B[ÑB[ÓB]ˆEÜ!ŸJ™J×3Ñ3°EÓ:¸1Ñ=ˆMä˜Z¬¨T°MÑ-AÓ)BÓCˆJä�K‰K˜
 fÑ,´E·M±MÈ&Ð`eÔf‰ñgùó1ùò Is   ”H>¡H>ÂIc                 ó”  — t        j                  t        «      }|j                  «       D ]d  \  }}t	        |«      dkD  r:|| vr6dj                  |j                  d«      dd «      }t	        |«      dkD  r|| vrŒ6||   j                  | |   «       Œf |j                  «       D ��cg c]  \  }}t        |«      dhk(  sŒ|‘Œ c}}S c c}}w )zT
    Returns the list of shard files containing only weights offloaded to disk.
    r   rþ   Nr�   rÄ  )	rš  r   r  r®   r  r  rß  r‹  r  )r¸  r  Úfiles_contentrã  r  ÚfnameÚdevicess          r‹   rì  rì  ®  sÇ   € ô  ×+Ñ+¬DÓ1€MØ!+×!1Ñ!1Ó!3ò @Ñˆ�XÜ�+Ó Ò" {¸*Ñ'DØŸ(™( ;×#4Ñ#4°SÓ#9¸#¸2Ð#>Ó?ˆKô �+Ó Ò" {¸*Ò'Dà�hÑ×&Ñ& z°+Ñ'>Õ?ð@ð
 )6×(;Ñ(;Ó(=×Z‘n�e˜WÄÀWÃÐRXÐQYÓAYŠEÓZÐZùÓZs   Â$CÂ<Cc                   ól   — e Zd ZdZeeedœZd„ Zd„ Z	d„ Z
d„ Zd„ Zd„ Zed	ed
efd„«       Zdee   fd„Zy)ÚAttentionInterfacea_  
    Dict-like object keeping track of allowed attention functions. You can easily add a new attention function
    with a call to `register()`. If a model needs to locally overwrite an existing attention function, say `sdpa`,
    it needs to declare a new instance of this class inside the `modeling_<model>.py`, and declare it on that instance.
    )r  r  r  c                 ó   — i | _         y r¨   ©Ú_local_mappingrs  s    r‹   rÎ  zAttentionInterface.__init__Ê  s
   € Ø ˆÕrŠ   c                 óZ   — || j                   v r| j                   |   S | j                  |   S r¨   )rê  Ú_global_mapping©ro  r0  s     r‹   Ú__getitem__zAttentionInterface.__getitem__Í  s2   € à�$×%Ñ%Ñ%Ø×&Ñ& sÑ+Ð+Ø×#Ñ# CÑ(Ð(rŠ   c                 ó>   — | j                   j                  ||i«       y r¨   )rê  r   )ro  r0  rß  s      r‹   Ú__setitem__zAttentionInterface.__setitem__Ó  s   € à×Ñ×"Ñ" C¨ <Õ0rŠ   c                 ó   — | j                   |= y r¨   ré  rí  s     r‹   Ú__delitem__zAttentionInterface.__delitem__×  s   € Ø×Ñ Ñ$rŠ   c                 óH   — t        i | j                  ¥| j                  ¥«      S r¨   )Úiterrì  rê  rs  s    r‹   Ú__iter__zAttentionInterface.__iter__Ú  s$   € äÐC�t×+Ñ+ÐC¨t×/BÑ/BÐCÓDÐDrŠ   c                 ó~   — t        | j                  j                  «       | j                  j                  «       z  «      S r¨   )r  rì  r  rê  rs  s    r‹   Ú__len__zAttentionInterface.__len__Þ  s0   € Ü�4×'Ñ'×,Ñ,Ó.°×1DÑ1D×1IÑ1IÓ1KÑKÓLÐLrŠ   r0  rß  c                 ó>   — | j                   j                  ||i«       y r¨   )rì  r   )r  r0  rß  s      r‹   ÚregisterzAttentionInterface.registerá  s   € à×Ñ×"Ñ" C¨ <Õ0rŠ   rÉ   c                 ó4   — t        | j                  «       «      S r¨   )r  r  rs  s    r‹   r	  zAttentionInterface.valid_keyså  s   € Ü�D—I‘I“KÓ Ð rŠ   N)r  rV  rW  rÀ  r.   r/   r0   rì  rÎ  rî  rð  rò  rõ  r÷  r^  rÜ   r   rù  r   r	  r‰   rŠ   r‹   rç  rç  »  sk   „ ñð 5Ø0Ø&ñ€Oò!ò)ò1ò%òEòMð ð1˜3ð 1 xò 1ó ð1ð!˜D ™Iô !rŠ   rç  r  )TT)Fr  T)rk  r¼  )
NNNNNNFNNNr¨   r½  )rŠ  ("  rš  rÖ  rÓ  r  Úimportlib.metadatar%  r%  rO  r  r×  r†   rÊ  rò  rí  r„  r   Úcollections.abcr   Ú
contextlibr   Údataclassesr   Úenumr   r   r	   Ú	threadingr
   Útypingr   r   r   r   r   r   r   r   r   r   Úzipfiler   r‚   Útorch.distributed.tensorÚhuggingface_hubr   Ú	packagingr   r   r   Útorch.distributionsr   Útorch.nnr   r   Útorch.utils.checkpointr   Útransformers.utilsr   Útorchao.quantizationr   Úactivationsr!   Úconfiguration_utilsr"   Údynamic_module_utilsr#   Ú
generationr$   r%   r&   Úintegrationsr'   r(   r)   Úintegrations.accelerater*   r+   Úintegrations.deepspeedr,   r-   Úintegrations.flash_attentionr.   Úintegrations.flex_attentionr/   Úintegrations.sdpa_attentionr0   Úintegrations.tensor_parallelr1   r2   Úloss.loss_utilsr3   Úpytorch_utilsr4   r5   r6   r7   r8   r9   r:   Ú
quantizersr;   r<   Úquantizers.quantizers_utilsr=   Úsafetensors_conversionr>   rÆ  r?   r@   rA   rB   rC   rD   rE   rF   rG   rH   rI   rJ   rK   rL   rM   rN   rO   rP   rQ   rR   rS   rT   rU   rV   rW   rX   rY   rZ   r[   r\   r]   r^   r_   r`   ra   rb   Ú	utils.hubrc   rd   Úutils.import_utilsre   rf   rg   rh   Úutils.quantization_configri   rj   r‡   rˆ   Úupperrk   rm   rx   rn   ro   Úaccelerate.hooksrp   Úaccelerate.utilsrq   rr   rs   rt   ru   rv   rw   rc  rY  r  Úaccelerate.utils.modelingrz   Úsafetensorsr{   Úsafetensors.torchr|   r  r}   r!  rý  Ú
get_loggerr  r  r¬   r·   r»   rƒ   r„   rÖ  rŒ   r‘   Ú!smdistributed.modelparallel.torchÚmodelparallelr  Úsmdistributed.modelparallelr“   ÚSMP_VERSIONr  r”   r•   r°   r˜   r™   rš   r›   rœ   r�   rž   rŸ   r    r¡   r¢   r£   r¤   r¥   r­   r´   r¸   r¼   rÄ   rÛ   rá   ræ   ró   rù   rû   r8  rÃ  Úuint8Úint8Úint16r(  rê   Úint32rí   Úfloat64Úint64r£  r]  Úuint16Úuint32Úuint64Úfloat8_e5m2rÜ   r_  rÙ   r  rs  r�   r{  rƒ  r˜  r�  r`  rå   r±  rµ  Úno_gradrâ  rç  r  r  r  r-  rF  rQ  rS  rÆ   r–   rä  rÀ  rP  rh  rw  rŒ  r’  rš  r¸  r  rë  rï  rì  rç  r  rX  r‰   rŠ   r‹   ú<module>r4     sø  ðô  Û Û Û 	Û Û Û Û Û Û 	Û 	Û Û Û Ý #Ý *Ý %Ý !Ý ß $Ý ß X× X× XÝ ã Û Ý >Ý ß Ý +ß /Ý -å 3ñ ÔÝ9å 'Ý 1Ý 4ß HÑ Hß XÑ Xß Mß ]Ý AÝ ?Ý ?÷õ *÷÷ ñ ÷ 5Ý =Ý 3÷%÷ %÷ %÷ %÷ %÷ %÷ %÷ %÷ %ó %÷L M÷ó ÷ Nð �zŠz�~Š~˜n¨cÓ2×8Ò8Ó:€Ø—J’J—N’NÐ#6¸Ó<×BÒBÓDÐ ñ Ôß@Ý3÷÷ ñ ð '˜Ÿš y×'9Ò'9×'AÑ'AÀ,Ó'OÓPÐØ˜]˜WŸ]š]¨6Ó2Ò2ÝIáÔÝ%Ý=Ý=ñ ÔÛà	ˆ×	Ò	˜HÓ	%€ð €Ø€ØÐ Ø$×0Ò0×=Ò=Ó?Ð òòñ Ôß3Ð3ÝFà - §¢¨kÓ :¸m¸g¿mºmÈFÓ>SÑ SÑà %ÐáÔÝ/ñ &Ð&CÐK\Ô]Ð ð —’× Ò Ø�wŠw�ŠØ—W’W×*Ò*Ø—’×"Ò"Ø—w’w×.Ò.Ø—g’g×,Ò,ØŸš×0Ò0Ø—w’w×.Ò.Ø�wŠw�ŠØ�gŠg�nŠnØ—g’g×,Ò,Ø—W’W×*Ò*Ø—w’w×.Ò.Ø—g’g×,Ò,ñÐ ð$ ñ4ó ð4ð2 ñó ðð ñ#ó ð#òð&% E¨"¯)ª)Ð5GÐ*GÑ$Hó %ð$¨¨r¯yªyÐ:LÐ/LÑ)Mó $ð$. 5¨¯ªÐ4FÐ)FÑ#Gó .òbNò
/óQTðj �JŠJØ
�+Š+Ø
�*Š*Ø�;Š;Ø�=Š=Ø�NŠNØ�;Š;Ø�=Š=Ø�=Š=Ø�;Š;Ø×"Ò"ñÐ ñ ˜WÔ%Ø$)×$7Ò$7Ð�yÑ!á˜WÔ%Ø %§¢Ð�uÑØ %§¢Ð�uÑØ %§¢Ð�uÑá˜WÔ%Ø$)×$7Ò$7Ð�yÑ!Ø$)×$5Ò$5Ð�yÑ!ð
 Ø7<Øñ	NØ˜3 §¢Ð+Ñ,ðNàðNð ˜5  e§l¢lÐ!2Ñ3Ñ4ðNð ó	Nòb&ð(�U—\‘\ð  có ñ "§)¢)ó ð,˜D  S¡™Nð ,¸¸SÀ%Ç,Á,Ð=NÑ8Oð ,ÐTYÐZ^Ð_bÐcfÑ_gÑZhÐjnÐorÑjsÐZsÑTtô ,ð>%˜T # c¡(™^ð %¸¸cÀ5Ç<Á<Ð>OÑ9Pð %ÐUZÐ[_Ð`cÐdgÑ`hÑ[iÐknÐorÑksÐ[sÑUtô %ð. 04Ø*.ñNØðNàðNð —‘ðNð ! §¢Ñ,ð	Nð
 ˜;Ñ'ðNð ˆ4�˜%Ÿ+š+Ñ&Ð&Ñ'ôNð>LÐ&7ð LÀSð LÐRW×R^ÑR^ô Lð €‡‚ƒð "&Ø)-Ø)-Ø(,Ø(,Ø*.Ø Ø/3Ø+/ØHLñD1ØðD1àðD1ð ðD1ð ˜‘9ð	D1ð
 # 3¨ 8™nðD1ð ˜‘ðD1ð " #™ðD1ð ! ™ðD1ð ! ™ðD1ð   ‘~ðD1ð ˜;Ñ'ðD1ð ðD1ð ! §¢Ñ,ðD1ð ˜d 3™iÑ(ðD1ð ÐDÑEðD1ð  ˆ8�D‰>˜8 D™>Ð)Ñ*ò!D1ó ñD1ñN˜sð ¨X°c©]ð Àcô ðv.Ø#+¨E°#°r·{²{Ð2BÑ,CÑ#Dðv.àðv.ð �c‰]ðv.ð ˜‰}ð	v.ð
 ðv.ð ðv.ð ðv.ð ðv.ð ðv.ð �d˜3 ˜8‘nÑ%ðv.ð ðv.ð �E˜#˜t˜)Ñ$Ñ%ðv.ñ ðv.ð ðv.ð ˜#‘ðv.ð  ˆ8�D˜‘IÑ ¨¡Ð.Ñ/ô!v.ðr	M+à˜%  U§[¢[°$Ð 6Ñ7Ñ8ðM+ð ˜t C™yÑ)ðM+ð ð	M+ð
 ˜t‘nðM+ð ˜‘ðM+ð ðM+ð Ð˜X e§k¢kÑ2°H¸U¿[º[Ñ4IÐIÑJôM+ð`<Øð<à˜˜s D˜yÑ)Ñ*ð<ð ˜‘ð<ð ˜;Ñ'ð	<ð
 ˜%Ÿ+š+Ñ&ð<ð ! §¢Ñ,ð<ð 
ô<ð~9)àð9)ð # 3™ið9)ð ˜#‘Yð	9)ð
 .2ð9)ð ˜;Ñ'ð9)ð ð9)ð ˆ4�‰9�d˜3‘iÐÑ ô9)ðx5.Øð5.à˜‘ð5.ð ˜t C™yÑ)ð5.ð "ð	5.ð
 !  c ™Nð5.ð ð5.ð ð5.ð ˆ4�‰9�d˜5  c ™?Ñ+Ð+Ñ,ô5.ôp�tõ ÷
tqò tqôp	N7Q�b—i’iÑ!1°?ÀNÐTdõ N7Qñbn (©×(CÒ(CÓD�Õ Ù×Ò×&Ò&Ð2Ù*9×*EÒ*E×*MÒ*M×*TÒ*TØ [¸|ð +Uó +�O×ÒÕ'ô
#˜Ÿ	š	õ #ôLB�b—i’iõ BôJ>˜Ÿ	š	õ >ðB ô3�kó 3ó ñ3ô<r�—	’	õ rôj`�b—i’iõ `ñF˜Ÿ	š	ð ¨dð ¸r¿yºyô ó<	ñ3g¡Oð 3gÈ$ô 3gól
[ô+!˜õ +!ò^ /AÓ.BÑ Ñ+Õ BrŠ   