§
    ‚Štj–å  ã                   óD  — d Z ddlZddlmZ ddlZddlZddlmZ ddlm	Z	 ddl
mZ ddlmZ dd	lmZmZmZ dd
lmZ ddlmZ ddlmZ ddlmZ ddlmZmZmZmZm Z m!Z! ddl"m#Z#m$Z$ ddl%m&Z& ddl'm(Z(m)Z)m*Z*m+Z+ ddl,m-Z- ddl.m/Z/m0Z0 ddl1m2Z2 ddl3m4Z4  e+j5        e6¦  «        Z7dZ8dLde9de9de:dej;        fd„Z<dej;        de9de9fd „Z=	 	 dMd!e>e9e9f         d"e:d#e9d$ej?        dz  d%e9dej@        fd&„ZA G d'„ d(ejB        ¦  «        ZC	 	 dNd*ejD        d+ej;        d,ej;        d-ej;        d$ej;        dz  d.e:dz  d/e:fd0„ZE G d1„ d2ejD        ¦  «        ZF G d3„ d4e¦  «        ZG G d5„ d6e¦  «        ZHe) G d7„ d8e$¦  «        ¦   «         ZI G d9„ d:eI¦  «        ZJ G d;„ d<eI¦  «        ZKe) G d=„ d>eI¦  «        ¦   «         ZL e)d?¬@¦  «         G dA„ dBe4eI¦  «        ¦   «         ZM G dC„ dDeI¦  «        ZN e)dE¬@¦  «         G dF„ dGeIe¦  «        ¦   «         ZO e)dH¬@¦  «         G dI„ dJeI¦  «        ¦   «         ZPg dK¢ZQdS )OzPyTorch Whisper model.é    N)ÚCallable)Únn)ÚCrossEntropyLossé   )Úinitialization)ÚACT2FN)ÚCacheÚDynamicCacheÚEncoderDecoderCache)ÚGenerationMixin)Úcreate_causal_mask)ÚFlashAttentionKwargs)ÚGradientCheckpointingLayer)ÚBaseModelOutputÚ)BaseModelOutputWithPastAndCrossAttentionsÚ!CausalLMOutputWithCrossAttentionsÚSeq2SeqLMOutputÚSeq2SeqModelOutputÚSequenceClassifierOutput)ÚALL_ATTENTION_FUNCTIONSÚPreTrainedModel)ÚUnpack)ÚTransformersKwargsÚauto_docstringÚcan_return_tupleÚlogging)Úmerge_with_config_defaults)ÚOutputRecorderÚcapture_outputsé   )ÚWhisperConfig)ÚWhisperGenerationMixiné'  ÚlengthÚchannelsÚmax_timescaleÚreturnc                 óÄ  — |dz  dk    rt          d|› d�¦  «        ‚t          j        |¦  «        |dz  dz
  z  }t          j        | t          j        |dz  ¦  «        z  ¦  «        }t          j        | ¦  «                             dd¦  «        |                     dd¦  «        z  }t          j        |                     ¦   «         | 	                    ¦   «         gd¬¦  «        S )z*Returns sinusoids for positional embeddingé   r   zVNumber of channels has to be divisible by 2 for sinusoidal positional embeddings, got z
 channels.r    éÿÿÿÿ©Údim)
Ú
ValueErrorÚmathÚlogÚtorchÚexpÚarangeÚviewÚcatÚsinÚcos)r$   r%   r&   Úlog_timescale_incrementÚinv_timescalesÚscaled_times         új/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/whisper/modeling_whisper.pyÚ	sinusoidsr;   7   sÛ   € à�!�|�qÒÐÝØyÐemÐyÐyÐyñ
ô 
ð 	
õ #œh }Ñ5Ô5¸ÀQ¹ÈÑ9JÑKÐÝ”YÐ 7Ð7½%¼,ÀxÐSTÁ}Ñ:UÔ:UÑUÑVÔV€NÝ”,˜vÑ&Ô&×+Ò+¨B°Ñ2Ô2°^×5HÒ5HÈÈBÑ5OÔ5OÑO€KÝŒ9�k—o’oÑ'Ô'¨¯ªÑ):Ô):Ð;ÀÐCÑCÔCÐCó    Ú	input_idsÚpad_token_idÚdecoder_start_token_idc                 óô   — |                       | j        ¦  «        }| dd…dd…f                              ¦   «         |dd…dd…f<   ||dd…df<   |€t          d¦  «        ‚|                     |dk    |¦  «         |S )z1
    Shift input ids one token to the right.
    Nr*   r    r   z1self.model.config.pad_token_id has to be defined.iœÿÿÿ)Ú	new_zerosÚshapeÚcloner-   Úmasked_fill_)r=   r>   r?   Úshifted_input_idss       r:   Úshift_tokens_rightrF   D   s˜   € ð "×+Ò+¨I¬OÑ<Ô<ÐØ(¨¨¨¨C¨R¨C¨Ô0×6Ò6Ñ8Ô8Ð�a�a�a˜˜˜�eÑØ4Ð�a�a�a˜�dÑàÐÝÐLÑMÔMÐMà×"Ò"Ð#4¸Ò#<¸lÑKÔKÐKàÐr<   rB   Ú	mask_probÚmask_lengthÚattention_maskÚ	min_masksc                 ó@  ‡‡‡‡‡— | \  }Š‰dk     rt          d¦  «        ‚‰‰k    rt          d‰› d‰› d�¦  «        ‚t          j                             d¦  «                             ¦   «         Šˆˆˆˆˆfd„}|�9|                     ¦   «                              d¦  «                             ¦   «         nˆfd	„t          |¦  «        D ¦   «         }t          j	        |‰ft          ¬
¦  «        }g }	 |‰¦  «        }
|
dk    r|S |D ]·} ||¦  «        }t          j                             t          j        |‰dz
  z
  ¦  «        |d¬¦  «        }t          |¦  «        dk    r‰dz
  }n|d         }t          j        |t          j        |
|z
  t          j        ¬
¦  «        |z  g¦  «        }|	                     |¦  «         Œ¸t          j        |	¦  «        }	t          j        |	dd…dd…df         ||
‰f¦  «        }	|	                     ||
‰z  ¦  «        }	t          j        ‰¦  «        dddd…f         }t          j        |||
‰f¦  «                             ||
‰z  ¦  «        }|	|z   }	|	                     ¦   «         ‰dz
  k    r‰dz
  |	|	‰dz
  k    <   t          j        ||	dd¦  «         |S )an  
    Computes random mask spans for a given shape. Used to implement [SpecAugment: A Simple Data Augmentation Method for
    ASR](https://huggingface.co/papers/1904.08779). Note that this method is not optimized to run on TPU and should be run on
    CPU as part of the preprocessing during training.

    Args:
        shape: The shape for which to compute masks. This should be of a tuple of size 2 where
               the first element is the batch size and the second element is the length of the axis to span.
        mask_prob:  The percentage of the whole axis (between 0 and 1) which will be masked. The number of
                    independently generated mask spans of length `mask_length` is computed by
                    `mask_prob*shape[1]/mask_length`. Note that due to overlaps, `mask_prob` is an upper bound and the
                    actual percentage will be smaller.
        mask_length: size of the mask
        min_masks: minimum number of masked spans
        attention_mask: A (right-padded) attention mask which independently shortens the feature axis of
                        each batch dimension.
    r    z&`mask_length` has to be bigger than 0.zO`mask_length` has to be smaller than `sequence_length`, but got `mask_length`: z and `sequence_length`: ú`c                 ó¸   •— t          ‰| z  ‰z  ‰z   ¦  «        }t          |‰¦  «        }|‰z  ‰k    r‰‰z  }| ‰dz
  z
  |k     rt          | ‰dz
  z
  d¦  «        }|S )z;Given input length, compute how many spans should be maskedr    r   )ÚintÚmax)Úinput_lengthÚnum_masked_spanÚepsilonrH   rG   rJ   Úsequence_lengths     €€€€€r:   Úcompute_num_masked_spanz6_compute_mask_indices.<locals>.compute_num_masked_span{   s~   ø€ å˜i¨,Ñ6¸ÑDÀwÑNÑOÔOˆÝ˜o¨yÑ9Ô9ˆð ˜[Ñ(¨?Ò:Ð:Ø-°Ñ<ˆOð ˜;¨™?Ñ+¨oÒ=Ð=Ý! ,°+À±/Ñ"BÀAÑFÔFˆOàÐr<   Nr*   c                 ó   •— g | ]}‰‘ŒS © rV   )Ú.0Ú_rS   s     €r:   ú
<listcomp>z)_compute_mask_indices.<locals>.<listcomp>Ž   s   ø€ Ð9Ð9Ð9 !ˆoÐ9Ð9Ð9r<   )Údtyper   F)Úreplace)r-   ÚnpÚrandomÚrandÚitemÚdetachÚsumÚtolistÚrangeÚzerosÚboolÚchoicer2   ÚlenÚconcatenateÚonesÚint32ÚappendÚarrayÚbroadcast_toÚreshaperO   Úput_along_axis)rB   rG   rH   rI   rJ   Ú
batch_sizerT   Úinput_lengthsÚspec_aug_maskÚspec_aug_mask_idxsÚmax_num_masked_spanrP   rQ   Úspec_aug_mask_idxÚdummy_mask_idxÚoffsetsrR   rS   s    `` `           @@r:   Ú_compute_mask_indicesrx   U   sP  øøøøø€ ð0 #(Ñ€J�à�Q‚€ÝÐAÑBÔBÐBà�_Ò$Ð$Ýð:Ð^ið :ð :Ø'6ð:ð :ð :ñ
ô 
ð 	
õ Œi�nŠn˜QÑÔ×$Ò$Ñ&Ô&€Gðð ð ð ð ð ð ð ð ð$ Ð%ð 	×ÒÑÔ×#Ò# BÑ'Ô'×.Ò.Ñ0Ô0Ð0à9Ð9Ð9Ð9¥u¨ZÑ'8Ô'8Ð9Ñ9Ô9ð õ ”H˜j¨/Ð:Å$ÐGÑGÔG€MØÐà1Ð1°/ÑBÔBÐà˜aÒÐØÐà%ð 5ð 5ˆà1Ð1°,Ñ?Ô?ˆõ œI×,Ò,ÝŒI�l k°A¡oÑ6Ñ7Ô7¸ÐRWð -ñ 
ô 
Ðõ Ð Ñ!Ô! QÒ&Ð&ð -¨qÑ0ˆNˆNà.¨qÔ1ˆNåœNØ¥¤Ð(;¸oÑ(MÕUWÔU]Ð ^Ñ ^Ô ^ÐaoÑ oÐpñ
ô 
Ðð 	×!Ò!Ð"3Ñ4Ô4Ð4Ð4åœÐ"4Ñ5Ô5Ðõ œØ˜1˜1˜1˜a˜a˜a ˜:Ô&¨Ð5HÈ+Ð(Vñô Ðð ,×3Ò3°JÐ@SÐVaÑ@aÑbÔbÐõ Œi˜Ñ$Ô$ T¨4°°° ]Ô3€GÝŒo˜g¨
Ð4GÈÐ'UÑVÔV×^Ò^ØÐ'¨+Ñ5ñô €Gð ,¨gÑ5Ðð ×ÒÑÔ /°AÑ"5Ò5Ð5ØGVÐYZÑGZÐÐ-°À!Ñ0CÒCÑDõ Ô�mÐ%7¸¸BÑ?Ô?Ð?àÐr<   c                   ó<   ‡ — e Zd Zddedededz  fˆ fd„Zd	d„Zˆ xZS )
ÚWhisperPositionalEmbeddingNÚnum_positionsÚembedding_dimÚpadding_idxc                 óL   •— t          ¦   «                              ||¦  «         d S ©N)ÚsuperÚ__init__)Úselfr{   r|   r}   Ú	__class__s       €r:   r�   z#WhisperPositionalEmbedding.__init__Í   s#   ø€ Ý‰Œ×Ò˜¨Ñ6Ô6Ð6Ð6Ð6r<   r   c                 óZ   — |€| j         |||j        d         z   …         S | j         |         S ©Nr    )ÚweightrB   )r‚   r=   Úpast_key_values_lengthÚposition_idss       r:   Úforwardz"WhisperPositionalEmbedding.forwardÐ   s8   € ØÐØ”;Ð5Ð8NÐQZÔQ`ÐabÔQcÑ8cÐcÔdÐdà”;˜|Ô,Ð,r<   r   )r   N)Ú__name__Ú
__module__Ú__qualname__rN   r�   r‰   Ú__classcell__©rƒ   s   @r:   rz   rz   Ì   sp   ø€ € € € € ð7ð 7 cð 7¸#ð 7ÈCÐRVÉJð 7ð 7ð 7ð 7ð 7ð 7ð-ð -ð -ð -ð -ð -ð -ð -r<   rz   ç        ÚmoduleÚqueryÚkeyÚvalueÚscalingÚdropoutc                 ó®  — |€|                      d¦  «        dz  }t          j        ||                     dd¦  «        ¦  «        |z  }|�||z   }t          j                             |d¬¦  «        }t          j                             ||| j        ¬¦  «        }t          j        ||¦  «        }	|	                     dd¦  «         	                    ¦   «         }	|	|fS )Nr*   ç      à¿r)   r   r+   ©ÚpÚtrainingr    )
Úsizer0   ÚmatmulÚ	transposer   Ú
functionalÚsoftmaxr•   rš   Ú
contiguous)
r�   r‘   r’   r“   rI   r”   r•   ÚkwargsÚattn_weightsÚattn_outputs
             r:   Úeager_attention_forwardr¤   ×   sÆ   € ð €Ø—*’*˜R‘.”. DÑ(ˆå”<  s§}¢}°Q¸Ñ':Ô':Ñ;Ô;¸gÑE€LØÐ!Ø# nÑ4ˆå”=×(Ò(¨¸2Ð(Ñ>Ô>€Lå”=×(Ò(¨¸È6Ì?Ð(Ñ[Ô[€LÝ”,˜|¨UÑ3Ô3€KØ×'Ò'¨¨1Ñ-Ô-×8Ò8Ñ:Ô:€Kà˜Ð$Ð$r<   c                   ó  ‡ — e Zd ZdZ	 	 	 	 	 	 ddededed	ed
edededz  dedz  fˆ fd„Z	 	 	 	 dde	j
        de	j
        dz  dedz  de	j
        dz  dedee         dee	j
        e	j
        dz  ee	j
                 dz  f         fd„Zˆ xZS )ÚWhisperAttentionz=Multi-headed attention from 'Attention Is All You Need' paperr�   FTNÚ	embed_dimÚ	num_headsr•   Ú
is_decoderÚbiasÚ	is_causalÚ	layer_idxÚconfigc	                 óp  •— t          ¦   «                              ¦   «          || _        || _        || _        ||z  | _        || _        | j        |z  | j        k    rt          d| j        › d|› d�¦  «        ‚| j        dz  | _        || _	        || _
        |€*|r(t                               d| j        j        › d�¦  «         || _        t!          j        ||d¬¦  «        | _        t!          j        |||¬¦  «        | _        t!          j        |||¬¦  «        | _        t!          j        |||¬¦  «        | _        d S )	Nz;embed_dim must be divisible by num_heads (got `embed_dim`: z and `num_heads`: z).r—   zInstantiating a decoder z³ without passing `layer_idx` is not recommended and will to errors during the forward call, if caching is used. Please make sure to provide a `layer_idx` when creating this class.F©rª   )r€   r�   r§   r¨   r•   Úhead_dimr­   r-   r”   r©   r«   ÚloggerÚwarning_oncerƒ   rŠ   r¬   r   ÚLinearÚk_projÚv_projÚq_projÚout_proj)
r‚   r§   r¨   r•   r©   rª   r«   r¬   r­   rƒ   s
            €r:   r�   zWhisperAttention.__init__ô   sW  ø€ õ 	‰Œ×ÒÑÔÐØ"ˆŒØ"ˆŒØˆŒØ! YÑ.ˆŒØˆŒàŒM˜IÑ%¨$¬.Ò8Ð8Ýð3ÈdÌnð 3ð 3Ø%.ð3ð 3ð 3ñô ð ð ”} dÑ*ˆŒØ$ˆŒØ"ˆŒàÐ ÐÝ×Òð,¨4¬>Ô+Bð ,ð ,ð ,ñô ð ð
 #ˆŒå”i 	¨9¸5ÐAÑAÔAˆŒÝ”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝ”i 	¨9¸4Ð@Ñ@Ô@ˆŒÝœ	 )¨Y¸TÐBÑBÔBˆŒˆˆr<   Úhidden_statesÚkey_value_statesÚpast_key_valuesrI   Úoutput_attentionsr¡   r'   c                 ó¸  — |du}|j         dd…         }g |¢d‘| j        ‘R }	|                      |¦  «        | j        z                       |	¦  «                             dd¦  «                             ¦   «         }
|�Tt          |t          ¦  «        r?|j	         
                    | j        ¦  «        }|rd|j	        | j        <   |j        }n|j        }|�|n|}|r3|r1|r/|j        | j                 j        }|j        | j                 j        }nÓ|d         d| j        | j        f}|                      |¦  «                             |¦  «                             dd¦  «                             ¦   «         }|                      |¦  «                             |¦  «                             dd¦  «                             ¦   «         }|�|                     ||| j        ¦  «        \  }}t+          j        | j        j        t2          ¦  «        } || |
|||f| j        sdn| j        d|d	œ|¤Ž\  }} |j        g |¢d‘R Ž                      ¦   «         }|                      |¦  «        }||fS )
z#Input shape: Batch x Time x ChannelNr*   r    r)   Tr   r�   ç      ð?)r•   r”   r»   )rB   r°   r¶   r”   r3   r�   r    Ú
isinstancer   Ú
is_updatedÚgetr¬   Úcross_attention_cacheÚself_attention_cacheÚlayersÚkeysÚvaluesr¨   r´   rµ   Úupdater   Úget_interfacer­   Ú_attn_implementationr¤   rš   r•   rn   r·   )r‚   r¸   r¹   rº   rI   r»   r¡   Úis_cross_attentionÚinput_shapeÚhidden_shapeÚquery_statesr¿   Úcurrent_statesÚ
key_statesÚvalue_statesÚkv_shapeÚattention_interfacer£   r¢   s                      r:   r‰   zWhisperAttention.forward  sš  € ð .°TÐ9Ðà#Ô)¨#¨2¨#Ô.ˆØ8˜Ð8 bÐ8¨$¬-Ð8Ð8ˆð Ÿš MÑ2Ô2°T´\ÑA×GÒGÈÑUÔU×_Ò_Ð`aÐcdÑeÔe×pÒpÑrÔrˆð Ð&­:°oÕGZÑ+[Ô+[Ð&Ø(Ô3×7Ò7¸¼ÑGÔGˆJØ!ð Gà=A�Ô*¨4¬>Ñ:Ø"1Ô"G��à"1Ô"F�ð .>Ð-IÐ)Ð)È}ˆØð 	l /ð 	l°jð 	là(Ô/°´Ô?ÔDˆJØ*Ô1°$´.ÔAÔHˆLˆLð
 $ Aœ¨¨D¬N¸D¼MÐJˆHØŸš ^Ñ4Ô4×9Ò9¸(ÑCÔC×MÒMÈaÐQRÑSÔS×^Ò^Ñ`Ô`ˆJØŸ;š; ~Ñ6Ô6×;Ò;¸HÑEÔE×OÒOÐPQÐSTÑUÔU×`Ò`ÑbÔbˆLØÐ*Ø+:×+AÒ+AÀ*ÈlÐ\`Ô\jÑ+kÔ+kÑ(�
˜Lå(?Ô(MØŒKÔ,Õ.Eñ)
ô )
Ðð %8Ð$7ØØØØØð
%
ð  $œ}Ð>�C�C°$´,ØØ/ð
%
ð 
%
ð ð
%
ð 
%
Ñ!ˆ�\ð *�kÔ)Ð;¨;Ð;¸Ð;Ð;Ð;×FÒFÑHÔHˆØ—m’m KÑ0Ô0ˆà˜LÐ(Ð(r<   )r�   FTFNN)NNNF)rŠ   r‹   rŒ   Ú__doc__rN   Úfloatre   r!   r�   r0   ÚTensorr	   r   r   Útupler‰   r�   rŽ   s   @r:   r¦   r¦   ñ   sy  ø€ € € € € ØGÐGð Ø ØØØ $Ø'+ð&Cð &Càð&Cð ð&Cð ð	&Cð
 ð&Cð ð&Cð ð&Cð ˜‘:ð&Cð  Ñ$ð&Cð &Cð &Cð &Cð &Cð &CðV 15Ø(,Ø.2Ø"'ðH)ð H)à”|ðH)ð  œ,¨Ñ-ðH)ð  ™ð	H)ð
 œ tÑ+ðH)ð  ðH)ð Ð-Ô.ðH)ð 
ˆuŒ|˜Uœ\¨DÑ0°%¸¼Ô2EÈÑ2LÐLÔ	MðH)ð H)ð H)ð H)ð H)ð H)ð H)ð H)r<   r¦   c                   óf   ‡ — e Zd Zdefˆ fd„Zdej        dej        dee         dej        fd„Z	ˆ xZ
S )ÚWhisperEncoderLayerr­   c                 ó  •— t          ¦   «                              ¦   «          |j        | _        t	          | j        |j        |j        |¬¦  «        | _        t          j	        | j        ¦  «        | _
        |j        | _        t          |j                 | _        |j        | _        t          j        | j        |j        ¦  «        | _        t          j        |j        | j        ¦  «        | _        t          j	        | j        ¦  «        | _        d S )N)r§   r¨   r•   r­   )r€   r�   Úd_modelr§   r¦   Úencoder_attention_headsÚattention_dropoutÚ	self_attnr   Ú	LayerNormÚself_attn_layer_normr•   r   Úactivation_functionÚactivation_fnÚactivation_dropoutr³   Úencoder_ffn_dimÚfc1Úfc2Úfinal_layer_norm©r‚   r­   rƒ   s     €r:   r�   zWhisperEncoderLayer.__init__i  sÐ   ø€ Ý‰Œ×ÒÑÔÐØœˆŒå)Ø”nØÔ4ØÔ,Øð	
ñ 
ô 
ˆŒõ %'¤L°´Ñ$@Ô$@ˆÔ!Ø”~ˆŒÝ# FÔ$>Ô?ˆÔØ"(Ô";ˆÔÝ”9˜Tœ^¨VÔ-CÑDÔDˆŒÝ”9˜VÔ3°T´^ÑDÔDˆŒÝ "¤¨T¬^Ñ <Ô <ˆÔÐÐr<   r¸   rI   r¡   r'   c                 óº  — |}|                       |¦  «        } | j        d||dœ|¤Ž\  }}t          j                             || j        | j        ¬¦  «        }||z   }|}|                      |¦  «        }|                      |                      |¦  «        ¦  «        }t          j                             || j	        | j        ¬¦  «        }|  
                    |¦  «        }t          j                             || j        | j        ¬¦  «        }||z   }|j        t          j        k    r9t          j        |j        ¦  «        j        dz
  }t          j        || |¬¦  «        }|S )a>  
        Args:
            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
            attention_mask (`torch.FloatTensor`): attention mask of size
                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
        )r¸   rI   r˜   iè  )ÚminrO   rV   )rÞ   rÜ   r   rž   r•   rš   rå   rà   rã   rá   rä   rZ   r0   Úfloat16ÚfinforO   Úclamp)r‚   r¸   rI   r¡   ÚresidualrX   Úclamp_values          r:   r‰   zWhisperEncoderLayer.forward{  s[  € ð !ˆØ×1Ò1°-Ñ@Ô@ˆØ)˜4œ>ð 
Ø'Ø)ð
ð 
ð ð
ð 
Ñˆ�qõ
 œ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆà ˆØ×-Ò-¨mÑ<Ô<ˆØ×*Ò*¨4¯8ª8°MÑ+BÔ+BÑCÔCˆÝœ×-Ò-¨m¸tÔ?VÐaeÔanÐ-ÑoÔoˆØŸš Ñ/Ô/ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆàÔ¥%¤-Ò/Ð/Ýœ+ mÔ&9Ñ:Ô:Ô>ÀÑEˆKÝ!œK¨¸K¸<È[ÐYÑYÔYˆMàÐr<   )rŠ   r‹   rŒ   r!   r�   r0   rÔ   r   r   r‰   r�   rŽ   s   @r:   r×   r×   h  sŠ   ø€ € € € € ð=˜}ð =ð =ð =ð =ð =ð =ð$"à”|ð"ð œð"ð Ð+Ô,ð	"ð
 
Œð"ð "ð "ð "ð "ð "ð "ð "r<   r×   c                   óÀ   ‡ — e Zd Zddededz  fˆ fd„Z	 	 	 	 	 ddej        dej        dz  dej        dz  d	ej        dz  d
edz  de	dz  de
e         dej        fd„Zˆ xZS )ÚWhisperDecoderLayerNr­   r¬   c           	      ó¨  •— t          ¦   «                              ¦   «          |j        | _        t	          | j        |j        |j        dd||¬¦  «        | _        |j        | _        t          |j
                 | _        |j        | _        t          j        | j        ¦  «        | _        t	          | j        |j        |j        d||¬¦  «        | _        t          j        | j        ¦  «        | _        t          j        | j        |j        ¦  «        | _        t          j        |j        | j        ¦  «        | _        t          j        | j        ¦  «        | _        d S )NT)r§   r¨   r•   r©   r«   r¬   r­   )r•   r©   r¬   r­   )r€   r�   rÙ   r§   r¦   Údecoder_attention_headsrÛ   rÜ   r•   r   rß   rà   rá   r   rÝ   rÞ   Úencoder_attnÚencoder_attn_layer_normr³   Údecoder_ffn_dimrã   rä   rå   )r‚   r­   r¬   rƒ   s      €r:   r�   zWhisperDecoderLayer.__init__¡  s   ø€ Ý‰Œ×ÒÑÔÐØœˆŒå)Ø”nØÔ4ØÔ,ØØØØð
ñ 
ô 
ˆŒð ”~ˆŒÝ# FÔ$>Ô?ˆÔØ"(Ô";ˆÔå$&¤L°´Ñ$@Ô$@ˆÔ!Ý,ØŒNØÔ*ØÔ,ØØØð
ñ 
ô 
ˆÔõ (*¤|°D´NÑ'CÔ'CˆÔ$Ý”9˜Tœ^¨VÔ-CÑDÔDˆŒÝ”9˜VÔ3°T´^ÑDÔDˆŒÝ "¤¨T¬^Ñ <Ô <ˆÔÐÐr<   Tr¸   rI   Úencoder_hidden_statesÚencoder_attention_maskrº   Ú	use_cacher¡   r'   c                 óÞ  — |}|                       |¦  «        } | j        |f||dœ|¤Ž\  }}	t          j                             || j        | j        ¬¦  «        }||z   }|�]|}|                      |¦  «        } | j        |f|||dœ|¤Ž\  }}	t          j                             || j        | j        ¬¦  «        }||z   }|}|                      |¦  «        }|  	                    |  
                    |¦  «        ¦  «        }t          j                             || j        | j        ¬¦  «        }|                      |¦  «        }t          j                             || j        | j        ¬¦  «        }||z   }|S )að  
        Args:
            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
            attention_mask (`torch.FloatTensor`): attention mask of size
                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
            encoder_hidden_states (`torch.FloatTensor`):
                cross attention input to the layer of shape `(batch, seq_len, embed_dim)`
            encoder_attention_mask (`torch.FloatTensor`): encoder attention mask of size
                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
            past_key_values (`Cache`): cached past key and value projection states
        )rº   rI   r˜   N)r¹   rI   rº   )rÞ   rÜ   r   rž   r•   rš   ró   rò   rå   rà   rã   rá   rä   )
r‚   r¸   rI   rõ   rö   rº   r÷   r¡   rì   rX   s
             r:   r‰   zWhisperDecoderLayer.forwardÀ  s§  € ð* !ˆØ×1Ò1°-Ñ@Ô@ˆð *˜4œ>Øð
à+Ø)ð
ð 
ð ð	
ð 
Ñˆ�qõ œ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆð !Ð,Ø$ˆHØ ×8Ò8¸ÑGÔGˆMØ0˜tÔ0Øð à!6Ø5Ø /ð	 ð  ð
 ð ð  ÑˆM˜1õ œM×1Ò1°-À4Ä<ÐZ^ÔZgÐ1ÑhÔhˆMØ$ }Ñ4ˆMð !ˆØ×-Ò-¨mÑ<Ô<ˆØ×*Ò*¨4¯8ª8°MÑ+BÔ+BÑCÔCˆÝœ×-Ò-¨m¸tÔ?VÐaeÔanÐ-ÑoÔoˆØŸš Ñ/Ô/ˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆØ  =Ñ0ˆàÐr<   r   )NNNNT)rŠ   r‹   rŒ   r!   rN   r�   r0   rÔ   r   re   r   r   r‰   r�   rŽ   s   @r:   rï   rï      sõ   ø€ € € € € ð=ð =˜}ð =¸¸t¹ð =ð =ð =ð =ð =ð =ðD /3Ø59Ø6:Ø6:Ø!%ð9ð 9à”|ð9ð œ tÑ+ð9ð  %œ|¨dÑ2ð	9ð
 !&¤¨tÑ 3ð9ð -¨tÑ3ð9ð ˜$‘;ð9ð Ð+Ô,ð9ð 
Œð9ð 9ð 9ð 9ð 9ð 9ð 9ð 9r<   rï   c                   ó’   ‡ — e Zd ZU eed<   dZdZdZdZddgZ	dZ
dZdZdZ ej        ¦   «         ˆ fd„¦   «         Zd	ej        fd
„Zˆ xZS )ÚWhisperPreTrainedModelr­   ÚmodelÚinput_features)ÚaudioÚtextTr×   rï   c                 ó€  •— t          ¦   «                              |¦  «         t          |t          ¦  «        r7t	          j        |j        j        t          |j        j        j	        Ž ¦  «         d S t          |t          ¦  «        r8| j        j        r.t	          j        |j        d| j        j        dz   z  ¦  «         d S d S d S )Nr½   r    )r€   Ú_init_weightsr¾   ÚWhisperEncoderÚinitÚcopy_Úembed_positionsr†   r;   rB   ÚWhisperForAudioClassificationr­   Úuse_weighted_layer_sumÚ	constant_Úlayer_weightsÚnum_hidden_layers)r‚   r�   rƒ   s     €r:   r   z$WhisperPreTrainedModel._init_weights
  s¿   ø€ å‰Œ×Ò˜fÑ%Ô%Ð%Ý�f�nÑ-Ô-ð 	`ÝŒJ�vÔ-Ô4µiÀÔAWÔA^ÔAdÐ6eÑfÔfÐfÐfÐfÝ˜Õ =Ñ>Ô>ð 	`ØŒ{Ô1ð `Ý”˜vÔ3°S¸D¼KÔ<YÐ\]Ñ<]Ñ5^Ñ_Ô_Ð_Ð_Ð_ð	`ð 	`ð`ð `r<   rq   c                 ó   — |dz
  dz  dz   }|S )zH
        Computes the output length of the convolutional layers
        r    r)   rV   )r‚   rq   s     r:   Ú _get_feat_extract_output_lengthsz7WhisperPreTrainedModel._get_feat_extract_output_lengths  s   € ð '¨Ñ*¨qÑ0°1Ñ4ˆàÐr<   )rŠ   r‹   rŒ   r!   Ú__annotations__Úbase_model_prefixÚmain_input_nameÚinput_modalitiesÚsupports_gradient_checkpointingÚ_no_split_modulesÚ_supports_flash_attnÚ_supports_sdpaÚ_supports_flex_attnÚ_can_compile_fullgraphr0   Úno_gradr   Ú
LongTensorr  r�   rŽ   s   @r:   rú   rú   ü  sµ   ø€ € € € € € àÐÐÑØÐØ&€OØ(ÐØ&*Ð#Ø.Ð0EÐFÐØÐØ€NØÐà!Ðà€U„]�_„_ð`ð `ð `ð `ñ „_ð`ð¸eÔ>Nð ð ð ð ð ð ð ð r<   rú   c                   ó¨   ‡ — e Zd ZdZeedœZdZdefˆ fd„Z	d„ Z
dej        fd„Zd	ej        fd
„Zee	 ddee         defd„¦   «         ¦   «         Zˆ xZS )r  z°
    Transformer encoder consisting of *config.encoder_layers* self attention layers. Each layer is a
    [`WhisperEncoderLayer`].

    Args:
        config: WhisperConfig
    )r¸   Ú
attentions)rý   r­   c                 óè  •‡— t          ¦   «                              ‰¦  «         ‰j        | _        ‰j        | _        ‰j        }‰j        | _        ‰j        | _        ‰j	        | _	        ‰j
        rt          j        |¦  «        nd| _        t          j        | j        |dd¬¦  «        | _        t          j        ||ddd¬¦  «        | _        t          j        | j	        |¦  «        | _        | j                             d¦  «         t          j        ˆfd„t-          ‰j        ¦  «        D ¦   «         ¦  «        | _        t          j        ‰j        ¦  «        | _        d| _        |                      ¦   «          d S )	Nr½   r   r    )Úkernel_sizeÚpaddingr)   )r  Ústrider  Fc                 ó.   •— g | ]}t          ‰¦  «        ‘ŒS rV   )r×   )rW   rX   r­   s     €r:   rY   z+WhisperEncoder.__init__.<locals>.<listcomp><  s"   ø€ Ð$gÐ$gÐ$gÀQÕ%8¸Ñ%@Ô%@Ð$gÐ$gÐ$gr<   )r€   r�   r•   Úencoder_layerdropÚ	layerdroprÙ   Únum_mel_binsr>   r}   Úmax_source_positionsÚscale_embeddingr.   ÚsqrtÚembed_scaler   ÚConv1dÚconv1Úconv2Ú	Embeddingr  Úrequires_grad_Ú
ModuleListrc   Úencoder_layersrÃ   rÝ   Ú
layer_normÚgradient_checkpointingÚ	post_init)r‚   r­   r§   rƒ   s    ` €r:   r�   zWhisperEncoder.__init__+  sB  øø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø”~ˆŒØÔ1ˆŒà”Nˆ	Ø"Ô/ˆÔØ!Ô.ˆÔØ$*Ô$?ˆÔ!Ø39Ô3IÐR�4œ9 YÑ/Ô/Ð/ÈsˆÔå”Y˜tÔ0°)ÈÐTUÐVÑVÔVˆŒ
Ý”Y˜y¨)ÀÈ1ÐVWÐXÑXÔXˆŒ
å!œ|¨DÔ,EÀyÑQÔQˆÔØÔ×+Ò+¨EÑ2Ô2Ð2å”mÐ$gÐ$gÐ$gÐ$gÍ%ÐPVÔPeÑJfÔJfÐ$gÑ$gÔ$gÑhÔhˆŒÝœ, v¤~Ñ6Ô6ˆŒà&+ˆÔ#à�ŠÑÔÐÐÐr<   c                 óP   — |                       ¦   «         D ]	}d|_        Œ
d| _        d S ©NF)Ú
parametersÚrequires_gradÚ_requires_grad)r‚   Úparams     r:   Ú_freeze_parametersz!WhisperEncoder._freeze_parametersC  s4   € Ø—_’_Ñ&Ô&ð 	(ð 	(ˆEØ"'ˆEÔÐØ#ˆÔÐÐr<   r'   c                 ó   — | j         S r   ©r'  ©r‚   s    r:   Úget_input_embeddingsz#WhisperEncoder.get_input_embeddingsH  s
   € ØŒzÐr<   r“   c                 ó   — || _         d S r   r8  ©r‚   r“   s     r:   Úset_input_embeddingsz#WhisperEncoder.set_input_embeddingsK  s   € ØˆŒ
ˆ
ˆ
r<   Nr¡   c           	      ó‚  — | j         j        | j        j        d         z  | j        j        d         z  }|j        d         |k    r$t          d|› d|j        d         › d|› d�¦  «        ‚t          j         	                    |                      |¦  «        ¦  «        }t          j         	                    |                      |¦  «        ¦  «        }| 
                    ddd¦  «        }t          j        | j        j        |j        ¬	¦  «        }||                      |¦  «        z   }t          j                             || j        | j        ¬
¦  «        }t%          | j        ¦  «        D ];\  }}	d}
| j        r!t          j        g ¦  «        }|| j        k     rd}
|
s
 |	|dfi |¤Ž}Œ<|                      |¦  «        }t/          |¬¦  «        S )a0  
        Args:
            input_features (`torch.LongTensor` of shape `(batch_size, feature_size, sequence_length)`):
                Float values of mel features extracted from the raw speech waveform. Raw speech waveform can be
                obtained by loading a `.flac` or `.wav` audio file into an array of type `list[float]`, a
                `numpy.ndarray` or a `torch.Tensor`, *e.g.* via the torchcodec library (`pip install torchcodec`) or
                the soundfile library (`pip install soundfile`). To prepare the array into
                `input_features`, the [`AutoFeatureExtractor`] should be used for extracting the mel features, padding
                and conversion into a tensor of type `torch.FloatTensor`. See [`~WhisperFeatureExtractor.__call__`]
            attention_mask (`torch.Tensor`)`, *optional*):
                Whisper does not support masking of the `input_features`, this argument is preserved for compatibility,
                but it is not used. By default the silence in the input log mel spectrogram are ignored.
        r   r*   z7Whisper expects the mel input features to be of length z, but found z-. Make sure to pad the input mel features to ú.r)   r    ©Údevicer˜   FTN)Úlast_hidden_state)r­   r"  r'  r  r(  rB   r-   r   rž   ÚgeluÚpermuter0   r2   r  Únum_embeddingsrA  r•   rš   Ú	enumeraterÃ   r^   r   r-  r   )r‚   rü   rI   r¡   Úexpected_seq_lengthÚinputs_embedsÚall_positionsr¸   ÚidxÚencoder_layerÚto_dropÚdropout_probabilitys               r:   r‰   zWhisperEncoder.forwardN  s  € ð, #œkÔ>ÀÄÔARÐSTÔAUÑUÐX\ÔXbÔXiÐjkÔXlÑlÐØÔ Ô#Ð':Ò:Ð:Ýð IÐJ]ð  Ið  IÐkyÔkð  ACô  lDð  Ið  Ið  sFð  Ið  Ið  Iñô ð õ œ×*Ò*¨4¯:ª:°nÑ+EÔ+EÑFÔFˆÝœ×*Ò*¨4¯:ª:°mÑ+DÔ+DÑEÔEˆà%×-Ò-¨a°°AÑ6Ô6ˆÝœ TÔ%9Ô%HÐQ^ÔQeÐfÑfÔfˆà%¨×(<Ò(<¸]Ñ(KÔ(KÑKˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆå"+¨D¬KÑ"8Ô"8ð 	ð 	ÑˆC�àˆGØŒ}ð #Ý&+¤j°¡n¤nÐ#Ø&¨¬Ò7Ð7Ø"�Gàð Ø - Ø!Øð!ð !ð ð!ð !�øð Ÿš¨Ñ6Ô6ˆåØ+ð
ñ 
ô 
ð 	
r<   r   )rŠ   r‹   rŒ   rÒ   r×   r¦   Ú_can_record_outputsr  r!   r�   r6  r   ÚModuler:  r=  r   r   r   r   r   r‰   r�   rŽ   s   @r:   r  r    s  ø€ € € € € ðð ð -Ø&ðð Ðð "Ðð˜}ð ð ð ð ð ð ð0$ð $ð $ð
 b¤ið ð ð ð ð¨"¬)ð ð ð ð ð  Øð ð6
ð 6
ð Ð+Ô,ð	6
ð
 
ð6
ð 6
ð 6
ñ „_ñ  Ôð6
ð 6
ð 6
ð 6
ð 6
r<   r  c                   ó¸   ‡ — e Zd ZdZe eedd¬¦  «         eedd¬¦  «        dœZdZdZ	d	e
fˆ fd
„Zee	 	 	 	 	 	 	 ddee         defd„¦   «         ¦   «         Zˆ xZS )ÚWhisperDecoderzœ
    Transformer decoder consisting of *config.decoder_layers* layers. Each layer is a [`WhisperDecoderLayer`]

    Args:
        config: WhisperConfig
    r    rÜ   )ÚindexÚ
layer_namerò   )r¸   r  Úcross_attentionsr=   )rþ   r­   c                 ó„  •‡— t          ¦   «                              ‰¦  «         ‰j        | _        ‰j        | _        ‰j        | _        ‰j        | _        ‰j        | _        ‰j	        rt          j        ‰j        ¦  «        nd| _        t          j        ‰j        ‰j        | j        ¦  «        | _        t%          | j        ‰j        ¦  «        | _        t          j        ˆfd„t+          ‰j        ¦  «        D ¦   «         ¦  «        | _        t          j        ‰j        ¦  «        | _        d| _        |                      ¦   «          d S )Nr½   c                 ó0   •— g | ]}t          ‰|¦  «        ‘ŒS rV   )rï   )rW   r¬   r­   s     €r:   rY   z+WhisperDecoder.__init__.<locals>.<listcomp>§  s$   ø€ ÐbÐbÐb¸	Õ  ¨Ñ3Ô3ÐbÐbÐbr<   F)r€   r�   r•   Údecoder_layerdropr   r>   r}   Úmax_target_positionsr"  r#  r.   r$  rÙ   r%  r   r)  Ú
vocab_sizeÚembed_tokensrz   r  r+  rc   Údecoder_layersrÃ   rÝ   r-  r.  r/  ræ   s    `€r:   r�   zWhisperDecoder.__init__š  s  øø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø”~ˆŒØÔ1ˆŒØ!Ô.ˆÔØ$*Ô$?ˆÔ!Ø$*Ô$?ˆÔ!Ø8>Ô8NÐW�4œ9 V¤^Ñ4Ô4Ð4ÐTWˆÔåœL¨Ô):¸F¼NÈDÔL\Ñ]Ô]ˆÔÝ9¸$Ô:SÐU[ÔUcÑdÔdˆÔå”mØbÐbÐbÐbÅUÈ6ÔK`ÑEaÔEaÐbÑbÔbñ
ô 
ˆŒõ œ, v¤~Ñ6Ô6ˆŒà&+ˆÔ#à�ŠÑÔÐÐÐr<   Nr¡   r'   c                 ó&  — |du |duz  rt          d¦  «        ‚|€|                      |¦  «        }|r[|€Y|€| j        j        r6t	          t          | j        ¬¦  «        t          | j        ¬¦  «        ¦  «        nt          | j        ¬¦  «        }|�|                     ¦   «         nd}	|€]t          j        |j	        d         |j
        ¬¦  «        |	z   }|                     d¦  «                             |j	        d         d¦  «        }|�|                      ||	|¬¦  «        }
n|                      ||	|¬¦  «        }
||
                     |j
        ¦  «        z   }t          j                             || j        | j        ¬¦  «        }t'          | j        ||||¬	¦  «        }t)          | j        ¦  «        D ]?\  }}| j        r t          j        g ¦  «        }|| j        k     rŒ, ||||fd|r|nd|d
œ|¤Ž}Œ@|                      |¦  «        }t3          ||¬¦  «        S )ax  
        Args:
            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you
                provide it.

                Indices can be obtained using [`WhisperTokenizer`]. See [`PreTrainedTokenizer.encode`] and
                [`PreTrainedTokenizer.__call__`] for details.

                [What are input IDs?](../glossary#input-ids)
            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:

                - 1 for tokens that are **not masked**,
                - 0 for tokens that are **masked**.

                [What are attention masks?](../glossary#attention-mask)
            encoder_hidden_states (`torch.FloatTensor` of shape `(batch_size, encoder_sequence_length, hidden_size)`, *optional*):
                Sequence of hidden-states at the output of the last layer of the encoder. Used in the cross-attention
                of the decoder.
            past_key_values (`EncoderDecoderCache` or `tuple(tuple(torch.FloatTensor))`, *optional*):
                It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).

                If `past_key_values` are used, the user can optionally input only the last `decoder_input_ids` (those
                that don't have their past key value states given to this model) of shape `(batch_size, 1)` instead of
                all `decoder_input_ids` of shape `(batch_size, sequence_length)`.
            inputs_embeds (`torch.FloatTensor` of
                shape `(batch_size, sequence_length, hidden_size)`, *optional*): Optionally, instead of passing
                `input_ids` you can choose to directly pass an embedded representation. This is useful if you want more
                control over how to convert `input_ids` indices into associated vectors than the model's internal
                embedding lookup matrix.
        NzTYou cannot specify both decoder_input_ids and decoder_inputs_embeds at the same time)r­   r   r    r@  )r‡   rˆ   r˜   )r­   rH  rI   rº   rˆ   )rö   rº   r÷   )rB  rº   )r-   rZ  r­   Úis_encoder_decoderr   r
   Úget_seq_lengthr0   r2   rB   rA  Ú	unsqueezeÚrepeatr  Útor   rž   r•   rš   r   rF  rÃ   r^   r   r-  r   )r‚   r=   rI   rõ   rº   rH  rˆ   r÷   r¡   r‡   Ú	positionsr¸   Úcausal_maskrJ  Údecoder_layerrM  s                   r:   r‰   zWhisperDecoder.forward°  s�  € ðZ ˜Ð -°tÐ";Ñ<ð 	uÝÐsÑtÔtÐtàÐ Ø ×-Ò-¨iÑ8Ô8ˆMàð 	˜Ð0ð )Ð4¸¼Ô8VÐ4õ $¥L¸¼Ð$DÑ$DÔ$DÅlÐZ^ÔZeÐFfÑFfÔFfÑgÔgÐgå!¨¬Ð5Ñ5Ô5ð ð FUÐE` ×!?Ò!?Ñ!AÔ!AÐ!AÐfgÐàÐÝ œ<¨Ô(;¸AÔ(>À}ÔG[Ð\Ñ\Ô\Ð_uÑuˆLØ'×1Ò1°!Ñ4Ô4×;Ò;¸MÔ<OÐPQÔ<RÐTUÑVÔVˆLð Ð Ø×,Ò,ØÐ2HÐWcð -ñ ô ˆIˆIð ×,Ò,ØÐ6LÐ[gð -ñ ô ˆIð &¨	¯ª°]Ô5IÑ(JÔ(JÑJˆÝœ×-Ò-¨m¸t¼|ÐVZÔVcÐ-ÑdÔdˆå(Ø”;Ø'Ø)Ø+Ø%ð
ñ 
ô 
ˆõ #,¨D¬KÑ"8Ô"8ð 	ð 	ÑˆC�àŒ}ð Ý&+¤j°¡n¤nÐ#Ø&¨¬Ò7Ð7Øà)˜MØØØ%ðð (,Ø3<Ð F  À$Ø#ðð ð ðð ˆMˆMð Ÿš¨Ñ6Ô6ˆå8Ø+Ø+ð
ñ 
ô 
ð 	
r<   ©NNNNNNN)rŠ   r‹   rŒ   rÒ   rï   r   r¦   rN  r  r  r!   r�   r   r   r   r   r   r‰   r�   rŽ   s   @r:   rQ  rQ  ‰  s  ø€ € € € € ðð ð -Ø$�nÐ%5¸QÈ;ÐWÑWÔWØ*˜NÐ+;À1ÐQ_Ð`Ñ`Ô`ðð Ðð "€OØ Ðð˜}ð ð ð ð ð ð ð,  Øð ØØ"ØØØØði
ð i
ð Ð+Ô,ði
ð 
3ði
ð i
ð i
ñ „_ñ  Ôði
ð i
ð i
ð i
ð i
r<   rQ  c                   ó�  ‡ — e Zd Zdefˆ fd„Zd„ Zd„ Zd„ Z	 ddej	        dej
        dz  fd	„Zee	 	 	 	 	 	 	 	 	 ddej	        dz  dej
        dz  d
ej
        dz  dej
        dz  deeej	                          dz  dedz  deej	                 dz  deej
                 dz  dedz  deej                 ez  fd„¦   «         ¦   «         Zˆ xZS )ÚWhisperModelr­   c                 óÂ   •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          |¦  «        | _        |                      ¦   «          d S r   )r€   r�   r  ÚencoderrQ  Údecoderr/  ræ   s     €r:   r�   zWhisperModel.__init__   sO   ø€ Ý‰Œ×Ò˜Ñ Ô Ð å% fÑ-Ô-ˆŒÝ% fÑ-Ô-ˆŒà�ŠÑÔÐÐÐr<   c                 ó   — | j         j        S r   ©rj  rZ  r9  s    r:   r:  z!WhisperModel.get_input_embeddings(  ó   € ØŒ|Ô(Ð(r<   c                 ó   — || j         _        d S r   rl  r<  s     r:   r=  z!WhisperModel.set_input_embeddings+  ó   € Ø$)ˆŒÔ!Ð!Ð!r<   c                 ó8   — | j                              ¦   «          dS ©z©
        Calling this function will disable the gradient computation for the Whisper encoder so that its parameters will
        not be updated during training.
        N©ri  r6  r9  s    r:   Úfreeze_encoderzWhisperModel.freeze_encoder.  ó   € ð
 	Œ×'Ò'Ñ)Ô)Ð)Ð)Ð)r<   Nrü   rI   c                 ó~  — t          | j        dd¦  «        s|S |                     ¦   «         \  }}}| j        j        dk    r‡| j        r€t          ||f| j        j        | j        j        || j        j        ¬¦  «        }t          j	        ||j
        t          j        ¬¦  «        }|dd…df                              d|d¦  «        }d||<   | j        j        dk    re| j        r^t          ||f| j        j        | j        j        | j        j        ¬¦  «        }t          j	        ||j
        t          j        ¬¦  «        }d||<   |S )	z¢
        Masks extracted features along time axis and/or along feature axis according to
        [SpecAugment](https://huggingface.co/papers/1904.08779).
        Úapply_spec_augmentTr   )rG   rH   rI   rJ   )rA  rZ   Nr*   )rG   rH   rJ   )Úgetattrr­   r›   Úmask_time_probrš   rx   Úmask_time_lengthÚmask_time_min_masksr0   ÚtensorrA  re   ÚexpandÚmask_feature_probÚmask_feature_lengthÚmask_feature_min_masks)r‚   rü   rI   rp   Úhidden_sizerS   Úmask_time_indicesÚmask_feature_indicess           r:   Ú_mask_input_featuresz!WhisperModel._mask_input_features5  s[  € õ �t”{Ð$8¸$Ñ?Ô?ð 	"Ø!Ð!ð 4B×3FÒ3FÑ3HÔ3HÑ0ˆ
�K àŒ;Ô%¨Ò)Ð)¨d¬mÐ)å 5Ø˜_Ð-Øœ+Ô4Ø œKÔ8Ø-Øœ+Ô9ð!ñ !ô !Ðõ !&¤Ð->À~ÔG\ÕdiÔdnÐ oÑ oÔ oÐØ 1°!°!°!°T°'Ô :× AÒ AÀ"ÀkÐSUÑ VÔ VÐØ01ˆNÐ,Ñ-àŒ;Ô(¨1Ò,Ð,°´Ð,å#8Ø˜[Ð)Øœ+Ô7Ø œKÔ;Øœ+Ô<ð	$ñ $ô $Ð õ $)¤<Ð0DÈ^ÔMbÕjoÔjtÐ#uÑ#uÔ#uÐ Ø34ˆNÐ/Ñ0àÐr<   Údecoder_input_idsÚdecoder_attention_maskÚencoder_outputsrº   Údecoder_inputs_embedsÚdecoder_position_idsr÷   r'   c
                 óÌ  — |€&|                       ||¬¦  «        } | j        |fi |
¤Ž}nct          |t          ¦  «        sNt          |d         t	          |¦  «        dk    r|d         ndt	          |¦  «        dk    r|d         nd¬¦  «        } | j        d	|||d         ||||	dœ|
¤Ž}t          |j        |j        |j	        |j
        |j        |j        |j	        |j
        ¬¦  «        S )
ay	  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`WhisperTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are decoder input IDs?](../glossary#decoder-input-ids)

            Whisper uses the `decoder_start_token_id` as the starting token for `decoder_input_ids` generation. If
            `past_key_values` is used, optionally only the last `decoder_input_ids` have to be input (see
            `past_key_values`).
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.

            If you want to change padding behavior, you should read
            [`modeling_whisper._prepare_decoder_attention_mask`] and modify to your needs. See diagram 1 in [the BART
            paper](https://huggingface.co/papers/1910.13461) for more information on the default strategy.
        decoder_position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
            config.n_positions - 1]`.

            [What are position IDs?](../glossary#position-ids)

        Example:
         ```python
         >>> import torch
         >>> from transformers import AutoFeatureExtractor, WhisperModel
         >>> from datasets import load_dataset

         >>> model = WhisperModel.from_pretrained("openai/whisper-base")
         >>> feature_extractor = AutoFeatureExtractor.from_pretrained("openai/whisper-base")
         >>> ds = load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation")
         >>> inputs = feature_extractor(ds[0]["audio"]["array"], return_tensors="pt")
         >>> input_features = inputs.input_features
         >>> decoder_input_ids = torch.tensor([[1, 1]]) * model.config.decoder_start_token_id
         >>> last_hidden_state = model(input_features, decoder_input_ids=decoder_input_ids).last_hidden_state
         >>> list(last_hidden_state.shape)
         [1, 2, 512]
         ```N)rI   r   r    r)   ©rB  r¸   r  )r=   rI   rõ   rº   rH  rˆ   r÷   )rB  rº   Údecoder_hidden_statesÚdecoder_attentionsrT  Úencoder_last_hidden_staterõ   Úencoder_attentionsrV   )rƒ  ri  r¾   r   rg   rj  r   rB  rº   r¸   r  rT  )r‚   rü   rI   r„  r…  r†  rº   r‡  rˆ  r÷   r¡   Údecoder_outputss               r:   r‰   zWhisperModel.forward`  sD  € ðp Ð"Ø!×6Ò6°~ÐVdÐ6ÑeÔeˆNà*˜dœlØðð àðð ˆOˆOõ ˜O­_Ñ=Ô=ð 	Ý-Ø"1°!Ô"4Ý47¸Ñ4HÔ4HÈ1Ò4LÐ4L˜o¨aÔ0Ð0ÐRVÝ14°_Ñ1EÔ1EÈÒ1IÐ1I˜?¨1Ô-Ð-Ètðñ ô ˆOð '˜$œ,ð 	
Ø'Ø1Ø"1°!Ô"4Ø+Ø/Ø-Øð	
ð 	
ð ð	
ð 	
ˆõ "Ø-Ô?Ø+Ô;Ø"1Ô"?Ø.Ô9Ø,Ô=Ø&5Ô&GØ"1Ô"?Ø.Ô9ð	
ñ 	
ô 	
ð 		
r<   r   )	NNNNNNNNN)rŠ   r‹   rŒ   r!   r�   r:  r=  rs  r0   ÚFloatTensorr  rƒ  r   r   rÕ   r	   re   rÔ   r   r‰   r�   rŽ   s   @r:   rg  rg    sÐ  ø€ € € € € ð˜}ð ð ð ð ð ð ð)ð )ð )ð*ð *ð *ð*ð *ð *ð 37ð)ð )àÔ)ð)ð Ô(¨4Ñ/ð)ð )ð )ð )ðV Øð 48Ø26Ø59Ø:>ØBFØ(,ØAEØ?CØ!%ðY
ð Y
àÔ)¨DÑ0ðY
ð Ô(¨4Ñ/ðY
ð !Ô+¨dÑ2ð	Y
ð
 !&Ô 0°4Ñ 7ðY
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðY
ð  ™ðY
ð  % UÔ%6Ô7¸$Ñ>ðY
ð $ EÔ$4Ô5¸Ñ<ðY
ð ˜$‘;ðY
ð 
ˆuŒ|Ô	Ð1Ñ	1ðY
ð Y
ð Y
ñ „^ñ ÔðY
ð Y
ð Y
ð Y
ð Y
r<   rg  zh
    The Whisper Model with a language modeling head. Can be used for automatic speech recognition.
    )Úcustom_introc                   óš  ‡ — e Zd ZdZddiZdefˆ fd„Zd„ Zd„ Zde	j
        fd	„Zd
„ Zee	 	 	 	 	 	 	 	 	 	 ddej        dz  dej        dz  dej        dz  dej        dz  deeej                          dz  dedz  deej                 dz  deej                 dz  dej        dz  dedz  deej                 ez  fd„¦   «         ¦   «         Zˆ xZS )ÚWhisperForConditionalGenerationrû   úproj_out.weightú!model.decoder.embed_tokens.weightr­   c                 óþ   •— t          ¦   «                              |¦  «         t          |¦  «        | _        t	          j        |j        |j        d¬¦  «        | _        |j	        | _	        |  
                    ¦   «          d S ©NFr¯   )r€   r�   rg  rû   r   r³   rÙ   rY  Úproj_outrX  r/  ræ   s     €r:   r�   z(WhisperForConditionalGeneration.__init__Ç  sj   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ý! &Ñ)Ô)ˆŒ
Ýœ	 &¤.°&Ô2CÈ%ÐPÑPÔPˆŒØ$*Ô$?ˆÔ!ð 	�ŠÑÔÐÐÐr<   c                 ó   — | j         S r   ©r˜  r9  s    r:   Úget_output_embeddingsz5WhisperForConditionalGeneration.get_output_embeddingsÐ  ó
   € ØŒ}Ðr<   c                 ó   — || _         d S r   rš  ©r‚   Únew_embeddingss     r:   Úset_output_embeddingsz5WhisperForConditionalGeneration.set_output_embeddingsÓ  ó   € Ø&ˆŒˆˆr<   r'   c                 ó4   — | j                              ¦   «         S r   ©rû   r:  r9  s    r:   r:  z4WhisperForConditionalGeneration.get_input_embeddingsÖ  ó   € ØŒz×.Ò.Ñ0Ô0Ð0r<   c                 óB   — | j         j                             ¦   «          dS rq  )rû   ri  r6  r9  s    r:   rs  z.WhisperForConditionalGeneration.freeze_encoderÙ  s!   € ð
 	Œ
Ô×-Ò-Ñ/Ô/Ð/Ð/Ð/r<   Nrü   rI   r„  r…  r†  rº   r‡  rˆ  Úlabelsr÷   c                 óz  — |	�e|	j         d         | j        k    r&t          d|	j         d         › d| j        › d�¦  «        ‚|€'|€%t          |	| j        j        | j        j        ¦  «        } | j        |f||||||||
dœ|¤Ž}|                      |j	        ¦  «        }d}|	�et          ¦   «         }|	                     |j        ¦  «        }	 ||                     d| j        j        ¦  «        |	                     d¦  «        ¦  «        }t!          |||j        |j        |j        |j        |j        |j        |j        ¬¦	  «	        S )	a†  
        decoder_input_ids (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Indices of decoder input sequence tokens in the vocabulary.

            Indices can be obtained using [`WhisperTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are decoder input IDs?](../glossary#decoder-input-ids)

            Whisper uses the `decoder_start_token_id` as the starting token for `decoder_input_ids` generation. If
            `past_key_values` is used, optionally only the last `decoder_input_ids` have to be input (see
            `past_key_values`).
        decoder_attention_mask (`torch.LongTensor` of shape `(batch_size, target_sequence_length)`, *optional*):
            Default behavior: generate a tensor that ignores pad tokens in `decoder_input_ids`. Causal mask will also
            be used by default.

            If you want to change padding behavior, you should read
            [`modeling_whisper._prepare_decoder_attention_mask`] and modify to your needs. See diagram 1 in [the BART
            paper](https://huggingface.co/papers/1910.13461) for more information on the default strategy.
        decoder_position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
            config.n_positions - 1]`.

            [What are position IDs?](../glossary#position-ids)
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the language modeling loss. Indices should either be in `[0, ..., config.vocab_size]`
            or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored (masked), the loss is
            only computed for the tokens with labels in `[0, ..., config.vocab_size]`. `sequence_length` should be smaller than or equal to `config.max_target_positions`.

        Example:

        ```python
        >>> import torch
        >>> from transformers import AutoProcessor, WhisperForConditionalGeneration
        >>> from datasets import load_dataset

        >>> processor = AutoProcessor.from_pretrained("openai/whisper-tiny.en")
        >>> model = WhisperForConditionalGeneration.from_pretrained("openai/whisper-tiny.en")

        >>> ds = load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation")

        >>> inputs = processor(ds[0]["audio"]["array"], return_tensors="pt")
        >>> input_features = inputs.input_features

        >>> generated_ids = model.generate(inputs=input_features)

        >>> transcription = processor.batch_decode(generated_ids, skip_special_tokens=True)[0]
        >>> transcription
        ' Mr. Quilter is the apostle of the middle classes, and we are glad to welcome his gospel.'
        ```Nr    zLabels' sequence length z- cannot exceed the maximum allowed length of z tokens.)rI   r„  r†  r…  rº   r‡  rˆ  r÷   r*   )	ÚlossÚlogitsrº   r‹  rŒ  rT  r�  rõ   rŽ  )rB   rX  r-   rF   r­   r>   r?   rû   r˜  rB  r   ra  rA  r3   rY  rn   r   rº   r‹  rŒ  rT  r�  rõ   rŽ  )r‚   rü   rI   r„  r…  r†  rº   r‡  rˆ  r¦  r÷   r¡   ÚoutputsÚ	lm_logitsr¨  Úloss_fcts                   r:   r‰   z'WhisperForConditionalGeneration.forwardà  s’  € ðD ÐØŒ|˜AŒ Ô!:Ò:Ð:Ý ð Q¨v¬|¸A¬ð  Qð  QÐmqô  nGð  Qð  Qð  Qñô ð ð !Ð(Ð-BÐ-JÝ$6Ø˜DœKÔ4°d´kÔ6Xñ%ô %Ð!ð '1 d¤jØð'
à)Ø/Ø+Ø#9Ø+Ø"7Ø!5Øð'
ð '
ð ð'
ð '
ˆð —M’M 'Ô";Ñ<Ô<ˆ	àˆØÐÝ'Ñ)Ô)ˆHà—Y’Y˜yÔ/Ñ0Ô0ˆFØ�8˜IŸNšN¨2¨t¬{Ô/EÑFÔFÈÏÊÐWYÑHZÔHZÑ[Ô[ˆDåØØØ#Ô3Ø")Ô"?Ø&Ô9Ø$Ô5Ø&-Ô&GØ")Ô"?Ø&Ô9ð

ñ 

ô 

ð 
	
r<   )
NNNNNNNNNN)rŠ   r‹   rŒ   r  Ú_tied_weights_keysr!   r�   r›  r   r   rO  r:  rs  r   r   r0   r�  r  rÕ   r	   re   rÔ   r   r‰   r�   rŽ   s   @r:   r“  r“  ¾  sÚ  ø€ € € € € ð  ÐØ+Ð-PÐQÐð˜}ð ð ð ð ð ð ðð ð ð'ð 'ð 'ð1 b¤ið 1ð 1ð 1ð 1ð0ð 0ð 0ð Øð 48Ø26Ø59Ø:>ØBFØ(,ØAEØ?CØ*.Ø!%ði
ð i
àÔ)¨DÑ0ði
ð Ô(¨4Ñ/ði
ð !Ô+¨dÑ2ð	i
ð
 !&Ô 0°4Ñ 7ði
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ði
ð  ™ði
ð  % UÔ%6Ô7¸$Ñ>ði
ð $ EÔ$4Ô5¸Ñ<ði
ð Ô  4Ñ'ði
ð ˜$‘;ði
ð 
ˆuŒ|Ô	˜Ñ	.ði
ð i
ð i
ñ „^ñ Ôði
ð i
ð i
ð i
ð i
r<   r“  c                   ó4   ‡ — e Zd ZdZˆ fd„Zd„ Zd„ Zd„ Zˆ xZS )ÚWhisperDecoderWrapperz½
    This wrapper class is a helper class to correctly load pretrained checkpoints when the causal language model is
    used in combination with the [`EncoderDecoderModel`] framework.
    c                 ó¨   •— t          ¦   «                              |¦  «         d|_        t          |¦  «        | _        |                      ¦   «          d S r1  )r€   r�   r]  rQ  rj  r/  ræ   s     €r:   r�   zWhisperDecoderWrapper.__init__T  sH   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø$)ˆÔ!Ý% fÑ-Ô-ˆŒØ�ŠÑÔÐÐÐr<   c                 ó   — | j         j        S r   rl  r9  s    r:   r:  z*WhisperDecoderWrapper.get_input_embeddingsZ  rm  r<   c                 ó   — || j         _        d S r   rl  r<  s     r:   r=  z*WhisperDecoderWrapper.set_input_embeddings]  ro  r<   c                 ó   —  | j         |i |¤ŽS r   )rj  )r‚   Úargsr¡   s      r:   r‰   zWhisperDecoderWrapper.forward`  s   € ØˆtŒ|˜TÐ, VÐ,Ð,Ð,r<   )	rŠ   r‹   rŒ   rÒ   r�   r:  r=  r‰   r�   rŽ   s   @r:   r¯  r¯  N  so   ø€ € € € € ðð ð
ð ð ð ð ð)ð )ð )ð*ð *ð *ð-ð -ð -ð -ð -ð -ð -r<   r¯  zx
    Whisper decoder with a language modeling head on top (linear layer with weights tied to the input embeddings).
    c                   ó  ‡ — e Zd ZddiZdZˆ fd„Zd„ Zd„ Zdej	        fd„Z
d	„ Zee	 	 	 	 	 	 	 ddej        d
z  dej        d
z  deej                 d
z  ded
z  dej        d
z  dej        d
z  ded
z  deez  fd„¦   «         ¦   «         Zˆ xZS )ÚWhisperForCausalLMr”  r•  r=   c                 óô   •— t          ¦   «                              |¦  «         d|_        t          |¦  «        | _        t          j        |j        |j        d¬¦  «        | _	        |  
                    ¦   «          d S r—  )r€   r�   r]  r¯  rû   r   r³   r€  rY  r˜  r/  ræ   s     €r:   r�   zWhisperForCausalLM.__init__m  sh   ø€ Ý‰Œ×Ò˜Ñ Ô Ð Ø$)ˆÔ!Ý*¨6Ñ2Ô2ˆŒ
åœ	 &Ô"4°fÔ6GÈeÐTÑTÔTˆŒð 	�ŠÑÔÐÐÐr<   c                 ó   — | j         S r   rš  r9  s    r:   r›  z(WhisperForCausalLM.get_output_embeddingsw  rœ  r<   c                 ó   — || _         d S r   rš  rž  s     r:   r   z(WhisperForCausalLM.set_output_embeddingsz  r¡  r<   r'   c                 ó4   — | j                              ¦   «         S r   r£  r9  s    r:   r:  z'WhisperForCausalLM.get_input_embeddings}  r¤  r<   c                 ó:   — | j                              |¦  «         d S r   )rû   r=  r<  s     r:   r=  z'WhisperForCausalLM.set_input_embeddings€  s   € ØŒ
×'Ò'¨Ñ.Ô.Ð.Ð.Ð.r<   NrI   r†  rº   rH  r¦  r÷   c           
      óâ  — t          |t          t          t          f¦  «        r|d         } | j        j        d||||||dœ|¤Ž}	|                      |	d         ¦  «        }
d}|�e|                     |
j        ¦  «        }t          ¦   «         } ||
 
                    d| j        j        ¦  «        | 
                    d¦  «        ¦  «        }t          ||
|	j        |	j        |	j        |	j        ¬¦  «        S )aF  
        encoder_outputs (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
            Sequence of hidden-states at the output of the last layer of the encoder. Used in the cross-attention
            if the model is configured as a decoder.
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
            config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.

        Example:

        ```python
        >>> from transformers import WhisperForCausalLM, WhisperForConditionalGeneration, WhisperProcessor
        >>> import torch
        >>> from datasets import load_dataset

        >>> processor = WhisperProcessor.from_pretrained("openai/whisper-large-v2")
        >>> model = WhisperForConditionalGeneration.from_pretrained("openai/whisper-large-v2")

        >>> assistant_model = WhisperForCausalLM.from_pretrained("distil-whisper/distil-large-v2")

        >>> ds = load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation")
        >>> sample = ds[0]["audio"]
        >>> input_features = processor(
        ...     sample["array"], sampling_rate=sample["sampling_rate"], return_tensors="pt"
        ... ).input_features

        >>> predicted_ids = model.generate(input_features, assistant_model=assistant_model)

        >>> # decode token ids to text
        >>> transcription = processor.batch_decode(predicted_ids, skip_special_tokens=True)[0]
        >>> transcription
        ' Mr. Quilter is the apostle of the middle classes and we are glad to welcome his gospel.'
        ```r   )r=   rI   rõ   rº   rH  r÷   Nr*   )r¨  r©  rº   r¸   r  rT  rV   )r¾   r   rÕ   Úlistrû   rj  r˜  ra  rA  r   r3   r­   rY  r   rº   r¸   r  rT  )r‚   r=   rI   r†  rº   rH  r¦  r÷   r¡   rª  r©  r¨  r¬  s                r:   r‰   zWhisperForCausalLM.forwardƒ  s
  € õ` �o­½ÅÐ'EÑFÔFð 	1Ø-¨aÔ0ˆOð %�$”*Ô$ð 
ØØ)Ø"1Ø+Ø'Øð
ð 
ð ð
ð 
ˆð —’˜w qœzÑ*Ô*ˆàˆØÐØ—Y’Y˜vœ}Ñ-Ô-ˆFÝ'Ñ)Ô)ˆHØ�8˜FŸKšK¨¨D¬KÔ,BÑCÔCÀVÇ[Â[ÐQSÁ_Ä_ÑUÔUˆDå0ØØØ#Ô3Ø!Ô/ØÔ)Ø$Ô5ð
ñ 
ô 
ð 	
r<   re  )rŠ   r‹   rŒ   r­  r  r�   r›  r   r   rO  r:  r=  r   r   r0   r  rÔ   rÕ   r�  r	   re   r   r‰   r�   rŽ   s   @r:   r¶  r¶  d  su  ø€ € € € € ð ,Ð-PÐQÐØ!€Oðð ð ð ð ðð ð ð'ð 'ð 'ð1 b¤ið 1ð 1ð 1ð 1ð/ð /ð /ð Øð .2Ø.2Ø;?Ø(,Ø26Ø*.Ø!%ðK
ð K
àÔ# dÑ*ðK
ð œ tÑ+ðK
ð ˜uÔ0Ô1°DÑ8ð	K
ð
  ™ðK
ð Ô(¨4Ñ/ðK
ð Ô  4Ñ'ðK
ð ˜$‘;ðK
ð 
Ð2Ñ	2ðK
ð K
ð K
ñ „^ñ ÔðK
ð K
ð K
ð K
ð K
r<   r¶  zž
    Whisper Encoder Model with a sequence classification head on top (a linear layer over the pooled output) for tasks
    like SUPERB Keyword Spotting.
    c                   óô   ‡ — e Zd Zˆ fd„Zd„ Zdej        fd„Zdej        fd„Ze	e
	 	 	 ddej        dz  d	eeej                          dz  d
ej        dz  deej                 ez  fd„¦   «         ¦   «         Zˆ xZS )r  c                 ó¨  •— t          ¦   «                              |¦  «         t          |¦  «        | _        |j        dz   }|j        r.t          j        t          j	        |¦  «        |z  ¦  «        | _
        t          j        |j        |j        ¦  «        | _        t          j        |j        |j        ¦  «        | _        |                      ¦   «          d S r…   )r€   r�   r  ri  r	  r  r   Ú	Parameterr0   ri   r  r³   r€  Úclassifier_proj_sizeÚ	projectorÚ
num_labelsÚ
classifierr/  )r‚   r­   Ú
num_layersrƒ   s      €r:   r�   z&WhisperForAudioClassification.__init__Ú  s®   ø€ Ý‰Œ×Ò˜Ñ Ô Ð å% fÑ-Ô-ˆŒØÔ-°Ñ1ˆ
ØÔ(ð 	SÝ!#¤­e¬j¸Ñ.DÔ.DÀzÑ.QÑ!RÔ!RˆDÔÝœ 6Ô#5°vÔ7RÑSÔSˆŒÝœ) FÔ$?ÀÔARÑSÔSˆŒð 	�ŠÑÔÐÐÐr<   c                 ó8   — | j                              ¦   «          dS )zí
        Calling this function will disable the gradient computation for the Whisper encoder so that its parameters will
        not be updated during training. Only the projection layers and classification head will be updated.
        Nrr  r9  s    r:   rs  z,WhisperForAudioClassification.freeze_encoderç  rt  r<   r'   c                 ó4   — | j                              ¦   «         S r   )ri  r:  r9  s    r:   r:  z2WhisperForAudioClassification.get_input_embeddingsî  s   € ØŒ|×0Ò0Ñ2Ô2Ð2r<   r“   c                 ó:   — | j                              |¦  «         d S r   )ri  r=  r<  s     r:   r=  z2WhisperForAudioClassification.set_input_embeddingsñ  s   € ØŒ×)Ò)¨%Ñ0Ô0Ð0Ð0Ð0r<   Nrü   r†  r¦  c                 ó°  — | j         j        rd|d<   |€ | j        |fi |¤Ž}nct          |t          ¦  «        sNt	          |d         t          |¦  «        dk    r|d         ndt          |¦  «        dk    r|d         nd¬¦  «        }| j         j        rx|t                   }t          j        |d¬¦  «        }t          j
                             | j        d	¬¦  «        }||                     d	dd¦  «        z                       d¬¦  «        }n|d         }|                      |¦  «        }|                     d¬¦  «        }|                      |¦  «        }d}	|�et%          ¦   «         }
|                     |j        ¦  «        } |
|                     d	| j         j        ¦  «        |                     d	¦  «        ¦  «        }	t-          |	||j        |j        ¬
¦  «        S )a¥  
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the sequence classification/regression loss. Indices should be in `[0, ...,
            config.num_labels - 1]`. If `config.num_labels == 1` a regression loss is computed (Mean-Square loss), If
            `config.num_labels > 1` a classification loss is computed (Cross-Entropy).

        Example:

        ```python
        >>> import torch
        >>> from transformers import AutoFeatureExtractor, WhisperForAudioClassification
        >>> from datasets import load_dataset

        >>> feature_extractor = AutoFeatureExtractor.from_pretrained("sanchit-gandhi/whisper-medium-fleurs-lang-id")
        >>> model = WhisperForAudioClassification.from_pretrained("sanchit-gandhi/whisper-medium-fleurs-lang-id")

        >>> ds = load_dataset("google/fleurs", "all", split="validation", streaming=True)
        >>> sample = next(iter(ds))

        >>> inputs = feature_extractor(
        ...     sample["audio"]["array"], sampling_rate=sample["audio"]["sampling_rate"], return_tensors="pt"
        ... )
        >>> input_features = inputs.input_features

        >>> with torch.no_grad():
        ...     logits = model(input_features).logits

        >>> predicted_class_ids = torch.argmax(logits).item()
        >>> predicted_label = model.config.id2label[predicted_class_ids]
        >>> predicted_label
        'Afrikaans'
        ```TÚoutput_hidden_statesNr   r    r)   rŠ  r+   r*   )r¨  r©  r¸   r  )r­   r  ri  r¾   r   rg   Ú_HIDDEN_STATES_START_POSITIONr0   Ústackr   rž   rŸ   r  r3   ra   rÂ  ÚmeanrÄ  r   ra  rA  rÃ  r   r¸   r  )r‚   rü   r†  r¦  r¡   r¸   Únorm_weightsÚpooled_outputr©  r¨  r¬  s              r:   r‰   z%WhisperForAudioClassification.forwardô  sõ  € ðT Œ;Ô-ð 	2Ø-1ˆFÐ)Ñ*àÐ"Ø*˜dœlØðð àðð ˆOˆOõ ˜O­_Ñ=Ô=ð 	Ý-Ø"1°!Ô"4Ý47¸Ñ4HÔ4HÈ1Ò4LÐ4L˜o¨aÔ0Ð0ÐRVÝ14°_Ñ1EÔ1EÈÒ1IÐ1I˜?¨1Ô-Ð-Ètðñ ô ˆOð Œ;Ô-ð 	/Ø+Õ,IÔJˆMÝ!œK¨¸1Ð=Ñ=Ô=ˆMÝœ=×0Ò0°Ô1CÈÐ0ÑLÔLˆLØ*¨\×->Ò->¸rÀ1ÀaÑ-HÔ-HÑH×MÒMÐRSÐMÑTÔTˆMˆMà+¨AÔ.ˆMàŸš }Ñ5Ô5ˆØ%×*Ò*¨qÐ*Ñ1Ô1ˆà—’ Ñ/Ô/ˆàˆØÐÝ'Ñ)Ô)ˆHà—Y’Y˜vœ}Ñ-Ô-ˆFØ�8˜FŸKšK¨¨D¬KÔ,BÑCÔCÀVÇ[Â[ÐQSÁ_Ä_ÑUÔUˆDå'ØØØ)Ô7Ø&Ô1ð	
ñ 
ô 
ð 	
r<   )NNN)rŠ   r‹   rŒ   r�   rs  r   rO  r:  r=  r   r   r0   r  rÕ   r�  rÔ   r   r‰   r�   rŽ   s   @r:   r  r  Ó  s  ø€ € € € € ðð ð ð ð ð*ð *ð *ð3 b¤ið 3ð 3ð 3ð 3ð1¨"¬)ð 1ð 1ð 1ð 1ð Øð 37ØBFØ*.ð	P
ð P
àÔ(¨4Ñ/ðP
ð ˜u UÔ%6Ô7Ô8¸4Ñ?ðP
ð Ô  4Ñ'ð	P
ð 
ˆuŒ|Ô	Ð7Ñ	7ðP
ð P
ð P
ñ „^ñ ÔðP
ð P
ð P
ð P
ð P
r<   r  )r¶  r“  rg  rú   r  )r#   )Nr   )Nr�   )RrÒ   r.   Úcollections.abcr   Únumpyr\   r0   r   Útorch.nnr   Ú r   r  Úactivationsr   Úcache_utilsr	   r
   r   Ú
generationr   Úmasking_utilsr   Úmodeling_flash_attention_utilsr   Úmodeling_layersr   Úmodeling_outputsr   r   r   r   r   r   Úmodeling_utilsr   r   Úprocessing_utilsr   Úutilsr   r   r   r   Úutils.genericr   Úutils.output_capturingr   r   Úconfiguration_whisperr!   Úgeneration_whisperr"   Ú
get_loggerrŠ   r±   rË  rN   rÓ   rÔ   r;   rF   rÕ   r  Úndarrayrx   r)  rz   rO  r¤   r¦   r×   rï   rú   r  rQ  rg  r“  r¯  r¶  r  Ú__all__rV   r<   r:   ú<module>rå     sJ  ðð Ð à €€€Ø $Ð $Ð $Ð $Ð $Ð $à Ð Ð Ð Ø €€€Ø Ð Ð Ð Ð Ð Ø %Ð %Ð %Ð %Ð %Ð %à &Ð &Ð &Ð &Ð &Ð &Ø !Ð !Ð !Ð !Ð !Ð !Ø CÐ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CÐ CØ )Ð )Ð )Ð )Ð )Ð )Ø /Ð /Ð /Ð /Ð /Ð /ðð ð ð ð ð ð :Ð 9Ð 9Ð 9Ð 9Ð 9ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð GÐ FÐ FÐ FÐ FÐ FÐ FÐ FØ &Ð &Ð &Ð &Ð &Ð &Ø RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RØ 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø EÐ EÐ EÐ EÐ EÐ EÐ EÐ EØ 0Ð 0Ð 0Ð 0Ð 0Ð 0Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6ð 
ˆÔ	˜HÑ	%Ô	%€à !Ð ð	Dð 	D�cð 	D Sð 	D¸ð 	DÈ5Ì<ð 	Dð 	Dð 	Dð 	Dð %¤,ð ¸cð Ð[^ð ð ð ð ð* /3Øðtð tØ��c�Œ?ðtàðtð ðtð Ô$ tÑ+ð	tð
 ðtð „Zðtð tð tð tðn-ð -ð -ð -ð - ¤ñ -ô -ð -ð" !Øð%ð %ØŒIð%àŒ<ð%ð 
Œð%ð Œ<ð	%ð
 ”L 4Ñ'ð%ð �T‰\ð%ð ð%ð %ð %ð %ð4s)ð s)ð s)ð s)ð s)�r”yñ s)ô s)ð s)ðn5ð 5ð 5ð 5ð 5Ð4ñ 5ô 5ð 5ðpYð Yð Yð Yð YÐ4ñ Yô Yð Yðx ðð ð ð ð ˜_ñ ô ñ „ðð>j
ð j
ð j
ð j
ð j
Ð+ñ j
ô j
ð j
ðZR
ð R
ð R
ð R
ð R
Ð+ñ R
ô R
ð R
ðj ð\
ð \
ð \
ð \
ð \
Ð)ñ \
ô \
ñ „ð\
ð~ €ððñ ô ð
H
ð H
ð H
ð H
ð H
Ð&<Ð>Tñ H
ô H
ñô ð
H
ðV-ð -ð -ð -ð -Ð2ñ -ô -ð -ð, €ððñ ô ð
g
ð g
ð g
ð g
ð g
Ð/°ñ g
ô g
ñô ð
g
ðT €ððñ ô ðm
ð m
ð m
ð m
ð m
Ð$:ñ m
ô m
ñô ðm
ð`ð ð €€€r<   