§
    ‚Štjý  ã                   óT   — d Z ddlmZ ddlmZ e G d„ de¦  «        ¦   «         ZdgZdS )z$Speech processor class for SpeechT5.é   )ÚProcessorMixin)Úauto_docstringc                   ó:   ‡ — e Zd Zˆ fd„Zed„ ¦   «         Zd„ Zˆ xZS )ÚSpeechT5Processorc                 óL   •— t          ¦   «                              ||¦  «         d S )N)ÚsuperÚ__init__)ÚselfÚfeature_extractorÚ	tokenizerÚ	__class__s      €ún/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/speecht5/processing_speecht5.pyr	   zSpeechT5Processor.__init__   s$   ø€ Ý‰Œ×ÒÐ*¨IÑ6Ô6Ð6Ð6Ð6ó    c                 óh  — |                      dd ¦  «        }|                      dd ¦  «        }|                      dd ¦  «        }|                      dd ¦  «        }|                      dd ¦  «        }|�|�t          d¦  «        ‚|�|�t          d¦  «        ‚|€|€|€|€t          d¦  «        ‚|� | j        |g|¢R d|i|¤Ž}n|� | j        |fi |¤Ž}nd }|� | j        |||d	œ|¤Ž}	|	d
         }
n|� | j        |fi |¤Ž}	|	d         }
nd }	|€|	S |	�!|
|d<   |	                     d¦  «        }|�||d<   |S )NÚaudioÚtextÚtext_targetÚaudio_targetÚsampling_ratez\Cannot process both `audio` and `text` inputs. Did you mean `audio_target` or `text_target`?z\Cannot process both `audio_target` and `text_target` inputs. Did you mean `audio` or `text`?zaYou need to specify either an `audio`, `audio_target`, `text`, or `text_target` input to process.)r   r   Úinput_valuesÚ	input_idsÚlabelsÚattention_maskÚdecoder_attention_mask)ÚpopÚ
ValueErrorr   r   Úget)r
   ÚargsÚkwargsr   r   r   r   r   ÚinputsÚtargetsr   r   s               r   Ú__call__zSpeechT5Processor.__call__   sÊ  € à—
’
˜7 DÑ)Ô)ˆØ�zŠz˜& $Ñ'Ô'ˆØ—j’j °Ñ5Ô5ˆØ—z’z .°$Ñ7Ô7ˆØŸ
š
 ?°DÑ9Ô9ˆàÐ Ð!1ÝØnñô ð ð Ð#¨Ð(?ÝØnñô ð ð ˆ=˜\Ð1°d°lÀ{ÐGZÝØsñô ð ð ÐØ+�TÔ+¨EÐ`°DÐ`Ð`Ð`ÈÐ`ÐY_Ð`Ð`ˆFˆFØÐØ#�T”^ DÐ3Ð3¨FÐ3Ð3ˆFˆFàˆFàÐ#Ø,�dÔ,È¸,Ð]jÐuÐuÐntÐuÐuˆGØ˜^Ô,ˆFˆFØÐ$Ø$�d”n [Ð;Ð;°FÐ;Ð;ˆGØ˜[Ô)ˆFˆFàˆGàˆ>ØˆNàÐØ%ˆF�8Ñà%,§[¢[Ð1AÑ%BÔ%BÐ"Ø%Ð1Ø3I�Ð/Ñ0àˆr   c                 óª  — |                      dd¦  «        }|                      dd¦  «        }|                      dd¦  «        }|�|�t          d¦  «        ‚|€|€|€t          d¦  «        ‚|� | j        j        |g|¢R i |¤Ž}n|� | j        j        |fi |¤Ž}nd}|�Œd|v st          |t          ¦  «        r&d|d         v r | j        j        |fi |¤Ž}|d         }nO| j        j        }| j        j        | j        _         | j        j        |g|¢R i |¤Ž}|| j        _        |d         }nd}|€|S |�!||d<   | 	                    d¦  «        }	|	�|	|d	<   |S )
au  
        Collates the audio and text inputs, as well as their targets, into a padded batch.

        Audio inputs are padded by SpeechT5FeatureExtractor's [`~SpeechT5FeatureExtractor.pad`]. Text inputs are padded
        by SpeechT5Tokenizer's [`~SpeechT5Tokenizer.pad`].

        Valid input combinations are:

        - `input_ids` only
        - `input_values` only
        - `labels` only, either log-mel spectrograms or text tokens
        - `input_ids` and log-mel spectrogram `labels`
        - `input_values` and text `labels`

        Please refer to the docstring of the above two methods for more information.
        r   Nr   r   z:Cannot process both `input_values` and `input_ids` inputs.zZYou need to specify either an `input_values`, `input_ids`, or `labels` input to be padded.é    r   r   )
r   r   r   Úpadr   Ú
isinstanceÚlistÚfeature_sizeÚnum_mel_binsr   )
r
   r   r   r   r   r   r    r!   Úfeature_size_hackr   s
             r   r%   zSpeechT5Processor.padJ   sÍ  € ð" —z’z .°$Ñ7Ô7ˆØ—J’J˜{¨DÑ1Ô1ˆ	Ø—’˜H dÑ+Ô+ˆàÐ#¨	Ð(=ÝÐYÑZÔZÐZØÐ IÐ$5¸&¸.ÝØlñô ð ð Ð#Ø/�TÔ+Ô/°ÐN¸tÐNÐNÐNÀvÐNÐNˆFˆFØÐ"Ø'�T”^Ô'¨	Ð<Ð<°VÐ<Ð<ˆFˆFàˆFàÐØ˜fÐ$Ð$­°F½DÑ)AÔ)AÐ$ÀkÐU[Ð\]ÔU^ÐF^ÐF^Ø,˜$œ.Ô,¨VÐ>Ð>°vÐ>Ð>�Ø  Ô-��à$(Ô$:Ô$GÐ!Ø6:Ô6LÔ6Y�Ô&Ô3Ø4˜$Ô0Ô4°VÐM¸dÐMÐMÐMÀfÐMÐM�Ø6G�Ô&Ô3Ø  Ô0��àˆGàˆ>ØˆNàÐØ%ˆF�8Ñà%,§[¢[Ð1AÑ%BÔ%BÐ"Ø%Ð1Ø3I�Ð/Ñ0àˆr   )Ú__name__Ú
__module__Ú__qualname__r	   r   r"   r%   Ú__classcell__)r   s   @r   r   r      sc   ø€ € € € € ð7ð 7ð 7ð 7ð 7ð ð.ð .ñ „^ð.ð`:ð :ð :ð :ð :ð :ð :r   r   N)Ú__doc__Úprocessing_utilsr   Úutilsr   r   Ú__all__© r   r   ú<module>r4      s~   ðð +Ð *à .Ð .Ð .Ð .Ð .Ð .Ø #Ð #Ð #Ð #Ð #Ð #ð ðoð oð oð oð o˜ñ oô oñ „ðoðd Ð
€€€r   