§
    ‚Štj™3  ã                   óÎ   — d dl mZmZmZ ddlmZ ddlmZ ddlm	Z	 ddl
mZmZ ddlmZ  e	¦   «         rd d	lZdd
lmZ ddlmZ dZ G d„ ded¬¦  «        Z G d„ de¦  «        Zd	S )é    )ÚAnyÚ	TypedDictÚoverloadé   )Ú
AudioInput)ÚGenerationConfig)Úis_torch_available)ÚChatÚChatTypeé   )ÚPipelineN)Ú%MODEL_FOR_TEXT_TO_SPECTROGRAM_MAPPING)ÚSpeechT5HifiGanzmicrosoft/speecht5_hifiganc                   ó(   — e Zd ZU dZeed<   eed<   dS )ÚAudioOutputz›
    audio (`AudioInput`):
        The generated audio waveform.
    sampling_rate (`int`):
        The sampling rate of the generated audio waveform.
    ÚaudioÚsampling_rateN)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   Ú__annotations__Úint© ó    úb/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/pipelines/text_to_audio.pyr   r   !   s6   € € € € € € ðð ð ÐÐÑØÐÐÑÐÐr   r   F)Útotalc                   ó@  ‡ — e Zd ZdZdZdZdZdZdZ e	d¬¦  «        Z
dddœˆ fd„
Zd	„ Zd
„ Zedededefd„¦   «         Zedee         dedee         fd„¦   «         Zedededefd„¦   «         Zedee         dedee         fd„¦   «         Zˆ fd„Z	 	 	 dd„Zd„ Zˆ xZS )ÚTextToAudioPipelineaô  
    Text-to-audio generation pipeline using any `AutoModelForTextToWaveform` or `AutoModelForTextToSpectrogram`. This
    pipeline generates an audio file from an input text and optional other conditional inputs.

    Unless the model you're using explicitly sets these generation parameters in its configuration files
    (`generation_config.json`), the following default values will be used:
    - max_new_tokens: 256

    Example:

    ```python
    >>> from transformers import pipeline

    >>> pipe = pipeline(model="suno/bark-small")
    >>> output = pipe("Hey it's HuggingFace on the phone!")

    >>> audio = output["audio"]
    >>> sampling_rate = output["sampling_rate"]
    ```

    Learn more about the basics of using a pipeline in the [pipeline tutorial](../pipeline_tutorial)

    <Tip>

    You can specify parameters passed to the model by using [`TextToAudioPipeline.__call__.forward_params`] or
    [`TextToAudioPipeline.__call__.generate_kwargs`].

    Example:

    ```python
    >>> from transformers import pipeline

    >>> music_generator = pipeline(task="text-to-audio", model="facebook/musicgen-small")

    >>> # diversify the music generation by adding randomness with a high temperature and set a maximum music length
    >>> generate_kwargs = {
    ...     "do_sample": True,
    ...     "temperature": 0.7,
    ...     "max_new_tokens": 35,
    ... }

    >>> outputs = music_generator("Techno music with high melodic riffs", generate_kwargs=generate_kwargs)
    ```

    </Tip>

    This pipeline can currently be loaded from [`pipeline`] using the following task identifiers: `"text-to-speech"` or
    `"text-to-audio"`.

    See the list of available models on [huggingface.co/models](https://huggingface.co/models?filter=text-to-speech).
    TNFé   )Úmax_new_tokens)Úvocoderr   c                óŽ  •—  t          ¦   «         j        |i |¤Ž d | _        | j        j        t          j        ¦   «         v r?|€6t          j        t          ¦  «         
                    | j        j        ¦  «        n|| _        | j        j        j        dv rd | _        || _        | j        �| j        j        j        | _        | j        €Á| j        j        }| j        j                             dd ¦  «        }|�C|                     d„ |                     ¦   «                              ¦   «         D ¦   «         ¦  «         dD ]M}t+          ||d ¦  «        }|�|| _        Œt+          |dd ¦  «        �t+          |j        |d ¦  «        }|�|| _        ŒN| j        €4| j        �/t/          | j        d¦  «        r| j        j        j        | _        d S d S d S d S )N)ÚmusicgenÚspeecht5Úgeneration_configc                 ó   — i | ]
\  }}|®||“ŒS ©Nr   )Ú.0ÚkÚvs      r   ú
<dictcomp>z0TextToAudioPipeline.__init__.<locals>.<dictcomp>†   s$   € Ð^Ð^Ð^©¨¨1ÐPQÐP]˜q !ÐP]ÐP]ÐP]r   )Úsample_rater   Úcodec_configÚfeature_extractor)ÚsuperÚ__init__r"   ÚmodelÚ	__class__r   Úvaluesr   Úfrom_pretrainedÚDEFAULT_VOCODER_IDÚtoÚdeviceÚconfigÚ
model_typeÚ	processorr   Ú__dict__ÚgetÚupdateÚto_dictÚitemsÚgetattrr.   Úhasattrr/   )	Úselfr"   r   ÚargsÚkwargsr9   Ú
gen_configÚsampling_rate_namer3   s	           €r   r1   zTextToAudioPipeline.__init__m   sÞ  ø€ Ø�‰ŒÔ˜$Ð) &Ð)Ð)Ð)àˆŒØŒ:ÔÕ#HÔ#OÑ#QÔ#QÐQÐQð �?õ  Ô/Õ0BÑCÔC×FÒFÀtÄzÔGXÑYÔYÐYàð ŒLð Œ:ÔÔ'Ð+CÐCÐCà!ˆDŒNà*ˆÔØŒ<Ð#Ø!%¤Ô!4Ô!BˆDÔàÔÐ%ð ”ZÔ&ˆFØœÔ,×0Ò0Ð1DÀdÑKÔKˆJØÐ%Ø—’Ð^Ð^°
×0BÒ0BÑ0DÔ0D×0JÒ0JÑ0LÔ0LÐ^Ñ^Ô^Ñ_Ô_Ð_à&Fð ;ð ;Ð"Ý '¨Ð0BÀDÑ IÔ I�Ø Ð,Ø)6�DÔ&Ð&Ý˜V ^°TÑ:Ô:ÐFÝ$+¨FÔ,?ÐASÐUYÑ$ZÔ$Z�MØ$Ð0Ø-:˜Ô*øð ÔÐ%¨$¬.Ð*DÍÐQUÔQ_ÐatÑIuÔIuÐ*DØ!%¤Ô!AÔ!OˆDÔÐÐð &Ð%Ð*DÐ*DÐ*DÐ*Dr   c                 óL  — t          |t          ¦  «        r|g}| j        j        j        dk    rPd}t          | j        d¦  «        rt          | j        j        dd¦  «        }|ddddœ}| 	                    |¦  «         |}| j
        �| j
        n| j        }t          |t          ¦  «        r |j        |j        fdddœ|¤Ž}ne| j        j        j        d	k    r"d
„ |D ¦   «         }|                     dd¦  «         | j        j        j        dk    rd„ |D ¦   «         } ||fi |¤ddi¤Ž}|S )NÚbarkr    Úsemantic_configÚmax_input_semantic_lengthFT)Ú
max_lengthÚadd_special_tokensÚreturn_attention_maskÚreturn_token_type_ids)ÚtokenizeÚreturn_dictÚcsmc                 óF   — g | ]}|                      d ¦  «        sd|› �n|‘ŒS )ú[z[0]©Ú
startswith©r)   Úts     r   ú
<listcomp>z2TextToAudioPipeline.preprocess.<locals>.<listcomp>µ   s3   € ÐPÐPÐPÀa¨¯ª°cÑ):Ô):ÐA˜	˜a˜	˜	˜	ÀÐPÐPÐPr   rM   Údiac                 óF   — g | ]}|                      d ¦  «        sd|› �n|‘ŒS )rT   z[S1] rU   rW   s     r   rY   z2TextToAudioPipeline.preprocess.<locals>.<listcomp>¸   s3   € ÐRÐRÐRÈ¨1¯<ª<¸Ñ+<Ô+<ÐC˜ ˜˜˜À!ÐRÐRÐRr   Úreturn_tensorsÚpt)Ú
isinstanceÚstrr2   r9   r:   rB   r&   rA   rJ   r>   r;   Ú	tokenizerr
   Úapply_chat_templateÚmessagesÚ
setdefault)rC   ÚtextrE   rL   Ú
new_kwargsÚpreprocessorÚoutputs          r   Ú
preprocesszTextToAudioPipeline.preprocess•   s€  € Ý�d�CÑ Ô ð 	Ø�6ˆDàŒ:ÔÔ'¨6Ò1Ð1ð ˆJÝ�tÔ-Ð/@ÑAÔAð oÝ$ TÔ%;Ô%KÐMhÐjmÑnÔn�
à(Ø&+Ø)-Ø).ð	ð ˆJð ×Ò˜fÑ%Ô%Ð%ØˆFà)-¬Ð)C�t”~�~ÈÌˆÝ�d�DÑ!Ô!ð 	GØ5�\Ô5Ø”ðàØ ðð ð ð	ð ˆFˆFð ŒzÔ Ô+¨uÒ4Ð4ØPÐPÈ4ÐPÑPÔP�Ø×!Ò!Ð"6¸Ñ=Ô=Ð=ØŒzÔ Ô+¨uÒ4Ð4ØRÐRÈTÐRÑRÔR�Ø!�\ $ÐFÐF¨&ÐFÐFÀÐFÐFÐFˆFàˆr   c                 óf  — |                       || j        ¬¦  «        }|d         }|d         }| j                             ¦   «         r‡|                       || j        ¬¦  «        }d|vr
| j        |d<   |                     |¦  «         |                     ddi¦  «         | j        j        j        dv r	d|vrd|d<    | j        j        di |¤|¤Ž}nHt          |¦  «        r$t          d	|                     ¦   «         › �¦  «        ‚ | j        di |¤|¤Žd
         }| j        �|                      |¦  «        }|S )N)r8   Úforward_paramsÚgenerate_kwargsr&   Úreturn_dict_in_generateT)rR   Úoutput_audiozñYou're using the `TextToAudioPipeline` with a forward-only model, but `generate_kwargs` is non empty. For forward-only TTA models, please use `forward_params` instead of `generate_kwargs`. For reference, the `generate_kwargs` used here are: r   r   )Ú_ensure_tensor_on_devicer8   r2   Úcan_generater&   r>   r9   r:   ÚgenerateÚlenÚ
ValueErrorÚkeysr"   )rC   Úmodel_inputsrE   rj   rk   rg   s         r   Ú_forwardzTextToAudioPipeline._forward½   s�  € à×.Ò.¨v¸d¼kÐ.ÑJÔJˆØÐ 0Ô1ˆØ Ð!2Ô3ˆàŒ:×"Ò"Ñ$Ô$ð 	Eà"×;Ò;¸OÐTXÔT_Ð;Ñ`Ô`ˆOð #¨/Ð9Ð9Ø7;Ô7M�Ð 3Ñ4ð ×!Ò! /Ñ2Ô2Ð2ð ×!Ò!Ð#<¸dÐ"CÑDÔDÐDàŒzÔ Ô+¨wÐ6Ð6ð "¨Ð7Ð7Ø59�N >Ñ2à(�T”ZÔ(ÐJÐJ¨<ÐJ¸>ÐJÐJˆFˆFå�?Ñ#Ô#ð Ý ðdàKZ×K_ÒK_ÑKaÔKaðdð dñô ð ð
  �T”ZÐAÐA ,ÐA°.ÐAÐAÀ!ÔDˆFàŒ<Ð#à—\’\ &Ñ)Ô)ˆFàˆr   Útext_inputsrj   Úreturnc                 ó   — d S r(   r   ©rC   rv   rj   s      r   Ú__call__zTextToAudioPipeline.__call__ç   s   € ØPSÐPSr   c                 ó   — d S r(   r   ry   s      r   rz   zTextToAudioPipeline.__call__ê   s   € Ø\_Ð\_r   c                 ó   — d S r(   r   ry   s      r   rz   zTextToAudioPipeline.__call__í   s   € ØUXÐUXr   c                 ó   — d S r(   r   ry   s      r   rz   zTextToAudioPipeline.__call__ð   s   € ØadÐadr   c                 ó8   •—  t          ¦   «         j        |fi |¤ŽS )aL  
        Generates speech/audio from the inputs. See the [`TextToAudioPipeline`] documentation for more information.

        Args:
            text_inputs (`str`, `list[str]`, `ChatType`, or `list[ChatType]`):
                One or several texts to generate. If strings or a list of string are passed, this pipeline will
                generate the corresponding text. Alternatively, a "chat", in the form of a list of dicts with "role"
                and "content" keys, can be passed, or a list of such chats. When chats are passed, the model's chat
                template will be used to format them before passing them to the model.
            forward_params (`dict`, *optional*):
                Parameters passed to the model generation/forward method. `forward_params` are always passed to the
                underlying model.
            generate_kwargs (`dict`, *optional*):
                The dictionary of ad-hoc parametrization of `generate_config` to be used for the generation call. For a
                complete overview of generate, check the [following
                guide](https://huggingface.co/docs/transformers/en/main_classes/text_generation). `generate_kwargs` are
                only passed to the underlying model if the latter is a generative model.

        Return:
            `AudioOutput` or a list of `AudioOutput`, which is a `TypedDict` with two keys:

            - **audio** (`np.ndarray` of shape `(nb_channels, audio_length)`) -- The generated audio waveform.
            - **sampling_rate** (`int`) -- The sampling rate of the generated audio waveform.
        )r0   rz   )rC   rv   rj   r3   s      €r   rz   zTextToAudioPipeline.__call__ó   s$   ø€ ð2  �u‰wŒwÔ Ð>Ð>¨~Ð>Ð>Ð>r   c                 ó²   — t          | dd ¦  «        �
| j        |d<   t          | dd ¦  «        �| j        |d<   | j        |d<   |r|ni |r|ni dœ}|€i }i }|||fS )NÚassistant_modelÚassistant_tokenizerr`   )rj   rk   )rA   r€   r`   r�   )rC   Úpreprocess_paramsrj   rk   ÚparamsÚpostprocess_paramss         r   Ú_sanitize_parametersz(TextToAudioPipeline._sanitize_parameters  sš   € õ �4Ð*¨DÑ1Ô1Ð=Ø15Ô1EˆOÐ-Ñ.Ý�4Ð.°Ñ5Ô5ÐAØ+/¬>ˆO˜KÑ(Ø59Ô5MˆOÐ1Ñ2ð 1?ÐF˜n˜nÀBØ2AÐI˜˜Àrð
ð 
ˆð
 Ð$Ø "ÐØÐà  &Ð*<Ð<Ð<r   c                 ó  — d}t          |t          ¦  «        rd|v r	|d         }n(d}|d         }nt          |t          ¦  «        r|d         }|r!| j        �| j                             |¦  «        }t          |t
          ¦  «        r*d„ |D ¦   «         }t          |¦  «        dk    r|n|d         }nE|                     dt          j	        ¬	¦  «         
                    ¦   «                              ¦   «         }t          || j        ¬
¦  «        S )NFr   TÚ	sequencesr   c                 ó˜   — g | ]G}|                      d t          j        ¬¦  «                             ¦   «                              ¦   «         ‘ŒHS )Úcpu©r8   Údtype)r7   ÚtorchÚfloatÚnumpyÚsqueeze)r)   Úels     r   rY   z3TextToAudioPipeline.postprocess.<locals>.<listcomp>4  sC   € Ð^Ð^Ð^ÐRT�R—U’U %­u¬{�UÑ;Ô;×AÒAÑCÔC×KÒKÑMÔMÐ^Ð^Ð^r   r   r‰   rŠ   )r   r   )r^   ÚdictÚtupler;   ÚdecodeÚlistrq   r7   rŒ   r�   rŽ   r�   r   r   )rC   r   Úneeds_decodings      r   ÚpostprocesszTextToAudioPipeline.postprocess%  s  € ØˆÝ�e�TÑ"Ô"ð 	Ø˜%ÐÐØ˜gœ��à!%�Ø˜kÔ*��Ý˜�uÑ%Ô%ð 	Ø˜!”HˆEàð 	1˜dœnÐ8Ø”N×)Ò)¨%Ñ0Ô0ˆEå�e�TÑ"Ô"ð 	PØ^Ð^ÐX]Ð^Ñ^Ô^ˆEÝ  ™ZœZ¨!š^˜^�E�E°°q´ˆEˆEà—H’H Eµ´�HÑ=Ô=×CÒCÑEÔE×MÒMÑOÔOˆEåØØÔ,ð
ñ 
ô 
ð 	
r   )NNN)r   r   r   r   Ú_pipeline_calls_generateÚ_load_processorÚ_load_image_processorÚ_load_feature_extractorÚ_load_tokenizerr   Ú_default_generation_configr1   rh   ru   r   r_   r   r   rz   r”   r   r…   r–   Ú__classcell__)r3   s   @r   r   r   -   s§  ø€ € € € € ð2ð 2ðh  $ÐØ€OØ!ÐØ#ÐØ€Oð "2Ð!1Øð"ñ "ô "Ðð '+¸$ð &Pð &Pð &Pð &Pð &Pð &Pð &PðP&ð &ð &ðP(ð (ð (ðT ØS CÐS¸3ÐSÀ;ÐSÐSÐSñ „XØSàØ_ D¨¤IÐ_ÀÐ_ÈÈkÔIZÐ_Ð_Ð_ñ „XØ_àØX HÐXÀÐXÈÐXÐXÐXñ „XØXàØd D¨¤NÐdÀcÐdÈdÐS^ÔN_ÐdÐdÐdñ „XØdð?ð ?ð ?ð ?ð ?ð: ØØð	=ð =ð =ð =ð.
ð 
ð 
ð 
ð 
ð 
ð 
r   r   )Útypingr   r   r   Úaudio_utilsr   Ú
generationr   Úutilsr	   Úutils.chat_template_utilsr
   r   Úbaser   rŒ   Úmodels.auto.modeling_autor   Ú!models.speecht5.modeling_speecht5r   r6   r   r   r   r   r   ú<module>r¦      s;  ðð ,Ð +Ð +Ð +Ð +Ð +Ð +Ð +Ð +Ð +à $Ð $Ð $Ð $Ð $Ð $Ø )Ð )Ð )Ð )Ð )Ð )Ø &Ð &Ð &Ð &Ð &Ð &Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6Ð 6Ð 6Ø Ð Ð Ð Ð Ð ð ÐÑÔð DØ€L€L€LàQÐQÐQÐQÐQÐQØCÐCÐCÐCÐCÐCà1Ð ð	ð 	ð 	ð 	ð 	�) 5ð 	ñ 	ô 	ð 	ðO
ð O
ð O
ð O
ð O
˜(ñ O
ô O
ð O
ð O
ð O
r   