§
    ‚Štjá@  ã                  óä   — d dl mZ d dlZd dlZd dlZd dlmZ d dlmZm	Z	m
Z
 erddlmZ  G d„ d¦  «        Z G d	„ d
e¦  «        Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        ZdS )é    )ÚannotationsN)ÚQueue)ÚTYPE_CHECKINGÚAnyÚcasté   )ÚPreTrainedTokenizerBasec                  ó   — e Zd ZdZd„ Zd„ ZdS )ÚBaseStreamerzG
    Base class from which `.generate()` streamers should inherit.
    c                ó   — t          ¦   «         ‚)z;Function that is called by `.generate()` to push new tokens©ÚNotImplementedError©ÚselfÚvalues     ú_/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/generation/streamers.pyÚputzBaseStreamer.put!   ó   € å!Ñ#Ô#Ð#ó    c                ó   — t          ¦   «         ‚)zHFunction that is called by `.generate()` to signal the end of generationr   ©r   s    r   ÚendzBaseStreamer.end%   r   r   N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   © r   r   r   r      s<   € € € € € ðð ð$ð $ð $ð$ð $ð $ð $ð $r   r   c                  ó8   — e Zd ZdZddd	„Zd
„ Zd„ Zddd„Zd„ ZdS )ÚTextStreamera¤  
    Simple text streamer that prints the token(s) to stdout as soon as entire words are formed.

    Parameters:
        tokenizer (`AutoTokenizer`):
            The tokenizer used to decode the tokens.
        skip_prompt (`bool`, *optional*, defaults to `False`):
            Whether to skip the prompt to `.generate()` or not. Useful e.g. for chatbots.
        decode_kwargs (`dict`, *optional*):
            Additional keyword arguments to pass to the tokenizer's `decode` method.

    Examples:

        ```python
        >>> from transformers import AutoModelForCausalLM, AutoTokenizer, TextStreamer

        >>> tok = AutoTokenizer.from_pretrained("openai-community/gpt2")
        >>> model = AutoModelForCausalLM.from_pretrained("openai-community/gpt2")
        >>> inputs = tok(["An increasing sequence: one,"], return_tensors="pt")
        >>> streamer = TextStreamer(tok)

        >>> # Despite returning the usual output, the streamer will also print the generated text to stdout.
        >>> _ = model.generate(**inputs, streamer=streamer, max_new_tokens=20)
        An increasing sequence: one, two, three, four, five, six, seven, eight, nine, ten, eleven,
        ```
    FÚ	tokenizerr	   Úskip_promptÚboolÚdecode_kwargsr   c                óZ   — || _         || _        || _        g | _        d| _        d| _        d S )Nr   T)r    r!   r#   Útoken_cacheÚ	print_lenÚnext_tokens_are_prompt)r   r    r!   r#   s       r   Ú__init__zTextStreamer.__init__F   s8   € Ø"ˆŒØ&ˆÔØ*ˆÔð ')ˆÔØˆŒØ&*ˆÔ#Ð#Ð#r   c                ó€  — t          |j        ¦  «        dk    r |j        d         dk    rt          d¦  «        ‚t          |j        ¦  «        dk    r|d         }| j        r| j        r	d| _        dS | j                             |                     ¦   «         ¦  «         t          t           | j
        j        | j        fi | j        ¤Ž¦  «        }|                     d¦  «        r|| j        d…         }g | _        d| _        nªt          |¦  «        dk    rU|                      t!          |d         ¦  «        ¦  «        r-|| j        d…         }| xj        t          |¦  «        z  c_        nB|| j        |                     d¦  «        dz   …         }| xj        t          |¦  «        z  c_        |                      |¦  «         dS )	zm
        Receives tokens, decodes them, and prints them to stdout as soon as they form entire words.
        é   r   z'TextStreamer only supports batch size 1FNú
éÿÿÿÿú )ÚlenÚshapeÚ
ValueErrorr!   r'   r%   ÚextendÚtolistr   Ústrr    Údecoder#   Úendswithr&   Ú_is_chinese_charÚordÚrfindÚon_finalized_text)r   r   ÚtextÚprintable_texts       r   r   zTextStreamer.putP   sª  € õ ˆuŒ{ÑÔ˜aÒÐ E¤K°¤N°QÒ$6Ð$6ÝÐFÑGÔGÐGÝ�”ÑÔ Ò!Ð!Ø˜!”HˆEàÔð 	 Ô ;ð 	Ø*/ˆDÔ'ØˆFð 	Ô×Ò §¢¡¤Ñ/Ô/Ð/Ý•CÐ.˜œÔ.¨tÔ/?ÐVÐVÀ4ÔCUÐVÐVÑWÔWˆð �=Š=˜ÑÔð 	2Ø! $¤.Ð"2Ð"2Ô3ˆNØ!ˆDÔØˆDŒNˆNå�‰YŒY˜Š]ˆ]˜t×4Ò4µS¸¸b¼±]´]ÑCÔCˆ]Ø! $¤.Ð"2Ð"2Ô3ˆNØˆNŒN�c .Ñ1Ô1Ñ1ˆNŒNˆNð " $¤.°4·:²:¸c±?´?ÀQÑ3FÐ"FÔGˆNØˆNŒN�c .Ñ1Ô1Ñ1ˆNŒNà×Ò˜~Ñ.Ô.Ð.Ð.Ð.r   c                ó  — t          | j        ¦  «        dk    rNt          t           | j        j        | j        fi | j        ¤Ž¦  «        }|| j        d…         }g | _        d| _        nd}d| _        |  	                    |d¬¦  «         dS )z;Flushes any remaining cache and prints a newline to stdout.r   NÚ T)Ú
stream_end)
r.   r%   r   r3   r    r4   r#   r&   r'   r9   )r   r:   r;   s      r   r   zTextStreamer.endr   s–   € õ ˆtÔÑ Ô  1Ò$Ð$Ý�Ð2˜Tœ^Ô2°4Ô3CÐZÐZÀtÔGYÐZÐZÑ[Ô[ˆDØ! $¤.Ð"2Ð"2Ô3ˆNØ!ˆDÔØˆDŒNˆNàˆNà&*ˆÔ#Ø×Ò˜~¸$ÐÑ?Ô?Ð?Ð?Ð?r   r:   r3   r>   c                ó2   — t          |d|sdnd¬¦  «         dS )zNPrints the new text to stdout. If the stream is ending, also prints a newline.Tr=   N)Úflushr   )Úprint©r   r:   r>   s      r   r9   zTextStreamer.on_finalized_text€   s&   € åˆd˜$¨jÐ$B B B¸dÐCÑCÔCÐCÐCÐCr   c                óÊ   — |dk    r|dk    sT|dk    r|dk    sH|dk    r|dk    s<|dk    r|dk    s0|d	k    r|d
k    s$|dk    r|dk    s|dk    r|dk    s|dk    r|dk    rdS dS )z6Checks whether CP is the codepoint of a CJK character.i N  iÿŸ  i 4  i¿M  i   iß¦ i § i?· i@· i¸ i ¸ i¯Î i ù  iÿú  i ø iú TFr   )r   Úcps     r   r6   zTextStreamer._is_chinese_char„   s–   € ð �6Š\ˆ\˜b Fšl˜lØ�f’�  v¢ Ø�g’� "¨¢- -Ø�g’� "¨¢- -Ø�g’� "¨¢- -Ø�g’� "¨¢- -Ø�f’�  v¢ Ø�g’� "¨¢- -à�4àˆur   N©F)r    r	   r!   r"   r#   r   ©r:   r3   r>   r"   )	r   r   r   r   r(   r   r   r9   r6   r   r   r   r   r   *   s†   € € € € € ðð ð6+ð +ð +ð +ð +ð /ð  /ð  /ðD@ð @ð @ðDð Dð Dð Dð Dðð ð ð ð r   r   c                  ó@   ‡ — e Zd ZdZ	 	 ddˆ fd„Zddd„Zd„ Zd„ Zˆ xZS )ÚTextIteratorStreamerac  
    Streamer that stores print-ready text in a queue, to be used by a downstream application as an iterator. This is
    useful for applications that benefit from accessing the generated text in a non-blocking way (e.g. in an interactive
    Gradio demo).

    Parameters:
        tokenizer (`AutoTokenizer`):
            The tokenizer used to decode the tokens.
        skip_prompt (`bool`, *optional*, defaults to `False`):
            Whether to skip the prompt to `.generate()` or not. Useful e.g. for chatbots.
        timeout (`float`, *optional*):
            The timeout for the text queue. If `None`, the queue will block indefinitely. Useful to handle exceptions
            in `.generate()`, when it is called in a separate thread.
        decode_kwargs (`dict`, *optional*):
            Additional keyword arguments to pass to the tokenizer's `decode` method.

    Examples:

        ```python
        >>> from transformers import AutoModelForCausalLM, AutoTokenizer, TextIteratorStreamer
        >>> from threading import Thread

        >>> tok = AutoTokenizer.from_pretrained("openai-community/gpt2")
        >>> model = AutoModelForCausalLM.from_pretrained("openai-community/gpt2")
        >>> inputs = tok(["An increasing sequence: one,"], return_tensors="pt")
        >>> streamer = TextIteratorStreamer(tok)

        >>> # Run the generation in a separate thread, so that we can fetch the generated text in a non-blocking way.
        >>> generation_kwargs = dict(inputs, streamer=streamer, max_new_tokens=20)
        >>> thread = Thread(target=model.generate, kwargs=generation_kwargs)
        >>> thread.start()
        >>> generated_text = ""
        >>> for new_text in streamer:
        ...     generated_text += new_text
        >>> generated_text
        'An increasing sequence: one, two, three, four, five, six, seven, eight, nine, ten, eleven,'
        ```
    FNr    r	   r!   r"   Útimeoutúfloat | Noner#   r   c                ó€   •—  t          ¦   «         j        ||fi |¤Ž t          ¦   «         | _        d | _        || _        d S ©N)Úsuperr(   r   Ú
text_queueÚstop_signalrI   )r   r    r!   rI   r#   Ú	__class__s        €r   r(   zTextIteratorStreamer.__init__Å   sD   ø€ ð 	�‰ŒÔ˜ KÐAÐA°=ÐAÐAÐAÝ™'œ'ˆŒØˆÔØˆŒˆˆr   r:   r3   r>   c                óœ   — | j                              || j        ¬¦  «         |r(| j                              | j        | j        ¬¦  «         dS dS )ú\Put the new text in the queue. If the stream is ending, also put a stop signal in the queue.©rI   N)rN   r   rI   rO   rB   s      r   r9   z&TextIteratorStreamer.on_finalized_textÑ   sZ   € àŒ×Ò˜D¨$¬,ÐÑ7Ô7Ð7Øð 	HØŒO×Ò Ô 0¸$¼,ÐÑGÔGÐGÐGÐGð	Hð 	Hr   c                ó   — | S rL   r   r   s    r   Ú__iter__zTextIteratorStreamer.__iter__×   ó   € Øˆr   c                óx   — | j                              | j        ¬¦  «        }|| j        k    rt	          ¦   «         ‚|S ©NrS   )rN   ÚgetrI   rO   ÚStopIterationr   s     r   Ú__next__zTextIteratorStreamer.__next__Ú   s9   € Ø”×#Ò#¨D¬LÐ#Ñ9Ô9ˆØ�DÔ$Ò$Ð$Ý‘/”/Ð!àˆLr   ©FN©r    r	   r!   r"   rI   rJ   r#   r   rE   rF   )	r   r   r   r   r(   r9   rU   r[   Ú__classcell__©rP   s   @r   rH   rH   �   s‘   ø€ € € € € ð%ð %ðT "Ø $ð	
ð 
ð 
ð 
ð 
ð 
ð 
ðHð Hð Hð Hð Hðð ð ðð ð ð ð ð ð r   rH   c                  ó@   ‡ — e Zd ZdZ	 	 ddˆ fd„Zddd„Zd„ Zd„ Zˆ xZS )ÚAsyncTextIteratorStreamera¢  
    Streamer that stores print-ready text in a queue, to be used by a downstream application as an async iterator.
    This is useful for applications that benefit from accessing the generated text asynchronously (e.g. in an
    interactive Gradio demo).

    Parameters:
        tokenizer (`AutoTokenizer`):
            The tokenizer used to decode the tokens.
        skip_prompt (`bool`, *optional*, defaults to `False`):
            Whether to skip the prompt to `.generate()` or not. Useful e.g. for chatbots.
        timeout (`float`, *optional*):
            The timeout for the text queue. If `None`, the queue will block indefinitely. Useful to handle exceptions
            in `.generate()`, when it is called in a separate thread.
        decode_kwargs (`dict`, *optional*):
            Additional keyword arguments to pass to the tokenizer's `decode` method.

    Raises:
        TimeoutError: If token generation time exceeds timeout value.

    Examples:

        ```python
        >>> from transformers import AutoModelForCausalLM, AutoTokenizer, AsyncTextIteratorStreamer
        >>> from threading import Thread
        >>> import asyncio

        >>> tok = AutoTokenizer.from_pretrained("openai-community/gpt2")
        >>> model = AutoModelForCausalLM.from_pretrained("openai-community/gpt2")
        >>> inputs = tok(["An increasing sequence: one,"], return_tensors="pt")

        >>> # Run the generation in a separate thread, so that we can fetch the generated text in a non-blocking way.
        >>> async def main():
        ...     # Important: AsyncTextIteratorStreamer must be initialized inside a coroutine!
        ...     streamer = AsyncTextIteratorStreamer(tok)
        ...     generation_kwargs = dict(inputs, streamer=streamer, max_new_tokens=20)
        ...     thread = Thread(target=model.generate, kwargs=generation_kwargs)
        ...     thread.start()
        ...     generated_text = ""
        ...     async for new_text in streamer:
        ...         generated_text += new_text
        >>>     print(generated_text)
        >>> asyncio.run(main())
        An increasing sequence: one, two, three, four, five, six, seven, eight, nine, ten, eleven,
        ```
    FNr    r	   r!   r"   rI   rJ   r#   r   c                óN  •—  t          ¦   «         j        ||fi |¤Ž t          j        ¦   «         | _        d | _        || _        t          j        ¦   «         | _        t          t          dd ¦  «        }t          j        dk    ot          |¦  «        | _        | j        r|nd | _        d S )NrI   )é   é   )rM   r(   Úasyncior   rN   rO   rI   Úget_running_loopÚloopÚgetattrÚsysÚversion_infoÚcallableÚhas_asyncio_timeoutÚasyncio_timeout)r   r    r!   rI   r#   Útimeout_contextrP   s         €r   r(   z"AsyncTextIteratorStreamer.__init__  s›   ø€ ð 	�‰ŒÔ˜ KÐAÐA°=ÐAÐAÐAÝ!œ-™/œ/ˆŒØˆÔØˆŒÝÔ,Ñ.Ô.ˆŒ	Ý!¥'¨9°dÑ;Ô;ˆÝ#&Ô#3°wÒ#>Ð#\Å8ÈOÑC\ÔC\ˆÔ Ø26Ô2JÐT˜˜ÐPTˆÔÐÐr   r:   r3   r>   c                ó¬   — | j                              | j        j        |¦  «         |r,| j                              | j        j        | j        ¦  «         dS dS )rR   N)rg   Úcall_soon_threadsaferN   Ú
put_nowaitrO   rB   s      r   r9   z+AsyncTextIteratorStreamer.on_finalized_text!  sZ   € àŒ	×&Ò& t¤Ô'AÀ4ÑHÔHÐHØð 	YØŒI×*Ò*¨4¬?Ô+EÀtÔGWÑXÔXÐXÐXÐXð	Yð 	Yr   c                ó   — | S rL   r   r   s    r   Ú	__aiter__z#AsyncTextIteratorStreamer.__aiter__'  rV   r   c              ƒ  óÔ  K  — 	 | j         rk| j        �d|                      | j        ¦  «        4 ƒd {V —† | j                             ¦   «         ƒ d {V —†}d d d ¦  «        ƒd {V —† n# 1 ƒd {V —†swxY w Y   n8t          j        | j                             ¦   «         | j        ¬¦  «        ƒ d {V —†}|| j        k    rt          ¦   «         ‚|S # t
          j	        $ r t          ¦   «         ‚w xY wrX   )
rl   rm   rI   rN   rY   re   Úwait_forrO   ÚStopAsyncIterationÚTimeoutErrorr   s     r   Ú	__anext__z#AsyncTextIteratorStreamer.__anext__*  sŽ  è è € ð	ØÔ'ð \¨DÔ,@Ð,LØ×/Ò/°´Ñ=Ô=ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8Ø"&¤/×"5Ò"5Ñ"7Ô"7Ð7Ð7Ð7Ð7Ð7Ð7�Eð8ð 8ð 8ñ 8ô 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8ð 8øøøð 8ð 8ð 8ð 8øõ &Ô.¨t¬×/BÒ/BÑ/DÔ/DÈdÌlÐ[Ñ[Ô[Ð[Ð[Ð[Ð[Ð[Ð[�ð ˜Ô(Ò(Ð(Ý(Ñ*Ô*Ð*à�øõ Ô#ð 	!ð 	!ð 	!Ý‘.”.Ð ð	!øøøs.   „.C	 ² A$ÁC	 Á$
A.Á.C	 Á1A.Á2<C	 Ã	C'r\   r]   rE   rF   )	r   r   r   r   r(   r9   rs   rx   r^   r_   s   @r   ra   ra   â   s˜   ø€ € € € € ð,ð ,ðb "Ø $ð	Uð Uð Uð Uð Uð Uð Uð Yð Yð Yð Yð Yðð ð ðð ð ð ð ð ð r   ra   c                  óJ   ‡ — e Zd ZdZ	 	 ddˆ fd„Zd„ Zd„ Zˆ fd„Zˆ fd„Zˆ xZ	S )ÚTextDiffusionStreamera€  
    Streamer that prints text diffusion outputs. Intermediate diffusion steps (drafts) are temporary
    and overwritten by subsequent drafts, and removed when confirmed text is printed.

    <Tip warning={true}>

    If you're running on an environment like tmux, the draft text may fail to overwrite itself.

    </Tip>


    Parameters:
        tokenizer (`AutoTokenizer`):
            The tokenized used to decode the tokens.
        skip_prompt (`bool`, *optional*, defaults to `False`):
            Whether to skip the prompt to `.generate()` or not. Useful e.g. for chatbots.
        sleep_time (`float`, *optional*):
            Time to sleep between diffusion drafts, which may be helpful to visualize intermediate outputs.
        decode_kwargs (`dict`, *optional*):
            Additional keyword arguments to pass to the tokenizer's `decode` method.

    Examples:

        ```python
        >>> from transformers import DiffusionGemmaForBlockDiffusion, AutoProcessor, TextDiffusionStreamer

        >>> model = DiffusionGemmaForBlockDiffusion.from_pretrained(
        ...     "google/diffusiongemma-26B-A4B-it", device_map="auto",
        ... )
        >>> processor = AutoProcessor.from_pretrained("google/diffusiongemma-26B-A4B-it")

        >>> chat = [{"role": "user", "content": "Why is the sky blue?"},]
        >>> input_ids = processor.apply_chat_template(
        ...     chat, tokenize=True, return_tensors="pt", add_generation_prompt=True
        ... )
        >>> streamer = TextDiffusionStreamer(tokenizer=processor.tokenizer)
        >>> model.generate(input_ids.to(model.device), max_new_tokens=512, streamer=streamer)
        ```
    FNr    r	   r!   r"   Ú
sleep_timerJ   r#   r   c                óh   •—  t          ¦   «         j        ||fi |¤Ž d| _        d| _        || _        d S )NF)rM   r(   Ú
_has_draftÚ_takes_logitsr{   )r   r    r!   r{   r#   rP   s        €r   r(   zTextDiffusionStreamer.__init__c  sB   ø€ ð 	�‰ŒÔ˜ KÐAÐA°=ÐAÐAÐAØˆŒð #ˆÔØ$ˆŒˆˆr   c                óJ   — | j         rt          ddd¬¦  «         d| _         d S d S )Nz8[Jr=   T©r   r@   F)r}   rA   r   s    r   Ú_clear_draftz"TextDiffusionStreamer._clear_draftr  s6   € ØŒ?ð 	$å�- R¨tÐ4Ñ4Ô4Ð4Ø#ˆDŒOˆOˆOð	$ð 	$r   c                ó°  — |                       ¦   «          t          |j        ¦  «        dk    r |j        d         dk    rt          d¦  «        ‚t          |j        ¦  «        dk    r|d         } | j        j        |fi | j        ¤Ž}t          ddd¬¦  «         t          d|› d	�dd¬¦  «         d| _        | j	        �t          j        | j	        ¦  «         d
S d
S )z‰
        Receives the full sequence of draft tokens, decodes them, and prints them in yellow.
        Overwrites previous draft.
        r*   r   z0TextDiffusionStreamer only supports batch size 1z7r=   Tr€   z[33mz[0mN)r�   r.   r/   r0   r    r4   r#   rA   r}   r{   ÚtimeÚsleep)r   r   Úkwargsr:   s       r   Ú	put_draftzTextDiffusionStreamer.put_draftx  sî   € ð
 	×ÒÑÔÐåˆuŒ{ÑÔ˜aÒÐ E¤K°¤N°QÒ$6Ð$6ÝÐOÑPÔPÐPÝ�”ÑÔ Ò!Ð!Ø˜!”HˆEà$ˆtŒ~Ô$ UÐAÐA¨dÔ.@ÐAÐAˆõ 	ˆg˜2 TÐ*Ñ*Ô*Ð*åÐ&˜Ð&Ð&Ð&¨B°dÐ;Ñ;Ô;Ð;ØˆŒØŒ?Ð&ÝŒJ�t”Ñ'Ô'Ð'Ð'Ð'ð 'Ð&r   c                ór   •— |                       ¦   «          t          ¦   «                              |¦  «         dS )zEReceives confirmed tokens, clears draft, and prints them permanently.N)r�   rM   r   )r   r   rP   s     €r   r   zTextDiffusionStreamer.putŽ  s1   ø€ à×ÒÑÔÐÝ‰Œ�Š�EÑÔÐÐÐr   c                óp   •— |                       ¦   «          t          ¦   «                              ¦   «          dS )z1Flushes any remaining cache and prints a newline.N)r�   rM   r   )r   rP   s    €r   r   zTextDiffusionStreamer.end“  s*   ø€ à×ÒÑÔÐÝ‰Œ�Š‰Œˆˆˆr   r\   )r    r	   r!   r"   r{   rJ   r#   r   )
r   r   r   r   r(   r�   r†   r   r   r^   r_   s   @r   rz   rz   :  s¥   ø€ € € € € ð&ð &ðV "Ø#'ð	%ð %ð %ð %ð %ð %ð %ð$ð $ð $ð(ð (ð (ð,ð ð ð ð ð
ð ð ð ð ð ð ð ð r   rz   )Ú
__future__r   re   ri   rƒ   Úqueuer   Útypingr   r   r   Útokenization_utils_baser	   r   r   rH   ra   rz   r   r   r   ú<module>r�      s€  ðð #Ð "Ð "Ð "Ð "Ð "à €€€Ø 
€
€
€
Ø €€€Ø Ð Ð Ð Ð Ð Ø +Ð +Ð +Ð +Ð +Ð +Ð +Ð +Ð +Ð +ð ð BØAÐAÐAÐAÐAÐAð$ð $ð $ð $ð $ñ $ô $ð $ðpð pð pð pð p�<ñ pô pð pðfBð Bð Bð Bð B˜<ñ Bô Bð BðJUð Uð Uð Uð U ñ Uô Uð Uðp\ð \ð \ð \ð \˜Lñ \ô \ð \ð \ð \r   