Ë
    µŒjg$  ã                   ó¬   — d dl mZ d dlmZmZmZmZ d dlmZ d dl	m
Z
 d dlmZ d dlmZmZ d dlmZ  G d„ d	ee«      Z G d
„ de«      Z G d„ de
«      Zy)é    )ÚEnum)ÚAnyÚIteratorÚListÚOptional)ÚCallbackManagerForLLMRun)ÚLLM)ÚGenerationChunk)Ú	BaseModelÚ
ConfigDict)Úenforce_stop_tokensc                   ó   — e Zd ZdZdZdZy)ÚDevicez,The device to use for inference, cuda or cpuÚcudaÚcpuN)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   © ó    úp/var/www/html/Fitness-lenito-AI-main/venv/lib/python3.12/site-packages/langchain_community/llms/titan_takeoff.pyr   r      s   „ Ù6à€DØ
�Cr   r   c                   ó˜   — e Zd ZU dZ ed¬«      Zeed<   	 ej                  Z
eed<   	 dZeed<   	 dZee   ed	<   	 d
Zeed<   	 dZeed<   y)ÚReaderConfigzAConfiguration for the reader to be deployed in Titan Takeoff API.r   )Úprotected_namespacesÚ
model_nameÚdeviceÚprimaryÚconsumer_groupNÚtensor_paralleli   Úmax_seq_lengthé   Úmax_batch_size)r   r   r   r   r   Úmodel_configÚstrÚ__annotations__r   r   r   r   r    r   Úintr!   r#   r   r   r   r   r      se   … ÙKáØô€Lð ƒOØ&à—[‘[€FˆFÓ Ø6à#€N�CÓ#Ø5à%)€O�X˜c‘]Ó)ØIà€N�CÓØKà€N�CÓØ@r   r   c                   ó   ‡ — e Zd ZU dZdZeed<   	 dZeed<   	 dZ	eed<   	 dZ
eed	<   	 d
Zeed<   	 ddddg fdededed	edee   f
ˆ fd„Zedefd„«       Z	 	 ddedeee      dee   dedef
d„Z	 	 ddedeee      dee   dedee   f
d„Zˆ xZS )ÚTitanTakeoffa¬  Titan Takeoff API LLMs.

    Titan Takeoff is a wrapper to interface with Takeoff Inference API for
    generative text to text language models.

    You can use this wrapper to send requests to a generative language model
    and to deploy readers with Takeoff.

    Examples:
        This is an example how to deploy a generative language model and send
        requests.

        .. code-block:: python
            # Import the TitanTakeoff class from community package
            import time
            from langchain_community.llms import TitanTakeoff

            # Specify the embedding reader you'd like to deploy
            reader_1 = {
                "model_name": "TheBloke/Llama-2-7b-Chat-AWQ",
                "device": "cuda",
                "tensor_parallel": 1,
                "consumer_group": "llama"
            }

            # For every reader you pass into models arg Takeoff will spin
            # up a reader according to the specs you provide. If you don't
            # specify the arg no models are spun up and it assumes you have
            # already done this separately.
            llm = TitanTakeoff(models=[reader_1])

            # Wait for the reader to be deployed, time needed depends on the
            # model size and your internet speed
            time.sleep(60)

            # Returns the query, ie a List[float], sent to `llama` consumer group
            # where we just spun up the Llama 7B model
            print(embed.invoke(
                "Where can I see football?", consumer_group="llama"
            ))

            # You can also send generation parameters to the model, any of the
            # following can be passed in as kwargs:
            # https://docs.titanml.co/docs/next/apis/Takeoff%20inference_REST_API/generate#request
            # for instance:
            print(embed.invoke(
                "Where can I see football?", consumer_group="llama", max_new_tokens=100
            ))
    zhttp://localhostÚbase_urli¸  Úporti¹  Ú	mgmt_portFÚ	streamingNÚclientÚmodelsc                 ó
  •— t         ‰| �  ||||¬«       	 ddlm}  || j
                  | j                  | j                  ¬«      | _        |D ]  }| j                  j                  |«       Œ y# t        $ r t	        d«      ‚w xY w)a�  Initialize the Titan Takeoff language wrapper.

        Args:
            base_url (str, optional): The base URL where the Takeoff
                Inference Server is listening. Defaults to `http://localhost`.
            port (int, optional): What port is Takeoff Inference API
                listening on. Defaults to 3000.
            mgmt_port (int, optional): What port is Takeoff Management API
                listening on. Defaults to 3001.
            streaming (bool, optional): Whether you want to by default use the
                generate_stream endpoint over generate to stream responses.
                Defaults to False. In reality, this is not significantly different
                as the streamed response is buffered and returned similar to the
                non-streamed response, but the run manager is applied per token
                generated.
            models (List[ReaderConfig], optional): Any readers you'd like to
                spin up on. Defaults to [].

        Raises:
            ImportError: If you haven't installed takeoff-client, you will
            get an ImportError. To remedy run `pip install 'takeoff-client==0.4.0'`
        )r*   r+   r,   r-   r   )ÚTakeoffClientzjtakeoff-client is required for TitanTakeoff. Please install it with `pip install 'takeoff-client>=0.4.0'`.)r+   r,   N)
ÚsuperÚ__init__Útakeoff_clientr1   ÚImportErrorr*   r+   r,   r.   Úcreate_reader)	Úselfr*   r+   r,   r-   r/   r1   ÚmodelÚ	__class__s	           €r   r3   zTitanTakeoff.__init__o   s‹   ø€ ô< 	‰ÑØ D°IÈð 	ô 	
ð	Ý4ñ $Ø�M‰M §	¡	°T·^±^ô
ˆŒó ˆEØ�K‰K×%Ñ% eÕ,ñ øô ò 	ÜðPóð ð	ús   –A- Á-BÚreturnc                  ó   — y)zReturn type of llm.Útitan_takeoffr   )r7   s    r   Ú	_llm_typezTitanTakeoff._llm_type�   s   € ð r   ÚpromptÚstopÚrun_managerÚkwargsc                 óÖ   — | j                   r,d}| j                  |||¬«      D ]  }||j                  z  }Œ |S  | j                  j                  |fi |¤Ž}|d   }|�t        ||«      }|S )aµ  Call out to Titan Takeoff (Pro) generate endpoint.

        Args:
            prompt: The prompt to pass into the model.
            stop: Optional list of stop words to use when generating.
            run_manager: Optional callback manager to use when streaming.

        Returns:
            The string generated by the model.

        Example:
            .. code-block:: python

                model = TitanTakeoff()

                prompt = "What is the capital of the United Kingdom?"

                # Use of model(prompt), ie `__call__` was deprecated in LangChain 0.1.7,
                # use model.invoke(prompt) instead.
                response = model.invoke(prompt)

        Ú )r>   r?   r@   Útext)r-   Ú_streamrD   r.   Úgenerater   )	r7   r>   r?   r@   rA   Útext_outputÚchunkÚresponserD   s	            r   Ú_callzTitanTakeoff._call¢   s…   € ð: �>Š>ØˆKØŸ™ØØØ'ð &ö �ð
 ˜uŸz™zÑ)‘ðð Ðà'�4—;‘;×'Ñ'¨Ñ9°&Ñ9ˆØ˜ÑˆàÐÜ& t¨TÓ2ˆDØˆr   c              +   ó  K  —  | j                   j                  |fi |¤Ž}d}|D ]   }||j                  z  }d|v sŒ|j                  d«      rd}t	        |j                  dd«      «      dk(  r&|j                  dd«      \  }}	|j                  d«      }|sŒqt        |¬«      }
d}|r|j                  |
j                  ¬«       |
–— Œ¢ |r?t        |j                  dd«      ¬«      }
|r|j                  |
j                  ¬«       |
–— y	y	­w)
a²  Call out to Titan Takeoff (Pro) stream endpoint.

        Args:
            prompt: The prompt to pass into the model.
            stop: Optional list of stop words to use when generating.
            run_manager: Optional callback manager to use when streaming.

        Yields:
            A dictionary like object containing a string token.

        Example:
            .. code-block:: python

                model = TitanTakeoff()

                prompt = "What is the capital of the United Kingdom?"
                response = model.stream(prompt)

                # OR

                model = TitanTakeoff(streaming=True)

                response = model.invoke(prompt)

        rC   zdata:é   é   Ú
)rD   )Útokenz</s>N)r.   Úgenerate_streamÚdataÚ
startswithÚlenÚsplitÚrstripr
   Úon_llm_new_tokenrD   Úreplace)r7   r>   r?   r@   rA   rI   ÚbufferrD   ÚcontentÚ_rH   s              r   rE   zTitanTakeoff._streamÐ   s  è ø€ ð@ /�4—;‘;×.Ñ.¨vÑ@¸Ñ@ˆØˆÛˆDØ�d—i‘iÑˆFØ˜&Ò à×$Ñ$ WÔ-Ø�FÜ�v—|‘| G¨QÓ/Ó0°AÒ5Ø!'§¡¨g°qÓ!9‘J�G˜QØ$Ÿ^™^¨DÓ1�FâÜ+°Ô8�EØ�FÙ"Ø#×4Ñ4¸5¿:¹:Ð4ÔFØ“Kð ñ$ Ü#¨¯©¸ÀÓ)CÔDˆEÙØ×,Ñ,°5·:±:Ð,Ô>Ø‹Kð	 ùs   ‚8D	»AD	ÂA4D	)NN)r   r   r   r   r*   r%   r&   r+   r'   r,   r-   Úboolr.   r   r   r   r3   Úpropertyr=   r   r   rJ   r   r
   rE   Ú__classcell__)r9   s   @r   r)   r)   -   sQ  ø… ñ0ðd '€HˆcÓ&ØWà€Dˆ#ÓØEà€IˆsÓØPà€IˆtÓØ8à€FˆCÓØEð +ØØØØ%'ñ,-àð,-ð ð,-ð ð	,-ð
 ð,-ð �\Ñ"õ,-ð\ ð˜3ò ó ðð %)Ø:>ñ	,àð,ð �t˜C‘yÑ!ð,ð Ð6Ñ7ð	,ð
 ð,ð 
ó,ðb %)Ø:>ñ	8àð8ð �t˜C‘yÑ!ð8ð Ð6Ñ7ð	8ð
 ð8ð 
�/Ñ	"÷8r   r)   N)Úenumr   Útypingr   r   r   r   Úlangchain_core.callbacksr   Ú#langchain_core.language_models.llmsr	   Úlangchain_core.outputsr
   Úpydanticr   r   Úlangchain_community.llms.utilsr   r%   r   r   r)   r   r   r   Ú<module>re      sG   ðÝ ß 0Ó 0å =Ý 3Ý 2ß *å >ôˆS�$ô ôA�9ô Aô4[�3õ [r   