ó
    üÞ jà.  ã                   ó¾   • S r SSKJrJrJrJrJr  SSKJr  SSK	J
r
  SSKJrJrJr  SSKJrJr   " S S\5      r " S	 S
\5      rS\\\4   S\4S jr " S S\5      rg)zIContains the `LLMEvaluator` class for building LLM-as-a-judge evaluators.é    )ÚAnyÚCallableÚOptionalÚUnionÚcast)Ú	BaseModel)Ú	warn_beta)ÚEvaluationResultÚEvaluationResultsÚRunEvaluator)ÚExampleÚRunc                   ó`   • \ rS rSr% Sr\\S'   \\   \S'   \\S'   Sr\	\S'   Sr
\\   \S	'   S
rg)ÚCategoricalScoreConfigé   z&Configuration for a categorical score.ÚkeyÚchoicesÚdescriptionFÚinclude_explanationNÚexplanation_description© )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__ÚstrÚ__annotations__Úlistr   Úboolr   r   Ú__static_attributes__r   ó    Ú\/var/www/html/gaurav/venv/lib/python3.13/site-packages/langsmith/evaluation/llm_evaluator.pyr   r      s4   ‡ Ù0à	ƒHØ�#‰YÓØÓØ %Ð˜Ó%Ø-1Ð˜X c™]Ö1r"   r   c                   ól   • \ rS rSr% Sr\\S'   Sr\\S'   Sr	\\S'   \\S'   S	r
\\S
'   Sr\\   \S'   Srg)ÚContinuousScoreConfigé   z%Configuration for a continuous score.r   r   Úminé   Úmaxr   Fr   Nr   r   )r   r   r   r   r   r   r   r'   Úfloatr)   r   r    r   r   r!   r   r"   r#   r%   r%      s<   ‡ Ù/à	ƒHØ€CˆƒNØ€CˆƒNØÓØ %Ð˜Ó%Ø-1Ð˜X c™]Ö1r"   r%   Úscore_configÚreturnc                 ó  • 0 n[        U [        5      (       a1  SU R                  SSR                  U R                  5       S3S.US'   OZ[        U [        5      (       a:  SU R
                  U R                  SU R
                   S	U R                   S
3S.US'   O[        S5      eU R                  (       a!  SU R                  c  SOU R                  S.US'   U R                  U R                  SUU R                  (       a  SS/S.$ S/S.$ )NÚstringz%The score for the evaluation, one of z, Ú.)ÚtypeÚenumr   ÚscoreÚnumberz&The score for the evaluation, between z and z, inclusive.)r0   ÚminimumÚmaximumr   z9Invalid score type. Must be 'categorical' or 'continuous'zThe explanation for the score.)r0   r   ÚexplanationÚobject)Útitler   r0   Ú
propertiesÚrequired)Ú
isinstancer   r   Újoinr%   r'   r)   Ú
ValueErrorr   r   r   r   )r+   r9   s     r#   Ú_create_score_json_schemar>   !   s,  € ð "$€JÜ�,Ô 6×7Ñ7àØ ×(Ñ(ØBØ�y‰y˜×-Ñ-Ó.Ð/¨qð2ñ
ˆ
�7Òô 
�LÔ"7×	8Ñ	8àØ#×'Ñ'Ø#×'Ñ'ØCØ×ÑÐ   l×&6Ñ&6Ð%7°|ðEñ	
ˆ
�7Òô ÐTÓUÐUà×'×'àð  ×7Ñ7Ñ?ñ 1à!×9Ñ9ñ%
ˆ
�=Ñ!ð ×!Ñ!Ø#×/Ñ/ØØ à(4×(H×(HˆW�mÐ$ñð ð PWÈiñð r"   c                   óæ  • \ rS rSrSrSSSS.S\\\\\\4      4   S\\	\
4   S	\\\\\   /\4      S
\S\4
S jjr\SS.S\S\\\\\\4      4   S\\	\
4   S	\\\\\   /\4      4S jj5       rS\\\\\\4      4   S\\	\
4   S	\\\\\   /\4      S\4S jr\ SS\S\\   S\\\4   4S jj5       r\ SS\S\\   S\\\4   4S jj5       rS\S\\   S\4S jrS\S\\\4   4S jrSrg)ÚLLMEvaluatoréL   z¨A class for building LLM-as-a-judge evaluators.

.. deprecated:: 0.5.0

   LLMEvaluator is deprecated. Use openevals instead: https://github.com/langchain-ai/openevals
Nzgpt-4oÚopenai)Úmap_variablesÚ
model_nameÚmodel_providerÚprompt_templater+   rC   rD   rE   c                ó†   •  SSK Jn  U" SXES.UD6n	U R                  XX95        g! [         a  n[        S5      UeSnAff = f)a5  Initialize the `LLMEvaluator`.

Args:
    prompt_template (Union[str, List[Tuple[str, str]]): The prompt
        template to use for the evaluation. If a string is provided, it is
        assumed to be a human / user message.
    score_config (Union[CategoricalScoreConfig, ContinuousScoreConfig]):
        The configuration for the score, either categorical or continuous.
    map_variables (Optional[Callable[[Run, Example], dict]], optional):
        A function that maps the run and example to the variables in the
        prompt.

        If `None`, it is assumed that the prompt only requires 'input',
        'output', and 'expected'.
    model_name (Optional[str], optional): The model to use for the evaluation.
    model_provider (Optional[str], optional): The model provider to use
        for the evaluation.
r   )Úinit_chat_modelzmLLMEvaluator requires langchain to be installed. Please install langchain by running `pip install langchain`.N)ÚmodelrE   r   )Úlangchain.chat_modelsrH   ÚImportErrorÚ_initialize)
ÚselfrF   r+   rC   rD   rE   ÚkwargsrH   ÚeÚ
chat_models
             r#   Ú__init__ÚLLMEvaluator.__init__T   sd   € ð8	õñ %ð 
Øñ
Ø?Eñ
ˆ
ð 	×Ñ˜¸ÕRøô ó 	ÜðOóð ðûð	ús   ‚% ¥
A ¯;»A )rC   rI   c                óL   • U R                  U 5      nUR                  X#XA5        U$ )a*  Create an `LLMEvaluator` instance from a `BaseChatModel` instance.

Args:
    model (BaseChatModel): The chat model instance to use for the evaluation.
    prompt_template (Union[str, List[Tuple[str, str]]): The prompt
        template to use for the evaluation. If a string is provided, it is
        assumed to be a system message.
    score_config (Union[CategoricalScoreConfig, ContinuousScoreConfig]):
        The configuration for the score, either categorical or continuous.
    map_variables (Optional[Callable[[Run, Example]], dict]], optional):
        A function that maps the run and example to the variables in the
        prompt.

        If `None`, it is assumed that the prompt only requires 'input',
        'output', and 'expected'.

Returns:
    LLMEvaluator: An instance of `LLMEvaluator`.
)Ú__new__rL   )ÚclsrI   rF   r+   rC   Úinstances         r#   Ú
from_modelÚLLMEvaluator.from_model€   s'   € ð8 —;‘;˜sÓ#ˆØ×Ñ˜_¸MÔQØˆr"   rP   c                 ó\  •  SSK Jn  SSKJn  [        XE5      (       a  [        US5      (       d  [        S5      e[        U[        5      (       a  UR                  SU4/5      U l
        OUR                  U5      U l
        [        U R                  R                  5      1 S	k-
  (       a  U(       d  [        S
5      eX0l        X l        [        U R                  5      U l        UR#                  U R                   5      nU R                  U-  U l        g! [         a  n[	        S5      UeSnAff = f)a•  Shared initialization code for `__init__` and `from_model`.

Args:
    prompt_template (Union[str, List[Tuple[str, str]]): The prompt template.
    score_config (Union[CategoricalScoreConfig, ContinuousScoreConfig]):
        The score configuration.
    map_variables (Optional[Callable[[Run, Example]], dict]]):
        Function to map variables.
    chat_model (BaseChatModel): The chat model instance.
r   )ÚBaseChatModel)ÚChatPromptTemplatez|LLMEvaluator requires langchain-core to be installed. Please install langchain-core by running `pip install langchain-core`.NÚwith_structured_outputzRchat_model must be an instance of BaseLanguageModel and support structured output.Úhuman>   ÚinputÚoutputÚexpectedzrmap_inputs must be provided if the prompt template contains variables other than 'input', 'output', and 'expected')Ú*langchain_core.language_models.chat_modelsrZ   Úlangchain_core.promptsr[   rK   r;   Úhasattrr=   r   Úfrom_messagesÚpromptÚsetÚinput_variablesrC   r+   r>   Úscore_schemar\   Úrunnable)rM   rF   r+   rC   rP   rZ   r[   rO   s           r#   rL   ÚLLMEvaluator._initialize    s  € ð"	ÝPÝAô �z×1Ñ1Ü˜
Ð$<×=Ñ=äðCóð ô
 �o¤s×+Ñ+Ø,×:Ñ:¸WÀoÐ<VÐ;WÓXˆD�Kà,×:Ñ:¸?ÓKˆDŒKäˆt�{‰{×*Ñ*Ó+Ò.M×MÞ Ü ðMóð ð +Ôà(ÔÜ5°d×6GÑ6GÓHˆÔà×6Ñ6°t×7HÑ7HÓIˆ
ØŸ™ jÑ0ˆ�øôA ó 	ÜðYóð ðûð	ús   ‚D Ä
D+ÄD&Ä&D+ÚrunÚexampler,   c                 ó˜   • U R                  X5      n[        [        U R                  R	                  U5      5      nU R                  U5      $ )zEvaluate a run.)Ú_prepare_variablesr   Údictri   ÚinvokeÚ_parse_output©rM   rk   rl   Ú	variablesr_   s        r#   Úevaluate_runÚLLMEvaluator.evaluate_runÖ   s@   € ð
 ×+Ñ+¨CÓ9ˆ	ÜœD $§-¡-×"6Ñ"6°yÓ"AÓBˆØ×!Ñ! &Ó)Ð)r"   c              ƒ   ó´   #   • U R                  X5      n[        [        U R                  R	                  U5      I Sh  v•N 5      nU R                  U5      $  N7f)zAsynchronously evaluate a run.N)rn   r   ro   ri   Úainvokerq   rr   s        r#   Úaevaluate_runÚLLMEvaluator.aevaluate_runß   sL   é € ð
 ×+Ñ+¨CÓ9ˆ	ÜœD¨¯©×(=Ñ(=¸iÓ(H×"HÓIˆØ×!Ñ! &Ó)Ð)ñ #Iùs   ‚:A¼A
½Ac                 ó  • U R                   (       a  U R                  X5      $ 0 nSU R                  R                  ;   aq  [        UR                  5      S:X  a  [        S5      e[        UR                  5      S:w  a  [        S5      e[        UR                  R                  5       5      S   US'   SU R                  R                  ;   a�  UR                  (       d  [        S5      e[        UR                  5      S:X  a  [        S5      e[        UR                  5      S:w  a  [        S5      e[        UR                  R                  5       5      S   US'   S	U R                  R                  ;   a”  U(       a  UR                  (       d  [        S
5      e[        UR                  5      S:X  a  [        S5      e[        UR                  5      S:w  a  [        S5      e[        UR                  R                  5       5      S   US	'   U$ )z'Prepare variables for model invocation.r^   r   zHNo input keys are present in run.inputs but the prompt requires 'input'.r(   zWMultiple input keys are present in run.inputs. Please provide a map_variables function.r_   zKNo output keys are present in run.outputs but the prompt requires 'output'.zYMultiple output keys are present in run.outputs. Please provide a map_variables function.r`   zMNo example or example outputs is provided but the prompt requires 'expected'.zQNo output keys are present in example.outputs but the prompt requires 'expected'.z]Multiple output keys are present in example.outputs. Please provide a map_variables function.)	rC   re   rg   ÚlenÚinputsr=   r   ÚvaluesÚoutputs)rM   rk   rl   rs   s       r#   rn   ÚLLMEvaluator._prepare_variablesè   sÔ  € à××Ø×%Ñ% cÓ3Ð3àˆ	Ø�d—k‘k×1Ñ1Ó1Ü�3—:‘:‹ !Ó#Ü ð(óð ô �3—:‘:‹ !Ó#Ü ð0óð ô "& c§j¡j×&7Ñ&7Ó&9Ó!:¸1Ñ!=ˆI�gÑà�t—{‘{×2Ñ2Ó2Ø—;—;Ü ð)óð ô �3—;‘;Ó 1Ó$Ü ð)óð ô �3—;‘;Ó 1Ó$Ü ð8óð ô #' s§{¡{×'9Ñ'9Ó';Ó"<¸QÑ"?ˆI�hÑà˜Ÿ™×4Ñ4Ó4Þ '§/§/Ü ð+óð ô �7—?‘?Ó# qÓ(Ü ð+óð ô �7—?‘?Ó# qÓ(Ü ð8óð ô %)¨¯©×)?Ñ)?Ó)AÓ$BÀ1Ñ$EˆI�jÑ!àÐr"   r_   c                 óT  • [        U R                  [        5      (       a5  US   nUR                  SS5      n[	        U R                  R
                  X#S9$ [        U R                  [        5      (       a5  US   nUR                  SS5      n[	        U R                  R
                  XCS9$ g)z1Parse the model output into an evaluation result.r2   r6   N)r   ÚvalueÚcomment)r   r2   r‚   )r;   r+   r   Úgetr
   r   r%   )rM   r_   r�   r6   r2   s        r#   rq   ÚLLMEvaluator._parse_output!  s    € ä�d×'Ñ'Ô)?×@Ñ@Ø˜7‘OˆEØ Ÿ*™* ]°DÓ9ˆKÜ#Ø×%Ñ%×)Ñ)°ñð ô ˜×)Ñ)Ô+@×AÑAØ˜7‘OˆEØ Ÿ*™* ]°DÓ9ˆKÜ#Ø×%Ñ%×)Ñ)°ñð ð Br"   )rC   re   ri   r+   rh   )N)r   r   r   r   r   r   r   r   Útupler   r%   r   r   r   r   ro   rQ   Úclassmethodr   rW   rL   r	   r
   r   rt   rx   rn   rq   r!   r   r"   r#   r@   r@   L   s2  † ñð MQØ"Ø&ò*Sð ˜s D¨¨s°C¨x©Ñ$9Ð9Ñ:ð*Sð Ð2Ð4IÐIÑJð	*Sð
   ¨#¨x¸Ñ/@Ð)AÀ4Ð)GÑ HÑIð*Sð ð*Sð õ*SðX ð MQòàðð ˜s D¨¨s°C¨x©Ñ$9Ð9Ñ:ð	ð
 Ð2Ð4IÐIÑJðð   ¨#¨x¸Ñ/@Ð)AÀ4Ð)GÑ HÑIôó ðð>41à˜s D¨¨s°C¨x©Ñ$9Ð9Ñ:ð41ð Ð2Ð4IÐIÑJð41ð   ¨#¨x¸Ñ/@Ð)AÀ4Ð)GÑ HÑIð	41ð
 ô41ðl à59ñ*Øð*Ø!)¨'Ñ!2ð*à	ÐÐ!2Ð2Ñ	3ô*ó ð*ð à59ñ*Øð*Ø!)¨'Ñ!2ð*à	ÐÐ!2Ð2Ñ	3ô*ó ð*ð7 cð 7°H¸WÑ4Eð 7È$ô 7ðr Dð ¨UÐ3CÐEVÐ3VÑ-W÷ r"   r@   N)r   Útypingr   r   r   r   r   Úpydanticr   Ú#langsmith._internal._beta_decoratorr	   Úlangsmith.evaluationr
   r   r   Úlangsmith.schemasr   r   r   r%   ro   r>   r@   r   r"   r#   Ú<module>rŒ      se   ðÙ Oç 7Õ 7å å 9ß RÑ Rß *ô2˜Yô 2ô2˜Iô 2ð(ØÐ.Ð0EÐEÑFð(à	ô(ôVb�<õ br"   