
    j              -       |   d dl Z d dlZd dlmZmZmZmZmZmZm	Z	m
Z
mZmZ ddlmZ ddlmZ ddlmZmZ ddlmZ ddlmZ dd	lmZ dd
lmZ ddlmZ ddlmZ ddlm Z m!Z!m"Z"m#Z#m$Z$m%Z% ddl&m'Z'm(Z( ddl)m*Z*m+Z+m,Z, ddl-m.Z.m/Z/ ddlm0Z0 ddlm1Z1m2Z2m3Z3 ddlm4Z4 ddl5m6Z6 e
rddl5mZ7  e jp                  e9      Z:dZ;dZ<deejz                  ej|                  f   dee?   deee@      dee?   fdZAdeejz                  ej|                  f   dee?   deee@      dee%j                     dee@   de	eej                     ee?   f   fdZDd ej                  ddfd!ZFd"ee1   d#ee0j                     dee(j                     fd$ZI	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dLd&eejz                  ej|                  ej                  f   d'e2d(eee'j                        d)eee.j                        d*ee@   d+ee@   d,ee@   d-eee@ef      d.e?dee?   d/e?d0eej                     d1eeej                        d2ee3   deee@      dee%j                     d3e?d"eee1      d4eee@      dee@   de"j                  f*d5ZOddddddddddd%dddd6d&ejz                  d'e2d7eej                     deee@      dee@   d*ee@   d+ee@   d,ee@   d-eee@ef      d1eeej                        d4eee@      d.e?d/e?d8ee@   d9ee@   d:ee@   dd;f"d<ZQd7ej                  d ej                  d&eejz                  ej|                  f   d'e2d(ee'j                     d,ee@   d.e?dee?   d/e?d2ee3   deee@      dee%j                     d3e?d"ee1   dee@   de"j                  f d=ZRddd>d7ej                  d ej                  d&ejz                  d'e2d,ee@   d.e?d/e?d8ee@   deee@      dee@   de	e"j                  eSf   fd?ZT	 	 	 	 	 	 dMd+e@d(ee'j                     d)eee.j                        d@e?d.e?d2ee3   dAee@   d"eee1      de"j                  fdBZUdCe+j                  dDeee@ef      deee@ef   gee@ef   f   fdEZW	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dNd&eejz                  ej|                  ej                  f   dDeee@ef      dCeee@e+j                  f      d(eee'j                        d)eee.j                        d*ee@   d+ee@   d,ee@   d-eee@ef      d.e?dee?   d/e?d0eej                     deee@      dee%j                     d3e?d"eee1      d4eee@      dee@   de"j                  f(dFZX	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dLd9e@d&eejz                  ej|                  ej                  f   d'e2d(eee'j                        d)eee.j                        d*ee@   d+ee@   d,ee@   d-eee@ef      d.e?dee?   d/e?d0eej                     d1eeej                        d2ee3   deee@      dee%j                     d3e?d"eee1      d4eee@      dee@   de"j                  f,dGZY	 	 	 	 	 	 dOdHeee@ef      d'e2d(eee'j                        d)eee.j                        d,ee@   d.e?d2ee3   d@e?de"j                  fdIZ[d)eee.j                        d(eee'j                        d,ee@   dee'j                     fdJZ\d+ee@   d*ee@   dee@   fdKZ]y)P    N)
AnyCallableDictIteratorListOptionalTupleTYPE_CHECKINGUnioncast   )base_prompt)opik_client)dataset
experiment)dataset_item)helpers)execution_policy)chat_prompt_template)types)evaluation_suite   )asyncio_supportengineevaluation_resultreportrest_operationssamplers)base_metricscore_result)ModelCapabilities
base_modelmodels_factory)scorer_functionscorer_wrapper_metric)test_result)ExperimentScoreFunctionLLMTaskScoringKeyMappingType)url_helpers)suite_result_constructorz>https://www.comet.com/docs/opik/evaluation/evaluate_multimodal   dataset_
nb_samplesdataset_item_idsreturnc                     |t        |      S |$| j                  t        || j                        S |S | j                  S )z;Calculate the total number of items that will be evaluated.)lendataset_items_countminr-   r.   r/   s      r/Users/manta/Documents/Projects/TheRoad-I1/backend/.venv/lib/python3.12/site-packages/opik/evaluation/evaluator.py_calculate_total_itemsr7   4   sM     ##$$''3z8#?#?@@'''    dataset_samplerdataset_filter_stringc                    |+| j                  ||t        |      }t        | ||      }||fS t        j	                  d       t        | j                  ||t        |            }|j                  |      }t        |      t        |      fS )z
    Resolve dataset items for evaluation.

    Handles streaming vs sampling, and calculates total item count.

    Returns:
        Tuple of (items iterator, total item count or None).
    r.   r/   
batch_sizefilter_stringr5   z)Dataset streaming disabled due to sampler)	-__internal_api__stream_items_as_dataclasses__$EVALUATION_STREAM_DATASET_BATCH_SIZEr7   LOGGERinfolistsampleiterr2   )r-   r.   r/   r9   r:   
items_itertotal
items_lists           r6   _resolve_dataset_itemsrI   E   s     KK!-;/	 L 

 '!-

 5  
KK;<>>!-;/	 	? 	
J !''
3J
S_,,r8   r   c                     	 | j                   j                  | j                  g       y # t        $ r% t        j                  d| j                  d       Y y w xY w)N)idszKFailed to notify backend about the experiment completion. Experiment ID: %sTexc_info)experiments_rest_clientfinish_experimentsid	ExceptionrA   debug)r   s    r6   *_try_notifying_about_experiment_completionrS   o   sQ    
**==:==/=R 
YMM 	 	

s   '* +AAexperiment_scoring_functionstest_resultsc                     | r|sg S g }| D ]>  }	  ||      }t        |t              r|j                  |       n|j                  |       @ |S # t        $ r"}t
        j                  d|d       Y d}~id}~ww xY w)z2Compute experiment-level scores from test results.z&Failed to compute experiment score: %sTrL   N)
isinstancerC   extendappendrQ   rA   warning)rT   rU   
all_scoresscore_functionscoreses         r6   _compute_experiment_scoresr_   |   s    
 (|	13J6	#L1F&$'!!&)!!&) 7   	NN8   	s   ;A	A9A44A9   r   taskscoring_metricsscoring_functionsexperiment_name_prefixexperiment_nameproject_nameexperiment_configverbosetask_threadspromptpromptsscoring_key_mappingtrial_countexperiment_tagsc                    t        | t        j                        r| j                  } |g n|}t	        j
                  ||      }t        j                         }t        ||      }|j                  || j                  |||t         | j                         dd            }t        |||      }t        ||| |||||	|
||||||      S )u  
    Performs task evaluation on a given dataset. You can use either `scoring_metrics` or `scorer_functions` to calculate
    evaluation metrics. The scorer functions doesn't require `scoring_key_mapping` and use reserved parameters
    to receive inputs and outputs from the task.

    Args:
        dataset: An Opik Dataset or DatasetVersion instance

        task: A callable object that takes dict with dataset item content
            as input and returns dict which will later be used for scoring.

        experiment_name_prefix: The prefix to be added to automatically generated experiment names to make them unique
            but grouped under the same prefix. For example, if you set `experiment_name_prefix="my-experiment"`,
            the first experiment created will be named `my-experiment-<unique-random-part>`.

        experiment_name: The name of the experiment associated with evaluation run.
            If None, a generated name will be used.

        project_name: The name of the project. If not provided, traces and spans will be logged to the `Default Project`

        experiment_config: The dictionary with parameters that describe experiment

        scoring_metrics: List of metrics to calculate during evaluation.
            Each metric has `score(...)` method, arguments for this method
            are taken from the `task` output, check the signature
            of the `score` method in metrics that you need to find out which keys
            are mandatory in `task`-returned dictionary.
            If no value provided, the experiment won't have any scoring metrics.

        scoring_functions: List of scorer functions to be executed during evaluation.
            Each scorer function includes a scoring method that accepts predefined
            arguments supplied by the evaluation engine:
                • dataset_item — a dictionary containing the dataset item content,
                • task_outputs — a dictionary containing the LLM task output.
                • task_span - the data collected during the LLM task execution [optional].

        verbose: an integer value that controls evaluation output logs such as summary and tqdm progress bar.
            0 - no outputs, 1 - outputs are enabled (default), 2 - outputs are enabled and detailed statistics
            are displayed.

        nb_samples: number of samples to evaluate. If no value is provided, all samples in the dataset will be evaluated.

        task_threads: number of thread workers to run tasks. If set to 1, no additional
            threads are created, all tasks executed in the current thread sequentially.
            are executed sequentially in the current thread.
            Use more than 1 worker if your task object is compatible with sharing across threads.

        prompt: Prompt object to link with experiment. Deprecated, use `prompts` argument instead.

        prompts: A list of Prompt objects to link with experiment.

        scoring_key_mapping: A dictionary that allows you to rename keys present in either the dataset item or the task output
            so that they match the keys expected by the scoring metrics. For example if you have a dataset item with the following content:
            {"user_question": "What is Opik ?"} and a scoring metric that expects a key "input", you can use scoring_key_mapping
            `{"input": "user_question"}` to map the "user_question" key to "input".

        dataset_item_ids: list of dataset item ids to evaluate. If not provided, all samples in the dataset will be evaluated.

        dataset_sampler: An instance of a dataset sampler that will be used to sample dataset items for evaluation.
            If not provided, all samples in the dataset will be evaluated.

        trial_count: number of times to run the task and evaluate the task output for every dataset item.

        experiment_scoring_functions: List of callable functions that compute experiment-level scores.
            Each function takes a list of TestResult objects and returns a list of ScoreResult objects.
            These scores are computed after all test results are collected and represent aggregate
            metrics across the entire experiment.

        experiment_tags: Optional list of tags to associate with the experiment.

        dataset_filter_string: Optional OQL filter string to filter dataset items.
            Supports filtering by tags, data fields, metadata, etc.

            Supported columns include:
            - `id`, `source`, `trace_id`, `span_id`: String fields
            - `data`: Dictionary field (use dot notation, e.g., "data.category")
            - `tags`: List field (use "contains" operator)
            - `created_at`, `last_updated_at`: DateTime fields (ISO 8601 format)
            - `created_by`, `last_updated_by`: String fields

            Examples:
            - `tags contains "failed"` - Items with 'failed' tag
            - `data.category = "test"` - Items with specific data field value
            - `created_at >= "2024-01-01T00:00:00Z"` - Items created after date
    Nrj   rk   re   rd   rP   namedataset_namerg   rk   tagsdataset_version_idrc   rb   rf   clientr   r   ra   rb   rf   rh   r.   ri   rl   r/   r9   rm   rT   r:   )rW   r   EvaluationSuiter   experiment_helpershandle_prompt_argsr   get_client_cached_use_or_create_experiment_namecreate_experimentrs   getattrget_version_info_wrap_scoring_functions_evaluate_task)r   ra   rb   rc   rd   re   rf   rg   rh   r.   ri   rj   rk   rl   r/   r9   rm   rT   rn   r:   checked_promptsry   r   s                          r6   evaluater      s    Z '+;;<// +28T ! );;O
 **,F4'5O
 ))\\+"#;7#;#;#=tTJ * J .+'!O '!!/)'%A3 r8   )ry   r/   r:   rd   re   rf   rg   rk   rn   rh   ri   evaluator_modeloptimization_idexperiment_typery   r   r   r   z!suite_types.EvaluationSuiteResultc                <   |t        j                         }t        ||      }t        || j                  ||	d|
d      }||xs d|d<   ||d<    |j
                  di |}|dk\  rUt        j                  |j                  | j                  |j                  j                  	      }t        j                  |       t        ||| |||||||

      \  }}t        j                  |      }|dk\  r.t        j                   | j                  ||||j"                         |S )aI  
    Run evaluation on a dataset configured as an evaluation suite.

    This function is designed for evaluation suites where evaluators and execution
    policies are stored in the dataset itself. Unlike the general `evaluate` function,
    this function:
    - Does not accept scoring_metrics (they come from the dataset)
    - Does not accept trial_count (it comes from the dataset's execution_policy)
    - Does not accept dataset_sampler or nb_samples (suites evaluate all items)

    Returns:
        EvaluationSuiteResult with pass/fail status for each item and the suite.
    Nrq   r   )rs   rt   rg   rk   evaluation_methodru   rv   trialtyper   r   experiment_id
dataset_idurl_override)
ry   r   r   ra   rf   rh   ri   r   r/   r:   )rh   experiment_url )r   r}   r~   dictrs   r   r*   get_experiment_url_by_idrP   configr   r   display_evaluation_in_progress_evaluate_suite_taskr+   build_suite_resultdisplay_suite_resultsr   )r   ra   ry   r/   r:   rd   re   rf   rg   rk   rn   rh   ri   r   r   r   create_experiment_kwargsexperiment_r   eval_result
total_timesuite_results                         r6   evaluate_suiter   ;  s?   @ ~..04'5O
 04\\+,0 "+:+Eg (6E !23*&**F-EFK!|$==%..zz33

 	--n=2!!')3K ,>>{KL!|$$LL'66	
 r8   c                    t        j                          }t        j                         5  t        |||
||      \  }}t	        j
                  ||      }t        j                  | |||      }|j                  ||||	d |||      }d d d        t        j                          |z
  }t        |      }|dk\  r"t        j                  |j                  |||       t        j                  |j                  |j                  | j                   j"                        }t        j$                  |       | j'                          t)        |       |r |j*                  |	       t-        j.                  |j                  |j                  |j                  ||||
      }|dk\  r!t        j0                  |j                  |       |S # 1 sw Y   =xY w)Nr-   r.   r/   r9   r:   runs_per_itempass_thresholdry   rf   workersrh   dataset_itemsra   rb   rl   r   r   default_execution_policytotal_itemsrT   rU   r   r   r   score_resultsr   r   re   rU   r   rm   experiment_scoresr   rt   evaluation_results)timer   )async_http_connections_expire_immediatelyrI   dataset_execution_policyExecutionPolicyr   EvaluationEnginerun_and_scorer_   r   display_experiment_resultsrs   r*   r   rP   r   r   display_experiment_linkflushrS   log_experiment_scoresr   EvaluationResult$display_evaluation_scores_statistics)ry   r   r   ra   rb   rf   rh   r.   ri   rl   r/   r9   rm   rT   r:   
start_timerF   rG   policyevaluation_enginerU   r   computed_experiment_scoresr   evaluation_result_s                            r6   r   r     s   $ J		B	B	D2!-+"7

E *99%&

 #33% 	
 )66$+ 3 "%+ 7 	
' 
E< z)J "<%A!"
 !|))LL*l4N	
 !99 mm::]]//N "".A
LLN.z: "(
((7QR*;;:: mm"!%4 !|33 1	

 W 
E	Ds   AGG)r/   r:   c        
            t        j                          }
t        j                         5   |j                  |      } |j                         } |j
                  d |t        |	      }|j                  }t        j                  | |||      }|j                  |||d ||||d	      }d d d        t        j                          |
z
  }t        j                  |j                  |j                  | j                  j                        }t!        j"                  |j                  |j                  |j$                  |dg       }| j'                          t)        |       ||fS # 1 sw Y   xY w)Nr<   r   F)	r   ra   rb   rl   r   r   r   r   show_scores_in_progress_barr   r   r   )r   r   r   get_evaluatorsget_execution_policyr?   r@   r3   r   r   r   r*   r   rP   r   r   r   r   rs   r   rS   )ry   r   r   ra   rf   rh   ri   r   r/   r:   r   rb   r   rF   rG   r   rU   r   r   r   s                       r6   r   r     sX    J		B	B	D0'00A77779JWJJ-;/	

 ++"33% 	
 )66$+ $+"%5(- 7 

# 
E: z)J 99 mm::]]//N +;;:: mm"!% LLN.z:z))g 
E	Ds   A=EE'scoring_threadsr   c           	         |g n|}t        j                          }t        j                         }	|r(t        j	                  d       |	j                  |      }
nt        j                  |	|       }
|	j                  |
j                        }t        j                  |
||      }|d   j                  }t        j                  |	|      }t        |||	      }t        j                         5  t!        j"                  |	|||
      }|j%                  |||      }ddd       t        j                          |z
  }t'        |      }|dk\  r"t)        j*                  |j,                  |||       t/        j0                  |
j2                  |j2                  |	j4                  j6                        }t)        j8                  |       t;        |
       |r |
j<                  |       t?        j@                  |j2                  |
j2                  |
j,                  ||d|      }|dk\  r!t)        jB                  |j,                  |       |S # 1 sw Y   -xY w)uL	  Update the existing experiment with new evaluation metrics. You can use either `scoring_metrics` or `scorer_functions` to calculate
    evaluation metrics. The scorer functions doesn't require `scoring_key_mapping` and use reserved parameters
    to receive inputs and outputs from the task.

    Args:
        experiment_name: The name of the experiment to update.

        scoring_metrics: List of metrics to calculate during evaluation.
            Each metric has `score(...)` method, arguments for this method
            are taken from the `task` output, check the signature
            of the `score` method in metrics that you need to find out which keys
            are mandatory in `task`-returned dictionary.

        scoring_functions: List of scorer functions to be executed during evaluation.
            Each scorer function includes a scoring method that accepts predefined
            arguments supplied by the evaluation engine:
                • dataset_item — a dictionary containing the dataset item content,
                • task_outputs — a dictionary containing the LLM task output.
                • task_span - the data collected during the LLM task execution [optional].

        scoring_threads: amount of thread workers to run scoring metrics.

        verbose: an integer value that controls evaluation output logs such as summary and tqdm progress bar.

        scoring_key_mapping: A dictionary that allows you to rename keys present in either the dataset item or the task output
            so that they match the keys expected by the scoring metrics. For example, if you have a dataset item with the following content:
            {"user_question": "What is Opik ?"} and a scoring metric that expects a key "input", you can use scoring_key_mapping
            `{"input": "user_question"}` to map the "user_question" key to "input".

        experiment_id: The ID of the experiment to evaluate. If not provided, the experiment will be evaluated based on the experiment name.

        experiment_scoring_functions: List of callable functions that compute experiment-level scores.
            Each function takes a list of TestResult objects and returns a list of ScoreResult objects.
            These scores are computed after all test results are collected and represent aggregate
            metrics across the entire experiment.
    Nz5Getting experiment by id. Experiment name is ignored.)rP   )ry   re   )rs   )r   r-   rl   r   )ry   trace_idrw   r   )
test_casesrb   rl   r   r   r   r   r   r   r   r   )"r   r   r}   rA   rB   get_experiment_by_idr   get_experiment_with_unique_nameget_datasetrt   get_experiment_test_casesr   get_trace_project_namer   r   r   r   r   score_test_casesr_   r   r   rs   r*   r   rP   r   r   r   rS   r   r   r   r   )re   rb   rc   r   rh   rl   r   rT   r   ry   r   r-   r   first_trace_idrf   r   rU   r   r   r   r   s                        r6   evaluate_experimentr   <  s5   ^ +28T ! J**,FKL00M0B
$DD?

 !!z'>'>!?H ::/J
  ]++N"99L
 .+'!O 
	B	B	D"33%#	
 )99!+ 3 : 
 
E z)J "<%A!"
 !|))MM&		
 !99 mm;;]]//N "".A.z: "(
((7QR*;;;; mm"!%4 !|33!1	

 w 
E	Ds   ,.IImodelmessagesc                 :    t        t        j                  t        j                  t         dd             t        j                  t         dd             d      t        j                  |d      j                         }|D ch c]  }j                  |d      s| }}|rAdj                  t        |            }t        j                  dt         dd      |t               dt         t"        t$        f   d	t         t"        t$        f   f fd
}|S c c}w )N
model_name)visionvideoF)r   validate_placeholdersz, zModel '%s' does not support %s content. Multimedia parts will be flattened to text placeholders. See %s for supported models and customization options.unknownprompt_variablesr0   c                     | j                  d      }j                  | |      }t        j                  |      5 }||j                  d   j
                  j                  dcd d d        S # 1 sw Y   y xY w)Nr   )	variablessupported_modalitiestemplate_type)model_providerr   r   )inputoutput)getformatr"   get_provider_responsechoicesmessagecontent)r   template_type_overrideprocessed_messages
llm_outputchat_prompt_template_r   r   s       r6   _prompt_evaluation_taskz>_build_prompt_evaluation_task.<locals>._prompt_evaluation_task  s~    !1!5!5f!=299&!50 : 
 -- +=
+$,,Q/77??
 
 
s   &A..A7)r   prompt_typesSupportedModalitiesr!   supports_visionr   supports_videor   ChatPromptTemplaterequired_modalitiesr   joinsortedrA   rZ   MODALITY_SUPPORT_DOC_URLr   strr   )	r   r   r   modalityunsupported_modalitiesmodalities_listr   r   r   s	   `      @@r6   _build_prompt_evaluation_taskr     s     (('77|T2 '55|T2		

 1CC 0CCE ,+H#''%8 	+   ))F+A$BC[E<3$	
$sCx. T#s(^   #"As   Dc                    t        | t        j                        r| j                  } |g n|}t        |t              rt        j                  |      }n't        |t        j                        st        d      |}|||j                  d}nd|vr||d<   d|vr|j                  |d<   t        j                         }|r|gnd}t        ||      }|j                  || j                  |||t!         | j"                         dd      	      }t%        |||
      }t'        j&                         }t)        j*                         5  t-        | |
|||      \  }}t/        j0                  ||      }t3        j4                  ||||	      }|j7                  |t9        ||      |dd|||      }ddd       t'        j&                         |z
  }t;        |      }|	dk\  r"t=        j>                  | j                  |||       tA        jB                  |jD                  | jD                  |jF                  jH                        }t=        jJ                  |       |jM                          tO        |       |r |jP                  |       tS        jT                  |jD                  | jD                  |j                  ||||      } |	dk\  r!t=        jV                  | j                  |        | S # 1 sw Y   =xY w)u3  
    Performs prompt evaluation on a given dataset.

    Args:
        dataset: An Opik Dataset or DatasetVersion instance

        messages: A list of prompt messages to evaluate.

        model: The name of the model to use for evaluation. Defaults to "gpt-3.5-turbo".

        scoring_metrics: List of metrics to calculate during evaluation.
            The LLM input and output will be passed as arguments to each metric `score(...)` method.

        scoring_functions: List of scorer functions to be executed during evaluation.
            Each scorer function includes a scoring method that accepts predefined
            arguments supplied by the evaluation engine:
                • dataset_item — a dictionary containing the dataset item content,
                • task_outputs — a dictionary containing the LLM task output.
                • task_span - the data collected during the LLM task execution [optional].

        experiment_name_prefix: The prefix to be added to automatically generated experiment names to make them unique
            but grouped under the same prefix. For example, if you set `experiment_name_prefix="my-experiment"`,
            the first experiment created will be named `my-experiment-<unique-random-part>`.

        experiment_name: name of the experiment.

        project_name: The name of the project to log data

        experiment_config: configuration of the experiment.

        verbose: an integer value that controls evaluation output logs such as summary and tqdm progress bar.

        nb_samples: number of samples to evaluate.

        task_threads: amount of thread workers to run scoring metrics.

        prompt: Prompt object to link with experiment.

        dataset_item_ids: list of dataset item ids to evaluate. If not provided, all samples in the dataset will be evaluated.

        dataset_sampler: An instance of a dataset sampler that will be used to sample dataset items for evaluation.
            If not provided, all samples in the dataset will be evaluated.

        trial_count: number of times to execute the prompt and evaluate the LLM output for every dataset item.

        experiment_scoring_functions: List of callable functions that compute experiment-level scores.
            Each function takes a list of TestResult objects and returns a list of ScoreResult objects.
            These scores are computed after all test results are collected and represent aggregate
            metrics across the entire experiment.

        experiment_tags: List of tags to be associated with the experiment.

        dataset_filter_string: Optional OQL filter string to filter dataset items.
            Supports filtering by tags, data fields, metadata, etc.

            Supported columns include:
            - `id`, `source`, `trace_id`, `span_id`: String fields
            - `data`: Dictionary field (use dot notation, e.g., "data.category")
            - `tags`: List field (use "contains" operator)
            - `created_at`, `last_updated_at`: DateTime fields (ISO 8601 format)
            - `created_by`, `last_updated_by`: String fields

            Examples:
            - `tags contains "failed"` - Items with 'failed' tag
            - `data.category = "test"` - Items with specific data field value
            - `created_at >= "2024-01-01T00:00:00Z"` - Items created after date
    N)r   z<`model` must be either a string or an OpikBaseModel instance)prompt_templater   r   r   rq   rP   rr   rw   r   r   r   )r   r   r   r   r   r   r   r   )r   r   re   rU   r   rm   r   r   r   ),rW   r   rz   r   r   r#   r   r"   OpikBaseModel
ValueErrorr   r   r}   r~   r   rs   r   r   r   r   r   r   rI   r   r   r   r   r   r   r_   r   r   r*   r   rP   r   r   r   r   rS   r   r   r   r   )!r   r   r   rb   rc   rd   re   rf   rg   rh   r.   ri   rj   r/   r9   rm   rT   rn   r:   
opik_modelry   rk   r   r   rF   rG   r   r   rU   r   r   r   r   s!                                    r6   evaluate_promptr    s   t '+;;<// +28T ! %#''59
z778WXX
 '**

 $553;/0++)3)>)>g&**,F vhdG4'5O
 ))\\+"#;7#;#;#=tTJ * J .+'!O J		B	B	D2!-+"7

E *99%&

 #33% 	
 )66$.Z(S+ $ "%+ 7 	
' 
E< z)J "<%A!"
 !|))LL*l4N	
 !99 mm::]]//N "".A
LLN.z: "(
((7QR*;; mm::"!%4 !|33 1	

 W 
E	Ds   A(K%%K/c                    t        |t        j                        r|j                  }|g n|}|g }t	        j
                  ||      }t        |||      }t        j                         }t        ||      }|j                  ||j                  ||d| |t         |j                         dd            }t        |||||||	|
|||||||      S )	u'  
    Performs task evaluation on a given dataset.

    Args:
        optimization_id: The ID of the optimization associated with the experiment.

        dataset: An Opik Dataset or DatasetVersion instance

        task: A callable object that takes dict with dataset item content
            as input and returns dict which will later be used for scoring.

        scoring_functions: List of scorer functions to be executed during evaluation.
            Each scorer function includes a scoring method that accepts predefined
            arguments supplied by the evaluation engine:
                • dataset_item — a dictionary containing the dataset item content,
                • task_outputs — a dictionary containing the LLM task output.
                • task_span - the data collected during the LLM task execution [optional].

        experiment_name_prefix: The prefix to be added to automatically generated experiment names to make them unique
                    but grouped under the same prefix. For example, if you set `experiment_name_prefix="my-experiment"`,
                    the first experiment created will be named `my-experiment-<unique-random-part>`.

        experiment_name: The name of the experiment associated with evaluation run.
            If None, a generated name will be used.

        project_name: The name of the project. If not provided, traces and spans will be logged to the `Default Project`

        experiment_config: The dictionary with parameters that describe experiment

        scoring_metrics: List of metrics to calculate during evaluation.
            Each metric has `score(...)` method, arguments for this method
            are taken from the `task` output, check the signature
            of the `score` method in metrics that you need to find out which keys
            are mandatory in `task`-returned dictionary.
            If no value provided, the experiment won't have any scoring metrics.

        verbose: an integer value that controls evaluation output logs such as summary and tqdm progress bar.
            0 - no outputs, 1 - outputs are enabled (default).

        nb_samples: number of samples to evaluate. If no value is provided, all samples in the dataset will be evaluated.

        task_threads: number of thread workers to run tasks. If set to 1, no additional
            threads are created, all tasks executed in the current thread sequentially.
            are executed sequentially in the current thread.
            Use more than 1 worker if your task object is compatible with sharing across threads.

        prompt: Prompt object to link with experiment. Deprecated, use `prompts` argument instead.

        prompts: A list of Prompt objects to link with experiment.

        scoring_key_mapping: A dictionary that allows you to rename keys present in either the dataset item or the task output
            so that they match the keys expected by the scoring metrics. For example if you have a dataset item with the following content:
            {"user_question": "What is Opik ?"} and a scoring metric that expects a key "input", you can use scoring_key_mapping
            `{"input": "user_question"}` to map the "user_question" key to "input".

        dataset_item_ids: list of dataset item ids to evaluate. If not provided, all samples in the dataset will be evaluated.

        dataset_sampler: An instance of a dataset sampler that will be used to sample dataset items for evaluation.
            If not provided, all samples in the dataset will be evaluated.

        trial_count: number of times to execute the prompt and evaluate the LLM output for every dataset item.

        experiment_scoring_functions: List of callable functions that compute experiment-level scores.
            Each function takes a list of TestResult objects and returns a list of ScoreResult objects.
            These scores are computed after all test results are collected and represent aggregate
            metrics across the entire experiment.

        experiment_tags: A list of tags to associate with the experiment.

        dataset_filter_string: Optional OQL filter string to filter dataset items.
            Supports filtering by tags, data fields, metadata, etc.

            Supported columns include:
            - `id`, `source`, `trace_id`, `span_id`: String fields
            - `data`: Dictionary field (use dot notation, e.g., "data.category")
            - `tags`: List field (use "contains" operator)
            - `created_at`, `last_updated_at`: DateTime fields (ISO 8601 format)
            - `created_by`, `last_updated_by`: String fields

            Examples:
            - `tags contains "failed"` - Items with 'failed' tag
            - `data.category = "test"` - Items with specific data field value
            - `created_at >= "2024-01-01T00:00:00Z"` - Items created after date
    Nrp   rw   rq   r   rP   )rs   rt   rg   rk   r   r   ru   rv   rx   )rW   r   rz   r   r{   r|   r   r   r}   r~   r   rs   r   r   r   )r   r   ra   rb   rc   rd   re   rf   rg   rh   r.   ri   rj   rk   rl   r/   r9   rm   rT   rn   r:   r   ry   r   s                           r6   evaluate_optimization_trialr    s   Z '+;;<// +28T ! (;;O .+'!O **,F4'5O
 ))\\+'"#;7#;#;#=tTJ * 	J '!!/)'%A3 r8   itemsc                 J   t        |||      }|s+t        j                  d       t        j                  g       S t        j                         }t        j                         5  t        |       D 	
cg c]  \  }	}
t        j                  ddd|	 i|
! }}	}
t        j                  dd      }t        j                  ||||      }|j!                  t#        |      |||d	d	|t%        |      
      }d	d	d	       t        j                        S c c}
}	w # 1 sw Y   %xY w)u
  
    Lightweight evaluation function that evaluates a task on dataset items (as dictionaries)
    without requiring a Dataset object or creating an experiment.

    This function is useful for optimization scenarios where you need to evaluate many
    candidate solutions quickly using Opik's metric infrastructure. It creates traces for
    tracking but doesn't require experiment setup or dataset management.

    Args:
        items: List of dataset item contents (dictionaries with the data to evaluate).

        task: A callable object that takes dict with dataset item content
            as input and returns dict which will later be used for scoring.

        scoring_metrics: List of metrics to calculate during evaluation.
            Each metric's `score(...)` method will be called with arguments taken from
            the dataset item and task output.

        scoring_functions: List of scorer functions to be executed during evaluation.
            Each scorer function accepts predefined arguments:
                • dataset_item — a dictionary containing the dataset item content,
                • task_outputs — a dictionary containing the LLM task output.

        project_name: The name of the project for logging traces.

        verbose: Controls evaluation output logs and progress bars.
            0 - no outputs (default), 1 - enable outputs.

        scoring_key_mapping: A dictionary that allows you to rename keys present in either
            the dataset item or the task output to match the keys expected by scoring metrics.

        scoring_threads: Number of thread workers to run scoring metrics.

    Returns:
        EvaluationResultOnDictItems object containing test results and providing methods
        to aggregate scores, similar to the regular evaluation result.

    Example:
        ```python
        import opik
        from opik.evaluation.metrics import Equals

        items = [
            {"input": "What is 2+2?", "expected_output": "4"},
            {"input": "What is 3+3?", "expected_output": "6"},
        ]

        def my_task(item):
            # Your LLM call here
            question = item["input"]
            # ... call model ...
            return {"output": model_output}

        result = opik.evaluate_on_dict_items(
            items=items,
            task=my_task,
            scoring_metrics=[Equals()],
            scoring_key_mapping={"reference": "expected_output"},
        )

        # Access individual test results
        for test_result in result.test_results:
            print(f"Score: {test_result.score_results[0].value}")

        # Get aggregated statistics
        aggregated = result.aggregate_evaluation_scores()
        print(f"Mean equals score: {aggregated['equals_metric'].mean}")
        ```
    rw   z0No scoring metrics provided for items evaluation)rU   rP   
temp_item_r   r   r   Nr   r   )r   rA   rZ   r   EvaluationResultOnDictItemsr   r}   r   r   	enumerater   DatasetItemr   r   r   r   r   rE   r2   )r  ra   rb   rc   rf   rh   rl   r   ry   iitemr   r   r   rU   s                  r6   evaluate_on_dict_itemsr    s6   ` .+'!O IJ <<"MM**,F		B	B	D %U+
+4 $$A*QC(8ADA+ 	 
 *99A
 #33%#	
 )66}-+ 3 %+M* 7 	
 
E4 88! 3
 
E	Ds   $D3$DADDD"c                 l    | r-t        j                  | |      }|r|j                  |       n|}|r|S g S )N)rf   )r%   wrap_scorer_functionsrX   )rc   rb   rf   function_metricss       r6   r   r     sC    
 0FFL
 ""#34.O-?525r8   c                 :    | r| S |rt        j                  |      S y )N)r{   generate_unique_experiment_namerq   s     r6   r~   r~     s+     !AA"
 	
 r8   )NNNNNNr   Nr`   NNNNNr   NNN)Nr`   r   NNN)NNNNNNNr   Nr`   NNNr   NNN)NNNr   Nr`   )^loggingr   typingr   r   r   r   r   r   r	   r
   r   r   api_objects.promptr   api_objectsr   r   r   api_objects.datasetr   api_objects.experimentr   r{   r   r   api_objects.prompt.chatr   r   r   r    r   r   r   r   r   r   metricsr   r    modelsr!   r"   r#   scorersr$   r%   r&   r'   r(   r)   r*   $api_objects.dataset.evaluation_suiter+   suite_types	getLogger__name__rA   r   r@   DatasetDatasetVersionintr   r7   BaseDatasetSamplerr
  rI   
ExperimentrS   
TestResultScoreResultr_   rz   
BaseMetricScorerFunction
BasePromptr   r   Opikr   r   floatr   r   r   r   r  r  r  r  r   r~   r   r8   r6   <module>r.     s       - % - . B N : 6 2  / A A ;  J J  KK			8	$D  (+ $(GOOW%;%;;<(( tCy)( c]	("'-GOOW%;%;;<'-'- tCy)'- h99:	'-
 $C='- 8L,,-x}<='-T

%%

	

"&'>"?{--. 
,
"
"#@ ?CHL,0%)"&26 $/36:;?,0=ALP+/+/-a//1A1Q1QQa 	a
 d;#9#9:;a  _%C%C DEa %SMa c]a 3-a  S#X/a a a a [++,a d;1123a  ""78!a" tCy)#a$ h99:%a& 'a( #+40G+H"I)a* d3i(+a, $C=-a. ''/aP *.,0+/,0%)"&266:+/%)%)%)#W__W
W [%%&	W
 tCy)W $C=W %SMW c]W 3-W  S#X/W d;1123W d3i(W W W c]W  c]!W" c]#W$ )%Wt__ %%_ 7??G$:$::;	_
 _ +001_ 3-_ _ _ _ ""78_ tCy)_ h99:_ _ #''>"?_  $C=!_" ''#_X -1+/B*B* %%B* __	B*
 B* 3-B* B* B* c]B* tCy)B* $C=B* --u45B*P IM;?#'LPKK+001K  _%C%C DEK 	K
 K ""78K C=K #+40G+H"IK ''K\4###4#/3DcN/C4#tCH~S#X./4#x =A>BHL,0%)"&26 $/3,0=ALP+/+/+Z//1A1Q1QQZ 4S>"	Z
 E#z77789Z d;#9#9:;Z  _%C%C DEZ %SMZ c]Z 3-Z  S#X/Z Z Z Z [++,Z  tCy)!Z" h99:#Z$ %Z& #+40G+H"I'Z( d3i()Z* $C=+Z, ''-ZF ?CHL,0%)"&26 $/36:;?,0=ALP+/+//ff//1A1Q1QQf
 f d;#9#9:;f  _%C%C DEf %SMf c]f 3-f  S#X/f f f f [++,f  d;1123!f" ""78#f$ tCy)%f& h99:'f( )f* #+40G+H"I+f, d3i(-f. $C=/f0 ''1fX ?CHL"&;?xS#Xx
x d;#9#9:;x  _%C%C DE	x
 3-x x ""78x x 22xv6_%C%C DE6d;#9#9:;6 3-6 
+
 
 !	6"c]<DSMc]r8   