diff --git a/README.md b/README.md index 9082cb4..239ed62 100644 --- a/README.md +++ b/README.md @@ -240,6 +240,22 @@ evaluators: threshold: 0.7 ``` +Rubric-based metrics take their rubrics from the same config, as `{id, text}` mappings or plain strings, and score each rubric per invocation: + +```yaml +evaluators: + - name: rubric_based_final_response_quality_v1 + type: builtin + judge_model: gemini-2.5-flash + rubrics: + - id: corrected_value_wins + text: The response uses the operator's latest correction, never a superseded value. + - id: no_invented_cause + text: The response states no root cause the conversation did not establish. +``` + +Rubrics on a matched eval case (`rubrics` on the case, or on one of its invocations) are added to the metric's own at run time; see the [eval set format](docs/eval-set-format.md#rubrics). + Evaluators with a `requirements.txt` get automatic virtual environment management. You can also use `type: remote` for community evaluators from GitHub, or `type: openai_eval` to delegate grading to the [OpenAI Evals API](https://developers.openai.com/api/reference/resources/evals/methods/create) (requires `pip install "agentevals-cli[openai]"`). See the [Custom Evaluators guide](docs/custom-evaluators.md) for the full protocol reference, SDK helpers, and how to contribute evaluators. diff --git a/docs/custom-evaluators.md b/docs/custom-evaluators.md index 8076270..47b7739 100644 --- a/docs/custom-evaluators.md +++ b/docs/custom-evaluators.md @@ -89,6 +89,21 @@ agentevals run traces/my_trace.json \ Each evaluator entry in the `evaluators` list uses the following fields. The `type` field determines which other fields are valid. +### `type: builtin` (ADK metrics) + +| Field | Required | Default | Description | +|---|---|---|---| +| `name` | yes | | The metric, for example `tool_trajectory_avg_score` or `rubric_based_final_response_quality_v1` | +| `type` | yes | | `builtin` | +| `threshold` | no | `0.5` | Score at or above this value means PASSED | +| `judge_model` | no | ADK default | Judge model for LLM-backed metrics | +| `trajectory_match_type` | no | `EXACT` | `EXACT`, `IN_ORDER` or `ANY_ORDER`, for `tool_trajectory_avg_score` | +| `rubrics` | for `rubric_based_*` | | A list of `{id, text}` mappings, or plain strings (given ids `rubric_0`, `rubric_1`, ...). Ids must be unique. An entry may also carry `type`, an ADK rubric type; unset, the metric's own type is used | +| `credential_ref` | no | | Name of a credential the judge key is resolved from (API runs) | +| `judge_base_url` | no | | Base URL of an OpenAI-compatible judge endpoint | + +A `rubric_based_*` metric with no rubric from the config, the matched eval case or its invocations reports an error rather than scoring nothing. Its result carries `details.per_invocation[].rubric_scores` and `details.overall_rubric_scores`, one entry per rubric id with its score and the judge's rationale. + ### `type: code` (local scripts) | Field | Required | Default | Description | diff --git a/docs/eval-set-format.md b/docs/eval-set-format.md index 9a29eff..cff238d 100644 --- a/docs/eval-set-format.md +++ b/docs/eval-set-format.md @@ -176,6 +176,22 @@ An eval case can have multiple invocations to represent a conversation. Each inv | `rubrics` | list[Rubric] | no | Scoring rubrics for this specific invocation | | `creation_timestamp` | float | no | Unix timestamp | +### Rubrics + +`rubrics` on an eval case apply to every invocation of that case; `rubrics` on an invocation apply to that invocation alone. Each rubric is an ADK `Rubric`: `rubric_id`, `rubric_content.text_property`, and an optional `type`. When a `rubric_based_*` metric runs, the runner adds the matched case's rubrics and each invocation's rubrics to the rubrics the metric's config declares, the way ADK's own eval service does, and a rubric without a `type` is given the metric's (`FINAL_RESPONSE_QUALITY` or `TOOL_USE_QUALITY`), since ADK applies invocation-level rubrics by type. Rubric ids must be distinct across the metric, the case and the invocation; a clash is reported as an error. With no rubric from any of the three, the metric reports an error. + +The `rubric_based_*` metrics grade each invocation on its own, with that invocation's prompt, steps and final response, so a case-level rubric is judged once per turn. In a multi-turn case a rubric that describes the end state ("produces the completed draft") is therefore judged against the intermediate turns as well, where it fails, and lowers the case's mean. Write such a rubric on the final invocation instead, and keep case-level rubrics for properties every turn must have. + +```json +{ + "eval_id": "resume-correction", + "rubrics": [ + {"rubric_id": "corrected_value_wins", "rubric_content": {"text_property": "The response uses the operator's latest correction."}} + ], + "conversation": [ ... ] +} +``` + ### Content Uses the Google GenAI `Content` format: diff --git a/src/agentevals/api/routes.py b/src/agentevals/api/routes.py index a958990..5c7c3b1 100644 --- a/src/agentevals/api/routes.py +++ b/src/agentevals/api/routes.py @@ -242,7 +242,7 @@ async def list_metrics(): requires_gcp=m.metric_name in METRICS_NEEDING_GCP, requires_rubrics=m.metric_name in _METRICS_NEEDING_RUBRICS, description=m.description or "No description available", - working=m.metric_name not in _METRICS_NEEDING_RUBRICS, + working=True, ) ) diff --git a/src/agentevals/builtin_metrics.py b/src/agentevals/builtin_metrics.py index 014a669..c746ec6 100644 --- a/src/agentevals/builtin_metrics.py +++ b/src/agentevals/builtin_metrics.py @@ -160,6 +160,19 @@ def _enrich_app_details(invocations: list[Invocation]) -> list[Invocation]: return [inv.model_copy(update={"app_details": app_details}) for inv in invocations] +METRICS_NEEDING_RUBRICS = { + "rubric_based_final_response_quality_v1", + "rubric_based_tool_use_quality_v1", +} + +# The ADK rubric type each rubric metric applies to invocation-level rubrics; +# a rubric written for the metric without a type is given this one. +_METRIC_RUBRIC_TYPES = { + "rubric_based_final_response_quality_v1": "FINAL_RESPONSE_QUALITY", + "rubric_based_tool_use_quality_v1": "TOOL_USE_QUALITY", +} + + def rubric_strings_to_objects(rubric_texts: list[str]) -> list[Rubric]: """Convert plain-text rubric strings into ADK Rubric objects.""" return [ @@ -171,14 +184,88 @@ def rubric_strings_to_objects(rubric_texts: list[str]) -> list[Rubric]: ] +def to_rubric_objects(rubrics: list[str | Rubric] | None) -> list[Rubric]: + """Plain strings become positional rubrics; ADK ``Rubric`` objects pass through.""" + if not rubrics: + return [] + texts = [r for r in rubrics if isinstance(r, str)] + objects = [r for r in rubrics if isinstance(r, Rubric)] + return rubric_strings_to_objects(texts) + objects + + +def attach_case_rubrics( + metric_name: str, + actual_invocations: list[Invocation], + expected_invocations: list[Invocation] | None, + case_rubrics: list[Rubric] | None, +) -> list[Invocation]: + """Give the actual invocations the eval case's rubrics, as ADK's own eval + service does: case-level rubrics on every invocation, invocation-level + rubrics from the expected invocation at the same position. A rubric + without a type gets the metric's, since ADK filters invocation rubrics by + type. Duplicate ids within one invocation are an error.""" + metric_type = _METRIC_RUBRIC_TYPES.get(metric_name) + attached = [] + for index, actual in enumerate(actual_invocations): + extra: list[Rubric] = [] + if case_rubrics: + extra.extend(case_rubrics) + if expected_invocations and index < len(expected_invocations) and expected_invocations[index].rubrics: + extra.extend(expected_invocations[index].rubrics or []) + if not extra: + attached.append(actual) + continue + merged: dict[str, Rubric] = {r.rubric_id: r for r in actual.rubrics or []} + for rubric in extra: + if rubric.rubric_id in merged: + raise ValueError( + f"Rubric id '{rubric.rubric_id}' is defined more than once for invocation {index} " + f"(eval case, invocation and metric rubrics must have distinct ids)." + ) + merged[rubric.rubric_id] = ( + rubric if rubric.type or not metric_type else rubric.model_copy(update={"type": metric_type}) + ) + attached.append(actual.model_copy(update={"rubrics": list(merged.values())})) + return attached + + +def extract_rubric_details(eval_result: EvaluationResult) -> dict[str, Any]: + """Per-rubric scores, per invocation and overall, from a rubric metric.""" + + def _score(rubric_score: Any) -> dict[str, Any]: + return { + "rubric_id": rubric_score.rubric_id, + "score": rubric_score.score, + "rationale": rubric_score.rationale, + } + + per_invocation = [] + for per_inv_result in eval_result.per_invocation_results: + actual_inv = per_inv_result.actual_invocation + per_invocation.append( + { + "invocation_id": actual_inv.invocation_id if actual_inv else None, + "rubric_scores": [_score(r) for r in per_inv_result.rubric_scores or []], + } + ) + return { + "per_invocation": per_invocation, + "overall_rubric_scores": [_score(r) for r in eval_result.overall_rubric_scores or []], + } + + def build_eval_metric( metric_name: str, judge_model: str | None, threshold: float | None, - rubrics: list[str] | None = None, + rubrics: list[str | Rubric] | None = None, match_type: str | None = None, ) -> EvalMetric: - """Construct an ADK ``EvalMetric`` with the appropriate criterion.""" + """Construct an ADK ``EvalMetric`` with the appropriate criterion. + + ``rubrics`` feed the ``rubric_based_*`` metrics' criterion: plain strings + get positional ids, ADK ``Rubric`` objects are used as given. + """ effective_threshold = threshold if threshold is not None else 0.5 criterion: BaseCriterion | None = None @@ -211,7 +298,7 @@ def build_eval_metric( judge_opts = JudgeModelOptions() if judge_model: judge_opts.judge_model = judge_model - rubric_objects = rubric_strings_to_objects(rubrics) if rubrics else [] + rubric_objects = to_rubric_objects(rubrics) criterion = RubricsBasedCriterion( threshold=effective_threshold, judge_model_options=judge_opts, @@ -370,9 +457,17 @@ async def evaluate_builtin_metric( match_type: str | None = None, credential_ref: str | None = None, judge_base_url: str | None = None, + rubrics: list[str | Rubric] | None = None, + case_rubrics: list[Rubric] | None = None, ) -> dict[str, Any]: """Evaluate a single built-in ADK metric. + ``rubrics`` are the metric's own (from the eval config); ``case_rubrics`` + are the matched eval case's, applied to every invocation, and each + expected invocation's own rubrics apply to the actual invocation at the + same position. A rubric metric with no rubric from any of these is an + error, not an empty evaluation. + Returns a dict with keys: metric_name, score, eval_status, per_invocation_scores, error, details. """ @@ -387,8 +482,24 @@ async def evaluate_builtin_metric( ), ) + if metric_name in METRICS_NEEDING_RUBRICS: + try: + actual_invocations = attach_case_rubrics( + metric_name, actual_invocations, expected_invocations, case_rubrics + ) + except ValueError as exc: + return MetricResult(metric_name=metric_name, error=str(exc)) + if not rubrics and not any(inv.rubrics for inv in actual_invocations): + return MetricResult( + metric_name=metric_name, + error=( + f"Metric '{metric_name}' requires rubrics: set 'rubrics' on the evaluator in the eval " + "config, or 'rubrics' on the matched eval case or its invocations." + ), + ) + try: - eval_metric = build_eval_metric(metric_name, judge_model, threshold, match_type=match_type) + eval_metric = build_eval_metric(metric_name, judge_model, threshold, rubrics=rubrics, match_type=match_type) evaluator: Evaluator = get_evaluator(eval_metric) if credential_ref: @@ -425,6 +536,8 @@ async def evaluate_builtin_metric( details = None if metric_name == "tool_trajectory_avg_score": details = extract_trajectory_details(eval_result) + elif metric_name in METRICS_NEEDING_RUBRICS: + details = extract_rubric_details(eval_result) return MetricResult( metric_name=metric_name, diff --git a/src/agentevals/config.py b/src/agentevals/config.py index cf3c194..7f38faa 100644 --- a/src/agentevals/config.py +++ b/src/agentevals/config.py @@ -4,11 +4,14 @@ from collections import Counter from pathlib import Path -from typing import Annotated, Any, Literal +from typing import TYPE_CHECKING, Annotated, Any, Literal from pydantic import BaseModel, ConfigDict, Field, field_validator from pydantic.alias_generators import to_camel +if TYPE_CHECKING: + from google.adk.evaluation.eval_rubrics import Rubric + def _normalize_trajectory_match_type(v: str | None) -> str | None: valid = {"EXACT", "IN_ORDER", "ANY_ORDER"} @@ -17,6 +20,53 @@ def _normalize_trajectory_match_type(v: str | None) -> str | None: return v.upper() if v is not None else v +class RubricDef(BaseModel): + """One rubric for a ``rubric_based_*`` metric, as written in an eval config. + + A rubric is a testable statement about the response or the tool use, with + an id that names it in the per-rubric scores. In YAML an entry is either a + mapping ``{id, text}`` or a plain string, which gets a positional id + (``rubric_0``, ``rubric_1``, ...). + """ + + model_config = ConfigDict(alias_generator=to_camel, populate_by_name=True, extra="forbid") + + id: str = Field(min_length=1, description="Unique id of the rubric within the metric.") + text: str = Field(min_length=1, description="The testable statement the judge assesses.") + type: str | None = Field( + default=None, + description=( + "Optional ADK rubric type (for example FINAL_RESPONSE_QUALITY). Left unset, the metric's own type is used." + ), + ) + + def to_adk(self, default_type: str | None = None) -> Rubric: + from google.adk.evaluation.eval_rubrics import Rubric, RubricContent + + return Rubric( + rubric_id=self.id, + rubric_content=RubricContent(text_property=self.text), + type=self.type or default_type, + ) + + +def _normalize_rubrics(value: Any) -> Any: + """Accept plain strings beside ``{id, text}`` mappings; reject duplicate ids.""" + if value is None: + return None + if not isinstance(value, list): + raise ValueError("'rubrics' must be a list of strings or {id, text} mappings") + normalized: list[Any] = [] + for index, entry in enumerate(value): + if isinstance(entry, str): + if not entry.strip(): + raise ValueError(f"Rubric {index} is empty") + normalized.append({"id": f"rubric_{index}", "text": entry}) + else: + normalized.append(entry) + return normalized + + class BuiltinMetricDef(BaseModel): """A built-in ADK metric, optionally with threshold/judge overrides.""" @@ -27,6 +77,13 @@ class BuiltinMetricDef(BaseModel): threshold: float | None = Field(default=None, ge=0, le=1) judge_model: str | None = None trajectory_match_type: str | None = None + rubrics: list[RubricDef] | None = Field( + default=None, + description=( + "Rubrics for rubric_based_* metrics: a list of {id, text} mappings or plain strings. " + "Rubrics on the matched eval case or its invocations are added at run time." + ), + ) credential_ref: str | None = Field( default=None, description="Logical name of a RunSpec.credential_refs entry whose resolved value is the judge API key.", @@ -41,6 +98,21 @@ class BuiltinMetricDef(BaseModel): def _validate_trajectory_match_type(cls, v: str | None) -> str | None: return _normalize_trajectory_match_type(v) + @field_validator("rubrics", mode="before") + @classmethod + def _normalize_rubrics(cls, v: Any) -> Any: + return _normalize_rubrics(v) + + @field_validator("rubrics") + @classmethod + def _unique_rubric_ids(cls, v: list[RubricDef] | None) -> list[RubricDef] | None: + if v is None: + return v + duplicates = sorted(rubric_id for rubric_id, count in Counter(r.id for r in v).items() if count > 1) + if duplicates: + raise ValueError("Rubric ids must be unique within a metric. Duplicate ids: " + ", ".join(duplicates)) + return v + class BaseEvaluatorDef(BaseModel): """Shared fields for all executable evaluator definitions.""" diff --git a/src/agentevals/custom_evaluators.py b/src/agentevals/custom_evaluators.py index 0e2f71d..356197e 100644 --- a/src/agentevals/custom_evaluators.py +++ b/src/agentevals/custom_evaluators.py @@ -432,6 +432,7 @@ async def evaluate_custom_evaluator( actual_invocations: list[Invocation], expected_invocations: list[Invocation] | None, performance_metrics: dict[str, Any] | None = None, + eval_case=None, ): """Evaluate a single custom evaluator and return a ``MetricResult``. @@ -446,6 +447,9 @@ async def evaluate_custom_evaluator( from .runner import MetricResult if isinstance(evaluator_def, BuiltinMetricDef): + from .builtin_metrics import _METRIC_RUBRIC_TYPES + + rubric_type = _METRIC_RUBRIC_TYPES.get(evaluator_def.name) return await evaluate_builtin_metric( metric_name=evaluator_def.name, actual_invocations=actual_invocations, @@ -455,6 +459,8 @@ async def evaluate_custom_evaluator( match_type=evaluator_def.trajectory_match_type, credential_ref=evaluator_def.credential_ref, judge_base_url=evaluator_def.judge_base_url, + rubrics=[r.to_adk(rubric_type) for r in evaluator_def.rubrics] if evaluator_def.rubrics else None, + case_rubrics=list(eval_case.rubrics) if eval_case is not None and eval_case.rubrics else None, ) if isinstance(evaluator_def, OpenAIEvalDef): diff --git a/src/agentevals/runner.py b/src/agentevals/runner.py index eae4066..7321219 100644 --- a/src/agentevals/runner.py +++ b/src/agentevals/runner.py @@ -9,7 +9,7 @@ from collections.abc import Awaitable, Callable from typing import Any -from google.adk.evaluation.eval_case import Invocation +from google.adk.evaluation.eval_case import EvalCase, Invocation from google.adk.evaluation.eval_set import EvalSet from pydantic import BaseModel, ConfigDict, Field from pydantic.alias_generators import to_camel @@ -279,8 +279,10 @@ async def _evaluate_trace( actual_invocations = conv_result.invocations expected_invocations: list[Invocation] | None = None + eval_case = None if eval_set: - expected_invocations = _find_expected_invocations(actual_invocations, eval_set) + eval_case = _find_eval_case(actual_invocations, eval_set) + expected_invocations = eval_case.conversation if eval_case and eval_case.conversation else None async def _append_result(result: MetricResult) -> MetricResult: trace_result.metric_results.append(result) @@ -300,6 +302,7 @@ async def _eval_with_semaphore(evaluator_def: EvaluatorDef) -> MetricResult: actual_invocations=actual_invocations, expected_invocations=expected_invocations, performance_metrics=performance_metrics, + eval_case=eval_case, ) result.duration_ms = (time.monotonic() - t0) * 1000 return await _append_result(result) @@ -311,39 +314,43 @@ async def _eval_with_semaphore(evaluator_def: EvaluatorDef) -> MetricResult: return trace_result -def _find_expected_invocations( +def _find_eval_case( actual_invocations: list[Invocation], eval_set: EvalSet, -) -> list[Invocation] | None: +) -> EvalCase | None: """Match actual invocations to an eval case. Uses the sole eval case if there's only one, otherwise matches by user content text.""" if not eval_set.eval_cases: return None if len(eval_set.eval_cases) == 1: - case = eval_set.eval_cases[0] - if case.conversation: - return case.conversation - return None + return eval_set.eval_cases[0] actual_user_text = _get_user_text(actual_invocations[0]) if actual_invocations else None if not actual_user_text: - case = eval_set.eval_cases[0] - return case.conversation if case.conversation else None + return eval_set.eval_cases[0] for case in eval_set.eval_cases: if not case.conversation: continue expected_user_text = _get_user_text(case.conversation[0]) if expected_user_text and _text_matches(actual_user_text, expected_user_text): - return case.conversation + return case logger.warning( "No matching eval case found for user text: '%s'. Using first eval case.", actual_user_text[:100], ) - case = eval_set.eval_cases[0] - return case.conversation if case.conversation else None + return eval_set.eval_cases[0] + + +def _find_expected_invocations( + actual_invocations: list[Invocation], + eval_set: EvalSet, +) -> list[Invocation] | None: + """The matched eval case's conversation, or None.""" + case = _find_eval_case(actual_invocations, eval_set) + return case.conversation if case and case.conversation else None def _get_user_text(invocation: Invocation) -> str | None: diff --git a/tests/test_rubrics.py b/tests/test_rubrics.py new file mode 100644 index 0000000..c48dde0 --- /dev/null +++ b/tests/test_rubrics.py @@ -0,0 +1,218 @@ +"""The public path for rubric_based_* metrics: rubrics in the eval config, +on the matched eval case and on its invocations, resolved the way ADK's own +eval service resolves them, with per-rubric scores in the result.""" + +from __future__ import annotations + +import asyncio +import textwrap + +import pytest +from google.adk.evaluation.eval_case import EvalCase, Invocation +from google.adk.evaluation.eval_rubrics import Rubric, RubricContent, RubricScore +from google.adk.evaluation.eval_set import EvalSet +from google.adk.evaluation.evaluator import EvalStatus, EvaluationResult, PerInvocationResult +from google.genai import types as genai_types + +from agentevals import builtin_metrics +from agentevals.builtin_metrics import ( + attach_case_rubrics, + build_eval_metric, + evaluate_builtin_metric, + extract_rubric_details, + to_rubric_objects, +) +from agentevals.config import BuiltinMetricDef, RubricDef, apply_builtin_overrides +from agentevals.custom_evaluators import evaluate_custom_evaluator +from agentevals.eval_config_loader import load_eval_config +from agentevals.runner import _find_eval_case, _find_expected_invocations + +METRIC = "rubric_based_final_response_quality_v1" + + +def _rubric(rubric_id: str, text: str = "a statement", rubric_type: str | None = None) -> Rubric: + return Rubric(rubric_id=rubric_id, rubric_content=RubricContent(text_property=text), type=rubric_type) + + +def _invocation(text: str = "hi", rubrics: list[Rubric] | None = None) -> Invocation: + return Invocation( + invocation_id=text, + user_content=genai_types.Content(role="user", parts=[genai_types.Part(text=text)]), + final_response=genai_types.Content(role="model", parts=[genai_types.Part(text="ok")]), + rubrics=rubrics, + ) + + +class TestConfig: + def test_rubrics_accept_mappings_and_strings(self): + metric = BuiltinMetricDef.model_validate( + { + "name": METRIC, + "type": "builtin", + "rubrics": [{"id": "grounded", "text": "No invented cause."}, "Uses the latest correction."], + } + ) + assert [(r.id, r.text) for r in metric.rubrics] == [ + ("grounded", "No invented cause."), + ("rubric_1", "Uses the latest correction."), + ] + assert metric.rubrics[0].to_adk("FINAL_RESPONSE_QUALITY").type == "FINAL_RESPONSE_QUALITY" + assert RubricDef(id="x", text="t", type="TOOL_USE_QUALITY").to_adk("FINAL_RESPONSE_QUALITY").type == ( + "TOOL_USE_QUALITY" + ) + + @pytest.mark.parametrize( + ("rubrics", "message"), + [ + ([{"id": "a", "text": "one"}, {"id": "a", "text": "two"}], "unique"), + ([" "], "empty"), + ("not a list", "must be a list"), + ([{"id": "a"}], "text"), + ([{"id": "a", "text": "one", "weight": 2}], "weight"), + ], + ) + def test_malformed_rubrics_are_refused(self, rubrics, message): + with pytest.raises(ValueError, match=message): + BuiltinMetricDef.model_validate({"name": METRIC, "type": "builtin", "rubrics": rubrics}) + + def test_rubrics_survive_run_level_overrides(self): + metric = BuiltinMetricDef(name=METRIC, rubrics=[RubricDef(id="a", text="t")]) + (updated,) = apply_builtin_overrides([metric], judge_model="gemini-2.5-flash", threshold=0.9) + assert updated.judge_model == "gemini-2.5-flash" + assert [r.id for r in updated.rubrics] == ["a"] + + def test_loader_reads_rubrics_from_yaml(self, tmp_path): + config_file = tmp_path / "eval_config.yaml" + config_file.write_text( + textwrap.dedent( + f""" + evaluators: + - name: {METRIC} + type: builtin + judge_model: gemini-2.5-flash + rubrics: + - id: corrected_value_wins + text: The response uses the operator's latest correction. + - The response states no root cause the conversation did not establish. + """ + ), + encoding="utf-8", + ) + (metric,) = load_eval_config(config_file).evaluators + assert isinstance(metric, BuiltinMetricDef) + assert [r.id for r in metric.rubrics] == ["corrected_value_wins", "rubric_1"] + + +class TestCriterion: + def test_strings_and_objects_both_reach_the_criterion(self): + metric = build_eval_metric(METRIC, "gemini-2.5-flash", 0.5, rubrics=["plain", _rubric("named", "typed")]) + rubrics = metric.criterion.rubrics + assert [(r.rubric_id, r.rubric_content.text_property) for r in rubrics] == [ + ("rubric_0", "plain"), + ("named", "typed"), + ] + assert to_rubric_objects(None) == [] + + +class TestCaseRubrics: + def test_case_rubrics_reach_every_invocation_and_get_the_metrics_type(self): + actual = [_invocation("a"), _invocation("b")] + attached = attach_case_rubrics(METRIC, actual, None, [_rubric("case")]) + for inv in attached: + (rubric,) = inv.rubrics + assert (rubric.rubric_id, rubric.type) == ("case", "FINAL_RESPONSE_QUALITY") + # The originals are untouched. + assert all(inv.rubrics is None for inv in actual) + + def test_invocation_rubrics_apply_at_their_position_and_keep_their_type(self): + actual = [_invocation("a"), _invocation("b")] + expected = [_invocation("a", [_rubric("first", rubric_type="CUSTOM")]), _invocation("b")] + attached = attach_case_rubrics(METRIC, actual, expected, None) + assert [(r.rubric_id, r.type) for r in attached[0].rubrics] == [("first", "CUSTOM")] + assert attached[1].rubrics is None + + def test_a_duplicate_id_between_case_and_invocation_is_an_error(self): + actual = [_invocation("a")] + expected = [_invocation("a", [_rubric("dup")])] + with pytest.raises(ValueError, match="defined more than once"): + attach_case_rubrics(METRIC, actual, expected, [_rubric("dup")]) + + def test_a_rubric_metric_without_any_rubric_is_an_error_before_any_judge_call(self): + result = asyncio.run( + evaluate_builtin_metric( + METRIC, + [_invocation("a")], + None, + judge_model="gemini-2.5-flash", + threshold=0.5, + ) + ) + assert result.score is None + assert "requires rubrics" in result.error + + def test_the_dispatch_passes_config_and_case_rubrics(self, monkeypatch): + seen = {} + + async def fake(**kwargs): + seen.update(kwargs) + from agentevals.runner import MetricResult + + return MetricResult(metric_name=kwargs["metric_name"], score=1.0, eval_status="PASSED") + + monkeypatch.setattr(builtin_metrics, "evaluate_builtin_metric", fake) + case = EvalCase(eval_id="c", conversation=[_invocation("a")], rubrics=[_rubric("case")]) + metric = BuiltinMetricDef(name=METRIC, rubrics=[RubricDef(id="cfg", text="t")]) + asyncio.run(evaluate_custom_evaluator(metric, [_invocation("a")], case.conversation, eval_case=case)) + assert [r.rubric_id for r in seen["rubrics"]] == ["cfg"] + assert seen["rubrics"][0].type == "FINAL_RESPONSE_QUALITY" + assert [r.rubric_id for r in seen["case_rubrics"]] == ["case"] + + +class TestMatching: + def test_the_matched_case_carries_its_rubrics(self): + eval_set = EvalSet( + eval_set_id="s", + eval_cases=[ + EvalCase(eval_id="one", conversation=[_invocation("first question")]), + EvalCase(eval_id="two", conversation=[_invocation("second question")], rubrics=[_rubric("r")]), + ], + ) + case = _find_eval_case([_invocation("second question")], eval_set) + assert case.eval_id == "two" and [r.rubric_id for r in case.rubrics] == ["r"] + assert _find_expected_invocations([_invocation("second question")], eval_set) is case.conversation + assert _find_eval_case([], EvalSet(eval_set_id="e", eval_cases=[])) is None + + +class TestDetails: + def test_per_rubric_scores_are_reported_per_invocation_and_overall(self): + actual = _invocation("a") + result = EvaluationResult( + overall_score=0.5, + overall_eval_status=EvalStatus.FAILED, + per_invocation_results=[ + PerInvocationResult( + actual_invocation=actual, + score=0.5, + eval_status=EvalStatus.FAILED, + rubric_scores=[ + RubricScore(rubric_id="grounded", score=1.0, rationale="no cause was invented"), + RubricScore(rubric_id="corrected", score=0.0, rationale="the stale value is current"), + ], + ) + ], + overall_rubric_scores=[ + RubricScore(rubric_id="grounded", score=1.0), + RubricScore(rubric_id="corrected", score=0.0), + ], + ) + details = extract_rubric_details(result) + assert details["per_invocation"] == [ + { + "invocation_id": "a", + "rubric_scores": [ + {"rubric_id": "grounded", "score": 1.0, "rationale": "no cause was invented"}, + {"rubric_id": "corrected", "score": 0.0, "rationale": "the stale value is current"}, + ], + } + ] + assert [r["rubric_id"] for r in details["overall_rubric_scores"]] == ["grounded", "corrected"]