Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
112 changes: 105 additions & 7 deletions src/google/adk/evaluation/vertex_ai_eval_facade.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@
from .eval_case import Invocation
from .eval_case import InvocationEvent
from .eval_case import InvocationEvents
from .eval_rubrics import RubricScore
from .evaluator import _validate_invocation_lengths
from .evaluator import EvalStatus
from .evaluator import EvaluationResult
Expand All @@ -40,6 +41,14 @@

logger = logging.getLogger("google_adk." + __name__)

# `rubric_id` of the RubricScore that carries the judge's overall explanation
# of a metric, as opposed to its verdict on one specific rubric.
_EXPLANATION_RUBRIC_ID = "explanation"

# `rubric_id` of the RubricScore that carries the error Vertex reported when it
# could not score a metric.
_ERROR_RUBRIC_ID = "error"

_ERROR_MESSAGE_SUFFIX = """
You should specify both project id and location. This metric uses Vertex Gen AI
Eval SDK, and it requires google cloud credentials.
Expand Down Expand Up @@ -132,6 +141,85 @@ def _get_score(self, eval_result: object) -> Optional[float]:

return None

def _get_rubric_scores(
self, eval_result: object
) -> Optional[list[RubricScore]]:
"""Returns the judge's rubric verdicts and explanation as RubricScores.

Every call to `_perform_eval` evaluates exactly one metric, so all the
metric results found in `eval_result` belong to `self._metric_name`.

Each rubric verdict (returned by adaptive-rubric metrics such as
`multi_turn_task_success_v1`) becomes one RubricScore. The metric's
explanation, and its error message when Vertex could not score it, are
appended as extra RubricScores that only carry a rationale.

Returns None when Vertex returned none of these details.
"""
rubric_scores: list[RubricScore] = []
for metric_result in _VertexAiEvalFacade._get_metric_results(eval_result):
rubric_verdicts = getattr(metric_result, "rubric_verdicts", None) or []
for index, rubric_verdict in enumerate(rubric_verdicts):
rubric_scores.append(
_VertexAiEvalFacade._map_rubric_verdict_to_rubric_score(
index, rubric_verdict
)
)

explanation = getattr(metric_result, "explanation", None)
if explanation:
rubric_scores.append(
RubricScore(
rubric_id=_EXPLANATION_RUBRIC_ID, rationale=str(explanation)
)
)

error_message = getattr(metric_result, "error_message", None)
if error_message:
rubric_scores.append(
RubricScore(
rubric_id=_ERROR_RUBRIC_ID, rationale=str(error_message)
)
)

return rubric_scores or None

@staticmethod
def _get_metric_results(eval_result: object) -> list[object]:
"""Returns all the per-metric results contained in a Vertex eval result."""
metric_results: list[object] = []
eval_case_results = getattr(eval_result, "eval_case_results", None) or []
for eval_case_result in eval_case_results:
candidate_results = (
getattr(eval_case_result, "response_candidate_results", None) or []
)
for candidate_result in candidate_results:
results_by_metric = getattr(candidate_result, "metric_results", None)
if isinstance(results_by_metric, dict):
metric_results.extend(results_by_metric.values())

return metric_results

@staticmethod
def _map_rubric_verdict_to_rubric_score(
index: int, rubric_verdict: object
) -> RubricScore:
rubric = getattr(rubric_verdict, "evaluated_rubric", None)
rubric_content = getattr(rubric, "content", None)
rubric_property = getattr(rubric_content, "property", None)
description = getattr(rubric_property, "description", None)
rubric_id = (
description or getattr(rubric, "rubric_id", None) or f"rubric_{index}"
)
reasoning = getattr(rubric_verdict, "reasoning", None)

return RubricScore(
rubric_id=str(rubric_id),
rationale=str(reasoning) if reasoning else None,
# Vertex omits `verdict` when a rubric is not met.
score=1.0 if getattr(rubric_verdict, "verdict", None) else 0.0,
)

def _get_eval_status(self, score: Optional[float]) -> EvalStatus:
if score is not None:
return (
Expand Down Expand Up @@ -197,12 +285,16 @@ def evaluate_invocations(
dataset=dataset, metrics=[self._metric_name]
)
score = self._get_score(eval_case_result)
# Each invocation is judged separately, so its explanation and verdicts
# stay on that invocation and are not aggregated into
# `overall_rubric_scores`.
per_invocation_results.append(
PerInvocationResult(
actual_invocation=actual,
expected_invocation=expected,
score=score,
eval_status=self._get_eval_status(score),
rubric_scores=self._get_rubric_scores(eval_case_result),
)
)

Expand Down Expand Up @@ -273,23 +365,29 @@ def evaluate_invocations(
)

score = self._get_score(eval_case_result)
rubric_scores = self._get_rubric_scores(eval_case_result)
per_invocation_results.append(
PerInvocationResult(
actual_invocation=actual_invocations[-1],
expected_invocation=expected_invocations[-1],
score=score,
eval_status=self._get_eval_status(score),
rubric_scores=rubric_scores,
)
)

if score is not None:
return EvaluationResult(
overall_score=score,
overall_eval_status=self._get_eval_status(score),
per_invocation_results=per_invocation_results,
)
if score is None and rubric_scores is None:
return EvaluationResult()

return EvaluationResult()
# The whole conversation is judged once, so the verdicts attached to the
# last turn are also the overall verdicts. They are kept even without a
# score, since they then carry the reason Vertex could not score it.
return EvaluationResult(
overall_score=score,
overall_eval_status=self._get_eval_status(score),
per_invocation_results=per_invocation_results,
overall_rubric_scores=rubric_scores,
)

@staticmethod
def _get_agent_data(
Expand Down
200 changes: 200 additions & 0 deletions tests/unittests/evaluation/test_vertex_ai_eval_facade.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@
from google.adk.evaluation.eval_case import Invocation
from google.adk.evaluation.eval_case import InvocationEvent
from google.adk.evaluation.eval_case import InvocationEvents
from google.adk.evaluation.eval_rubrics import RubricScore
from google.adk.evaluation.evaluator import EvalStatus
from google.adk.evaluation.vertex_ai_eval_facade import _MultiTurnVertexiAiEvalFacade
from google.adk.evaluation.vertex_ai_eval_facade import _SingleTurnVertexAiEvalFacade
Expand Down Expand Up @@ -259,6 +260,129 @@ def test_evaluate_invocations_metric_multiple_invocations(self, mocker):
assert evaluation_result.overall_eval_status == EvalStatus.FAILED
assert mock_perform_eval.call_count == num_invocations

def test_evaluate_invocations_reports_explanation_per_invocation(
self, mocker
):
"""The judge's explanation surfaces on the invocation it explains."""
mocker.patch("google.adk.dependencies.vertexai.vertexai.Client")
mock_perform_eval = mocker.patch(
"google.adk.evaluation.vertex_ai_eval_facade._VertexAiEvalFacade._perform_eval"
)
mock_perform_eval.return_value = _make_eval_result(
score=0.0,
metric_result=vertexai_types.EvalCaseMetricResult(
metric_name="safety_v1",
score=0.0,
explanation="Violated policies: PII & Demographic Data",
),
)
evaluator = _SingleTurnVertexAiEvalFacade(
threshold=0.8, metric_name=vertexai_types.PrebuiltMetric.SAFETY
)

evaluation_result = evaluator.evaluate_invocations([_make_invocation()])

assert evaluation_result.per_invocation_results[0].rubric_scores == [
RubricScore(
rubric_id="explanation",
rationale="Violated policies: PII & Demographic Data",
),
]
# Each invocation is judged separately, so nothing is aggregated.
assert evaluation_result.overall_rubric_scores is None

def test_evaluate_invocations_reports_error_when_not_scored(self, mocker):
"""The error Vertex returns for an unscored invocation is kept."""
mocker.patch("google.adk.dependencies.vertexai.vertexai.Client")
mock_perform_eval = mocker.patch(
"google.adk.evaluation.vertex_ai_eval_facade._VertexAiEvalFacade._perform_eval"
)
mock_perform_eval.return_value = _make_eval_result(
score=None,
metric_result=vertexai_types.EvalCaseMetricResult(
metric_name="response_evaluation_score",
error_message="400 INVALID_ARGUMENT",
),
)
evaluator = _SingleTurnVertexAiEvalFacade(
threshold=0.8, metric_name=vertexai_types.PrebuiltMetric.COHERENCE
)

evaluation_result = evaluator.evaluate_invocations([_make_invocation()])

per_invocation_result = evaluation_result.per_invocation_results[0]
assert per_invocation_result.eval_status == EvalStatus.NOT_EVALUATED
assert per_invocation_result.rubric_scores == [
RubricScore(rubric_id="error", rationale="400 INVALID_ARGUMENT")
]

def test_evaluate_invocations_without_details_has_no_rubric_scores(
self, mocker
):
"""A result carrying only a score yields no rubric scores."""
mocker.patch("google.adk.dependencies.vertexai.vertexai.Client")
mock_perform_eval = mocker.patch(
"google.adk.evaluation.vertex_ai_eval_facade._VertexAiEvalFacade._perform_eval"
)
mock_perform_eval.return_value = vertexai_types.EvaluationResult(
summary_metrics=[vertexai_types.AggregatedMetricResult(mean_score=0.9)],
eval_case_results=[],
)
evaluator = _SingleTurnVertexAiEvalFacade(
threshold=0.8, metric_name=vertexai_types.PrebuiltMetric.COHERENCE
)

evaluation_result = evaluator.evaluate_invocations([_make_invocation()])

assert evaluation_result.per_invocation_results[0].rubric_scores is None


def _make_invocation(invocation_id: str = "inv1") -> Invocation:
return Invocation(
invocation_id=invocation_id,
user_content=genai_types.Content(parts=[genai_types.Part(text="query")]),
final_response=genai_types.Content(
parts=[genai_types.Part(text="response")]
),
)


def _make_rubric_verdict(
description: str, verdict: bool | None, reasoning: str
) -> vertexai_types.RubricVerdict:
return vertexai_types.RubricVerdict(
evaluated_rubric=vertexai_types.evals.Rubric(
rubric_id=f"id-{description}",
content=vertexai_types.evals.RubricContent(
property=vertexai_types.evals.RubricContentProperty(
description=description
)
),
),
verdict=verdict,
reasoning=reasoning,
)


def _make_eval_result(
score: float | None,
metric_result: vertexai_types.EvalCaseMetricResult,
) -> vertexai_types.EvaluationResult:
return vertexai_types.EvaluationResult(
summary_metrics=[vertexai_types.AggregatedMetricResult(mean_score=score)],
eval_case_results=[
vertexai_types.EvalCaseResult(
eval_case_index=0,
response_candidate_results=[
vertexai_types.ResponseCandidateResult(
response_index=0,
metric_results={metric_result.metric_name: metric_result},
)
],
)
],
)


class TestVertexAiEvalFacade:
"""A class to help organize "patch" that are applicable to all tests."""
Expand Down Expand Up @@ -608,3 +732,79 @@ def test_evaluate_invocations_multi_turn_metric_passed(self, mocker):
assert agent_data.turns[0].turn_id == "inv1"
assert agent_data.turns[1].turn_id == "inv2"
assert len(agent_data.turns[1].events) == 3 # user, intermediate, agent

@pytest.mark.parametrize(
"metric",
[
vertexai_types.RubricMetric.MULTI_TURN_TASK_SUCCESS,
vertexai_types.RubricMetric.MULTI_TURN_TOOL_USE_QUALITY,
vertexai_types.RubricMetric.MULTI_TURN_TRAJECTORY_QUALITY,
],
)
def test_evaluate_invocations_reports_conversation_verdicts(
self, mocker, metric
):
"""Verdicts on the conversation surface on the last turn and overall."""
mocker.patch("google.adk.dependencies.vertexai.vertexai.Client")
mock_perform_eval = mocker.patch(
"google.adk.evaluation.vertex_ai_eval_facade._VertexAiEvalFacade._perform_eval"
)
mock_perform_eval.return_value = _make_eval_result(
score=0.5,
metric_result=vertexai_types.EvalCaseMetricResult(
metric_name=metric.name,
score=0.5,
rubric_verdicts=[
_make_rubric_verdict("Books the flight.", True, "Booked."),
# Vertex omits `verdict` when a rubric is not met.
_make_rubric_verdict("Confirms the date.", None, "Skipped."),
],
),
)
evaluator = _MultiTurnVertexiAiEvalFacade(threshold=0.8, metric_name=metric)

evaluation_result = evaluator.evaluate_invocations(
[_make_invocation("inv1"), _make_invocation("inv2")]
)

expected_rubric_scores = [
RubricScore(
rubric_id="Books the flight.", rationale="Booked.", score=1.0
),
RubricScore(
rubric_id="Confirms the date.", rationale="Skipped.", score=0.0
),
]
assert evaluation_result.overall_eval_status == EvalStatus.FAILED
assert evaluation_result.overall_rubric_scores == expected_rubric_scores
assert evaluation_result.per_invocation_results[0].rubric_scores is None
assert (
evaluation_result.per_invocation_results[1].rubric_scores
== expected_rubric_scores
)

def test_evaluate_invocations_reports_error_when_not_scored(self, mocker):
"""An unscored conversation keeps the error Vertex returned."""
mocker.patch("google.adk.dependencies.vertexai.vertexai.Client")
mock_perform_eval = mocker.patch(
"google.adk.evaluation.vertex_ai_eval_facade._VertexAiEvalFacade._perform_eval"
)
mock_perform_eval.return_value = _make_eval_result(
score=None,
metric_result=vertexai_types.EvalCaseMetricResult(
metric_name="multi_turn_task_success_v1",
error_message="Quota exceeded.",
),
)
evaluator = _MultiTurnVertexiAiEvalFacade(
threshold=0.8,
metric_name=vertexai_types.RubricMetric.MULTI_TURN_TASK_SUCCESS,
)

evaluation_result = evaluator.evaluate_invocations([_make_invocation()])

assert evaluation_result.overall_score is None
assert evaluation_result.overall_eval_status == EvalStatus.NOT_EVALUATED
assert evaluation_result.overall_rubric_scores == [
RubricScore(rubric_id="error", rationale="Quota exceeded.")
]
Loading