mirror of
https://github.com/encounter/adk-python.git
synced 2026-07-09 18:19:28 -07:00
feat: Introduce User Personas to the ADK evaluation framework
PiperOrigin-RevId: 871366815
This commit is contained in:
committed by
Copybara-Service
parent
976a238544
commit
6a808c60b3
@@ -108,6 +108,7 @@ community = [
|
||||
|
||||
eval = [
|
||||
# go/keep-sorted start
|
||||
"Jinja2>=3.1.4,<4.0.0", # For eval template rendering
|
||||
"google-cloud-aiplatform[evaluation]>=1.100.0",
|
||||
"pandas>=2.2.3",
|
||||
"rouge-score>=0.1.2",
|
||||
|
||||
@@ -14,9 +14,14 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic import field_validator
|
||||
|
||||
from .common import EvalBaseModel
|
||||
from .simulation.pre_built_personas import DEFAULT_USER_PERSONA_REGISTRY
|
||||
from .simulation.user_simulator_personas import UserPersona
|
||||
|
||||
|
||||
class ConversationScenario(EvalBaseModel):
|
||||
@@ -48,6 +53,18 @@ class ConversationScenario(EvalBaseModel):
|
||||
your overall goal is complete.
|
||||
"""
|
||||
|
||||
user_persona: Optional[UserPersona] = Field(default=None)
|
||||
"""User persona that the user simulator should adopt. If a persona id is specified instead, we will try to use one of our default personas."""
|
||||
|
||||
@field_validator("user_persona", mode="before")
|
||||
@classmethod
|
||||
def validate_user_persona(
|
||||
cls, value: Optional[UserPersona | str]
|
||||
) -> Optional[UserPersona]:
|
||||
if value is not None and isinstance(value, str):
|
||||
return DEFAULT_USER_PERSONA_REGISTRY.get_persona(value)
|
||||
return value
|
||||
|
||||
|
||||
class ConversationScenarios(EvalBaseModel):
|
||||
"""A simple container for the list of ConversationScenario.
|
||||
|
||||
@@ -31,6 +31,8 @@ from ...utils.feature_decorator import experimental
|
||||
from .._retry_options_utils import add_default_retry_options_if_not_present
|
||||
from ..conversation_scenarios import ConversationScenario
|
||||
from ..evaluator import Evaluator
|
||||
from .llm_backed_user_simulator_prompts import get_llm_backed_user_simulator_prompt
|
||||
from .llm_backed_user_simulator_prompts import is_valid_user_simulator_template
|
||||
from .user_simulator import BaseUserSimulatorConfig
|
||||
from .user_simulator import NextUserMessage
|
||||
from .user_simulator import Status
|
||||
@@ -41,63 +43,6 @@ logger = logging.getLogger("google_adk." + __name__)
|
||||
_AUTHOR_USER = "user"
|
||||
_STOP_SIGNAL = "</finished>"
|
||||
|
||||
_DEFAULT_USER_AGENT_INSTRUCTIONS = """You are a Simulated User designed to test an AI Agent.
|
||||
|
||||
Your single most important job is to react logically to the Agent's last message.
|
||||
The Conversation Plan is your canonical grounding, not a script; your response MUST be dictated by what the Agent just said.
|
||||
|
||||
# Primary Operating Loop
|
||||
|
||||
You MUST follow this three-step process while thinking:
|
||||
|
||||
Step 1: Analyze what the Agent just said or did. Specifically, is the Agent asking you a question, reporting a successful or unsuccessful operation, or saying something incorrect or unexpected?
|
||||
|
||||
Step 2: Choose one action based on your analysis:
|
||||
* ANSWER any questions the Agent asked.
|
||||
* ADVANCE to the next request as per the Conversation Plan if the Agent succeeds in satisfying your current request.
|
||||
* INTERVENE if the Agent is yet to complete your current request and the Conversation Plan requires you to modify it.
|
||||
* CORRECT the Agent if it is making a mistake or failing.
|
||||
* END the conversation if any of the below stopping conditions are met:
|
||||
- The Agent has completed all your requests from the Conversation Plan.
|
||||
- The Agent has failed to fulfill a request *more than once*.
|
||||
- The Agent has performed an incorrect operation and informs you that it is unable to correct it.
|
||||
- The Agent ends the conversation on its own by transferring you to a *human/live agent* (NOT another AI Agent).
|
||||
|
||||
Step 3: Formulate a response based on the chosen action and the below Action Protocols and output it.
|
||||
|
||||
# Action Protocols
|
||||
|
||||
**PROTOCOL: ANSWER**
|
||||
* Only answer the Agent's questions using information from the Conversation Plan.
|
||||
* Do NOT provide any additional information the Agent did not explicitly ask for.
|
||||
* If you do not have the information requested by the Agent, inform the Agent. Do NOT make up information that is not in the Conversation Plan.
|
||||
* Do NOT advance to the next request in the Conversation Plan.
|
||||
|
||||
**PROTOCOL: ADVANCE**
|
||||
* Make the next request from the Conversation Plan.
|
||||
* Skip redundant requests already fulfilled by the Agent.
|
||||
|
||||
**PROTOCOL: INTERVENE**
|
||||
* Change your current request as directed by the Conversation Plan with natural phrasing.
|
||||
|
||||
**PROTOCOL: CORRECT**
|
||||
* Challenge illogical or incorrect statements made by the Agent.
|
||||
* If the Agent did an incorrect operation, ask the Agent to fix it.
|
||||
* If this is the FIRST time the Agent failed to satisfy your request, ask the Agent to try again.
|
||||
|
||||
**PROTOCOL: END**
|
||||
* End the conversation only when any of the stopping conditions are met; do NOT end prematurely.
|
||||
* Output `{stop_signal}` to indicate that the conversation with the AI Agents is over.
|
||||
|
||||
# Conversation Plan
|
||||
|
||||
{conversation_plan}
|
||||
|
||||
# Conversation History
|
||||
|
||||
{conversation_history}
|
||||
"""
|
||||
|
||||
|
||||
class LlmBackedUserSimulatorConfig(BaseUserSimulatorConfig):
|
||||
"""Contains configurations required by an LLM backed user simulator."""
|
||||
@@ -130,13 +75,15 @@ prompt is also counted as an invocation.
|
||||
custom_instructions: Optional[str] = Field(
|
||||
default=None,
|
||||
description="""Custom instructions for the LlmBackedUserSimulator. The
|
||||
instructions must contain the following formatting placeholders:
|
||||
* {stop_signal} : text to be generated when the user simulator decides that the
|
||||
instructions must contain the following formatting placeholders following Jinja syntax:
|
||||
* {{ stop_signal }} : text to be generated when the user simulator decides that the
|
||||
conversation is over.
|
||||
* {conversation_plan} : the overall plan for the conversation that the user
|
||||
* {{ conversation_plan }} : the overall plan for the conversation that the user
|
||||
simulator must follow.
|
||||
* {conversation_history} : the conversation between the user and the agent so
|
||||
far.""",
|
||||
* {{ conversation_history }} : the conversation between the user and the agent so
|
||||
far.
|
||||
* {{ persona }} : Only needed if specifying user_persona in the conversation scenario.
|
||||
""",
|
||||
)
|
||||
|
||||
@field_validator("custom_instructions")
|
||||
@@ -144,18 +91,18 @@ instructions must contain the following formatting placeholders:
|
||||
def validate_custom_instructions(cls, value: Optional[str]) -> Optional[str]:
|
||||
if value is None:
|
||||
return value
|
||||
if not all(
|
||||
placeholder in value
|
||||
for placeholder in [
|
||||
"{stop_signal}",
|
||||
"{conversation_plan}",
|
||||
"{conversation_history}",
|
||||
]
|
||||
if not is_valid_user_simulator_template(
|
||||
value,
|
||||
required_params=[
|
||||
"stop_signal",
|
||||
"conversation_plan",
|
||||
"conversation_history",
|
||||
],
|
||||
):
|
||||
raise ValueError(
|
||||
"custom_instructions must contain each of the following formatting"
|
||||
" placeholders:"
|
||||
" {stop_signal}, {conversation_plan}, {conversation_history}"
|
||||
" placeholders using Jinja syntax: {{ stop_signal }}, {{"
|
||||
" conversation_plan }}, {{ conversation_history }}"
|
||||
)
|
||||
return value
|
||||
|
||||
@@ -180,11 +127,7 @@ class LlmBackedUserSimulator(UserSimulator):
|
||||
llm_registry = LLMRegistry()
|
||||
llm_class = llm_registry.resolve(self._config.model)
|
||||
self._llm = llm_class(model=self._config.model)
|
||||
self._instructions = (
|
||||
self._config.custom_instructions
|
||||
if self._config.custom_instructions
|
||||
else _DEFAULT_USER_AGENT_INSTRUCTIONS
|
||||
)
|
||||
self._user_persona = self._conversation_scenario.user_persona
|
||||
|
||||
@classmethod
|
||||
def _summarize_conversation(
|
||||
@@ -221,10 +164,12 @@ class LlmBackedUserSimulator(UserSimulator):
|
||||
# first invocation - send the static starting prompt
|
||||
return self._conversation_scenario.starting_prompt
|
||||
|
||||
user_agent_instructions = self._instructions.format(
|
||||
stop_signal=_STOP_SIGNAL,
|
||||
user_agent_instructions = get_llm_backed_user_simulator_prompt(
|
||||
conversation_plan=self._conversation_scenario.conversation_plan,
|
||||
conversation_history=rewritten_dialogue,
|
||||
stop_signal=_STOP_SIGNAL,
|
||||
custom_instructions=self._config.custom_instructions,
|
||||
user_persona=self._user_persona,
|
||||
)
|
||||
|
||||
llm_request = LlmRequest(
|
||||
|
||||
@@ -0,0 +1,217 @@
|
||||
# Copyright 2026 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import textwrap
|
||||
from typing import Optional
|
||||
|
||||
from .user_simulator_personas import UserPersona
|
||||
|
||||
_DEFAULT_USER_SIMULATOR_INSTRUCTIONS_TEMPLATE = """You are a Simulated User designed to test an AI Agent.
|
||||
|
||||
Your single most important job is to react logically to the Agent's last message.
|
||||
The Conversation Plan is your canonical grounding, not a script; your response MUST be dictated by what the Agent just said.
|
||||
|
||||
# Primary Operating Loop
|
||||
|
||||
You MUST follow this three-step process while thinking:
|
||||
|
||||
Step 1: Analyze what the Agent just said or did. Specifically, is the Agent asking you a question, reporting a successful or unsuccessful operation, or saying something incorrect or unexpected?
|
||||
|
||||
Step 2: Choose one action based on your analysis:
|
||||
* ANSWER any questions the Agent asked.
|
||||
* ADVANCE to the next request as per the Conversation Plan if the Agent succeeds in satisfying your current request.
|
||||
* INTERVENE if the Agent is yet to complete your current request and the Conversation Plan requires you to modify it.
|
||||
* CORRECT the Agent if it is making a mistake or failing.
|
||||
* END the conversation if any of the below stopping conditions are met:
|
||||
- The Agent has completed all your requests from the Conversation Plan.
|
||||
- The Agent has failed to fulfill a request *more than once*.
|
||||
- The Agent has performed an incorrect operation and informs you that it is unable to correct it.
|
||||
- The Agent ends the conversation on its own by transferring you to a *human/live agent* (NOT another AI Agent).
|
||||
|
||||
Step 3: Formulate a response based on the chosen action and the below Action Protocols and output it.
|
||||
|
||||
# Action Protocols
|
||||
|
||||
**PROTOCOL: ANSWER**
|
||||
* Only answer the Agent's questions using information from the Conversation Plan.
|
||||
* Do NOT provide any additional information the Agent did not explicitly ask for.
|
||||
* If you do not have the information requested by the Agent, inform the Agent. Do NOT make up information that is not in the Conversation Plan.
|
||||
* Do NOT advance to the next request in the Conversation Plan.
|
||||
|
||||
**PROTOCOL: ADVANCE**
|
||||
* Make the next request from the Conversation Plan.
|
||||
* Skip redundant requests already fulfilled by the Agent.
|
||||
|
||||
**PROTOCOL: INTERVENE**
|
||||
* Change your current request as directed by the Conversation Plan with natural phrasing.
|
||||
|
||||
**PROTOCOL: CORRECT**
|
||||
* Challenge illogical or incorrect statements made by the Agent.
|
||||
* If the Agent did an incorrect operation, ask the Agent to fix it.
|
||||
* If this is the FIRST time the Agent failed to satisfy your request, ask the Agent to try again.
|
||||
|
||||
**PROTOCOL: END**
|
||||
* End the conversation only when any of the stopping conditions are met; do NOT end prematurely.
|
||||
* Output `{{ stop_signal }}` to indicate that the conversation with the AI Agents is over.
|
||||
|
||||
# Conversation Plan
|
||||
|
||||
{{ conversation_plan }}
|
||||
|
||||
# Conversation History
|
||||
|
||||
{{ conversation_history }}
|
||||
"""
|
||||
|
||||
_USER_SIMULATOR_INSTRUCTIONS_WITH_PERSONA_TEMPLATE = """
|
||||
You are a Simulated User designed to test an AI Agent.
|
||||
|
||||
Your single most important job is to react logically to the Agent's last message while role-playing as the given Persona.
|
||||
The Conversation Plan is your canonical grounding, not a script; your response MUST be dictated by what the Agent just said.
|
||||
|
||||
# Persona Description
|
||||
|
||||
{{ persona.description }}
|
||||
This persona behaves in the following ways:
|
||||
{% for b in persona.behaviors %}
|
||||
## {{ b.name | render_string_filter}}
|
||||
{{ b.description | render_string_filter }}
|
||||
|
||||
Instructions:
|
||||
{{ b.get_behavior_instructions_str() | render_string_filter }}
|
||||
{% endfor %}
|
||||
# Conversation Plan
|
||||
|
||||
{{ conversation_plan }}
|
||||
|
||||
# Conversation History
|
||||
|
||||
{{ conversation_history }}
|
||||
""".strip()
|
||||
|
||||
|
||||
def is_valid_user_simulator_template(
|
||||
template_str: str, required_params: list[str]
|
||||
) -> bool:
|
||||
"""Checks if the given template_str is a valid jinja template."""
|
||||
from jinja2 import exceptions
|
||||
from jinja2 import meta
|
||||
from jinja2 import StrictUndefined
|
||||
from jinja2.sandbox import SandboxedEnvironment
|
||||
|
||||
# StrictUndefined allows us to check for all the given params.
|
||||
env = SandboxedEnvironment(undefined=StrictUndefined)
|
||||
try:
|
||||
# Check syntax of template
|
||||
template = env.parse(template_str)
|
||||
|
||||
# Find all variables the template expects
|
||||
undeclared_variables = meta.find_undeclared_variables(template)
|
||||
|
||||
# Check parameters in template
|
||||
missing_required = [
|
||||
v for v in required_params if v not in undeclared_variables
|
||||
]
|
||||
|
||||
return not (missing_required)
|
||||
|
||||
except (
|
||||
exceptions.TemplateSyntaxError,
|
||||
exceptions.UndefinedError,
|
||||
) as _:
|
||||
return False
|
||||
|
||||
|
||||
def _get_user_simulator_instructions_template(
|
||||
custom_instructions: Optional[str] = None,
|
||||
user_persona: Optional[UserPersona] = None,
|
||||
) -> str:
|
||||
"""Returns the appropriate instruction template for the user simulator."""
|
||||
if custom_instructions is None and user_persona is None:
|
||||
return _DEFAULT_USER_SIMULATOR_INSTRUCTIONS_TEMPLATE
|
||||
|
||||
if custom_instructions is None and user_persona is not None:
|
||||
return _USER_SIMULATOR_INSTRUCTIONS_WITH_PERSONA_TEMPLATE
|
||||
|
||||
if custom_instructions is not None and user_persona is None:
|
||||
return custom_instructions
|
||||
|
||||
if custom_instructions is not None and user_persona is not None:
|
||||
if not is_valid_user_simulator_template(
|
||||
custom_instructions,
|
||||
required_params=[
|
||||
"stop_signal",
|
||||
"conversation_plan",
|
||||
"conversation_history",
|
||||
"persona",
|
||||
],
|
||||
):
|
||||
raise ValueError(
|
||||
textwrap.dedent(
|
||||
"""Custom instructions using personas must contain the following formatting placeholders following Jinja syntax:
|
||||
* {{ stop_signal }} : text to be generated when the user simulator decides that the
|
||||
conversation is over.
|
||||
* {{ conversation_plan }} : the overall plan for the conversation that the user
|
||||
simulator must follow.
|
||||
* {{ conversation_history }} : the conversation between the user and the agent so far.
|
||||
* {{ persona }} : UserPersona for the simulator to use.
|
||||
"""
|
||||
)
|
||||
)
|
||||
|
||||
return custom_instructions
|
||||
|
||||
|
||||
def get_llm_backed_user_simulator_prompt(
|
||||
conversation_plan: str,
|
||||
conversation_history: str,
|
||||
stop_signal: str,
|
||||
custom_instructions: Optional[str] = None,
|
||||
user_persona: Optional[UserPersona] = None,
|
||||
):
|
||||
"""Formats the prompt for the llm-backed user simulator"""
|
||||
from jinja2 import DictLoader
|
||||
from jinja2 import pass_context
|
||||
from jinja2 import Template
|
||||
from jinja2.sandbox import SandboxedEnvironment
|
||||
|
||||
templates = {
|
||||
"user_instructions": _get_user_simulator_instructions_template(
|
||||
custom_instructions=custom_instructions,
|
||||
user_persona=user_persona,
|
||||
),
|
||||
}
|
||||
template_env = SandboxedEnvironment(loader=DictLoader(templates))
|
||||
|
||||
@pass_context
|
||||
def _render_string_filter(context, template_string):
|
||||
if not template_string:
|
||||
return ""
|
||||
return Template(template_string).render(context)
|
||||
|
||||
template_env.filters["render_string_filter"] = _render_string_filter
|
||||
|
||||
template_parameters = {
|
||||
"stop_signal": stop_signal,
|
||||
"conversation_plan": conversation_plan,
|
||||
"conversation_history": conversation_history,
|
||||
}
|
||||
if user_persona is not None:
|
||||
template_parameters["persona"] = user_persona
|
||||
|
||||
return template_env.get_template("user_instructions").render(
|
||||
template_parameters
|
||||
)
|
||||
@@ -0,0 +1,256 @@
|
||||
# Copyright 2026 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
from .user_simulator_personas import UserPersona
|
||||
|
||||
_LATEST_TURN_USER_SIMULATOR_EVALUATOR_PROMPT_TEMPLATE = """
|
||||
You are a data scientist tasked with evaluating the quality of a User Simulator that is interacting with an Agent.
|
||||
Your task is to determine if the Generated User Response is consistent with:
|
||||
- The Conversation Plan: A list of high-level goals that the User Simulator is expected to achieve in the conversation.
|
||||
- The Conversation History: The exchange between the User Simulator and the Agent so far.
|
||||
To determine this, we provide specific Evaluation Criteria that must be satisfied by the Generated User Response.
|
||||
|
||||
# Definition of Conversation Plan
|
||||
The Conversation Plan specifies the goals that the User Simulator must execute.
|
||||
The Conversation Plan also specifies the information and details that are needed to complete the goals.
|
||||
The Conversation Plan is sequential in nature and the User Simulator must ensure the sequence is followed.
|
||||
|
||||
# Definition of Conversation History
|
||||
The Conversation History is the actual dialogue between the User Simulator and the Agent.
|
||||
The Conversation History may not be complete, but the existing dialogue should adhere to the Conversation Plan.
|
||||
The Conversation History may contain instances where the User Simulator troubleshoots an incorrect/inappropriate response from the Agent in order to enforce the Conversation Plan.
|
||||
The Conversation History is finished only when the User Simulator outputs `{{ stop_signal }}` in its response. If this token is missing, the conversation between the User Simulator and the Agent has not finished, and more turns can be generated.
|
||||
|
||||
# Definition of Generated User Response
|
||||
The Generated User Response is a the next user response in the conversation between a User Simulator and an Agent.
|
||||
The Generated User Response was generated by the User Simulator based on a Conversation Plan and Conversation History.
|
||||
|
||||
# Evaluation Criteria
|
||||
Your task is to evaluate the Generated User Response on a PASS/FAIL basis looking for specific errors.
|
||||
The Generated User Response is marked as PASS unless it contains any of the Violations listed below, in which case it is marked as FAIL.
|
||||
|
||||
** CONVERSATION_PLAN_FOLLOWED **
|
||||
Does the Generated User Response stick to the Conversation Plan?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response repeats a high-level goal that was already completed in previous turns.
|
||||
- The Generated User Response provides details for a high-level goal that was already completed.
|
||||
- The Generated User Response response agrees to change the topic or perform a task not listed in the Conversation Plan.
|
||||
- The Generated User Response invents a new goal not present in the Conversation Plan.
|
||||
- The Generated User Response invents details (e.g., a made-up phone number or address) not provided in the Conversation Plan.
|
||||
|
||||
** STOP_CONDITION_FOLLOWED **
|
||||
Did the conversation end exactly when it was supposed to?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The conversation should have ended, but the Generated User Response did not use `{{ stop_signal }}`.
|
||||
- The Generated User Response used `{{ stop_signal }}`, but tasks in the Conversation Plan are still incomplete AND the Agent has not failed.
|
||||
- The Agent successfully transferred the User Simulator to a human/live agent, but the Generated User Response continued instead of using `{{ stop_signal }}`.
|
||||
|
||||
** USER_GOAL_ORIENTED **
|
||||
Is the User Simulator acting naturally, or is it "data dumping"?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response provides specific details for a high-level goal (email content, recipient address, phone numbers) BEFORE the Agent has explicitly asked for them.
|
||||
- The Generated User Response tries to accomplish more than one high-level task in a single turn.
|
||||
|
||||
** LIMITED_TROUBLESHOOTING **
|
||||
Does the User Simulator have the correct amount of patience? (Note: Please check the conversation history and count the number of Agent errors).
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response ends the conversation immediately after the first Agent error.
|
||||
- On the second Agent error, the Generated User Response response continues the conversation without using `{{ stop_signal }}`.
|
||||
- After the second Agent error, the Generated User Response tries to continue the conversation or continues addressing errors without using `{{ stop_signal }}`.
|
||||
|
||||
** RESPONSIVENESS **
|
||||
Does the User Simulator answer what is asked?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Agent asked a question (or multiple questions), and the Generated User Response failed to address one or all of them.
|
||||
- The Agent asked for information NOT in the Conversation Plan, and the Generated User Response made up an answer instead of stating, e.g., "I don't know" or "I don't have that info."
|
||||
|
||||
** CORRECTS_AGENT **
|
||||
Does the User Simulator catch the Agent's mistakes?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Agent provided incorrect information, but the Generated User Response continued as if it was correct.
|
||||
- The Agent made a dangerous assumption (e.g., sending an email without asking for the content first), and the Generated User Response continues without correcting the Agent.
|
||||
|
||||
** CONVERSATIONAL_TONE **
|
||||
Does the User Simulator sound like a human?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response uses overly complex sentence structures, or uses technical jargon inappropriately.
|
||||
- The Generated User Response is sterile and purely functional (direct commands) with no natural conversational framing.
|
||||
- The Generated User Response is too formal in nature, employing overly polite phrases and expressions.
|
||||
- The Generated User Response is a "wall of text" where a simple sentence would suffice.
|
||||
|
||||
# Output Format
|
||||
Format your response in the following JSON format:
|
||||
{
|
||||
"criteria": [
|
||||
{
|
||||
"name": "CRITERIA_NAME_1",
|
||||
"reasoning": "reasoning",
|
||||
"passes": True or False,
|
||||
},
|
||||
{
|
||||
"name": "CRITERIA_NAME_2",
|
||||
"reasoning": "reasoning",
|
||||
"passes": True or False,
|
||||
},
|
||||
...
|
||||
],
|
||||
"is_valid": True or False,
|
||||
}
|
||||
|
||||
# Conversation Plan
|
||||
{{ conversation_plan }}
|
||||
|
||||
# Conversation History
|
||||
{{ conversation_history }}
|
||||
|
||||
# Generated User Response
|
||||
{{ generated_user_response }}
|
||||
""".strip()
|
||||
|
||||
|
||||
_LATEST_TURN_USER_SIMULATOR_WITH_PERSONA_EVALUATOR_PROMPT_TEMPLATE = """
|
||||
You are a data scientist tasked with evaluating the quality of a User Simulator that is interacting with an Agent.
|
||||
Your task is to determine if the Generated User Response is consistent with:
|
||||
- The Conversation Plan: A list of high-level goals that the User Simulator is expected to achieve in the conversation.
|
||||
- The Conversation History: The exchange between the User Simulator and the Agent so far.
|
||||
- A Persona: A set of behaviours that the User Simulator is expected to exhibit in the conversation.
|
||||
To determine this, we provide specific Evaluation Criteria that you must use to evaluate the Generated User Response.
|
||||
|
||||
# Definition of Conversation Plan
|
||||
The Conversation Plan specifies the goals that the User Simulator must execute.
|
||||
The Conversation Plan also specifies the information and details that are needed to complete the goals.
|
||||
The Conversation Plan is sequential in nature and the User Simulator must ensure the sequence is followed.
|
||||
The Conversation Plan is not a script.
|
||||
|
||||
# Definition of Conversation History
|
||||
The Conversation History is the actual dialogue between the User Simulator and the Agent.
|
||||
The Conversation History may not be complete, but the exsisting dialogue should adhere to the Conversation Plan.
|
||||
The Conversation History may contain instances where the User Simulator troubleshoots an incorrect/inappropriate response from the Agent in order to enforce the Conversation Plan.
|
||||
The Conversation History is finished only when the User Simulator outputs `{{ stop_signal }}` in its response. If this token is missing, the conversation between the User Simulator and the Agent has not finished, and more turns can be generated.
|
||||
|
||||
# Definition of Persona
|
||||
The Persona is a description of how the User Simulator should behave in a conversation with the Agent.
|
||||
A Persona specifies behaviors, not goals.
|
||||
If the Persona contradicts the Conversation Plan, the Conversation Plan has precedence.
|
||||
|
||||
# Definition of Generated User Response
|
||||
The Generated User Response is the next user response in the conversation between a User Simulator and an Agent.
|
||||
The Generated User Response was generated by the User Simulator based on the Conversation Plan and Conversation History.
|
||||
|
||||
# Evaluation Criteria
|
||||
Your task is to evaluate the Generated User Response on a PASS/FAIL basis looking for specific errors.
|
||||
The Generated User Response is marked as PASS unless it contains any of the Violations listed below, in which case it is marked as FAIL.
|
||||
{% for b in persona.behaviors %}
|
||||
## Criteria: {{ b.name | render_string_filter}}
|
||||
{{ b.description | render_string_filter}}
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
{{ b.get_violation_rubrics_str() | render_string_filter}}
|
||||
{% endfor %}
|
||||
# Output Format
|
||||
Format your response in the following JSON format:
|
||||
{
|
||||
"criteria": [
|
||||
{
|
||||
"name": "CRITERIA_NAME_1",
|
||||
"reasoning": "reasoning",
|
||||
"passes": True or False,
|
||||
},
|
||||
{
|
||||
"name": "CRITERIA_NAME_2",
|
||||
"reasoning": "reasoning",
|
||||
"passes": True or False,
|
||||
},
|
||||
...
|
||||
],
|
||||
"is_valid": True if it passes all criteria, False otherwise
|
||||
}
|
||||
|
||||
# Conversation Plan
|
||||
{{ conversation_plan }}
|
||||
|
||||
# Conversation History
|
||||
{{ conversation_history }}
|
||||
|
||||
# Persona Description
|
||||
{{ persona.description }}
|
||||
The Evaluation Criteria above already specify how to evaluate whether the Generated User Response satisfies this persona.
|
||||
|
||||
# Generated User Response
|
||||
{{ generated_user_response }}
|
||||
""".strip()
|
||||
|
||||
|
||||
def _get_latest_turn_user_simulator_quality_prompt_template(
|
||||
user_persona: Optional[UserPersona] = None,
|
||||
) -> str:
|
||||
"""Returns the appropriate prompt for user simulator quality"""
|
||||
if user_persona is None:
|
||||
return _LATEST_TURN_USER_SIMULATOR_EVALUATOR_PROMPT_TEMPLATE
|
||||
return _LATEST_TURN_USER_SIMULATOR_WITH_PERSONA_EVALUATOR_PROMPT_TEMPLATE
|
||||
|
||||
|
||||
def get_per_turn_user_simulator_quality_prompt(
|
||||
conversation_plan: str,
|
||||
conversation_history: str,
|
||||
generated_user_response: str,
|
||||
stop_signal: str,
|
||||
user_persona: Optional[UserPersona] = None,
|
||||
):
|
||||
"""Formats the prompt for the per turn user simulator evaluator"""
|
||||
from jinja2 import DictLoader
|
||||
from jinja2 import Environment
|
||||
from jinja2 import pass_context
|
||||
from jinja2 import Template
|
||||
|
||||
templates = {
|
||||
"verifier_instructions": (
|
||||
_get_latest_turn_user_simulator_quality_prompt_template(
|
||||
user_persona=user_persona
|
||||
)
|
||||
),
|
||||
}
|
||||
template_env = Environment(loader=DictLoader(templates))
|
||||
|
||||
@pass_context
|
||||
def _render_string_filter(context, template_string):
|
||||
if not template_string:
|
||||
return ""
|
||||
return Template(template_string).render(context)
|
||||
|
||||
template_env.filters["render_string_filter"] = _render_string_filter
|
||||
|
||||
template_parameters = {
|
||||
"conversation_plan": conversation_plan,
|
||||
"conversation_history": conversation_history,
|
||||
"generated_user_response": generated_user_response,
|
||||
"stop_signal": stop_signal,
|
||||
}
|
||||
if user_persona is not None:
|
||||
template_parameters["persona"] = user_persona
|
||||
|
||||
return template_env.get_template("verifier_instructions").render(
|
||||
template_parameters
|
||||
)
|
||||
@@ -14,7 +14,6 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import ClassVar
|
||||
from typing import Optional
|
||||
@@ -43,117 +42,7 @@ from ..llm_as_judge import AutoRaterScore
|
||||
from ..llm_as_judge_utils import get_eval_status
|
||||
from ..llm_as_judge_utils import get_text_from_content
|
||||
from ..llm_as_judge_utils import Label
|
||||
|
||||
_LATEST_TURN_USER_SIMULATOR_EVALUATOR_PROMPT = """
|
||||
You are a data scientist tasked with evaluating the quality of a User Simulator that is interacting with an Agent.
|
||||
Your task is to determine if the Generated User Response is consistent with:
|
||||
- The Conversation Plan: A list of high-level goals that the User Simulator is expected to achieve in the conversation.
|
||||
- The Conversation History: The exchange between the User Simulator and the Agent so far.
|
||||
To determine this, we provide specific Evaluation Criteria that must be satisfied by the Generated User Response.
|
||||
|
||||
# Definition of Conversation Plan
|
||||
The Conversation Plan specifies the goals that the User Simulator must execute.
|
||||
The Conversation Plan also specifies the information and details that are needed to complete the goals.
|
||||
The Conversation Plan is sequential in nature and the User Simulator must ensure the sequence is followed.
|
||||
|
||||
# Definition of Conversation History
|
||||
The Conversation History is the actual dialogue between the User Simulator and the Agent.
|
||||
The Conversation History may not be complete, but the existing dialogue should adhere to the Conversation Plan.
|
||||
The Conversation History may contain instances where the User Simulator troubleshoots an incorrect/inappropriate response from the Agent in order to enforce the Conversation Plan.
|
||||
The Conversation History is finished only when the User Simulator outputs `{stop_signal}` in its response. If this token is missing, the conversation between the User Simulator and the Agent has not finished, and more turns can be generated.
|
||||
|
||||
# Definition of Generated User Response
|
||||
The Generated User Response is a the next user response in the conversation between a User Simulator and an Agent.
|
||||
The Generated User Response was generated by the User Simulator based on a Conversation Plan and Conversation History.
|
||||
|
||||
# Evaluation Criteria
|
||||
Your task is to evaluate the Generated User Response on a PASS/FAIL basis looking for specific errors.
|
||||
The Generated User Response is marked as PASS unless it contains any of the Violations listed below, in which case it is marked as FAIL.
|
||||
|
||||
** CONVERSATION_PLAN_FOLLOWED **
|
||||
Does the Generated User Response stick to the Conversation Plan?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response repeats a high-level goal that was already completed in previous turns.
|
||||
- The Generated User Response provides details for a high-level goal that was already completed.
|
||||
- The Generated User Response response agrees to change the topic or perform a task not listed in the Conversation Plan.
|
||||
- The Generated User Response invents a new goal not present in the Conversation Plan.
|
||||
- The Generated User Response invents details (e.g., a made-up phone number or address) not provided in the Conversation Plan.
|
||||
|
||||
** STOP_CONDITION_FOLLOWED **
|
||||
Did the conversation end exactly when it was supposed to?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The conversation should have ended, but the Generated User Response did not use `{stop_signal}`.
|
||||
- The Generated User Response used `{stop_signal}`, but tasks in the Conversation Plan are still incomplete AND the Agent has not failed.
|
||||
- The Agent successfully transferred the User Simulator to a human/live agent, but the Generated User Response continued instead of using `{stop_signal}`.
|
||||
|
||||
** USER_GOAL_ORIENTED **
|
||||
Is the User Simulator acting naturally, or is it "data dumping"?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response provides specific details for a high-level goal (email content, recipient address, phone numbers) BEFORE the Agent has explicitly asked for them.
|
||||
- The Generated User Response tries to accomplish more than one high-level task in a single turn.
|
||||
|
||||
** LIMITED_TROUBLESHOOTING **
|
||||
Does the User Simulator have the correct amount of patience? (Note: Please check the conversation history and count the number of Agent errors).
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response ends the conversation immediately after the first Agent error.
|
||||
- On the second Agent error, the Generated User Response response continues the conversation without using `{stop_signal}`.
|
||||
- After the second Agent error, the Generated User Response tries to continue the conversation or continues addressing errors without using `{stop_signal}`.
|
||||
|
||||
** RESPONSIVENESS **
|
||||
Does the User Simulator answer what is asked?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Agent asked a question (or multiple questions), and the Generated User Response failed to address one or all of them.
|
||||
- The Agent asked for information NOT in the Conversation Plan, and the Generated User Response made up an answer instead of stating, e.g., "I don't know" or "I don't have that info."
|
||||
|
||||
** CORRECTS_AGENT **
|
||||
Does the User Simulator catch the Agent's mistakes?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Agent provided incorrect information, but the Generated User Response continued as if it was correct.
|
||||
- The Agent made a dangerous assumption (e.g., sending an email without asking for the content first), and the Generated User Response continues without correcting the Agent.
|
||||
|
||||
** CONVERSATIONAL_TONE **
|
||||
Does the User Simulator sound like a human?
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
- The Generated User Response uses overly complex sentence structures, or uses technical jargon inappropriately.
|
||||
- The Generated User Response is sterile and purely functional (direct commands) with no natural conversational framing.
|
||||
- The Generated User Response is too formal in nature, employing overly polite phrases and expressions.
|
||||
- The Generated User Response is a "wall of text" where a simple sentence would suffice.
|
||||
|
||||
# Output Format
|
||||
Format your response in the following JSON format:
|
||||
{{
|
||||
"criteria": [
|
||||
{{
|
||||
"name": "CRITERIA_NAME_1",
|
||||
"reasoning": "reasoning",
|
||||
"passes": True or False,
|
||||
}},
|
||||
{{
|
||||
"name": "CRITERIA_NAME_2",
|
||||
"reasoning": "reasoning",
|
||||
"passes": True or False,
|
||||
}},
|
||||
...
|
||||
],
|
||||
"is_valid": True or False,
|
||||
}}
|
||||
|
||||
# Conversation Plan
|
||||
{conversation_plan}
|
||||
|
||||
# Conversation History
|
||||
{conversation_history}
|
||||
|
||||
# Generated User Response
|
||||
{generated_user_response}
|
||||
""".strip()
|
||||
from .per_turn_user_simulator_quality_prompts import get_per_turn_user_simulator_quality_prompt
|
||||
|
||||
|
||||
def _parse_llm_response(response: str) -> Label:
|
||||
@@ -167,7 +56,7 @@ def _parse_llm_response(response: str) -> Label:
|
||||
"""
|
||||
# Regex matching the label field in the response.
|
||||
is_valid_match = re.search(
|
||||
r'"is_valid":\s*\[*[\n\s]*"*([^"^\]^\s]*)"*[\n\s]*\]*\s*[,\n\}]',
|
||||
r'"is_valid":\s*\[*[\n\s]*"*([^"\]]*)"*[\n\s]*\]*\s*[,\n\}]',
|
||||
response,
|
||||
)
|
||||
|
||||
@@ -176,7 +65,7 @@ def _parse_llm_response(response: str) -> Label:
|
||||
return Label.NOT_FOUND
|
||||
|
||||
# Remove any trailing whitespace, commas, or end-brackets from the label.
|
||||
label = is_valid_match.group(1).strip(r"\s,\}").lower()
|
||||
label = is_valid_match.group(1).strip("}").replace(",", "").strip().lower()
|
||||
if label in [
|
||||
Label.INVALID.value,
|
||||
Label.ALMOST.value,
|
||||
@@ -193,7 +82,7 @@ def _parse_llm_response(response: str) -> Label:
|
||||
def _format_conversation_history(invocations: list[Invocation]) -> str:
|
||||
conversation_history = []
|
||||
for invocation in invocations:
|
||||
if invocation.user_content is not None:
|
||||
if invocation.user_content is not None and invocation.user_content.parts:
|
||||
conversation_history.append(
|
||||
f"user: {get_text_from_content(invocation.user_content)}"
|
||||
)
|
||||
@@ -244,8 +133,6 @@ class PerTurnUserSimulatorQualityV1(Evaluator):
|
||||
self._eval_metric = eval_metric
|
||||
self._criterion = self._deserialize_criterion(eval_metric)
|
||||
|
||||
self._prompt_template = _LATEST_TURN_USER_SIMULATOR_EVALUATOR_PROMPT
|
||||
|
||||
self._llm_options = self._criterion.judge_model_options
|
||||
self._stop_signal = self._criterion.stop_signal
|
||||
self._llm = self._setup_llm()
|
||||
@@ -336,11 +223,12 @@ class PerTurnUserSimulatorQualityV1(Evaluator):
|
||||
f"Encountered: {conversation_scenario}"
|
||||
)
|
||||
|
||||
return self._prompt_template.format(
|
||||
return get_per_turn_user_simulator_quality_prompt(
|
||||
conversation_plan=conversation_scenario.conversation_plan,
|
||||
conversation_history=_format_conversation_history(previous_invocations),
|
||||
generated_user_response=get_text_from_content(invocation.user_content),
|
||||
stop_signal=self._stop_signal,
|
||||
user_persona=conversation_scenario.user_persona,
|
||||
)
|
||||
|
||||
def _convert_llm_response_to_score(
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,130 @@
|
||||
# Copyright 2026 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Sequence
|
||||
|
||||
from pydantic import BaseModel
|
||||
from pydantic import Field
|
||||
|
||||
from ...errors.not_found_error import NotFoundError
|
||||
from ...utils.feature_decorator import experimental
|
||||
|
||||
logger = logging.getLogger("google_adk." + __name__)
|
||||
|
||||
|
||||
@experimental
|
||||
class UserBehavior(BaseModel):
|
||||
"""Container for the behavior of a persona."""
|
||||
|
||||
name: str = Field(description="Name of the UserBehavior")
|
||||
|
||||
description: str = Field(
|
||||
description=(
|
||||
"General description of the expected behavior. This will be used in"
|
||||
" bot the instructions for the user simulator and the user simulator"
|
||||
" evaluator."
|
||||
)
|
||||
)
|
||||
|
||||
behavior_instructions: list[str] = Field(
|
||||
description=(
|
||||
"Instructions the user should follow. These will be included in the"
|
||||
" instructions for the user simulator."
|
||||
)
|
||||
)
|
||||
|
||||
violation_rubrics: list[str] = Field(
|
||||
description=(
|
||||
"Rubrics to evaluate whether the user simulator presents the"
|
||||
" behavior. If the user response presents any of these violations,"
|
||||
" the evaluator will consider the user simulator response as invalid."
|
||||
)
|
||||
)
|
||||
|
||||
def get_behavior_instructions_str(self):
|
||||
"""Returns a string version of the violation rubrics."""
|
||||
return "\n".join(f" * {i}" for i in self.behavior_instructions)
|
||||
|
||||
def get_violation_rubrics_str(self):
|
||||
"""Returns a string version of the violation rubrics."""
|
||||
return "\n".join(f" * {v}" for v in self.violation_rubrics)
|
||||
|
||||
|
||||
@experimental
|
||||
class UserPersona(BaseModel):
|
||||
"""Container for a persona."""
|
||||
|
||||
id: str = Field(
|
||||
description=(
|
||||
"Human readable identifier for the UserPersona. Persona registries"
|
||||
" will refer to this identifier."
|
||||
)
|
||||
)
|
||||
|
||||
description: str = Field(
|
||||
description=(
|
||||
"Description for the UserPersona. This will be included in the"
|
||||
" instructions for the user simulator and its verifier."
|
||||
)
|
||||
)
|
||||
|
||||
behaviors: Sequence[UserBehavior] = Field(
|
||||
description=(
|
||||
"Sequence of UserBehaviors for the persona. These will be included in"
|
||||
" the instructions for the user simulator and its verifier."
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
@experimental
|
||||
class UserPersonaRegistry:
|
||||
"""A registry for UserPersona instances."""
|
||||
|
||||
def __init__(self):
|
||||
self._registry: dict[str, UserPersona] = {}
|
||||
|
||||
def get_persona(self, persona_id: str) -> UserPersona:
|
||||
"""Returns the User Persona associated with the given id."""
|
||||
if persona_id not in self._registry:
|
||||
raise NotFoundError(f"{persona_id} not found in registry.")
|
||||
|
||||
return self._registry[persona_id]
|
||||
|
||||
def register_persona(
|
||||
self,
|
||||
persona_id: str,
|
||||
user_persona: UserPersona,
|
||||
):
|
||||
"""Registers a user persona given the persona id.
|
||||
|
||||
If a mapping already exist, then it is updated.
|
||||
"""
|
||||
if persona_id in self._registry:
|
||||
logger.info(
|
||||
"Updating User Persona for %s from %s to %s",
|
||||
persona_id,
|
||||
self._registry[persona_id],
|
||||
user_persona,
|
||||
)
|
||||
|
||||
self._registry[persona_id] = user_persona
|
||||
|
||||
def get_registered_personas(
|
||||
self,
|
||||
) -> list[UserPersona]:
|
||||
"""Returns the list of User Personas registered so far."""
|
||||
return [persona for _, persona in self._registry.items()]
|
||||
@@ -763,7 +763,7 @@ def test_cli_add_eval_case_with_session(tmp_path: Path):
|
||||
eval_set_data = json.load(f)
|
||||
assert len(eval_set_data["eval_cases"]) == 1
|
||||
eval_case = eval_set_data["eval_cases"][0]
|
||||
assert eval_case["eval_id"] == "0a1a5048"
|
||||
assert eval_case["eval_id"] == "734909ff"
|
||||
assert eval_case["session_input"]["app_name"] == "test_app_add_2"
|
||||
|
||||
|
||||
|
||||
@@ -18,6 +18,8 @@ from google.adk.evaluation import conversation_scenarios
|
||||
from google.adk.evaluation.simulation.llm_backed_user_simulator import LlmBackedUserSimulator
|
||||
from google.adk.evaluation.simulation.llm_backed_user_simulator import LlmBackedUserSimulatorConfig
|
||||
from google.adk.evaluation.simulation.user_simulator import Status
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserBehavior
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserPersona
|
||||
from google.adk.events.event import Event
|
||||
from google.genai import types
|
||||
from pydantic import ValidationError
|
||||
@@ -94,7 +96,7 @@ def test_llm_backed_user_simulator_config_validation():
|
||||
config = LlmBackedUserSimulatorConfig(custom_instructions=None)
|
||||
assert config.custom_instructions is None
|
||||
valid_instructions = (
|
||||
"{stop_signal} {conversation_plan} {conversation_history}"
|
||||
"{{ stop_signal }} {{ conversation_plan }} {{ conversation_history }}"
|
||||
)
|
||||
config = LlmBackedUserSimulatorConfig(custom_instructions=valid_instructions)
|
||||
assert config.custom_instructions == valid_instructions
|
||||
@@ -144,12 +146,38 @@ def conversation_scenario():
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def user_persona():
|
||||
"""Provides a test user persona."""
|
||||
return UserPersona(
|
||||
id="test_persona",
|
||||
description="A test persona",
|
||||
behaviors=[
|
||||
UserBehavior(
|
||||
name="polite",
|
||||
description="is polite",
|
||||
behavior_instructions=["Always say please and thank you."],
|
||||
violation_rubrics=["is rude"],
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def conversation_scenario_with_persona(user_persona):
|
||||
"""Provides a test conversation scenario with a user persona."""
|
||||
return conversation_scenarios.ConversationScenario(
|
||||
starting_prompt="Hello",
|
||||
conversation_plan="test plan with persona",
|
||||
user_persona=user_persona,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def simulator(mock_llm_agent, conversation_scenario):
|
||||
"""Provides an LlmBackedUserSimulator instance for testing."""
|
||||
config = LlmBackedUserSimulatorConfig(
|
||||
model="test-model",
|
||||
model_configuration=types.GenerateContentConfig(),
|
||||
)
|
||||
sim = LlmBackedUserSimulator(
|
||||
config=config, conversation_scenario=conversation_scenario
|
||||
@@ -158,6 +186,19 @@ def simulator(mock_llm_agent, conversation_scenario):
|
||||
return sim
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def simulator_with_persona(mock_llm_agent, conversation_scenario_with_persona):
|
||||
"""Provides an LlmBackedUserSimulator instance for testing."""
|
||||
config = LlmBackedUserSimulatorConfig(
|
||||
model="test-model",
|
||||
)
|
||||
sim = LlmBackedUserSimulator(
|
||||
config=config, conversation_scenario=conversation_scenario_with_persona
|
||||
)
|
||||
sim._invocation_count = 1 # Bypass starting prompt by default for tests
|
||||
return sim
|
||||
|
||||
|
||||
class TestLlmBackedUserSimulator:
|
||||
"""Test cases for LlmBackedUserSimulator main methods."""
|
||||
|
||||
@@ -262,3 +303,27 @@ class TestLlmBackedUserSimulator:
|
||||
|
||||
assert next_user_message.status == Status.SUCCESS
|
||||
assert next_user_message.user_message == expected_user_message
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_next_user_message_with_persona_success(
|
||||
self, simulator_with_persona, mock_llm_agent, mocker
|
||||
):
|
||||
"""Tests get_next_user_message when the user message is generated successfully."""
|
||||
mock_llm_response = mocker.MagicMock()
|
||||
mock_llm_response.content = types.Content(
|
||||
parts=[types.Part(text="I need to book a flight.")]
|
||||
)
|
||||
mock_llm_agent.generate_content_async.return_value = to_async_iter(
|
||||
[mock_llm_response]
|
||||
)
|
||||
|
||||
next_user_message = await simulator_with_persona.get_next_user_message(
|
||||
events=_INPUT_EVENTS
|
||||
)
|
||||
|
||||
expected_user_message = types.Content(
|
||||
parts=[types.Part(text="I need to book a flight.")], role="user"
|
||||
)
|
||||
|
||||
assert next_user_message.status == Status.SUCCESS
|
||||
assert next_user_message.user_message == expected_user_message
|
||||
|
||||
@@ -0,0 +1,228 @@
|
||||
# Copyright 2026 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import textwrap
|
||||
|
||||
from google.adk.evaluation.simulation.llm_backed_user_simulator_prompts import _DEFAULT_USER_SIMULATOR_INSTRUCTIONS_TEMPLATE
|
||||
from google.adk.evaluation.simulation.llm_backed_user_simulator_prompts import _get_user_simulator_instructions_template
|
||||
from google.adk.evaluation.simulation.llm_backed_user_simulator_prompts import _USER_SIMULATOR_INSTRUCTIONS_WITH_PERSONA_TEMPLATE
|
||||
from google.adk.evaluation.simulation.llm_backed_user_simulator_prompts import get_llm_backed_user_simulator_prompt
|
||||
from google.adk.evaluation.simulation.llm_backed_user_simulator_prompts import is_valid_user_simulator_template
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserBehavior
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserPersona
|
||||
import pytest
|
||||
|
||||
_MOCK_DEFAULT_TEMPLATE = textwrap.dedent("""\
|
||||
Default template
|
||||
|
||||
# Conversation Plan
|
||||
{{conversation_plan}}
|
||||
|
||||
# Conversation History
|
||||
{{conversation_history}}
|
||||
|
||||
# Stop signal
|
||||
{{stop_signal}}
|
||||
""").strip()
|
||||
|
||||
_MOCK_PERSONA_TEMPLATE = textwrap.dedent("""\
|
||||
Persona template
|
||||
|
||||
# Persona Description
|
||||
{{persona.description}}
|
||||
{% for b in persona.behaviors %}
|
||||
## {{ b.name }}
|
||||
{{ b.description }}
|
||||
|
||||
Instructions:
|
||||
{{ b.get_behavior_instructions_str() }}
|
||||
{% endfor %}
|
||||
# Conversation Plan
|
||||
{{conversation_plan}}
|
||||
|
||||
# Conversation History
|
||||
{{conversation_history}}
|
||||
|
||||
# Stop signal
|
||||
{{stop_signal}}
|
||||
""").strip()
|
||||
|
||||
|
||||
class TestGetUserSimulatorInstructionsTemplate:
|
||||
"""Test cases for _get_user_simulator_instructions_template."""
|
||||
|
||||
def test_get_user_simulator_instructions_template_default(self):
|
||||
assert (
|
||||
_get_user_simulator_instructions_template()
|
||||
== _DEFAULT_USER_SIMULATOR_INSTRUCTIONS_TEMPLATE
|
||||
)
|
||||
|
||||
def test_get_user_simulator_instructions_template_with_custom_instructions(
|
||||
self,
|
||||
):
|
||||
custom_instructions = "custom instructions"
|
||||
assert (
|
||||
_get_user_simulator_instructions_template(
|
||||
custom_instructions=custom_instructions
|
||||
)
|
||||
== custom_instructions
|
||||
)
|
||||
|
||||
def test_get_user_simulator_instructions_template_with_persona(self):
|
||||
user_persona = UserPersona(
|
||||
id="test_persona", description="Test persona", behaviors=[]
|
||||
)
|
||||
assert (
|
||||
_get_user_simulator_instructions_template(user_persona=user_persona)
|
||||
== _USER_SIMULATOR_INSTRUCTIONS_WITH_PERSONA_TEMPLATE
|
||||
)
|
||||
|
||||
def test_get_user_simulator_instructions_template_with_bad_custom_instructions_raises_error(
|
||||
self,
|
||||
):
|
||||
custom_instructions = "custom instructions"
|
||||
user_persona = UserPersona(
|
||||
id="test_persona", description="Test persona", behaviors=[]
|
||||
)
|
||||
with pytest.raises(ValueError):
|
||||
_get_user_simulator_instructions_template(
|
||||
custom_instructions=custom_instructions, user_persona=user_persona
|
||||
)
|
||||
|
||||
|
||||
sample_persona = UserPersona(
|
||||
id="test_persona",
|
||||
description="Test persona description",
|
||||
behaviors=[
|
||||
UserBehavior(
|
||||
name="Test behavior",
|
||||
description="Test behavior description",
|
||||
behavior_instructions=["instruction 1", "instruction 2"],
|
||||
violation_rubrics=["rubric 1"],
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
class TestGetLlmBackedUserSimulatorPrompt:
|
||||
"""Test cases for get_llm_backed_user_simulator_prompt."""
|
||||
|
||||
def test_get_llm_backed_user_simulator_prompt_default(self, mocker):
|
||||
mocker.patch(
|
||||
"google.adk.evaluation.simulation.llm_backed_user_simulator_prompts._DEFAULT_USER_SIMULATOR_INSTRUCTIONS_TEMPLATE",
|
||||
_MOCK_DEFAULT_TEMPLATE,
|
||||
)
|
||||
prompt = get_llm_backed_user_simulator_prompt(
|
||||
conversation_plan="test plan",
|
||||
conversation_history="test history",
|
||||
stop_signal="test stop",
|
||||
)
|
||||
expected_prompt = textwrap.dedent("""\
|
||||
Default template
|
||||
|
||||
# Conversation Plan
|
||||
test plan
|
||||
|
||||
# Conversation History
|
||||
test history
|
||||
|
||||
# Stop signal
|
||||
test stop""").strip()
|
||||
|
||||
assert prompt == expected_prompt
|
||||
|
||||
def test_get_llm_backed_user_simulator_prompt_with_custom_instructions(self):
|
||||
custom_instructions = textwrap.dedent("""\
|
||||
Custom instructions:
|
||||
|
||||
# Past history
|
||||
{{conversation_plan}}
|
||||
|
||||
# Plan
|
||||
{{conversation_plan}}
|
||||
|
||||
# Finished!
|
||||
{{stop_signal}}""").strip()
|
||||
prompt = get_llm_backed_user_simulator_prompt(
|
||||
conversation_plan="test plan",
|
||||
conversation_history="test history",
|
||||
stop_signal="test stop",
|
||||
custom_instructions=custom_instructions,
|
||||
)
|
||||
|
||||
expected_prompt = textwrap.dedent("""\
|
||||
Custom instructions:
|
||||
|
||||
# Past history
|
||||
test plan
|
||||
|
||||
# Plan
|
||||
test plan
|
||||
|
||||
# Finished!
|
||||
test stop""").strip()
|
||||
assert prompt == expected_prompt
|
||||
|
||||
def test_get_llm_backed_user_simulator_prompt_with_persona(self, mocker):
|
||||
mocker.patch(
|
||||
"google.adk.evaluation.simulation.llm_backed_user_simulator_prompts._USER_SIMULATOR_INSTRUCTIONS_WITH_PERSONA_TEMPLATE",
|
||||
_MOCK_PERSONA_TEMPLATE,
|
||||
)
|
||||
prompt = get_llm_backed_user_simulator_prompt(
|
||||
conversation_plan="test plan",
|
||||
conversation_history="test history",
|
||||
stop_signal="test stop",
|
||||
user_persona=sample_persona,
|
||||
)
|
||||
expected_prompt = textwrap.dedent("""\
|
||||
Persona template
|
||||
|
||||
# Persona Description
|
||||
Test persona description
|
||||
|
||||
## Test behavior
|
||||
Test behavior description
|
||||
|
||||
Instructions:
|
||||
* instruction 1
|
||||
* instruction 2
|
||||
|
||||
# Conversation Plan
|
||||
test plan
|
||||
|
||||
# Conversation History
|
||||
test history
|
||||
|
||||
# Stop signal
|
||||
test stop""").strip()
|
||||
assert prompt == expected_prompt
|
||||
|
||||
|
||||
class TestIsValidUserSimulatorTemplate:
|
||||
"""Test cases for is_valid_user_simulator_template."""
|
||||
|
||||
def test_valid_template(self):
|
||||
template = "Hello {{ name }}"
|
||||
params = ["name"]
|
||||
assert is_valid_user_simulator_template(template, params) is True
|
||||
|
||||
def test_invalid_syntax(self):
|
||||
template = "Hello {{ name"
|
||||
params = ["name"]
|
||||
assert is_valid_user_simulator_template(template, params) is False
|
||||
|
||||
def test_missing_parameter(self):
|
||||
template = "Hello"
|
||||
params = ["name"]
|
||||
assert is_valid_user_simulator_template(template, params) is False
|
||||
+184
@@ -0,0 +1,184 @@
|
||||
# Copyright 2026 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import textwrap
|
||||
|
||||
from google.adk.evaluation.simulation.per_turn_user_simulator_quality_prompts import _get_latest_turn_user_simulator_quality_prompt_template
|
||||
from google.adk.evaluation.simulation.per_turn_user_simulator_quality_prompts import _LATEST_TURN_USER_SIMULATOR_EVALUATOR_PROMPT_TEMPLATE
|
||||
from google.adk.evaluation.simulation.per_turn_user_simulator_quality_prompts import _LATEST_TURN_USER_SIMULATOR_WITH_PERSONA_EVALUATOR_PROMPT_TEMPLATE
|
||||
from google.adk.evaluation.simulation.per_turn_user_simulator_quality_prompts import get_per_turn_user_simulator_quality_prompt
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserBehavior
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserPersona
|
||||
|
||||
_MOCK_DEFAULT_TEMPLATE = textwrap.dedent("""\
|
||||
Default template
|
||||
|
||||
# Conversation Plan
|
||||
{{conversation_plan}}
|
||||
|
||||
# Conversation History
|
||||
{{conversation_history}}
|
||||
|
||||
# Generated User Response
|
||||
{{generated_user_response}}
|
||||
|
||||
# Stop signal
|
||||
{{stop_signal}}
|
||||
""").strip()
|
||||
|
||||
_MOCK_PERSONA_TEMPLATE = textwrap.dedent("""\
|
||||
Persona template
|
||||
|
||||
# Persona Description
|
||||
{{persona.description}}
|
||||
{% for b in persona.behaviors %}
|
||||
## Criteria: {{ b.name | render_string_filter}}
|
||||
{{ b.description | render_string_filter}}
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
{{ b.get_violation_rubrics_str() | render_string_filter}}
|
||||
{% endfor %}
|
||||
# Conversation Plan
|
||||
{{conversation_plan}}
|
||||
|
||||
# Conversation History
|
||||
{{conversation_history}}
|
||||
|
||||
# Generated User Response
|
||||
{{generated_user_response}}
|
||||
|
||||
# Stop signal
|
||||
{{stop_signal}}
|
||||
""").strip()
|
||||
|
||||
|
||||
class TestGetLatestTurnUserSimulatorQualityPrompt:
|
||||
"""Test cases for get_latest_turn_user_simulator_quality_prompt."""
|
||||
|
||||
def test_get_get_latest_turn_user_simulator_quality_prompt_template_default(
|
||||
self,
|
||||
):
|
||||
prompt = _get_latest_turn_user_simulator_quality_prompt_template(
|
||||
user_persona=None
|
||||
)
|
||||
assert prompt == _LATEST_TURN_USER_SIMULATOR_EVALUATOR_PROMPT_TEMPLATE
|
||||
|
||||
def test_get_latest_turn_user_simulator_quality_prompt_template_with_persona(
|
||||
self,
|
||||
):
|
||||
"""Tests that the correct prompt is returned when a persona is provided."""
|
||||
persona = UserPersona(
|
||||
id="test_persona",
|
||||
description="Test persona description.",
|
||||
behaviors=[
|
||||
UserBehavior(
|
||||
name="test_behavior",
|
||||
description="Test behavior description.",
|
||||
behavior_instructions=["instruction1"],
|
||||
violation_rubrics=["violation1"],
|
||||
)
|
||||
],
|
||||
)
|
||||
prompt = _get_latest_turn_user_simulator_quality_prompt_template(
|
||||
user_persona=persona
|
||||
)
|
||||
assert (
|
||||
prompt
|
||||
== _LATEST_TURN_USER_SIMULATOR_WITH_PERSONA_EVALUATOR_PROMPT_TEMPLATE
|
||||
)
|
||||
|
||||
|
||||
class TestGetPerTurnUserSimulatorQualityPrompt:
|
||||
"""Test cases for get_per_turn_user_simulator_quality_prompt."""
|
||||
|
||||
def test_get_per_turn_user_simulator_quality_prompt_default(self, mocker):
|
||||
"""Tests that the correct prompt is returned when no persona is provided."""
|
||||
mocker.patch(
|
||||
"google.adk.evaluation.simulation.per_turn_user_simulator_quality_prompts._LATEST_TURN_USER_SIMULATOR_EVALUATOR_PROMPT_TEMPLATE",
|
||||
_MOCK_DEFAULT_TEMPLATE,
|
||||
)
|
||||
prompt = get_per_turn_user_simulator_quality_prompt(
|
||||
conversation_plan="plan",
|
||||
conversation_history="history",
|
||||
generated_user_response="response",
|
||||
stop_signal="stop",
|
||||
user_persona=None,
|
||||
)
|
||||
expected_prompt = textwrap.dedent("""\
|
||||
Default template
|
||||
|
||||
# Conversation Plan
|
||||
plan
|
||||
|
||||
# Conversation History
|
||||
history
|
||||
|
||||
# Generated User Response
|
||||
response
|
||||
|
||||
# Stop signal
|
||||
stop""").strip()
|
||||
assert prompt == expected_prompt
|
||||
|
||||
def test_get_per_turn_user_simulator_quality_prompt_with_persona(
|
||||
self, mocker
|
||||
):
|
||||
"""Tests that the correct prompt is returned when a persona is provided."""
|
||||
mocker.patch(
|
||||
"google.adk.evaluation.simulation.per_turn_user_simulator_quality_prompts._LATEST_TURN_USER_SIMULATOR_WITH_PERSONA_EVALUATOR_PROMPT_TEMPLATE",
|
||||
_MOCK_PERSONA_TEMPLATE,
|
||||
)
|
||||
persona = UserPersona(
|
||||
id="test_persona",
|
||||
description="Test persona description.",
|
||||
behaviors=[
|
||||
UserBehavior(
|
||||
name="test_behavior",
|
||||
description="Test behavior description.",
|
||||
behavior_instructions=["instruction1"],
|
||||
violation_rubrics=["violation1"],
|
||||
)
|
||||
],
|
||||
)
|
||||
prompt = get_per_turn_user_simulator_quality_prompt(
|
||||
conversation_plan="plan",
|
||||
conversation_history="history",
|
||||
generated_user_response="response",
|
||||
stop_signal="stop",
|
||||
user_persona=persona,
|
||||
)
|
||||
expected_prompt = textwrap.dedent("""\
|
||||
Persona template
|
||||
|
||||
# Persona Description
|
||||
Test persona description.
|
||||
|
||||
## Criteria: test_behavior
|
||||
Test behavior description.
|
||||
|
||||
Mark as FAIL if any of the following Violations occur:
|
||||
* violation1
|
||||
|
||||
# Conversation Plan
|
||||
plan
|
||||
|
||||
# Conversation History
|
||||
history
|
||||
|
||||
# Generated User Response
|
||||
response
|
||||
|
||||
# Stop signal
|
||||
stop""").strip()
|
||||
assert prompt == expected_prompt
|
||||
@@ -26,7 +26,10 @@ from google.adk.evaluation.llm_as_judge_utils import Label
|
||||
from google.adk.evaluation.simulation.per_turn_user_simulator_quality_v1 import _format_conversation_history
|
||||
from google.adk.evaluation.simulation.per_turn_user_simulator_quality_v1 import _parse_llm_response
|
||||
from google.adk.evaluation.simulation.per_turn_user_simulator_quality_v1 import PerTurnUserSimulatorQualityV1
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserBehavior
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserPersona
|
||||
from google.adk.models.llm_response import LlmResponse
|
||||
from google.genai import types
|
||||
from google.genai import types as genai_types
|
||||
import pytest
|
||||
|
||||
@@ -149,6 +152,54 @@ def test_parse_llm_response_label_valid(response_text):
|
||||
],
|
||||
"is_valid": "invalid",
|
||||
}
|
||||
```""",
|
||||
"""```json
|
||||
{
|
||||
"criteria": [
|
||||
{
|
||||
"name": "TEST_NAME",
|
||||
"reasoning": "test_resonining",
|
||||
"passes": False
|
||||
}
|
||||
],
|
||||
"is_valid": "almost",
|
||||
}
|
||||
```""",
|
||||
"""```json
|
||||
{
|
||||
"criteria": [
|
||||
{
|
||||
"name": "TEST_NAME",
|
||||
"reasoning": "test_resonining",
|
||||
"passes": False
|
||||
}
|
||||
],
|
||||
"is_valid": "partially_valid",
|
||||
}
|
||||
```""",
|
||||
"""```json
|
||||
{
|
||||
"criteria": [
|
||||
{
|
||||
"name": "TEST_NAME",
|
||||
"reasoning": "test_resonining",
|
||||
"passes": False
|
||||
}
|
||||
],
|
||||
"is_valid": "partially valid",
|
||||
}
|
||||
```""",
|
||||
"""```json
|
||||
{
|
||||
"criteria": [
|
||||
{
|
||||
"name": "TEST_NAME",
|
||||
"reasoning": "test_resonining",
|
||||
"passes": False
|
||||
}
|
||||
],
|
||||
"is_valid": "partially",
|
||||
}
|
||||
```""",
|
||||
],
|
||||
)
|
||||
@@ -158,16 +209,16 @@ def test_parse_llm_response_label_invalid(response_text):
|
||||
|
||||
|
||||
def create_test_template() -> str:
|
||||
return """This is a test template with stop signal: `{stop_signal}`.
|
||||
return """This is a test template with stop signal: `{{stop_signal}}`.
|
||||
|
||||
# Conversation Plan
|
||||
{conversation_plan}
|
||||
{{conversation_plan}}
|
||||
|
||||
# Conversation History
|
||||
{conversation_history}
|
||||
{{conversation_history}}
|
||||
|
||||
# Generated User Response
|
||||
{generated_user_response}
|
||||
{{generated_user_response}}
|
||||
""".strip()
|
||||
|
||||
|
||||
@@ -189,18 +240,19 @@ def _create_test_evaluator(
|
||||
),
|
||||
),
|
||||
)
|
||||
evaluator._prompt_template = create_test_template()
|
||||
return evaluator
|
||||
|
||||
|
||||
def _create_test_conversation_scenario(
|
||||
conversation_plan: str = "test conversation plan",
|
||||
starting_prompt: str = "test starting prompt",
|
||||
user_persona: UserPersona = None,
|
||||
) -> ConversationScenario:
|
||||
"""Returns a ConversationScenario."""
|
||||
return ConversationScenario(
|
||||
starting_prompt=starting_prompt,
|
||||
conversation_plan=conversation_plan,
|
||||
user_persona=user_persona,
|
||||
)
|
||||
|
||||
|
||||
@@ -243,48 +295,28 @@ def _create_test_invocations(
|
||||
return invocations
|
||||
|
||||
|
||||
def test_format_llm_prompt():
|
||||
evaluator = _create_test_evaluator(stop_signal="test stop signal")
|
||||
def test_format_llm_prompt_raises_error_if_previous_invocations_is_none():
|
||||
evaluator = _create_test_evaluator()
|
||||
with pytest.raises(
|
||||
ValueError, match="Previous invocations should have a set value"
|
||||
):
|
||||
evaluator._format_llm_prompt(
|
||||
invocation=_create_test_invocation("1"),
|
||||
conversation_scenario=_create_test_conversation_scenario(),
|
||||
previous_invocations=None,
|
||||
)
|
||||
|
||||
starting_prompt = "first user prompt."
|
||||
conversation_scenario = _create_test_conversation_scenario(
|
||||
conversation_plan="test conversation plan.",
|
||||
starting_prompt=starting_prompt,
|
||||
)
|
||||
invocation_history = _create_test_invocations([
|
||||
starting_prompt,
|
||||
"first agent response.",
|
||||
"second user prompt.",
|
||||
"second agent response.",
|
||||
"third user prompt.",
|
||||
"third agent response.",
|
||||
])
|
||||
|
||||
prompt = evaluator._format_llm_prompt(
|
||||
invocation=invocation_history[-1],
|
||||
conversation_scenario=conversation_scenario,
|
||||
previous_invocations=invocation_history[:-1],
|
||||
)
|
||||
|
||||
assert (
|
||||
prompt == """This is a test template with stop signal: `test stop signal`.
|
||||
|
||||
# Conversation Plan
|
||||
test conversation plan.
|
||||
|
||||
# Conversation History
|
||||
user: first user prompt.
|
||||
|
||||
model: first agent response.
|
||||
|
||||
user: second user prompt.
|
||||
|
||||
model: second agent response.
|
||||
|
||||
# Generated User Response
|
||||
third user prompt.
|
||||
""".strip()
|
||||
)
|
||||
def test_format_llm_prompt_raises_error_if_conversation_scenario_is_none():
|
||||
evaluator = _create_test_evaluator()
|
||||
with pytest.raises(
|
||||
ValueError, match="Conversation scenario should have a set value"
|
||||
):
|
||||
evaluator._format_llm_prompt(
|
||||
invocation=_create_test_invocation("1"),
|
||||
conversation_scenario=None,
|
||||
previous_invocations=[],
|
||||
)
|
||||
|
||||
|
||||
def test_convert_llm_response_to_score_pass():
|
||||
@@ -419,6 +451,19 @@ def test_aggregate_samples_failure():
|
||||
assert aggregation_result.eval_status == EvalStatus.FAILED
|
||||
|
||||
|
||||
def test_format_conversation_history_with_none_values():
|
||||
"""Tests that _format_conversation_history handles None values."""
|
||||
invocations = [
|
||||
Invocation(
|
||||
invocation_id="1",
|
||||
user_content=types.Content(),
|
||||
final_response=None,
|
||||
)
|
||||
]
|
||||
formatted_history = _format_conversation_history(invocations)
|
||||
assert formatted_history == ""
|
||||
|
||||
|
||||
def test_format_conversation_history():
|
||||
conversation_history = [
|
||||
"first user prompt.",
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
# Copyright 2026 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from google.adk.evaluation.simulation import pre_built_personas
|
||||
|
||||
|
||||
def test_get_default_persona_registry():
|
||||
"""Tests that the default persona registry can be loaded."""
|
||||
assert pre_built_personas.DEFAULT_USER_PERSONA_REGISTRY is not None
|
||||
@@ -0,0 +1,133 @@
|
||||
# Copyright 2026 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from google.adk.errors.not_found_error import NotFoundError
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserBehavior
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserPersona
|
||||
from google.adk.evaluation.simulation.user_simulator_personas import UserPersonaRegistry
|
||||
import pytest
|
||||
|
||||
|
||||
class TestUserBehavior:
|
||||
"""Test cases for UserBehavior."""
|
||||
|
||||
def test_create_user_behavior(self):
|
||||
"""Tests UserBehavior creation."""
|
||||
behavior = UserBehavior(
|
||||
name="test_behavior",
|
||||
description="Test behavior description.",
|
||||
behavior_instructions=["instruction1", "instruction2"],
|
||||
violation_rubrics=["violation1", "violation2"],
|
||||
)
|
||||
assert behavior.name == "test_behavior"
|
||||
assert behavior.description == "Test behavior description."
|
||||
assert behavior.behavior_instructions == ["instruction1", "instruction2"]
|
||||
assert behavior.violation_rubrics == ["violation1", "violation2"]
|
||||
|
||||
def test_get_behavior_instructions_str(self):
|
||||
"""Tests get_behavior_instructions_str method."""
|
||||
behavior = UserBehavior(
|
||||
name="test_behavior",
|
||||
description="Test behavior description.",
|
||||
behavior_instructions=["instruction1", "instruction2"],
|
||||
violation_rubrics=[],
|
||||
)
|
||||
assert (
|
||||
behavior.get_behavior_instructions_str()
|
||||
== " * instruction1\n * instruction2"
|
||||
)
|
||||
|
||||
def test_get_violation_rubrics_str(self):
|
||||
"""Tests get_violation_rubrics_str method."""
|
||||
behavior = UserBehavior(
|
||||
name="test_behavior",
|
||||
description="Test behavior description.",
|
||||
behavior_instructions=[],
|
||||
violation_rubrics=["violation1", "violation2"],
|
||||
)
|
||||
assert (
|
||||
behavior.get_violation_rubrics_str() == " * violation1\n * violation2"
|
||||
)
|
||||
|
||||
|
||||
class TestUserPersona:
|
||||
"""Test cases for UserPersona."""
|
||||
|
||||
def test_create_user_persona(self):
|
||||
"""Tests UserPersona creation."""
|
||||
behavior = UserBehavior(
|
||||
name="test_behavior",
|
||||
description="Test behavior description.",
|
||||
behavior_instructions=["instruction1"],
|
||||
violation_rubrics=["violation1"],
|
||||
)
|
||||
persona = UserPersona(
|
||||
id="test_persona",
|
||||
description="Test persona description.",
|
||||
behaviors=[behavior],
|
||||
)
|
||||
assert persona.id == "test_persona"
|
||||
assert persona.description == "Test persona description."
|
||||
assert persona.behaviors == [behavior]
|
||||
|
||||
|
||||
class TestUserPersonaRegistry:
|
||||
"""Test cases for UserPersonaRegistry."""
|
||||
|
||||
def test_register_and_get_persona(self):
|
||||
"""Tests register_persona and get_persona methods."""
|
||||
registry = UserPersonaRegistry()
|
||||
persona = UserPersona(
|
||||
id="test_persona", description="Test persona", behaviors=[]
|
||||
)
|
||||
registry.register_persona("persona1", persona)
|
||||
assert registry.get_persona("persona1") == persona
|
||||
|
||||
def test_get_persona_not_found(self):
|
||||
"""Tests get_persona for a non-existent persona."""
|
||||
registry = UserPersonaRegistry()
|
||||
with pytest.raises(NotFoundError, match="persona2 not found in registry."):
|
||||
registry.get_persona("persona2")
|
||||
|
||||
def test_update_persona(self):
|
||||
"""Tests updating an existing persona in the registry."""
|
||||
registry = UserPersonaRegistry()
|
||||
persona1 = UserPersona(
|
||||
id="test_persona1", description="Test persona 1", behaviors=[]
|
||||
)
|
||||
persona2 = UserPersona(
|
||||
id="test_persona2", description="Test persona 2", behaviors=[]
|
||||
)
|
||||
registry.register_persona("persona1", persona1)
|
||||
assert registry.get_persona("persona1") == persona1
|
||||
registry.register_persona("persona1", persona2)
|
||||
assert registry.get_persona("persona1") == persona2
|
||||
|
||||
def test_get_registered_personas(self):
|
||||
"""Tests get_registered_personas method."""
|
||||
registry = UserPersonaRegistry()
|
||||
persona1 = UserPersona(
|
||||
id="test_persona1", description="Test persona 1", behaviors=[]
|
||||
)
|
||||
persona2 = UserPersona(
|
||||
id="test_persona2", description="Test persona 2", behaviors=[]
|
||||
)
|
||||
registry.register_persona("persona1", persona1)
|
||||
registry.register_persona("persona2", persona2)
|
||||
registered_personas = registry.get_registered_personas()
|
||||
assert len(registered_personas) == 2
|
||||
assert persona1 in registered_personas
|
||||
assert persona2 in registered_personas
|
||||
Reference in New Issue
Block a user