mirror of
https://github.com/microsoft/agent-framework.git
synced 2026-06-16 21:04:09 +08:00
Python: Fix Eval samples (#4033)
* fix red team sample * Updated self-reflection * fix for workflow eval sample * fix test
This commit is contained in:
committed by
GitHub
Unverified
parent
6a39d5a652
commit
aab80d9ed9
@@ -1,24 +1,43 @@
|
||||
# Copyright (c) Microsoft. All rights reserved.
|
||||
# type: ignore
|
||||
|
||||
"""
|
||||
Script to run multi-agent travel planning workflow and evaluate agent responses.
|
||||
|
||||
This script:
|
||||
1. Executes the multi-agent workflow
|
||||
2. Displays response data summary
|
||||
3. Creates and runs evaluation with multiple evaluators
|
||||
4. Monitors evaluation progress and displays results
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from azure.ai.projects import AIProjectClient
|
||||
from azure.identity import DefaultAzureCredential
|
||||
from create_workflow import create_and_run_workflow
|
||||
from dotenv import load_dotenv
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from openai import OpenAI
|
||||
from openai.types import EvalCreateResponse
|
||||
from openai.types.evals import RunCreateResponse
|
||||
|
||||
"""
|
||||
Script to run multi-agent travel planning workflow and evaluate agent responses.
|
||||
|
||||
This script:
|
||||
1. Runs the multi-agent travel planning workflow
|
||||
2. Displays a summary of tracked agent responses
|
||||
3. Fetches and previews final agent responses
|
||||
4. Creates an evaluation with multiple evaluators
|
||||
5. Runs the evaluation on selected agent responses
|
||||
6. Monitors evaluation progress and displays results
|
||||
"""
|
||||
|
||||
|
||||
def create_openai_client() -> OpenAI:
|
||||
project_client = AIProjectClient(
|
||||
endpoint=os.environ["AZURE_AI_PROJECT_ENDPOINT"],
|
||||
credential=DefaultAzureCredential(),
|
||||
)
|
||||
return project_client.get_openai_client()
|
||||
|
||||
|
||||
def print_section(title: str):
|
||||
"""Print a formatted section header."""
|
||||
@@ -27,26 +46,26 @@ def print_section(title: str):
|
||||
print(f"{'=' * 80}")
|
||||
|
||||
|
||||
async def run_workflow():
|
||||
async def run_workflow(deployment_name: str | None = None) -> dict[str, Any]:
|
||||
"""Execute the multi-agent travel planning workflow.
|
||||
|
||||
Args:
|
||||
deployment_name: Optional model deployment name for the workflow agents
|
||||
|
||||
Returns:
|
||||
Dictionary containing workflow data with agent response IDs
|
||||
"""
|
||||
print_section("Step 1: Running Workflow")
|
||||
print("Executing multi-agent travel planning workflow...")
|
||||
print("This may take a few minutes...")
|
||||
|
||||
workflow_data = await create_and_run_workflow()
|
||||
workflow_data = await create_and_run_workflow(deployment_name=deployment_name)
|
||||
|
||||
print("Workflow execution completed")
|
||||
return workflow_data
|
||||
|
||||
|
||||
def display_response_summary(workflow_data: dict):
|
||||
def display_response_summary(workflow_data: dict) -> None:
|
||||
"""Display summary of response data."""
|
||||
print_section("Step 2: Response Data Summary")
|
||||
|
||||
print(f"Query: {workflow_data['query']}")
|
||||
print(f"\nAgents tracked: {len(workflow_data['agents'])}")
|
||||
|
||||
@@ -55,10 +74,8 @@ def display_response_summary(workflow_data: dict):
|
||||
print(f" {agent_name}: {response_count} response(s)")
|
||||
|
||||
|
||||
def fetch_agent_responses(openai_client, workflow_data: dict, agent_names: list):
|
||||
def fetch_agent_responses(openai_client: OpenAI, workflow_data: dict[str, Any], agent_names: list[str]) -> None:
|
||||
"""Fetch and display final responses from specified agents."""
|
||||
print_section("Step 3: Fetching Agent Responses")
|
||||
|
||||
for agent_name in agent_names:
|
||||
if agent_name not in workflow_data["agents"]:
|
||||
continue
|
||||
@@ -80,10 +97,9 @@ def fetch_agent_responses(openai_client, workflow_data: dict, agent_names: list)
|
||||
print(f" Error: {e}")
|
||||
|
||||
|
||||
def create_evaluation(openai_client, model_deployment: str):
|
||||
def create_evaluation(openai_client: OpenAI, deployment_name: str | None = "gpt-5.2") -> EvalCreateResponse:
|
||||
"""Create evaluation with multiple evaluators."""
|
||||
print_section("Step 4: Creating Evaluation")
|
||||
|
||||
deployment_name = os.environ.get("AZURE_AI_MODEL_DEPLOYMENT_NAME", deployment_name)
|
||||
data_source_config = {"type": "azure_ai_source", "scenario": "responses"}
|
||||
|
||||
testing_criteria = [
|
||||
@@ -91,25 +107,25 @@ def create_evaluation(openai_client, model_deployment: str):
|
||||
"type": "azure_ai_evaluator",
|
||||
"name": "relevance",
|
||||
"evaluator_name": "builtin.relevance",
|
||||
"initialization_parameters": {"deployment_name": model_deployment}
|
||||
"initialization_parameters": {"deployment_name": deployment_name},
|
||||
},
|
||||
{
|
||||
"type": "azure_ai_evaluator",
|
||||
"name": "groundedness",
|
||||
"evaluator_name": "builtin.groundedness",
|
||||
"initialization_parameters": {"deployment_name": model_deployment}
|
||||
"initialization_parameters": {"deployment_name": deployment_name},
|
||||
},
|
||||
{
|
||||
"type": "azure_ai_evaluator",
|
||||
"name": "tool_call_accuracy",
|
||||
"evaluator_name": "builtin.tool_call_accuracy",
|
||||
"initialization_parameters": {"deployment_name": model_deployment}
|
||||
"initialization_parameters": {"deployment_name": deployment_name},
|
||||
},
|
||||
{
|
||||
"type": "azure_ai_evaluator",
|
||||
"name": "tool_output_utilization",
|
||||
"evaluator_name": "builtin.tool_output_utilization",
|
||||
"initialization_parameters": {"deployment_name": model_deployment}
|
||||
"initialization_parameters": {"deployment_name": deployment_name},
|
||||
},
|
||||
]
|
||||
|
||||
@@ -126,10 +142,10 @@ def create_evaluation(openai_client, model_deployment: str):
|
||||
return eval_object
|
||||
|
||||
|
||||
def run_evaluation(openai_client, eval_object, workflow_data: dict, agent_names: list):
|
||||
def run_evaluation(
|
||||
openai_client: OpenAI, eval_object: EvalCreateResponse, workflow_data: dict[str, Any], agent_names: list[str]
|
||||
) -> RunCreateResponse:
|
||||
"""Run evaluation on selected agent responses."""
|
||||
print_section("Step 5: Running Evaluation")
|
||||
|
||||
selected_response_ids = []
|
||||
for agent_name in agent_names:
|
||||
if agent_name in workflow_data["agents"]:
|
||||
@@ -162,10 +178,8 @@ def run_evaluation(openai_client, eval_object, workflow_data: dict, agent_names:
|
||||
return eval_run
|
||||
|
||||
|
||||
def monitor_evaluation(openai_client, eval_object, eval_run):
|
||||
def monitor_evaluation(openai_client: OpenAI, eval_object: EvalCreateResponse, eval_run: RunCreateResponse):
|
||||
"""Monitor evaluation progress and display results."""
|
||||
print_section("Step 6: Monitoring Evaluation")
|
||||
|
||||
print("Waiting for evaluation to complete...")
|
||||
|
||||
while eval_run.status not in ["completed", "failed"]:
|
||||
@@ -187,29 +201,41 @@ def monitor_evaluation(openai_client, eval_object, eval_run):
|
||||
async def main():
|
||||
"""Main execution flow."""
|
||||
load_dotenv()
|
||||
openai_client = create_openai_client()
|
||||
|
||||
print("Travel Planning Workflow Evaluation")
|
||||
# Model configuration
|
||||
workflow_agent_model = os.environ.get("AZURE_AI_MODEL_DEPLOYMENT_NAME_WORKFLOW", "gpt-4.1-nano")
|
||||
eval_model = os.environ.get("AZURE_AI_MODEL_DEPLOYMENT_NAME_EVAL", "gpt-5.2")
|
||||
|
||||
workflow_data = await run_workflow()
|
||||
# Focus on these agents, uncomment other ones you want to have evals run on
|
||||
agents_to_evaluate = [
|
||||
"hotel-search-agent",
|
||||
"flight-search-agent",
|
||||
"activity-search-agent",
|
||||
# "booking-payment-agent",
|
||||
# "booking-info-aggregation-agent",
|
||||
# "travel-request-handler",
|
||||
# "booking-confirmation-agent",
|
||||
]
|
||||
|
||||
print_section("Travel Planning Workflow Evaluation")
|
||||
|
||||
print_section("Step 1: Running Workflow")
|
||||
workflow_data = await run_workflow(deployment_name=workflow_agent_model)
|
||||
|
||||
print_section("Step 2: Response Data Summary")
|
||||
display_response_summary(workflow_data)
|
||||
|
||||
project_client = AIProjectClient(
|
||||
endpoint=os.environ["AZURE_AI_PROJECT_ENDPOINT"],
|
||||
credential=DefaultAzureCredential(),
|
||||
api_version="2025-11-15-preview"
|
||||
)
|
||||
openai_client = project_client.get_openai_client()
|
||||
|
||||
agents_to_evaluate = ["hotel-search-agent", "flight-search-agent", "activity-search-agent"]
|
||||
|
||||
print_section("Step 3: Fetching Agent Responses")
|
||||
fetch_agent_responses(openai_client, workflow_data, agents_to_evaluate)
|
||||
|
||||
model_deployment = os.environ.get("AZURE_AI_MODEL_DEPLOYMENT_NAME", "gpt-4o-mini")
|
||||
eval_object = create_evaluation(openai_client, model_deployment)
|
||||
print_section("Step 4: Creating Evaluation")
|
||||
eval_object = create_evaluation(openai_client, deployment_name=eval_model)
|
||||
|
||||
print_section("Step 5: Running Evaluation")
|
||||
eval_run = run_evaluation(openai_client, eval_object, workflow_data, agents_to_evaluate)
|
||||
|
||||
print_section("Step 6: Monitoring Evaluation")
|
||||
monitor_evaluation(openai_client, eval_object, eval_run)
|
||||
|
||||
print_section("Complete")
|
||||
|
||||
Reference in New Issue
Block a user