diff --git a/dotnet/agent-framework-dotnet.slnx b/dotnet/agent-framework-dotnet.slnx
index e395627bc9..cf3cc1c773 100644
--- a/dotnet/agent-framework-dotnet.slnx
+++ b/dotnet/agent-framework-dotnet.slnx
@@ -403,6 +403,7 @@
+
diff --git a/dotnet/samples/02-agents/Evaluation/Evaluation_ExpectedOutputs/README.md b/dotnet/samples/02-agents/Evaluation/Evaluation_ExpectedOutputs/README.md
index 34f16865d2..77a2df6426 100644
--- a/dotnet/samples/02-agents/Evaluation/Evaluation_ExpectedOutputs/README.md
+++ b/dotnet/samples/02-agents/Evaluation/Evaluation_ExpectedOutputs/README.md
@@ -31,3 +31,4 @@ dotnet run --project .\Evaluation_ExpectedOutputs
- [Evaluation_SimpleEval](../Evaluation_SimpleEval/) — Simplest evaluation with built-in and custom checks
- [Evaluation_FoundryQuality](../../../05-end-to-end/Evaluation/Evaluation_FoundryQuality/) — Cloud-based quality evaluation with Foundry evaluators
+- [Evaluation_FoundryRubric](../../../05-end-to-end/Evaluation/Evaluation_FoundryRubric/) — Rubric (adaptive) evaluators with per-dimension scores
diff --git a/dotnet/samples/02-agents/Evaluation/Evaluation_Multimodal/README.md b/dotnet/samples/02-agents/Evaluation/Evaluation_Multimodal/README.md
index d02447651b..0b7dabc808 100644
--- a/dotnet/samples/02-agents/Evaluation/Evaluation_Multimodal/README.md
+++ b/dotnet/samples/02-agents/Evaluation/Evaluation_Multimodal/README.md
@@ -26,4 +26,5 @@ dotnet run --project .\Evaluation_Multimodal
- [Evaluation_SimpleEval](../Evaluation_SimpleEval/) — Simplest evaluation with built-in checks and `agent.EvaluateAsync()`
- [Evaluation_FoundryQuality](../../../05-end-to-end/Evaluation/Evaluation_FoundryQuality/) — Cloud-based quality evaluation with Foundry evaluators
+- [Evaluation_FoundryRubric](../../../05-end-to-end/Evaluation/Evaluation_FoundryRubric/) — Rubric (adaptive) evaluators with per-dimension scores
- [Evaluation_ConversationSplits](../../../05-end-to-end/Evaluation/Evaluation_ConversationSplits/) — Multi-turn conversation split strategies
diff --git a/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/Evaluation_FoundryRubric.csproj b/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/Evaluation_FoundryRubric.csproj
new file mode 100644
index 0000000000..6b4cb8f43e
--- /dev/null
+++ b/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/Evaluation_FoundryRubric.csproj
@@ -0,0 +1,15 @@
+
+
+
+ Exe
+ net10.0
+
+ enable
+ enable
+
+
+
+
+
+
+
diff --git a/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/Program.cs b/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/Program.cs
new file mode 100644
index 0000000000..5e1aef1736
--- /dev/null
+++ b/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/Program.cs
@@ -0,0 +1,141 @@
+// Copyright (c) Microsoft. All rights reserved.
+
+// This sample evaluates a pre-existing Azure AI Foundry agent against a rubric evaluator
+// that was authored in the Foundry portal.
+//
+// Rubric evaluators are LLM-as-judge evaluators with custom scoring dimensions you define
+// for your domain. agent-framework consumes pre-existing rubric evaluators — they are
+// authored in the Foundry portal (or via the dedicated SDK / REST surface) and referenced
+// here by name and version.
+//
+// Prerequisites:
+// - An Azure AI Foundry project with a deployed model.
+// - A registered Foundry agent in that project (the rubric was created against this agent).
+// - A rubric evaluator already created in the Foundry portal.
+// - .env (or environment) populated with the FOUNDRY_* variables below.
+//
+// IMPORTANT: FOUNDRY_PROJECT_ENDPOINT must be the project-scoped URL
+// https://.services.ai.azure.com/api/projects/
+// A bare Azure OpenAI endpoint silently fails eval submission with HTTP 500.
+
+using Azure.AI.Projects;
+using Azure.AI.Projects.Agents;
+using Azure.Identity;
+using Microsoft.Agents.AI;
+using Microsoft.Agents.AI.Foundry;
+using FoundryEvals = Microsoft.Agents.AI.Foundry.FoundryEvals;
+
+string projectEndpoint = Environment.GetEnvironmentVariable("FOUNDRY_PROJECT_ENDPOINT")
+ ?? throw new InvalidOperationException("FOUNDRY_PROJECT_ENDPOINT is not set.");
+string model = Environment.GetEnvironmentVariable("FOUNDRY_MODEL")
+ ?? throw new InvalidOperationException("FOUNDRY_MODEL is not set.");
+string agentName = Environment.GetEnvironmentVariable("FOUNDRY_AGENT_NAME")
+ ?? throw new InvalidOperationException("FOUNDRY_AGENT_NAME is not set.");
+string? agentVersion = Environment.GetEnvironmentVariable("FOUNDRY_AGENT_VERSION");
+string rubricName = Environment.GetEnvironmentVariable("FOUNDRY_RUBRIC_NAME")
+ ?? throw new InvalidOperationException("FOUNDRY_RUBRIC_NAME is not set.");
+string? rubricVersion = Environment.GetEnvironmentVariable("FOUNDRY_RUBRIC_VERSION");
+
+// WARNING: DefaultAzureCredential is convenient for development but requires careful
+// consideration in production. Prefer ManagedIdentityCredential (or a specific credential)
+// to avoid latency, unintended credential probing, and fallback security risks.
+AIProjectClient projectClient = new(new Uri(projectEndpoint), new DefaultAzureCredential());
+
+// 1. Connect to the pre-existing Foundry agent the rubric was created against.
+FoundryAgent agent;
+if (agentVersion is null)
+{
+ ProjectsAgentRecord agentRecord = await projectClient.AgentAdministrationClient.GetAgentAsync(agentName);
+ agent = projectClient.AsAIAgent(agentRecord);
+}
+else
+{
+ ProjectsAgentVersion versionRecord = await projectClient.AgentAdministrationClient.GetAgentVersionAsync(agentName, agentVersion);
+ agent = projectClient.AsAIAgent(versionRecord);
+}
+
+// 2. Reference the pre-existing rubric evaluator by name + version.
+// Always pin a version for reproducible CI runs; a versionless ref resolves to the
+// current version at run time and emits a Trace.TraceWarning on each criterion build.
+GeneratedEvaluatorRef rubric = rubricVersion is null
+ ? GeneratedEvaluatorRef.Latest(rubricName)
+ : new GeneratedEvaluatorRef(rubricName, rubricVersion);
+
+// 3. Mix the rubric with built-in evaluators in a single FoundryEvals config.
+// The implicit conversion lets you pass strings and refs interchangeably.
+FoundryEvals evals = new(
+ projectClient,
+ model,
+ rubric,
+ FoundryEvals.Relevance,
+ FoundryEvals.Coherence);
+
+// 4. Run two example queries against the agent and evaluate the outputs in one call.
+string[] queries =
+[
+ "What's the weather like in Seattle?",
+ "Should I bring an umbrella to London tomorrow?",
+];
+
+Console.WriteLine(new string('=', 60));
+Console.WriteLine($"Evaluating '{agent.Name}' with rubric '{rubricName}' (version {rubricVersion ?? "latest"})");
+Console.WriteLine(new string('=', 60));
+
+AgentEvaluationResults results = await agent.EvaluateAsync(queries, evals);
+
+Console.WriteLine($"Status: {results.Status}");
+Console.WriteLine($"Results: {results.Passed}/{results.Total} passed");
+if (results.ReportUrl is not null)
+{
+ Console.WriteLine($"Portal: {results.ReportUrl}");
+}
+
+Console.WriteLine(results.Passed == results.Total ? "[PASS] All passed" : $"[FAIL] {results.Failed} failed");
+
+// 5. Print per-dimension breakdown for each evaluated item — this is the unique value
+// of a rubric evaluator over the built-in numeric ones.
+Console.WriteLine();
+Console.WriteLine(new string('=', 60));
+Console.WriteLine("Per-dimension scores");
+Console.WriteLine(new string('=', 60));
+
+if (results.DetailedItems is { Count: > 0 })
+{
+ for (int i = 0; i < results.DetailedItems.Count; i++)
+ {
+ EvalItemResult item = results.DetailedItems[i];
+ Console.WriteLine($"Item {i + 1}{(i < queries.Length ? $" — \"{queries[i]}\"" : string.Empty)}");
+
+ foreach (EvalScoreResult score in item.Scores)
+ {
+ Console.WriteLine($" {score.Name}: {score.Score:F1}{(score.Passed is bool p ? (p ? " (pass)" : " (fail)") : string.Empty)}");
+ if (score.Dimensions is { Count: > 0 } dims)
+ {
+ foreach (RubricScore d in dims)
+ {
+ string scoreStr = d.Score is int s ? s.ToString() : "n/a";
+ Console.WriteLine($" - {d.Id}: {scoreStr} (weight={d.Weight}, applicable={d.Applicable})");
+ }
+ }
+ }
+
+ Console.WriteLine();
+ }
+}
+
+// 6. CI quality gate — fail the build if a critical dimension drops below threshold.
+// Replace "general_quality" with whatever dimension id your rubric actually defines.
+Console.WriteLine(new string('=', 60));
+Console.WriteLine("Per-dimension quality gate");
+Console.WriteLine(new string('=', 60));
+
+try
+{
+ results.AssertDimensionScoreAtLeast("general_quality", minScore: 3.0, evaluator: rubricName, requireApplicable: true);
+ Console.WriteLine($"[PASS] {results.ProviderName}: general_quality >= 3 on every item");
+}
+catch (InvalidOperationException ex)
+{
+ Console.WriteLine($"[FAIL] {results.ProviderName}: dimension gate tripped: {ex.Message}");
+ System.Environment.ExitCode = 1;
+}
diff --git a/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/README.md b/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/README.md
new file mode 100644
index 0000000000..a05cfca5e6
--- /dev/null
+++ b/dotnet/samples/05-end-to-end/Evaluation/Evaluation_FoundryRubric/README.md
@@ -0,0 +1,55 @@
+# Evaluation — Foundry Rubric
+
+This sample evaluates a pre-existing Azure AI Foundry agent against a **rubric evaluator**
+authored in the Foundry portal. Rubric evaluators are LLM-as-judge evaluators with custom
+scoring dimensions you define for your domain; agent-framework references them by name and
+version, mixes them with built-in evaluators, and exposes per-dimension scores you can gate
+CI on.
+
+## What this sample demonstrates
+
+- Connecting to a pre-existing Foundry agent (`AgentAdministrationClient.GetAgentAsync`).
+- Referencing a pre-existing rubric evaluator via `GeneratedEvaluatorRef(name, version)`.
+- Mixing the rubric with built-in evaluators (`Relevance`, `Coherence`) in one
+ `FoundryEvals` run.
+- Reading per-dimension breakdowns from `EvalScoreResult.Dimensions`.
+- Gating CI on a per-dimension threshold via
+ `AgentEvaluationResults.AssertDimensionScoreAtLeast(...)`.
+
+## Prerequisites
+
+- .NET 10 SDK or later.
+- Azure CLI installed and authenticated (`az login`).
+- An Azure AI Foundry project with a deployed model.
+- A registered Foundry agent in that project (the agent the rubric was created against).
+- A rubric evaluator created in the Foundry portal. Creating rubrics through the portal
+ currently requires picking a Foundry agent as the generation context, so this
+ prerequisite is implied by having a rubric at all.
+
+> [!IMPORTANT]
+> `FOUNDRY_PROJECT_ENDPOINT` **must** be the project-scoped URL
+> `https://.services.ai.azure.com/api/projects/`. A bare Azure OpenAI
+> endpoint silently fails eval submission with HTTP 500.
+
+> [!NOTE]
+> An **Eval Definition** (a saved bundle of testing_criteria with `"object": "eval"`) is
+> not the same as a **Rubric Evaluator** (a standalone evaluator with dimensions, weights,
+> and a version). `GeneratedEvaluatorRef` points at the latter.
+
+## Environment variables
+
+```powershell
+$env:FOUNDRY_PROJECT_ENDPOINT="https://your-resource.services.ai.azure.com/api/projects/your-project"
+$env:FOUNDRY_MODEL="gpt-4o-mini"
+$env:FOUNDRY_AGENT_NAME="your-agent-name"
+$env:FOUNDRY_AGENT_VERSION="1" # optional; omit for latest
+$env:FOUNDRY_RUBRIC_NAME="your-rubric-name"
+$env:FOUNDRY_RUBRIC_VERSION="1" # optional; omit for latest (CI: pin this)
+```
+
+## Run the sample
+
+```powershell
+cd dotnet/samples/05-end-to-end/Evaluation
+dotnet run --project .\Evaluation_FoundryRubric
+```
diff --git a/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalConverter.cs b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalConverter.cs
index 0754e2bc76..75fe99db74 100644
--- a/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalConverter.cs
+++ b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalConverter.cs
@@ -148,19 +148,68 @@ internal static WireEvalItemPayload ConvertEvalItem(EvalItem item, IConversation
///
/// Builds the testing_criteria array for evals.create().
///
- /// Evaluator names (short or fully-qualified).
+ ///
+ /// Evaluator specs — built-in evaluator names (short or fully-qualified) and/or
+ /// instances for pre-existing rubric evaluators.
+ ///
/// Model deployment name for the LLM judge.
///
/// Whether to include field-level data mapping (required for JSONL data source).
///
+ ///
+ /// Whether the mapped data items include tool definitions. Used to add a
+ /// tool_definitions mapping entry for rubric evaluators (built-in evaluators
+ /// derive this from their own membership).
+ ///
internal static List BuildTestingCriteria(
- IEnumerable evaluators,
+ IEnumerable evaluators,
string model,
- bool includeDataMapping = false)
+ bool includeDataMapping = false,
+ bool includeToolDefinitions = false)
{
var criteria = new List();
- foreach (var name in evaluators)
+ foreach (var spec in evaluators)
{
+ if (spec.IsRubric)
+ {
+ var @ref = spec.GeneratedRef!;
+ Dictionary? refMapping = null;
+ if (includeDataMapping)
+ {
+ // Rubric evaluators accept conversation arrays like agent evaluators,
+ // plus tool_definitions when items are tool-aware.
+ refMapping = new Dictionary
+ {
+ ["query"] = "{{item.query_messages}}",
+ ["response"] = "{{item.response_messages}}",
+ };
+
+ if (includeToolDefinitions)
+ {
+ refMapping["tool_definitions"] = "{{item.tool_definitions}}";
+ }
+ }
+
+ criteria.Add(new WireTestingCriterion
+ {
+ Name = @ref.DisplayName ?? @ref.Name,
+ EvaluatorName = @ref.Name,
+ EvaluatorVersion = @ref.Version,
+ InitializationParameters = new WireInitParams { DeploymentName = model },
+ DataMapping = refMapping,
+ });
+
+ if (@ref.Version is null)
+ {
+ System.Diagnostics.Trace.TraceWarning(
+ "GeneratedEvaluatorRef '{0}' has no pinned version; the eval run will resolve to whichever version is current at execution time. Pin the version for reproducible runs.",
+ @ref.Name);
+ }
+
+ continue;
+ }
+
+ var name = spec.BuiltinName!;
var qualified = ResolveEvaluator(name);
var shortName = name.StartsWith("builtin.", StringComparison.Ordinal)
? name.Substring("builtin.".Length)
@@ -248,8 +297,12 @@ internal static WireItemSchema BuildItemSchema(bool hasContext = false, bool has
/// Returns the subset of that require a ground-truth
/// (reference) value but cannot be evaluated because no item provided one.
///
+ ///
+ /// Rubric references () are skipped — they are not
+ /// ground-truth–dependent on the wire.
+ ///
internal static List FindMissingGroundTruthEvaluators(
- IEnumerable evaluators,
+ IEnumerable evaluators,
bool hasGroundTruth)
{
if (hasGroundTruth)
@@ -258,8 +311,14 @@ internal static List FindMissingGroundTruthEvaluators(
}
var missing = new List();
- foreach (var name in evaluators)
+ foreach (var spec in evaluators)
{
+ if (spec.IsRubric)
+ {
+ continue;
+ }
+
+ var name = spec.BuiltinName!;
if (GroundTruthEvaluators.Contains(ResolveEvaluator(name)))
{
missing.Add(name);
diff --git a/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalWireModels.cs b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalWireModels.cs
index c05232575c..f08b7e04f3 100644
--- a/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalWireModels.cs
+++ b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvalWireModels.cs
@@ -137,6 +137,9 @@ internal sealed class WireTestingCriterion
[JsonPropertyName("evaluator_name")]
public required string EvaluatorName { get; init; }
+ [JsonPropertyName("evaluator_version")]
+ public string? EvaluatorVersion { get; init; }
+
[JsonPropertyName("initialization_parameters")]
public required WireInitParams InitializationParameters { get; init; }
diff --git a/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvals.cs b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvals.cs
index 675ae38dfe..d1542fc4df 100644
--- a/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvals.cs
+++ b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvals.cs
@@ -43,7 +43,7 @@ public sealed class FoundryEvals : IAgentEvaluator
private readonly EvaluationClient _evaluationClient;
private readonly string _model;
- private readonly string[] _evaluatorNames;
+ private readonly FoundryEvaluatorSpec[] _evaluators;
private readonly IConversationSplitter? _splitter;
private readonly double _pollIntervalSeconds = 5.0;
private readonly double _timeoutSeconds = 300.0;
@@ -58,17 +58,21 @@ public sealed class FoundryEvals : IAgentEvaluator
/// The Azure AI Foundry project client.
/// Model deployment name for the LLM judge evaluator.
///
- /// Names of evaluators to use (e.g., , ).
- /// When empty, defaults to relevance and coherence.
+ /// Evaluator specs to use. Each entry can be a built-in evaluator name (string, for example
+ /// ) or a for a rubric evaluator
+ /// already registered in the Foundry project. When empty, defaults to relevance, coherence,
+ /// and task adherence.
///
- public FoundryEvals(AIProjectClient projectClient, string model, params string[] evaluators)
+ public FoundryEvals(AIProjectClient projectClient, string model, params FoundryEvaluatorSpec[] evaluators)
{
ArgumentNullException.ThrowIfNull(projectClient);
ArgumentException.ThrowIfNullOrWhiteSpace(model);
+ ArgumentNullException.ThrowIfNull(evaluators);
+ EnsureAllSpecsValid(evaluators, nameof(evaluators));
this._evaluationClient = projectClient.GetProjectOpenAIClient().GetEvaluationClient();
this._model = model;
- this._evaluatorNames = evaluators.Length > 0
+ this._evaluators = evaluators.Length > 0
? evaluators
: [Relevance, Coherence, TaskAdherence];
}
@@ -84,14 +88,14 @@ public FoundryEvals(AIProjectClient projectClient, string model, params string[]
/// or a custom implementation.
///
///
- /// Names of evaluators to use (e.g., , ).
- /// When empty, defaults to relevance and coherence.
+ /// Evaluator specs (built-in names and/or instances).
+ /// When empty, defaults to relevance, coherence, and task adherence.
///
public FoundryEvals(
AIProjectClient projectClient,
string model,
IConversationSplitter? splitter,
- params string[] evaluators)
+ params FoundryEvaluatorSpec[] evaluators)
: this(projectClient, model, evaluators)
{
this._splitter = splitter;
@@ -107,14 +111,16 @@ public FoundryEvals(
///
/// Seconds between status polls (default 5).
/// Maximum seconds to wait for completion (default 300).
- /// Evaluator names to use.
+ ///
+ /// Evaluator specs (built-in names and/or instances).
+ ///
public FoundryEvals(
AIProjectClient projectClient,
string model,
IConversationSplitter? splitter,
double pollIntervalSeconds,
double timeoutSeconds,
- params string[] evaluators)
+ params FoundryEvaluatorSpec[] evaluators)
: this(projectClient, model, splitter, evaluators)
{
ArgumentOutOfRangeException.ThrowIfLessThanOrEqual(pollIntervalSeconds, 0);
@@ -123,6 +129,81 @@ public FoundryEvals(
this._timeoutSeconds = timeoutSeconds;
}
+ // -----------------------------------------------------------------------
+ // string[] constructor overloads (source-compat with older API that took
+ // `params string[] evaluators` before FoundryEvaluatorSpec was introduced).
+ // `params` is intentionally omitted to avoid overload ambiguity with the
+ // spec-based ctors at zero-args; individual string literals still resolve
+ // through `params FoundryEvaluatorSpec[]` via implicit conversion.
+ // -----------------------------------------------------------------------
+
+ ///
+ /// Initializes a new instance of the class using built-in evaluator
+ /// names. Preserves source compatibility for callers that pass a array.
+ ///
+ /// The Azure AI Foundry project client.
+ /// Model deployment name for the LLM judge evaluator.
+ /// Built-in evaluator names (for example ).
+ public FoundryEvals(AIProjectClient projectClient, string model, string[] evaluators)
+ : this(projectClient, model, ToSpecs(evaluators))
+ {
+ }
+
+ ///
+ /// Initializes a new instance of the class with a splitter and
+ /// built-in evaluator names.
+ ///
+ /// The Azure AI Foundry project client.
+ /// Model deployment name for the LLM judge evaluator.
+ /// Default conversation splitter for multi-turn conversations.
+ /// Built-in evaluator names.
+ public FoundryEvals(
+ AIProjectClient projectClient,
+ string model,
+ IConversationSplitter? splitter,
+ string[] evaluators)
+ : this(projectClient, model, splitter, ToSpecs(evaluators))
+ {
+ }
+
+ ///
+ /// Initializes a new instance of the class with full configuration
+ /// and built-in evaluator names.
+ ///
+ /// The Azure AI Foundry project client.
+ /// Model deployment name for the LLM judge evaluator.
+ /// Default conversation splitter for multi-turn conversations.
+ /// Seconds between status polls.
+ /// Maximum seconds to wait for completion.
+ /// Built-in evaluator names.
+ public FoundryEvals(
+ AIProjectClient projectClient,
+ string model,
+ IConversationSplitter? splitter,
+ double pollIntervalSeconds,
+ double timeoutSeconds,
+ string[] evaluators)
+ : this(projectClient, model, splitter, pollIntervalSeconds, timeoutSeconds, ToSpecs(evaluators))
+ {
+ }
+
+ private static FoundryEvaluatorSpec[] ToSpecs(string[]? evaluators)
+ {
+ if (evaluators is null || evaluators.Length == 0)
+ {
+ return [];
+ }
+
+ var specs = new FoundryEvaluatorSpec[evaluators.Length];
+ for (int i = 0; i < evaluators.Length; i++)
+ {
+ specs[i] = evaluators[i]
+ ?? throw new ArgumentException($"Evaluator name at index {i} is null.", nameof(evaluators));
+ }
+
+ return specs;
+ }
+
// -----------------------------------------------------------------------
// IAgentEvaluator
// -----------------------------------------------------------------------
@@ -149,10 +230,10 @@ public async Task EvaluateAsync(
bool allHaveGroundTruth = payloads.Count > 0 && payloads.All(p => p.GroundTruth is not null);
// Filter out tool evaluators if no items have tools; auto-add ToolCallAccuracy if tools present
- var evaluators = FilterToolEvaluators(this._evaluatorNames, hasTools);
- if (hasTools && !evaluators.Any(e => FoundryEvalConverter.ToolEvaluators.Contains(FoundryEvalConverter.ResolveEvaluator(e))))
+ var evaluators = FilterToolEvaluators(this._evaluators, hasTools);
+ if (hasTools && !HasToolEvaluator(evaluators))
{
- evaluators = [.. evaluators, ToolCallAccuracy];
+ evaluators = [.. evaluators, (FoundryEvaluatorSpec)ToolCallAccuracy];
}
// Fail fast if a ground-truth evaluator (e.g. similarity) is requested but not
@@ -178,7 +259,7 @@ public async Task EvaluateAsync(
ItemSchema = FoundryEvalConverter.BuildItemSchema(hasContext, hasTools, hasGroundTruth),
},
TestingCriteria = FoundryEvalConverter.BuildTestingCriteria(
- evaluators, this._model, includeDataMapping: true),
+ evaluators, this._model, includeDataMapping: true, includeToolDefinitions: hasTools),
};
var createEvalJson = JsonSerializer.Serialize(createEvalPayload, s_jsonOptions);
@@ -270,6 +351,47 @@ public async Task EvaluateAsync(
// Static evaluation methods (traces and targets)
// -----------------------------------------------------------------------
+ ///
+ /// Source-compat overload of
+ /// that accepts a array of built-in evaluator names.
+ ///
+ /// The Azure AI Foundry project client.
+ /// Model deployment name for the LLM judge evaluator.
+ /// Evaluate specific Responses API response IDs.
+ /// Evaluate specific OTel trace IDs from App Insights.
+ /// Filter traces by agent ID (used with ).
+ /// Hours of trace history to evaluate.
+ /// Built-in evaluator names. Each is wrapped via .
+ /// Display name for the evaluation.
+ /// Seconds between status polls.
+ /// Maximum seconds to wait for completion.
+ /// Cancellation token.
+ /// Evaluation results with status, report URL, and per-item details.
+ public static Task EvaluateTracesAsync(
+ AIProjectClient projectClient,
+ string model,
+ IEnumerable? responseIds,
+ IEnumerable? traceIds,
+ string? agentId,
+ int lookbackHours,
+ string[]? evaluators = null,
+ string evalName = "Agent Framework Trace Eval",
+ double pollIntervalSeconds = 5.0,
+ double timeoutSeconds = 300.0,
+ CancellationToken cancellationToken = default)
+ => EvaluateTracesAsync(
+ projectClient,
+ model,
+ responseIds,
+ traceIds,
+ agentId,
+ lookbackHours,
+ ToSpecs(evaluators) is { Length: > 0 } specs ? specs : null,
+ evalName,
+ pollIntervalSeconds,
+ timeoutSeconds,
+ cancellationToken);
+
///
/// Evaluates agent behavior from Responses API response IDs, OTel traces, or agent activity.
///
@@ -287,7 +409,11 @@ public async Task EvaluateAsync(
/// Evaluate specific OTel trace IDs from App Insights.
/// Filter traces by agent ID (used with ).
/// Hours of trace history to evaluate (default 24).
- /// Evaluator names. Defaults to relevance, coherence, and task adherence.
+ ///
+ /// Evaluator specs. Each entry can be a built-in evaluator name (string) or a
+ /// for a rubric evaluator. Defaults to relevance,
+ /// coherence, and task adherence.
+ ///
/// Display name for the evaluation.
/// Seconds between status polls (default 5).
/// Maximum seconds to wait for completion (default 300).
@@ -300,7 +426,7 @@ public static async Task EvaluateTracesAsync(
IEnumerable? traceIds = null,
string? agentId = null,
int lookbackHours = 24,
- string[]? evaluators = null,
+ FoundryEvaluatorSpec[]? evaluators = null,
string evalName = "Agent Framework Trace Eval",
double pollIntervalSeconds = 5.0,
double timeoutSeconds = 300.0,
@@ -320,9 +446,10 @@ public static async Task EvaluateTracesAsync(
}
var evalClient = projectClient.GetProjectOpenAIClient().GetEvaluationClient();
- var resolvedEvaluators = evaluators is { Length: > 0 }
+ FoundryEvaluatorSpec[] resolvedEvaluators = evaluators is { Length: > 0 }
? evaluators
: [Relevance, Coherence, TaskAdherence];
+ EnsureAllSpecsValid(resolvedEvaluators, nameof(evaluators));
// Create the evaluation definition with the appropriate data source scenario
object dataSourceConfig;
@@ -429,6 +556,41 @@ public static async Task EvaluateTracesAsync(
};
}
+ ///
+ /// Source-compat overload of
+ /// that accepts a array of built-in evaluator names.
+ ///
+ /// The Azure AI Foundry project client.
+ /// Model deployment name for the LLM judge evaluator.
+ /// Target configuration (must include a "type" key).
+ /// Queries for Foundry to send to the target.
+ /// Built-in evaluator names. Each is wrapped via .
+ /// Display name for the evaluation.
+ /// Seconds between status polls.
+ /// Maximum seconds to wait for completion.
+ /// Cancellation token.
+ /// Evaluation results with status, report URL, and per-item details.
+ public static Task EvaluateFoundryTargetAsync(
+ AIProjectClient projectClient,
+ string model,
+ IDictionary target,
+ IEnumerable testQueries,
+ string[]? evaluators = null,
+ string evalName = "Agent Framework Target Eval",
+ double pollIntervalSeconds = 5.0,
+ double timeoutSeconds = 300.0,
+ CancellationToken cancellationToken = default)
+ => EvaluateFoundryTargetAsync(
+ projectClient,
+ model,
+ target,
+ testQueries,
+ ToSpecs(evaluators) is { Length: > 0 } specs ? specs : null,
+ evalName,
+ pollIntervalSeconds,
+ timeoutSeconds,
+ cancellationToken);
+
///
/// Evaluates a Foundry-registered agent or model deployment.
///
@@ -440,7 +602,10 @@ public static async Task EvaluateTracesAsync(
/// Model deployment name for the LLM judge evaluator.
/// Target configuration (must include a "type" key, e.g. "azure_ai_agent").
/// Queries for Foundry to send to the target.
- /// Evaluator names. Defaults to relevance, coherence, and task adherence.
+ ///
+ /// Evaluator specs (built-in names and/or instances).
+ /// Defaults to relevance, coherence, and task adherence.
+ ///
/// Display name for the evaluation.
/// Seconds between status polls (default 5).
/// Maximum seconds to wait for completion (default 300).
@@ -451,7 +616,7 @@ public static async Task EvaluateFoundryTargetAsync(
string model,
IDictionary target,
IEnumerable testQueries,
- string[]? evaluators = null,
+ FoundryEvaluatorSpec[]? evaluators = null,
string evalName = "Agent Framework Target Eval",
double pollIntervalSeconds = 5.0,
double timeoutSeconds = 300.0,
@@ -473,9 +638,10 @@ public static async Task EvaluateFoundryTargetAsync(
}
var evalClient = projectClient.GetProjectOpenAIClient().GetEvaluationClient();
- var resolvedEvaluators = evaluators is { Length: > 0 }
+ FoundryEvaluatorSpec[] resolvedEvaluators = evaluators is { Length: > 0 }
? evaluators
: [Relevance, Coherence, TaskAdherence];
+ EnsureAllSpecsValid(resolvedEvaluators, nameof(evaluators));
var createEvalPayload = new WireCreateEvalRequest
{
@@ -831,7 +997,13 @@ private static EvalItemResult ParseDetailedItem(JsonElement outputItem)
passed = pp.ValueKind == JsonValueKind.True;
}
- scores.Add(new EvalScoreResult(name, score, passed));
+ IReadOnlyList? dimensions = null;
+ if (r.TryGetProperty("sample", out var perResultSample))
+ {
+ dimensions = ParseRubricScores(perResultSample);
+ }
+
+ scores.Add(new EvalScoreResult(name, score, passed) { Dimensions = dimensions });
}
}
@@ -917,20 +1089,202 @@ private static EvalItemResult ParseDetailedItem(JsonElement outputItem)
return result;
}
- internal static string[] FilterToolEvaluators(string[] evaluators, bool hasTools)
+ private static readonly string[] s_rubricDimensionKeys = ["dimension_scores", "rubric_scores"];
+
+ ///
+ /// Extracts the per-dimension list from a result-level sample
+ /// payload, when present. Accepts several legacy/canonical shapes for forward compatibility
+ /// with provider SDK changes:
+ ///
+ ///
+ ///
+ /// -
+ /// sample.properties.dimension_scores (canonical Foundry shape).
+ ///
+ /// -
+ /// sample.properties.rubric_scores (preview / legacy key).
+ ///
+ /// -
+ /// Top-level sample.dimension_scores / sample.rubric_scores as a
+ /// defensive fallback.
+ ///
+ ///
+ /// Returns when no rubric scores are present (the evaluator was not
+ /// a rubric evaluator). Malformed entries (missing id, weight, or applicable)
+ /// are skipped without failing the whole list.
+ ///
+ internal static List? ParseRubricScores(JsonElement sample)
+ {
+ if (sample.ValueKind != JsonValueKind.Object)
+ {
+ return null;
+ }
+
+ // Prefer sample.properties. then fall back to top-level sample..
+ if (sample.TryGetProperty("properties", out var properties)
+ && properties.ValueKind == JsonValueKind.Object)
+ {
+ foreach (var key in s_rubricDimensionKeys)
+ {
+ if (properties.TryGetProperty(key, out var raw))
+ {
+ var parsed = ParseDimensionEntries(raw);
+ if (parsed.Count > 0)
+ {
+ return parsed;
+ }
+ }
+ }
+ }
+
+ foreach (var key in s_rubricDimensionKeys)
+ {
+ if (sample.TryGetProperty(key, out var raw))
+ {
+ var parsed = ParseDimensionEntries(raw);
+ if (parsed.Count > 0)
+ {
+ return parsed;
+ }
+ }
+ }
+
+ return null;
+ }
+
+ private static List ParseDimensionEntries(JsonElement raw)
+ {
+ var parsed = new List();
+ if (raw.ValueKind != JsonValueKind.Array)
+ {
+ return parsed;
+ }
+
+ foreach (var entry in raw.EnumerateArray())
+ {
+ if (entry.ValueKind != JsonValueKind.Object)
+ {
+ continue;
+ }
+
+ if (!entry.TryGetProperty("id", out var idProp)
+ || !entry.TryGetProperty("weight", out var weightProp)
+ || !entry.TryGetProperty("applicable", out var applicableProp))
+ {
+ continue;
+ }
+
+ string? id = idProp.ValueKind switch
+ {
+ JsonValueKind.String => idProp.GetString(),
+ JsonValueKind.Number => idProp.GetRawText(),
+ _ => null,
+ };
+ if (string.IsNullOrEmpty(id))
+ {
+ continue;
+ }
+
+ if (weightProp.ValueKind != JsonValueKind.Number
+ || !weightProp.TryGetInt32(out var weight))
+ {
+ continue;
+ }
+
+ if (applicableProp.ValueKind is not (JsonValueKind.True or JsonValueKind.False))
+ {
+ continue;
+ }
+
+ int? score = null;
+ if (entry.TryGetProperty("score", out var scoreProp)
+ && scoreProp.ValueKind == JsonValueKind.Number)
+ {
+ if (scoreProp.TryGetInt32(out var intScore))
+ {
+ score = intScore;
+ }
+ else if (scoreProp.TryGetDouble(out var doubleScore))
+ {
+ score = (int)doubleScore;
+ }
+ }
+
+ string reason = entry.TryGetProperty("reason", out var reasonProp)
+ && reasonProp.ValueKind == JsonValueKind.String
+ ? reasonProp.GetString() ?? string.Empty
+ : string.Empty;
+
+ parsed.Add(new RubricScore(
+ Id: id!,
+ Score: score,
+ Applicable: applicableProp.ValueKind == JsonValueKind.True,
+ Weight: weight,
+ Reason: reason));
+ }
+
+ return parsed;
+ }
+
+ internal static FoundryEvaluatorSpec[] FilterToolEvaluators(FoundryEvaluatorSpec[] evaluators, bool hasTools)
{
if (hasTools)
{
return evaluators;
}
- var filtered = Array.FindAll(evaluators, e =>
- !FoundryEvalConverter.ToolEvaluators.Contains(FoundryEvalConverter.ResolveEvaluator(e)));
+ var filtered = Array.FindAll(evaluators, spec =>
+ {
+ if (spec.IsRubric)
+ {
+ // Rubric refs are tool-aware but not tool-required; preserve them.
+ return true;
+ }
+
+ return !FoundryEvalConverter.ToolEvaluators.Contains(FoundryEvalConverter.ResolveEvaluator(spec.BuiltinName!));
+ });
return filtered.Length > 0
? filtered
: throw new ArgumentException(
"All configured evaluators require tool definitions, but no tool calls were found in the eval items. "
- + $"Tool evaluators: {string.Join(", ", evaluators)}. Either add tool call content to your EvalItems or remove tool-type evaluators.");
+ + $"Tool evaluators: {string.Join(", ", evaluators.Select(e => e.ToString()))}. Either add tool call content to your EvalItems or remove tool-type evaluators.");
+ }
+
+ ///
+ /// Validates every spec in — defensively guards against
+ /// default(FoundryEvaluatorSpec) values that would otherwise NRE deep in the
+ /// dispatch pipeline (e.g. on spec.BuiltinName!).
+ ///
+ internal static void EnsureAllSpecsValid(FoundryEvaluatorSpec[] evaluators, string paramName)
+ {
+ for (int i = 0; i < evaluators.Length; i++)
+ {
+ if (!evaluators[i].IsValid)
+ {
+ throw new ArgumentException(
+ $"Invalid {nameof(FoundryEvaluatorSpec)} at index {i}: must be constructed with either a built-in " +
+ $"evaluator name or a {nameof(GeneratedEvaluatorRef)}. The default struct value is not a valid spec.",
+ paramName);
+ }
+ }
+ }
+
+ private static bool HasToolEvaluator(FoundryEvaluatorSpec[] evaluators)
+ {
+ foreach (var spec in evaluators)
+ {
+ if (spec.IsRubric)
+ {
+ continue;
+ }
+
+ if (FoundryEvalConverter.ToolEvaluators.Contains(FoundryEvalConverter.ResolveEvaluator(spec.BuiltinName!)))
+ {
+ return true;
+ }
+ }
+
+ return false;
}
}
diff --git a/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvaluatorSpec.cs b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvaluatorSpec.cs
new file mode 100644
index 0000000000..0cb9225386
--- /dev/null
+++ b/dotnet/src/Microsoft.Agents.AI.Foundry/Evaluation/FoundryEvaluatorSpec.cs
@@ -0,0 +1,118 @@
+// Copyright (c) Microsoft. All rights reserved.
+
+using System;
+
+namespace Microsoft.Agents.AI.Foundry;
+
+///
+/// Specifies a single evaluator for a run — either a built-in
+/// Foundry evaluator (referenced by short or fully-qualified name) or a pre-existing rubric
+/// evaluator (referenced by ).
+///
+///
+///
+/// Both and are implicitly convertible
+/// to , so call sites can mix the two:
+///
+///
+/// var evals = new FoundryEvals(
+/// projectClient,
+/// "gpt-4o-mini",
+/// new GeneratedEvaluatorRef("policy-rubric", Version: "3"),
+/// FoundryEvals.Relevance,
+/// FoundryEvals.Coherence);
+///
+///
+public readonly struct FoundryEvaluatorSpec : IEquatable
+{
+ private FoundryEvaluatorSpec(string? builtinName, GeneratedEvaluatorRef? generatedRef)
+ {
+ this.BuiltinName = builtinName;
+ this.GeneratedRef = generatedRef;
+ }
+
+ ///
+ /// Initializes a new for a built-in evaluator by name
+ /// (for example "relevance" or "builtin.relevance").
+ ///
+ /// Built-in evaluator name.
+ public FoundryEvaluatorSpec(string builtinName)
+ : this(builtinName ?? throw new ArgumentNullException(nameof(builtinName)), null)
+ {
+ }
+
+ ///
+ /// Initializes a new for a generated rubric evaluator
+ /// previously registered with the provider.
+ ///
+ /// Reference to the rubric evaluator.
+ public FoundryEvaluatorSpec(GeneratedEvaluatorRef generatedRef)
+ : this(null, generatedRef ?? throw new ArgumentNullException(nameof(generatedRef)))
+ {
+ }
+
+ /// Gets the built-in evaluator name, or when this is a rubric reference.
+ public string? BuiltinName { get; }
+
+ /// Gets the rubric reference, or when this is a built-in evaluator.
+ public GeneratedEvaluatorRef? GeneratedRef { get; }
+
+ /// Gets whether this spec references a built-in evaluator.
+ public bool IsBuiltin => this.BuiltinName is not null;
+
+ /// Gets whether this spec references a generated rubric evaluator.
+ public bool IsRubric => this.GeneratedRef is not null;
+
+ /// Gets whether this spec is valid (i.e. references either a built-in or a rubric).
+ ///
+ /// Because is a struct, default(FoundryEvaluatorSpec)
+ /// is a syntactically-valid but semantically-invalid value (both and
+ /// are ). Call at
+ /// API boundaries to fail fast instead of NRE-ing later.
+ ///
+ public bool IsValid => this.BuiltinName is not null || this.GeneratedRef is not null;
+
+ /// Validates that this spec references either a built-in evaluator or a rubric.
+ /// Parameter name used in the thrown .
+ /// Thrown when neither nor is set.
+ public void EnsureValid(string? paramName = null)
+ {
+ if (!this.IsValid)
+ {
+ throw new ArgumentException(
+ $"Invalid {nameof(FoundryEvaluatorSpec)}: must be constructed with either a built-in evaluator name " +
+ $"or a {nameof(GeneratedEvaluatorRef)}. The default struct value is not a valid spec.",
+ paramName);
+ }
+ }
+
+ /// Implicit conversion from a built-in evaluator name.
+ public static implicit operator FoundryEvaluatorSpec(string builtinName) => new(builtinName);
+
+ /// Implicit conversion from a .
+ public static implicit operator FoundryEvaluatorSpec(GeneratedEvaluatorRef generatedRef) => new(generatedRef);
+
+ ///
+ public bool Equals(FoundryEvaluatorSpec other)
+ => this.BuiltinName == other.BuiltinName
+ && Equals(this.GeneratedRef, other.GeneratedRef);
+
+ ///
+ public override bool Equals(object? obj) => obj is FoundryEvaluatorSpec other && this.Equals(other);
+
+ ///
+ public override int GetHashCode()
+ => HashCode.Combine(this.BuiltinName, this.GeneratedRef);
+
+ /// Equality operator.
+ public static bool operator ==(FoundryEvaluatorSpec left, FoundryEvaluatorSpec right) => left.Equals(right);
+
+ /// Inequality operator.
+ public static bool operator !=(FoundryEvaluatorSpec left, FoundryEvaluatorSpec right) => !left.Equals(right);
+
+ ///
+ public override string ToString()
+ => this.IsRubric
+ ? $"GeneratedEvaluatorRef({this.GeneratedRef!.Name}@{this.GeneratedRef.Version ?? "latest"})"
+ : this.BuiltinName ?? "";
+}
diff --git a/dotnet/src/Microsoft.Agents.AI/Evaluation/AgentEvaluationResults.cs b/dotnet/src/Microsoft.Agents.AI/Evaluation/AgentEvaluationResults.cs
index f33d69a2e3..552479ad69 100644
--- a/dotnet/src/Microsoft.Agents.AI/Evaluation/AgentEvaluationResults.cs
+++ b/dotnet/src/Microsoft.Agents.AI/Evaluation/AgentEvaluationResults.cs
@@ -2,6 +2,7 @@
using System;
using System.Collections.Generic;
+using System.Globalization;
using System.Linq;
using Microsoft.Extensions.AI.Evaluation;
@@ -118,6 +119,338 @@ public void AssertAllPassed(string? message = null)
}
}
+ ///
+ /// Asserts that every per-evaluator score on every item is at least .
+ ///
+ ///
+ ///
+ /// Designed for CI gates on generated rubric evaluators (for example
+ /// results.AssertScoreAtLeast(0.80)). Walks across this
+ /// result and any from workflow evaluations.
+ ///
+ ///
+ /// When is , the assertion is a no-op for
+ /// this level. Providers that surface per-evaluator scores (such as Foundry) populate
+ /// ; providers that only emit aggregate
+ /// metrics do not.
+ ///
+ ///
+ /// Minimum acceptable score (inclusive).
+ ///
+ /// When set, only check scores whose matches.
+ ///
+ /// Optional custom failure message.
+ ///
+ /// Thrown when any matching score is below the threshold.
+ ///
+ public void AssertScoreAtLeast(double minScore, string? evaluator = null, string? message = null)
+ {
+ var offenders = new List();
+ CollectScoreOffenders(this, minScore, evaluator, offenders);
+
+ if (offenders.Count > 0)
+ {
+ throw new InvalidOperationException(
+ message ?? FormatOffenders(
+ $"{offenders.Count} score(s) below threshold {minScore.ToString(CultureInfo.InvariantCulture)}"
+ + (evaluator is not null ? $" for {evaluator}" : string.Empty),
+ offenders));
+ }
+ }
+
+ ///
+ /// Asserts that every item's score for the given rubric dimension is at least
+ /// .
+ ///
+ ///
+ /// Walks across
+ /// (and any ) looking for the named dimension. Non-applicable
+ /// dimensions are skipped by default; pass =
+ /// to fail when no applicable score is produced for an item. If dimension data exists
+ /// but the requested is never present, the assertion fails
+ /// to surface likely typos or evaluator mismatches.
+ ///
+ /// Dimension id — matches the rubric definition.
+ /// Minimum acceptable dimension score (inclusive).
+ ///
+ /// When set, only consider scores whose matches.
+ ///
+ ///
+ /// When , items with no applicable score for the dimension also fail
+ /// the assertion. Defaults to (skip).
+ ///
+ /// Optional custom failure message.
+ ///
+ /// Thrown when the dimension fails the threshold on any item.
+ ///
+ public void AssertDimensionScoreAtLeast(
+ string dimensionId,
+ double minScore,
+ string? evaluator = null,
+ bool requireApplicable = false,
+ string? message = null)
+ {
+ var offenders = new List();
+ var missing = new List();
+ CollectDimensionOffenders(this, dimensionId, minScore, evaluator, requireApplicable, offenders, missing);
+
+ var problems = new List();
+ bool hasAnyDimensionData = HasAnyDimensionData(this, evaluator);
+ if (hasAnyDimensionData && !HasDimension(this, dimensionId, evaluator))
+ {
+ problems.Add(
+ $"Dimension '{dimensionId}' was not found in results"
+ + (evaluator is not null ? $" for evaluator '{evaluator}'." : "."));
+ }
+
+ if (offenders.Count > 0)
+ {
+ problems.Add(FormatOffenders(
+ $"{offenders.Count} dimension score(s) for '{dimensionId}' below {minScore.ToString(CultureInfo.InvariantCulture)}",
+ offenders));
+ }
+
+ if (missing.Count > 0)
+ {
+ problems.Add(FormatOffenders(
+ $"Dimension '{dimensionId}' not applicable on {missing.Count} item(s)",
+ missing));
+ }
+
+ if (problems.Count > 0)
+ {
+ throw new InvalidOperationException(message ?? string.Join("; ", problems));
+ }
+ }
+
+ ///
+ /// Asserts that no item ended in a failed or errored state. Includes any sub-results
+ /// from workflow evaluations.
+ ///
+ /// Optional custom failure message.
+ ///
+ /// Thrown when any item failed or errored.
+ ///
+ public void AssertNoFailedItems(string? message = null)
+ {
+ var bad = new List();
+ CollectFailedItems(this, bad);
+
+ if (bad.Count > 0)
+ {
+ throw new InvalidOperationException(
+ message ?? FormatOffenders($"{bad.Count} item(s) failed or errored", bad));
+ }
+ }
+
+ private static void CollectScoreOffenders(
+ AgentEvaluationResults results,
+ double minScore,
+ string? evaluator,
+ List offenders)
+ {
+ if (results.DetailedItems is not null)
+ {
+ foreach (var item in results.DetailedItems)
+ {
+ foreach (var score in item.Scores)
+ {
+ if (evaluator is not null && score.Name != evaluator)
+ {
+ continue;
+ }
+
+ if (score.Score < minScore)
+ {
+ offenders.Add($"{item.ItemId}/{score.Name}={score.Score.ToString("F3", CultureInfo.InvariantCulture)}");
+ }
+ }
+ }
+ }
+
+ if (results.SubResults is not null)
+ {
+ foreach (var sub in results.SubResults.Values)
+ {
+ CollectScoreOffenders(sub, minScore, evaluator, offenders);
+ }
+ }
+ }
+
+ private static void CollectDimensionOffenders(
+ AgentEvaluationResults results,
+ string dimensionId,
+ double minScore,
+ string? evaluator,
+ bool requireApplicable,
+ List offenders,
+ List missing)
+ {
+ if (results.DetailedItems is not null)
+ {
+ foreach (var item in results.DetailedItems)
+ {
+ bool foundApplicable = false;
+ foreach (var score in item.Scores)
+ {
+ if (evaluator is not null && score.Name != evaluator)
+ {
+ continue;
+ }
+
+ if (score.Dimensions is null)
+ {
+ continue;
+ }
+
+ foreach (var rs in score.Dimensions)
+ {
+ if (rs.Id != dimensionId)
+ {
+ continue;
+ }
+
+ if (!rs.Applicable)
+ {
+ continue;
+ }
+
+ foundApplicable = true;
+ if (rs.Score is null || rs.Score.Value < minScore)
+ {
+ var actual = rs.Score is null
+ ? "null"
+ : rs.Score.Value.ToString(CultureInfo.InvariantCulture);
+ offenders.Add($"{item.ItemId}/{score.Name}/{dimensionId}={actual}");
+ }
+ }
+ }
+
+ if (requireApplicable && !foundApplicable)
+ {
+ missing.Add(item.ItemId);
+ }
+ }
+ }
+
+ if (results.SubResults is not null)
+ {
+ foreach (var sub in results.SubResults.Values)
+ {
+ CollectDimensionOffenders(sub, dimensionId, minScore, evaluator, requireApplicable, offenders, missing);
+ }
+ }
+ }
+
+ private static void CollectFailedItems(AgentEvaluationResults results, List bad)
+ {
+ if (results.DetailedItems is not null)
+ {
+ foreach (var item in results.DetailedItems)
+ {
+ if (item.IsFailed || item.IsError)
+ {
+ bad.Add($"{item.ItemId}:{item.Status}");
+ }
+ }
+ }
+
+ if (results.SubResults is not null)
+ {
+ foreach (var sub in results.SubResults.Values)
+ {
+ CollectFailedItems(sub, bad);
+ }
+ }
+ }
+
+ private static bool HasAnyDimensionData(AgentEvaluationResults results, string? evaluator)
+ {
+ if (results.DetailedItems is not null)
+ {
+ foreach (var item in results.DetailedItems)
+ {
+ foreach (var score in item.Scores)
+ {
+ if (evaluator is not null && score.Name != evaluator)
+ {
+ continue;
+ }
+
+ if (score.Dimensions is { Count: > 0 })
+ {
+ return true;
+ }
+ }
+ }
+ }
+
+ if (results.SubResults is not null)
+ {
+ foreach (var sub in results.SubResults.Values)
+ {
+ if (HasAnyDimensionData(sub, evaluator))
+ {
+ return true;
+ }
+ }
+ }
+
+ return false;
+ }
+
+ private static bool HasDimension(AgentEvaluationResults results, string dimensionId, string? evaluator)
+ {
+ if (results.DetailedItems is not null)
+ {
+ foreach (var item in results.DetailedItems)
+ {
+ foreach (var score in item.Scores)
+ {
+ if (evaluator is not null && score.Name != evaluator)
+ {
+ continue;
+ }
+
+ if (score.Dimensions is null)
+ {
+ continue;
+ }
+
+ if (score.Dimensions.Any(rs => rs.Id == dimensionId))
+ {
+ return true;
+ }
+ }
+ }
+ }
+
+ if (results.SubResults is not null)
+ {
+ foreach (var sub in results.SubResults.Values)
+ {
+ if (HasDimension(sub, dimensionId, evaluator))
+ {
+ return true;
+ }
+ }
+ }
+
+ return false;
+ }
+
+ private static string FormatOffenders(string prefix, List offenders)
+ {
+ const int MaxShown = 5;
+ if (offenders.Count <= MaxShown)
+ {
+ return $"{prefix}: {string.Join(", ", offenders)}";
+ }
+
+ var shown = string.Join(", ", offenders.GetRange(0, MaxShown));
+ return $"{prefix}: {shown} (+{offenders.Count - MaxShown} more)";
+ }
+
private static bool ItemPassed(EvaluationResult result)
{
foreach (var metric in result.Metrics.Values)
diff --git a/dotnet/src/Microsoft.Agents.AI/Evaluation/EvalItemResult.cs b/dotnet/src/Microsoft.Agents.AI/Evaluation/EvalItemResult.cs
index 64e317be2b..fb8014e2c3 100644
--- a/dotnet/src/Microsoft.Agents.AI/Evaluation/EvalItemResult.cs
+++ b/dotnet/src/Microsoft.Agents.AI/Evaluation/EvalItemResult.cs
@@ -66,7 +66,27 @@ public EvalItemResult(string itemId, string status, IReadOnlyListThe evaluator name that produced this score.
/// The numeric score value.
/// Whether the evaluator considered this a pass, or null if not determined.
-public record EvalScoreResult(string Name, double Score, bool? Passed = null);
+public record EvalScoreResult(string Name, double Score, bool? Passed = null)
+{
+ ///
+ /// Gets the per-dimension breakdown when this evaluator is a rubric-based evaluator.
+ ///
+ ///
+ ///
+ /// Rubric evaluators (for example, generated rubric evaluators authored in the Azure AI
+ /// Foundry portal) emit one per dimension per item alongside
+ /// the overall weighted . Each entry preserves the dimension's
+ /// applicability, weight, and the evaluator-supplied rationale.
+ ///
+ ///
+ /// Non-rubric evaluators (built-in quality, safety, or agent-behavior evaluators) leave
+ /// this property . Use
+ ///
+ /// to gate CI on a specific dimension across all items.
+ ///
+ ///
+ public IReadOnlyList? Dimensions { get; init; }
+}
///
/// Per-evaluator pass/fail breakdown from an evaluation run.
diff --git a/dotnet/src/Microsoft.Agents.AI/Evaluation/GeneratedEvaluatorRef.cs b/dotnet/src/Microsoft.Agents.AI/Evaluation/GeneratedEvaluatorRef.cs
new file mode 100644
index 0000000000..a73d9cf590
--- /dev/null
+++ b/dotnet/src/Microsoft.Agents.AI/Evaluation/GeneratedEvaluatorRef.cs
@@ -0,0 +1,55 @@
+// Copyright (c) Microsoft. All rights reserved.
+
+namespace Microsoft.Agents.AI;
+
+///
+/// A reference to a generated rubric evaluator that already exists in the provider's registry.
+///
+///
+///
+/// Pass instances of this class to a batch evaluator (for example
+/// Microsoft.Agents.AI.Foundry.FoundryEvals) to score items with a pre-existing rubric
+/// evaluator that was authored in the provider's portal or via the provider's dedicated SDK.
+/// Agent Framework is a consumer here: it does not create or modify the evaluator definition;
+/// it only references the persisted version by name.
+///
+///
+/// Pinning is strongly recommended so evaluation runs are reproducible.
+/// A resolves to whichever version is current at
+/// execution time; consuming evaluators are expected to emit a warning when a versionless
+/// reference is used. CI gates should always pass a concrete version.
+///
+///
+///
+/// Evaluator name as stored in the provider's registry (for example
+/// "reservation-policy-rubric"). Distinct from built-in evaluators such as
+/// "relevance".
+///
+///
+/// Pinned evaluator version. means "latest" — this is discouraged for
+/// reproducible runs and consumers may emit a warning when used.
+///
+///
+/// Optional human-readable name used in result summaries. Defaults to when
+/// unset.
+///
+public sealed record GeneratedEvaluatorRef(
+ string Name,
+ string? Version = null,
+ string? DisplayName = null)
+{
+ ///
+ /// Creates a versionless reference that resolves to the latest version of the evaluator at
+ /// run time.
+ ///
+ ///
+ /// Discouraged for reproducible runs. Prefer the primary constructor with an explicit
+ /// so CI and replay evaluations stay stable when the evaluator is
+ /// updated in the provider's registry.
+ ///
+ /// Evaluator name as stored in the provider's registry.
+ /// Optional human-readable name used in result summaries.
+ /// A new with unset.
+ public static GeneratedEvaluatorRef Latest(string name, string? displayName = null)
+ => new(name, Version: null, DisplayName: displayName);
+}
diff --git a/dotnet/src/Microsoft.Agents.AI/Evaluation/RubricScore.cs b/dotnet/src/Microsoft.Agents.AI/Evaluation/RubricScore.cs
new file mode 100644
index 0000000000..d873af2644
--- /dev/null
+++ b/dotnet/src/Microsoft.Agents.AI/Evaluation/RubricScore.cs
@@ -0,0 +1,34 @@
+// Copyright (c) Microsoft. All rights reserved.
+
+namespace Microsoft.Agents.AI;
+
+///
+/// A single dimension's score from a rubric-based evaluator run.
+///
+///
+///
+/// Rubric evaluators (such as the generated rubric evaluators produced by Azure AI Foundry's
+/// adaptive evals) emit one per dimension per item, alongside an
+/// overall weighted score. Attach instances to as
+/// a typed view of the per-dimension breakdown returned by the provider
+/// (e.g. properties.dimension_scores).
+///
+///
+/// Non-rubric evaluators (built-in quality, safety, or agent-behavior evaluators) leave
+/// as .
+///
+///
+/// Dimension identifier — matches the id defined on the rubric.
+///
+/// Numeric score for the dimension, or when the dimension was marked
+/// non-applicable for this item. Foundry rubric evaluators emit integer scores on a 1–5 scale.
+///
+/// Whether the dimension applied to this item.
+/// Dimension weight, mirroring the rubric definition.
+/// Short rationale produced by the evaluator.
+public sealed record RubricScore(
+ string Id,
+ int? Score,
+ bool Applicable,
+ int Weight,
+ string Reason);
diff --git a/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalConverterTests.cs b/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalConverterTests.cs
index aea1459e5e..4a54e21f9f 100644
--- a/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalConverterTests.cs
+++ b/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalConverterTests.cs
@@ -305,6 +305,71 @@ public void BuildTestingCriteria_WithoutDataMapping_OmitsMappingField()
Assert.Null(criteria[0].DataMapping);
}
+ [Fact]
+ public void BuildTestingCriteria_WithRubricRef_EmitsAzureAiEvaluatorWithVersion()
+ {
+ var rubric = new GeneratedEvaluatorRef("policy-rubric", Version: "3", DisplayName: "Policy");
+ var criteria = FoundryEvalConverter.BuildTestingCriteria(
+ [rubric], "gpt-4o-mini", includeDataMapping: true);
+
+ Assert.Single(criteria);
+ var entry = criteria[0];
+ Assert.Equal("azure_ai_evaluator", entry.Type);
+ Assert.Equal("Policy", entry.Name);
+ Assert.Equal("policy-rubric", entry.EvaluatorName);
+ Assert.Equal("3", entry.EvaluatorVersion);
+ Assert.Equal("gpt-4o-mini", entry.InitializationParameters.DeploymentName);
+
+ var mapping = entry.DataMapping;
+ Assert.NotNull(mapping);
+ Assert.Equal("{{item.query_messages}}", mapping["query"]);
+ Assert.Equal("{{item.response_messages}}", mapping["response"]);
+ Assert.False(mapping.ContainsKey("tool_definitions"));
+ }
+
+ [Fact]
+ public void BuildTestingCriteria_WithVersionlessRubricRef_OmitsVersionField()
+ {
+ var rubric = GeneratedEvaluatorRef.Latest("policy-rubric");
+ var criteria = FoundryEvalConverter.BuildTestingCriteria(
+ [rubric], "gpt-4o-mini", includeDataMapping: false);
+
+ Assert.Single(criteria);
+ var entry = criteria[0];
+ Assert.Equal("policy-rubric", entry.Name); // falls back to Name when DisplayName is null
+ Assert.Equal("policy-rubric", entry.EvaluatorName);
+ Assert.Null(entry.EvaluatorVersion);
+ Assert.Null(entry.DataMapping);
+ }
+
+ [Fact]
+ public void BuildTestingCriteria_RubricRefWithTools_IncludesToolDefinitions()
+ {
+ var rubric = new GeneratedEvaluatorRef("tool-aware-rubric", Version: "1");
+ var criteria = FoundryEvalConverter.BuildTestingCriteria(
+ [rubric], "gpt-4o-mini", includeDataMapping: true, includeToolDefinitions: true);
+
+ Assert.Single(criteria);
+ var mapping = criteria[0].DataMapping;
+ Assert.NotNull(mapping);
+ Assert.True(mapping.ContainsKey("tool_definitions"));
+ Assert.Equal("{{item.tool_definitions}}", mapping["tool_definitions"]);
+ }
+
+ [Fact]
+ public void BuildTestingCriteria_MixedSpecs_PreservesOrder()
+ {
+ var rubric = new GeneratedEvaluatorRef("policy-rubric", Version: "2");
+ var criteria = FoundryEvalConverter.BuildTestingCriteria(
+ ["relevance", rubric, "coherence"], "gpt-4o-mini", includeDataMapping: false);
+
+ Assert.Equal(3, criteria.Count);
+ Assert.Equal("builtin.relevance", criteria[0].EvaluatorName);
+ Assert.Equal("policy-rubric", criteria[1].EvaluatorName);
+ Assert.Equal("2", criteria[1].EvaluatorVersion);
+ Assert.Equal("builtin.coherence", criteria[2].EvaluatorName);
+ }
+
// ---------------------------------------------------------------
// FoundryEvalConverter.BuildItemSchema tests
// ---------------------------------------------------------------
@@ -391,6 +456,18 @@ public void FindMissingGroundTruthEvaluators_NoGroundTruthEvaluators_ReturnsEmpt
Assert.Empty(missing);
}
+ [Fact]
+ public void FindMissingGroundTruthEvaluators_IgnoresRubricRefs()
+ {
+ // Rubric refs are not ground-truth–dependent and must be skipped even when
+ // no items carry ExpectedOutput.
+ var rubric = new GeneratedEvaluatorRef("policy-rubric", Version: "1");
+ var missing = FoundryEvalConverter.FindMissingGroundTruthEvaluators(
+ [rubric, "relevance"], hasGroundTruth: false);
+
+ Assert.Empty(missing);
+ }
+
// ---------------------------------------------------------------
// FoundryEvalConverter.ConvertMessage DataContent test
// ---------------------------------------------------------------
diff --git a/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalsTests.cs b/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalsTests.cs
index a09dcf03fc..6b0c8ebbde 100644
--- a/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalsTests.cs
+++ b/dotnet/tests/Microsoft.Agents.AI.Foundry.UnitTests/FoundryEvalsTests.cs
@@ -1,6 +1,8 @@
// Copyright (c) Microsoft. All rights reserved.
using System;
+using System.Linq;
+using System.Text.Json;
namespace Microsoft.Agents.AI.Foundry.UnitTests;
@@ -13,7 +15,7 @@ public sealed class FoundryEvalsTests
public void FilterToolEvaluators_AllToolEvaluators_NoTools_ThrowsArgumentException()
{
// All configured evaluators are tool-type, but no items have tools.
- var evaluators = new[] { "tool_call_accuracy", "tool_selection" };
+ var evaluators = new FoundryEvaluatorSpec[] { "tool_call_accuracy", "tool_selection" };
var ex = Assert.Throws(
() => FoundryEvals.FilterToolEvaluators(evaluators, hasTools: false));
@@ -24,23 +26,229 @@ public void FilterToolEvaluators_AllToolEvaluators_NoTools_ThrowsArgumentExcepti
[Fact]
public void FilterToolEvaluators_MixedEvaluators_NoTools_FiltersToolOnes()
{
- var evaluators = new[] { "relevance", "tool_call_accuracy", "coherence" };
+ var evaluators = new FoundryEvaluatorSpec[] { "relevance", "tool_call_accuracy", "coherence" };
var result = FoundryEvals.FilterToolEvaluators(evaluators, hasTools: false);
Assert.Equal(2, result.Length);
- Assert.Contains("relevance", result);
- Assert.Contains("coherence", result);
- Assert.DoesNotContain("tool_call_accuracy", result);
+ Assert.Contains((FoundryEvaluatorSpec)"relevance", result);
+ Assert.Contains((FoundryEvaluatorSpec)"coherence", result);
+ Assert.DoesNotContain((FoundryEvaluatorSpec)"tool_call_accuracy", result);
}
[Fact]
public void FilterToolEvaluators_HasTools_ReturnsAllEvaluators()
{
- var evaluators = new[] { "relevance", "tool_call_accuracy" };
+ var evaluators = new FoundryEvaluatorSpec[] { "relevance", "tool_call_accuracy" };
var result = FoundryEvals.FilterToolEvaluators(evaluators, hasTools: true);
Assert.Equal(evaluators, result);
}
+
+ [Fact]
+ public void FilterToolEvaluators_PreservesRubricRefs_WhenNoTools()
+ {
+ // Rubric refs are tool-aware but never tool-required, so they must survive filtering
+ // when no items carry tool definitions.
+ var rubric = new GeneratedEvaluatorRef("policy-rubric", Version: "3");
+ var evaluators = new FoundryEvaluatorSpec[] { "relevance", rubric, "tool_call_accuracy" };
+
+ var result = FoundryEvals.FilterToolEvaluators(evaluators, hasTools: false);
+
+ Assert.Equal(2, result.Length);
+ Assert.Contains((FoundryEvaluatorSpec)"relevance", result);
+ Assert.Contains(result, s => s.IsRubric && s.GeneratedRef!.Name == "policy-rubric");
+ Assert.DoesNotContain((FoundryEvaluatorSpec)"tool_call_accuracy", result);
+ }
+
+ // ---------------------------------------------------------------
+ // FoundryEvals.ParseRubricScores tests
+ // ---------------------------------------------------------------
+
+ [Fact]
+ public void ParseRubricScores_CanonicalDimensionScoresKey_ParsesAllFields()
+ {
+ // Per Microsoft Learn docs, runtime output uses properties.dimension_scores.
+ const string Json = """
+ {
+ "properties": {
+ "dimension_scores": [
+ { "id": "intent_recognition", "score": 5, "applicable": true, "weight": 9, "reason": "Identified correctly." },
+ { "id": "general_quality", "score": 4, "applicable": true, "weight": 5, "reason": "Strong overall." }
+ ]
+ }
+ }
+ """;
+ using var doc = JsonDocument.Parse(Json);
+
+ var result = FoundryEvals.ParseRubricScores(doc.RootElement);
+
+ Assert.NotNull(result);
+ Assert.Equal(2, result!.Count);
+ Assert.Equal(["intent_recognition", "general_quality"], result.Select(r => r.Id));
+ Assert.Equal([5, 4], result.Select(r => r.Score));
+ Assert.Equal([9, 5], result.Select(r => r.Weight));
+ Assert.True(result[0].Applicable);
+ Assert.Equal("Identified correctly.", result[0].Reason);
+ }
+
+ [Fact]
+ public void ParseRubricScores_LegacyRubricScoresKey_StillSupported()
+ {
+ // Preview builds used the rubric_scores key; we still accept it for back-compat.
+ const string Json = """
+ {
+ "properties": {
+ "rubric_scores": [
+ { "id": "a", "score": 3, "applicable": true, "weight": 1, "reason": "r" }
+ ]
+ }
+ }
+ """;
+ using var doc = JsonDocument.Parse(Json);
+
+ var result = FoundryEvals.ParseRubricScores(doc.RootElement);
+
+ Assert.NotNull(result);
+ Assert.Single(result!);
+ Assert.Equal("a", result[0].Id);
+ }
+
+ [Fact]
+ public void ParseRubricScores_TopLevelKey_FallsBack()
+ {
+ // Defensive fallback when SDK shape omits the 'properties' wrapper.
+ const string Json = """
+ {
+ "dimension_scores": [
+ { "id": "x", "score": 2, "applicable": true, "weight": 1, "reason": "" }
+ ]
+ }
+ """;
+ using var doc = JsonDocument.Parse(Json);
+
+ var result = FoundryEvals.ParseRubricScores(doc.RootElement);
+
+ Assert.NotNull(result);
+ Assert.Single(result!);
+ Assert.Equal("x", result[0].Id);
+ }
+
+ [Fact]
+ public void ParseRubricScores_NoRubricKeys_ReturnsNull()
+ {
+ const string Json = """
+ { "properties": { "other_field": [] } }
+ """;
+ using var doc = JsonDocument.Parse(Json);
+
+ var result = FoundryEvals.ParseRubricScores(doc.RootElement);
+
+ Assert.Null(result);
+ }
+
+ [Fact]
+ public void ParseRubricScores_SkipsMalformedEntries()
+ {
+ // Entries missing weight or applicable are skipped, but well-formed siblings are kept.
+ const string Json = """
+ {
+ "properties": {
+ "dimension_scores": [
+ { "id": "good", "score": 3, "applicable": true, "weight": 1, "reason": "ok" },
+ { "id": "bad-no-weight", "score": 2, "applicable": true, "reason": "x" },
+ { "id": "bad-no-applicable", "score": 2, "weight": 1, "reason": "x" }
+ ]
+ }
+ }
+ """;
+ using var doc = JsonDocument.Parse(Json);
+
+ var result = FoundryEvals.ParseRubricScores(doc.RootElement);
+
+ Assert.NotNull(result);
+ Assert.Single(result!);
+ Assert.Equal("good", result[0].Id);
+ }
+
+ [Fact]
+ public void ParseRubricScores_NonApplicableDimension_KeepsNullScoreWhenMissing()
+ {
+ // Non-applicable dimensions can legitimately omit score (or set it to null).
+ const string Json = """
+ {
+ "properties": {
+ "dimension_scores": [
+ { "id": "skipped", "applicable": false, "weight": 5, "reason": "n/a" }
+ ]
+ }
+ }
+ """;
+ using var doc = JsonDocument.Parse(Json);
+
+ var result = FoundryEvals.ParseRubricScores(doc.RootElement);
+
+ Assert.NotNull(result);
+ Assert.Single(result!);
+ Assert.Equal("skipped", result[0].Id);
+ Assert.False(result[0].Applicable);
+ Assert.Null(result[0].Score);
+ }
+
+ // ---------------------------------------------------------------
+ // FoundryEvaluatorSpec validation tests
+ // ---------------------------------------------------------------
+
+ [Fact]
+ public void FoundryEvaluatorSpec_Default_IsNotValid()
+ {
+ var spec = default(FoundryEvaluatorSpec);
+ Assert.False(spec.IsValid);
+ Assert.Null(spec.BuiltinName);
+ Assert.Null(spec.GeneratedRef);
+ }
+
+ [Fact]
+ public void FoundryEvaluatorSpec_EnsureValid_DefaultThrows()
+ {
+ var spec = default(FoundryEvaluatorSpec);
+ var ex = Assert.Throws(() => spec.EnsureValid("evaluators"));
+ Assert.Equal("evaluators", ex.ParamName);
+ }
+
+ [Fact]
+ public void FoundryEvaluatorSpec_EnsureValid_BuiltinPasses()
+ {
+ var spec = (FoundryEvaluatorSpec)"relevance";
+ spec.EnsureValid(); // does not throw
+ }
+
+ [Fact]
+ public void FoundryEvaluatorSpec_EnsureValid_RubricPasses()
+ {
+ var spec = (FoundryEvaluatorSpec)new GeneratedEvaluatorRef("r", "1");
+ spec.EnsureValid(); // does not throw
+ }
+
+ [Fact]
+ public void EnsureAllSpecsValid_DefaultEntry_ThrowsWithParamName()
+ {
+ var specs = new FoundryEvaluatorSpec[] { "relevance", default };
+ var ex = Assert.Throws(
+ () => FoundryEvals.EnsureAllSpecsValid(specs, "evaluators"));
+ Assert.Equal("evaluators", ex.ParamName);
+ Assert.Contains("index 1", ex.Message);
+ }
+
+ [Fact]
+ public void EnsureAllSpecsValid_AllValid_DoesNotThrow()
+ {
+ var specs = new FoundryEvaluatorSpec[]
+ {
+ "relevance",
+ new GeneratedEvaluatorRef("policy", "1"),
+ };
+ FoundryEvals.EnsureAllSpecsValid(specs, "evaluators");
+ }
}
diff --git a/dotnet/tests/Microsoft.Agents.AI.UnitTests/EvaluationTests.cs b/dotnet/tests/Microsoft.Agents.AI.UnitTests/EvaluationTests.cs
index 071e9b723a..ff84b7d685 100644
--- a/dotnet/tests/Microsoft.Agents.AI.UnitTests/EvaluationTests.cs
+++ b/dotnet/tests/Microsoft.Agents.AI.UnitTests/EvaluationTests.cs
@@ -550,6 +550,343 @@ public void AgentEvaluationResults_SubResults_AllPassedChecksChildren()
Assert.False(results.AllPassed);
}
+ // ---------------------------------------------------------------
+ // RubricScore tests
+ // ---------------------------------------------------------------
+
+ [Fact]
+ public void RubricScore_Constructor_SetsAllProperties()
+ {
+ // Arrange & Act
+ var dim = new RubricScore("clarity", Score: 4, Applicable: true, Weight: 2, Reason: "clear");
+
+ // Assert
+ Assert.Equal("clarity", dim.Id);
+ Assert.Equal(4, dim.Score);
+ Assert.True(dim.Applicable);
+ Assert.Equal(2, dim.Weight);
+ Assert.Equal("clear", dim.Reason);
+ }
+
+ [Fact]
+ public void RubricScore_NonApplicable_AllowsNullScore()
+ {
+ // Arrange & Act
+ var dim = new RubricScore("safety", Score: null, Applicable: false, Weight: 1, Reason: "n/a");
+
+ // Assert
+ Assert.Null(dim.Score);
+ Assert.False(dim.Applicable);
+ }
+
+ [Fact]
+ public void EvalScoreResult_Dimensions_DefaultsToNull()
+ {
+ // Arrange & Act
+ var score = new EvalScoreResult("relevance", 0.8, Passed: true);
+
+ // Assert
+ Assert.Null(score.Dimensions);
+ }
+
+ [Fact]
+ public void EvalScoreResult_Dimensions_CanBeInitialized()
+ {
+ // Arrange
+ var dimensions = new List
+ {
+ new("clarity", 4, true, 1, "ok"),
+ new("safety", null, false, 1, "n/a"),
+ };
+
+ // Act
+ var score = new EvalScoreResult("custom-rubric", 0.75, Passed: true)
+ {
+ Dimensions = dimensions,
+ };
+
+ // Assert
+ Assert.NotNull(score.Dimensions);
+ Assert.Equal(2, score.Dimensions.Count);
+ Assert.Equal("clarity", score.Dimensions[0].Id);
+ Assert.False(score.Dimensions[1].Applicable);
+ }
+
+ // ---------------------------------------------------------------
+ // GeneratedEvaluatorRef tests
+ // ---------------------------------------------------------------
+
+ [Fact]
+ public void GeneratedEvaluatorRef_Constructor_DefaultsVersionAndDisplayNameToNull()
+ {
+ // Arrange & Act
+ var @ref = new GeneratedEvaluatorRef("policy-rubric");
+
+ // Assert
+ Assert.Equal("policy-rubric", @ref.Name);
+ Assert.Null(@ref.Version);
+ Assert.Null(@ref.DisplayName);
+ }
+
+ [Fact]
+ public void GeneratedEvaluatorRef_Constructor_AcceptsVersionAndDisplayName()
+ {
+ // Arrange & Act
+ var @ref = new GeneratedEvaluatorRef("policy-rubric", Version: "3", DisplayName: "Policy Quality");
+
+ // Assert
+ Assert.Equal("3", @ref.Version);
+ Assert.Equal("Policy Quality", @ref.DisplayName);
+ }
+
+ [Fact]
+ public void GeneratedEvaluatorRef_Latest_ProducesVersionlessReference()
+ {
+ // Arrange & Act
+ var @ref = GeneratedEvaluatorRef.Latest("policy-rubric", displayName: "Policy");
+
+ // Assert
+ Assert.Equal("policy-rubric", @ref.Name);
+ Assert.Null(@ref.Version);
+ Assert.Equal("Policy", @ref.DisplayName);
+ }
+
+ // ---------------------------------------------------------------
+ // Assertion helper tests (AssertScoreAtLeast / AssertDimensionScoreAtLeast / AssertNoFailedItems)
+ // ---------------------------------------------------------------
+
+ private static AgentEvaluationResults BuildResultsWithDetailed(params EvalItemResult[] detailed)
+ => new("test", Array.Empty())
+ {
+ DetailedItems = detailed,
+ };
+
+ [Fact]
+ public void AssertScoreAtLeast_AboveThreshold_DoesNotThrow()
+ {
+ // Arrange
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("relevance", 0.9, Passed: true),
+ new EvalScoreResult("coherence", 0.85, Passed: true),
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ results.AssertScoreAtLeast(0.8);
+ }
+
+ [Fact]
+ public void AssertScoreAtLeast_BelowThreshold_ThrowsWithOffender()
+ {
+ // Arrange
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("relevance", 0.5, Passed: false),
+ new EvalScoreResult("coherence", 0.9, Passed: true),
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ var ex = Assert.Throws(() => results.AssertScoreAtLeast(0.8));
+ Assert.Contains("item-1/relevance=0.500", ex.Message);
+ Assert.Contains("0.8", ex.Message);
+ }
+
+ [Fact]
+ public void AssertScoreAtLeast_EvaluatorFilter_OnlyChecksMatchingName()
+ {
+ // Arrange — coherence is below threshold but we filter to relevance.
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("relevance", 0.9, Passed: true),
+ new EvalScoreResult("coherence", 0.4, Passed: false),
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert — filtered to relevance, so no throw.
+ results.AssertScoreAtLeast(0.8, evaluator: "relevance");
+ }
+
+ [Fact]
+ public void AssertScoreAtLeast_RecursesIntoSubResults()
+ {
+ // Arrange
+ var failing = new EvalItemResult("sub-1", "pass", new[]
+ {
+ new EvalScoreResult("relevance", 0.2, Passed: false),
+ });
+ var sub = BuildResultsWithDetailed(failing);
+
+ var top = new AgentEvaluationResults("top", Array.Empty())
+ {
+ SubResults = new Dictionary { ["agent"] = sub },
+ };
+
+ // Act & Assert
+ var ex = Assert.Throws(() => top.AssertScoreAtLeast(0.8));
+ Assert.Contains("sub-1/relevance", ex.Message);
+ }
+
+ [Fact]
+ public void AssertDimensionScoreAtLeast_AboveThreshold_DoesNotThrow()
+ {
+ // Arrange
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("policy", 0.9, Passed: true)
+ {
+ Dimensions =
+ [
+ new RubricScore("clarity", Score: 4, Applicable: true, Weight: 1, Reason: "ok"),
+ new RubricScore("safety", Score: 5, Applicable: true, Weight: 1, Reason: "ok"),
+ ],
+ },
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ results.AssertDimensionScoreAtLeast("clarity", 3.0);
+ }
+
+ [Fact]
+ public void AssertDimensionScoreAtLeast_BelowThreshold_ThrowsWithOffender()
+ {
+ // Arrange
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("policy", 0.6, Passed: false)
+ {
+ Dimensions =
+ [
+ new RubricScore("clarity", Score: 2, Applicable: true, Weight: 1, Reason: "weak"),
+ ],
+ },
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ var ex = Assert.Throws(
+ () => results.AssertDimensionScoreAtLeast("clarity", 3.0, evaluator: "policy"));
+ Assert.Contains("item-1/policy/clarity=2", ex.Message);
+ }
+
+ [Fact]
+ public void AssertDimensionScoreAtLeast_NonApplicable_SkippedByDefault()
+ {
+ // Arrange — score=null + applicable=false should NOT trip the assertion by default.
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("policy", 1.0, Passed: true)
+ {
+ Dimensions =
+ [
+ new RubricScore("optional", Score: null, Applicable: false, Weight: 1, Reason: "n/a"),
+ ],
+ },
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ results.AssertDimensionScoreAtLeast("optional", 3.0);
+ }
+
+ [Fact]
+ public void AssertDimensionScoreAtLeast_RequireApplicable_ThrowsWhenMissing()
+ {
+ // Arrange — dimension never produced an applicable score on the item.
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("policy", 1.0, Passed: true)
+ {
+ Dimensions =
+ [
+ new RubricScore("optional", Score: null, Applicable: false, Weight: 1, Reason: "n/a"),
+ ],
+ },
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ var ex = Assert.Throws(
+ () => results.AssertDimensionScoreAtLeast("optional", 3.0, requireApplicable: true));
+ Assert.Contains("not applicable", ex.Message);
+ Assert.Contains("item-1", ex.Message);
+ }
+
+ [Fact]
+ public void AssertDimensionScoreAtLeast_UnknownDimension_ThrowsWhenDimensionDataExists()
+ {
+ // Arrange
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("policy", 1.0, Passed: true)
+ {
+ Dimensions =
+ [
+ new RubricScore("clarity", Score: 4, Applicable: true, Weight: 1, Reason: "ok"),
+ ],
+ },
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ var ex = Assert.Throws(
+ () => results.AssertDimensionScoreAtLeast("typo_dimension", 3.0));
+ Assert.Contains("typo_dimension", ex.Message);
+ Assert.Contains("not found", ex.Message);
+ }
+
+ [Fact]
+ public void AssertNoFailedItems_AllPassing_DoesNotThrow()
+ {
+ // Arrange
+ var detailed = new EvalItemResult("item-1", "pass", new[]
+ {
+ new EvalScoreResult("relevance", 0.9, Passed: true),
+ });
+ var results = BuildResultsWithDetailed(detailed);
+
+ // Act & Assert
+ results.AssertNoFailedItems();
+ }
+
+ [Fact]
+ public void AssertNoFailedItems_FailedOrErrored_ThrowsWithStatuses()
+ {
+ // Arrange
+ var failed = new EvalItemResult("item-1", "fail", new[]
+ {
+ new EvalScoreResult("relevance", 0.2, Passed: false),
+ });
+ var errored = new EvalItemResult("item-2", "errored", Array.Empty());
+ var results = BuildResultsWithDetailed(failed, errored);
+
+ // Act & Assert
+ var ex = Assert.Throws(() => results.AssertNoFailedItems());
+ Assert.Contains("item-1:fail", ex.Message);
+ Assert.Contains("item-2:errored", ex.Message);
+ }
+
+ [Fact]
+ public void AssertNoFailedItems_RecursesIntoSubResults()
+ {
+ // Arrange
+ var failed = new EvalItemResult("sub-1", "fail", new[]
+ {
+ new EvalScoreResult("relevance", 0.1, Passed: false),
+ });
+ var sub = BuildResultsWithDetailed(failed);
+ var top = new AgentEvaluationResults("top", Array.Empty())
+ {
+ SubResults = new Dictionary { ["agent"] = sub },
+ };
+
+ // Act & Assert
+ var ex = Assert.Throws(() => top.AssertNoFailedItems());
+ Assert.Contains("sub-1:fail", ex.Message);
+ }
+
// ---------------------------------------------------------------
// Mixed evaluator tests
// ---------------------------------------------------------------
diff --git a/python/packages/core/tests/core/test_local_eval.py b/python/packages/core/tests/core/test_local_eval.py
index e60fb35d51..595e1c8884 100644
--- a/python/packages/core/tests/core/test_local_eval.py
+++ b/python/packages/core/tests/core/test_local_eval.py
@@ -1113,3 +1113,114 @@ def test_evaluator_filter_isolates_offenders(self) -> None:
)
# The low-scoring "other" evaluator is filtered out; "policy" passes.
results.assert_dimension_score_at_least("clarity", 3, evaluator="policy")
+
+
+def _score_results(
+ *scores_per_item: list[EvalScoreResult],
+ sub_results: dict[str, EvalResults] | None = None,
+) -> EvalResults:
+ """Build an EvalResults shaped for score / status assertion tests."""
+ items = [
+ EvalItemResult(item_id=f"item-{i}", status="pass", scores=scores) for i, scores in enumerate(scores_per_item)
+ ]
+ return EvalResults(
+ provider="test",
+ eval_id="ev1",
+ run_id="run1",
+ result_counts={"passed": len(items), "failed": 0, "errored": 0, "total": len(items)},
+ items=items,
+ sub_results=sub_results or {},
+ )
+
+
+class TestAssertScoreAtLeast:
+ """Tests for EvalResults.assert_score_at_least (mirrors .NET coverage)."""
+
+ def test_all_above_threshold_passes(self) -> None:
+ results = _score_results(
+ [EvalScoreResult(name="relevance", score=0.9)],
+ [EvalScoreResult(name="relevance", score=0.85)],
+ )
+ # Should not raise.
+ results.assert_score_at_least(0.8)
+
+ def test_below_threshold_raises_with_offenders(self) -> None:
+ results = _score_results(
+ [EvalScoreResult(name="relevance", score=0.4)],
+ [EvalScoreResult(name="relevance", score=0.9)],
+ )
+ with pytest.raises(EvalNotPassedError) as exc:
+ results.assert_score_at_least(0.5)
+ msg = str(exc.value)
+ assert "item-0" in msg
+ assert "relevance" in msg
+ assert "0.400" in msg
+
+ def test_evaluator_filter_isolates_offenders(self) -> None:
+ results = _score_results(
+ [
+ EvalScoreResult(name="other", score=0.1),
+ EvalScoreResult(name="relevance", score=0.95),
+ ],
+ )
+ # The low-scoring "other" evaluator is filtered out; "relevance" passes.
+ results.assert_score_at_least(0.8, evaluator="relevance")
+
+ def test_recursion_into_sub_results(self) -> None:
+ sub = _score_results([EvalScoreResult(name="relevance", score=0.2)])
+ parent = _score_results(
+ [EvalScoreResult(name="relevance", score=0.9)],
+ sub_results={"sub_executor": sub},
+ )
+ with pytest.raises(EvalNotPassedError) as exc:
+ parent.assert_score_at_least(0.5)
+ # Offender from sub-result is surfaced.
+ assert "0.200" in str(exc.value)
+
+
+class TestAssertNoFailedItems:
+ """Tests for EvalResults.assert_no_failed_items (mirrors .NET coverage)."""
+
+ def test_all_passing_does_not_raise(self) -> None:
+ results = _score_results(
+ [EvalScoreResult(name="relevance", score=0.9)],
+ [EvalScoreResult(name="relevance", score=0.85)],
+ )
+ # Should not raise.
+ results.assert_no_failed_items()
+
+ def test_failed_and_errored_items_raise_with_statuses(self) -> None:
+ items = [
+ EvalItemResult(item_id="ok", status="pass", scores=[]),
+ EvalItemResult(item_id="bad", status="fail", scores=[]),
+ EvalItemResult(item_id="boom", status="error", scores=[], error_code="timeout"),
+ ]
+ results = EvalResults(
+ provider="test",
+ eval_id="ev1",
+ run_id="run1",
+ result_counts={"passed": 1, "failed": 1, "errored": 1, "total": 3},
+ items=items,
+ )
+ with pytest.raises(EvalNotPassedError) as exc:
+ results.assert_no_failed_items()
+ msg = str(exc.value)
+ assert "bad:fail" in msg
+ assert "boom:error" in msg
+
+ def test_recursion_into_sub_results(self) -> None:
+ sub_items = [EvalItemResult(item_id="sub-bad", status="fail", scores=[])]
+ sub = EvalResults(
+ provider="test",
+ eval_id="ev2",
+ run_id="run2",
+ result_counts={"passed": 0, "failed": 1, "errored": 0, "total": 1},
+ items=sub_items,
+ )
+ parent = _score_results(
+ [EvalScoreResult(name="relevance", score=0.9)],
+ sub_results={"sub_executor": sub},
+ )
+ with pytest.raises(EvalNotPassedError) as exc:
+ parent.assert_no_failed_items()
+ assert "sub-bad:fail" in str(exc.value)
diff --git a/python/packages/foundry/agent_framework_foundry/_foundry_evals.py b/python/packages/foundry/agent_framework_foundry/_foundry_evals.py
index 8059c2ce99..b4015f1e23 100644
--- a/python/packages/foundry/agent_framework_foundry/_foundry_evals.py
+++ b/python/packages/foundry/agent_framework_foundry/_foundry_evals.py
@@ -541,6 +541,8 @@ def _extract_rubric_scores(sample: Any) -> list[RubricScore] | None:
if props_dict is not None and props_dict is not properties:
containers.append(props_dict)
containers.append(sample_any)
+ else:
+ containers.append(sample)
for container in containers:
for key in _RUBRIC_DIMENSION_KEYS:
diff --git a/python/packages/foundry/tests/test_foundry_evals.py b/python/packages/foundry/tests/test_foundry_evals.py
index 8734650aaf..6e24a10ec7 100644
--- a/python/packages/foundry/tests/test_foundry_evals.py
+++ b/python/packages/foundry/tests/test_foundry_evals.py
@@ -2619,6 +2619,44 @@ def test_dimension_scores_via_attribute(self) -> None:
assert result[0].id == "policy_enforcement"
assert result[0].score == 1
+ def test_dimension_scores_directly_on_typed_sample_no_properties_wrapper(self) -> None:
+ """Typed SDK sample with ``dimension_scores`` directly on the instance (no ``properties``)."""
+
+ rs = MagicMock()
+ rs.id = "intent_recognition"
+ rs.score = 4
+ rs.applicable = True
+ rs.weight = 2
+ rs.reason = "ok"
+
+ # spec= restricts available attributes — no `properties`, just `dimension_scores`.
+ sample = MagicMock(spec=["dimension_scores"])
+ sample.dimension_scores = [rs]
+
+ result = _extract_rubric_scores(sample)
+ assert result is not None
+ assert result[0].id == "intent_recognition"
+ assert result[0].score == 4
+ assert result[0].weight == 2
+
+ def test_rubric_scores_directly_on_typed_sample_legacy_key(self) -> None:
+ """Same fallback works for the legacy ``rubric_scores`` key."""
+
+ rs = MagicMock()
+ rs.id = "policy"
+ rs.score = 2
+ rs.applicable = True
+ rs.weight = 1
+ rs.reason = "partial"
+
+ sample = MagicMock(spec=["rubric_scores"])
+ sample.rubric_scores = [rs]
+
+ result = _extract_rubric_scores(sample)
+ assert result is not None
+ assert result[0].id == "policy"
+ assert result[0].score == 2
+
# ---------------------------------------------------------------------------
# _poll_eval_run — timeout / failed / canceled paths