Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
21 commits
Select commit Hold shift + click to select a range
aabbf13
fix(orchestrator): carry worker failures into synthesis, abstain when…
arst Aug 26, 2026
7a77973
docs(orchestrator): mark the tautological quorum and stop overclaimin…
arst Aug 26, 2026
41fa1d2
fix(retry): rethrow cancellation and scope whole-turn retry to idempo…
arst Aug 26, 2026
2e0a099
fix(tool-auth): capability commit point, unconditional amount validat…
arst Aug 26, 2026
438ecdd
test(tool-auth): gate the race and the two lifecycle orderings, name …
arst Aug 26, 2026
3ba7171
fix(react): enforce the tool-call bound in the host, not the prompt
arst Aug 26, 2026
c14f112
fix(react): catch the tool-call budget by type, not by message text
arst Aug 26, 2026
4553fb9
fix(judge): strict verdicts, indeterminate on malformed output, multi…
arst Aug 26, 2026
35ec9a5
docs(judge): the probe is no longer a single swap
arst Aug 26, 2026
6396e22
fix(regression-evals): trace cases need human promotion; rename exact…
arst Aug 26, 2026
fc86e99
fix(regression-evals): report awaiting-signoff and awaiting-review co…
arst Aug 26, 2026
3880016
fix(regression-evals): derive the candidate count instead of hard-cod…
arst Aug 26, 2026
2ccc76c
fix(resource-aware): account for the fallback call and call the budge…
arst Aug 26, 2026
00e8800
ci: update and pin actions by commit sha
arst Aug 26, 2026
8ed43e4
fix(resource-aware): name the SK twin's budget soft too
arst Aug 26, 2026
c3c8f95
fix(react): enforce the tool-call budget with Terminate, not an excep…
arst Aug 26, 2026
f9b0000
fix(judge): an unreadable rubric verdict is Indeterminate, not a zero
arst Aug 26, 2026
79aa025
fix(judge): measure position bias per slot instead of folding the slo…
arst Aug 26, 2026
c97edd9
fix(twins,docs): narrow the SK fallback catch; draw the abstain/parti…
arst Aug 26, 2026
c0bb6ef
docs: stop the new mermaid branch from relabelling the worker registry
arst Aug 26, 2026
550f4f7
test(react): pin the auto-invocation loop to Terminate, not just the …
arst Aug 26, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 8 additions & 8 deletions .github/workflows/build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,8 @@ jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-dotnet@v4
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6.0.0
with:
dotnet-version: 10.0.x
- run: dotnet build "Agentic Patterns.slnx" --configuration Release
Expand All @@ -24,25 +24,25 @@ jobs:
contents: read
packages: write
steps:
- uses: actions/checkout@v4
- uses: docker/setup-qemu-action@v3
- uses: docker/setup-buildx-action@v3
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0
- uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
- if: github.event_name == 'push'
uses: docker/login-action@v3
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- id: meta
uses: docker/metadata-action@v5
uses: docker/metadata-action@dc802804100637a589fabce1cb79ff13a1411302 # v6.2.0
with:
images: ghcr.io/${{ github.repository }}
tags: |
type=raw,value=latest,enable={{is_default_branch}}
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
type=sha,prefix=sha-
- uses: docker/build-push-action@v6
- uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
with:
context: .
platforms: ${{ github.event_name == 'push' && 'linux/amd64,linux/arm64' || 'linux/amd64' }}
Expand Down
3 changes: 3 additions & 0 deletions AgenticPatterns.Tests/AgenticPatterns.Tests.csproj
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@
(and Program.cs's top-level-statement type) are internal and invisible across the
assembly boundary, so referencing it too would add nothing to test. -->
<ProjectReference Include="..\IdempotentToolCalls.AgentFramework\IdempotentToolCalls.AgentFramework.csproj"/>
<ProjectReference Include="..\LLMAsJudge.AgentFramework\LLMAsJudge.AgentFramework.csproj"/>
<ProjectReference Include="..\MCP.AgentFramework\MCP.AgentFramework.csproj"/>
<!-- NOT MCP.SemanticKernel: excluded conservatively, not because it conflicts - McpToolBinding
is mirrored into MCP.SemanticKernel's own namespace, so there is no name collision, and
Expand All @@ -36,7 +37,9 @@
<ProjectReference Include="..\PatternExplorer\PatternExplorer.csproj"/>
<ProjectReference Include="..\Planning.AgentFramework\Planning.AgentFramework.csproj"/>
<!-- NOT Planning.SemanticKernel: it defines the same global-namespace types as the AF flavor -->
<ProjectReference Include="..\ReasoningAndActing\ReasoningAndActing.csproj"/>
<ProjectReference Include="..\RedTeaming.AgentFramework\RedTeaming.AgentFramework.csproj"/>
<ProjectReference Include="..\RegressionEvals.AgentFramework\RegressionEvals.AgentFramework.csproj"/>
<ProjectReference Include="..\ResourceAwareOptimization.AgentFramework\ResourceAwareOptimization.AgentFramework.csproj"/>
<ProjectReference Include="..\SelfCorrectionLoop\SelfCorrectionLoop.csproj"/>
<ProjectReference Include="..\SelfCorrectionLoop.AgentFramework\SelfCorrectionLoop.AgentFramework.csproj"/>
Expand Down
67 changes: 67 additions & 0 deletions AgenticPatterns.Tests/CasePartitionTests.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
using RegressionEvals.AgentFramework;
using Xunit;

namespace AgenticPatterns.Tests;

public class CasePartitionTests
{
private static GoldenCase Case(string id, string reviewedBy) =>
new(id, "Q?", "A.", "contains", reviewedBy);

[Fact]
public void ReviewedCaseIsEvaluated()
{
var (evaluated, awaitingReview) = CasePartition.Partition([Case("reviewed", "alex")]);

Assert.Equal(["reviewed"], evaluated.Select(c => c.Id));
Assert.Empty(awaitingReview);
}

// This is the guarantee this task exists to pin: a case with no reviewer - exactly the shape
// ExtractTraceCase used to hand straight to the evaluator - must be EXCLUDED from evaluation,
// not merely present somewhere.
[Fact]
public void UnreviewedCaseIsExcludedFromEvaluationAndReportedAsAwaitingReview()
{
var (evaluated, awaitingReview) = CasePartition.Partition([Case("from-trace", reviewedBy: null!)]);

Assert.Empty(evaluated);
Assert.Equal(["from-trace"], awaitingReview.Select(c => c.Id));
}

[Fact]
public void EmptyReviewedByIsAlsoAwaitingReview()
{
var (evaluated, awaitingReview) = CasePartition.Partition([Case("blank", reviewedBy: "")]);

Assert.Empty(evaluated);
Assert.Single(awaitingReview);
}

[Fact]
public void MixedCorpusSplitsCorrectly()
{
var (evaluated, awaitingReview) = CasePartition.Partition(
[
Case("a", "alex"),
Case("b", null!),
Case("c", "jamie"),
Case("d", "")
]);

Assert.Equal(["a", "c"], evaluated.Select(c => c.Id));
Assert.Equal(["b", "d"], awaitingReview.Select(c => c.Id));
}

[Fact]
public void GateFailsWhenNothingWasEvaluatedEvenWithZeroFailures() =>
Assert.Equal(1, CasePartition.GateExitCode(evaluatedCount: 0, failureCount: 0));

[Fact]
public void GatePassesWhenSomethingWasEvaluatedAndNothingFailed() =>
Assert.Equal(0, CasePartition.GateExitCode(evaluatedCount: 3, failureCount: 0));

[Fact]
public void GateFailsOnAnyFailure() =>
Assert.Equal(1, CasePartition.GateExitCode(evaluatedCount: 3, failureCount: 1));
}
65 changes: 65 additions & 0 deletions AgenticPatterns.Tests/Fakes.cs
Original file line number Diff line number Diff line change
@@ -1,3 +1,6 @@
using System.Net;
using System.Text;
using System.Text.Json;
using CodeAct.AgentFramework.Execution;
using Microsoft.Extensions.AI;

Expand Down Expand Up @@ -77,3 +80,65 @@ public static T WithEnvironmentVariable<T>(string name, string? value, Func<T> b
finally { Environment.SetEnvironmentVariable(name, original); }
}
}

/// <summary>Stands in for the OpenAI endpoint at the <see cref="HttpMessageHandler"/> seam: no
/// port, no socket, no real network call. Answers every chat-completion request with an assistant
/// message that calls back whichever function the request offered, repeated
/// <paramref name="toolCallsPerTurn"/> times per response — so a Semantic Kernel auto-invocation
/// loop driven against this handler only ever stops if something (a filter) stops it.</summary>
internal sealed class ScriptedToolCallHttpHandler(int toolCallsPerTurn = 1) : HttpMessageHandler
{
private int _requestCount;

public int RequestCount => _requestCount;

/// <summary>Every request body this handler has answered, in order — lets a test assert the
/// model was never fed a budget-refusal message to paraphrase.</summary>
public List<string> RequestBodies { get; } = [];

protected override async Task<HttpResponseMessage> SendAsync(
HttpRequestMessage request, CancellationToken cancellationToken)
{
var body = request.Content is null
? ""
: await request.Content.ReadAsStringAsync(cancellationToken);
lock (RequestBodies) RequestBodies.Add(body);
var requestNumber = Interlocked.Increment(ref _requestCount);

using var doc = JsonDocument.Parse(body);
var toolName = doc.RootElement.GetProperty("tools")[0].GetProperty("function")
.GetProperty("name").GetString();

// Built via object graph + JsonSerializer, not a hand-assembled string: OpenAI's
// chat-completion response has enough nested braces that a raw string literal fights
// its own interpolation syntax.
var toolCalls = Enumerable.Range(0, toolCallsPerTurn).Select(i => new
{
id = $"call_{requestNumber}_{i}",
type = "function",
function = new { name = toolName, arguments = "{}" }
});
var responseBody = new
{
id = $"chatcmpl-{requestNumber}",
@object = "chat.completion",
created = 0,
model = "stub-model",
choices = new[]
{
new
{
index = 0,
message = new { role = "assistant", content = (string?)null, tool_calls = toolCalls },
finish_reason = "tool_calls"
}
},
usage = new { prompt_tokens = 1, completion_tokens = 1, total_tokens = 2 }
};

return new HttpResponseMessage(HttpStatusCode.OK)
{
Content = new StringContent(JsonSerializer.Serialize(responseBody), Encoding.UTF8, "application/json")
};
}
}
196 changes: 196 additions & 0 deletions AgenticPatterns.Tests/LlmAsJudgeTests.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,196 @@
using LLMAsJudge.AgentFramework;
using Microsoft.Extensions.AI;
using Microsoft.Extensions.AI.Evaluation;
using Xunit;

namespace AgenticPatterns.Tests;

public class LlmAsJudgeTests
{
[Theory]
[InlineData(null)] [InlineData("")] [InlineData("garbage")] [InlineData("{}")]
[InlineData("{\"winner\":\"a\"}")] [InlineData("{\"winner\":\"C\"}")]
public void AnythingUnexpectedIsIndeterminate(string? json) =>
Assert.Equal(Preference.Indeterminate, JudgeParsing.Parse(json));

[Fact] public void AParses() => Assert.Equal(Preference.A, JudgeParsing.Parse("{\"winner\":\"A\"}"));
[Fact] public void BParses() => Assert.Equal(Preference.B, JudgeParsing.Parse("{\"winner\":\"B\"}"));

// Malformed JSON that still throws on Deserialize (not just "returns null") - the brief's
// controller ruling: Parse must catch JsonException, not just handle null/missing keys.
[Theory]
[InlineData("[1,2,3]")]
[InlineData("not json at all {{{")]
public void MalformedJsonDoesNotThrow(string json) =>
Assert.Equal(Preference.Indeterminate, JudgeParsing.Parse(json));

[Theory]
[InlineData(Preference.A, true, true)]
[InlineData(Preference.B, true, false)]
[InlineData(Preference.A, false, false)]
[InlineData(Preference.B, false, true)]
public void ResolveTranslatesVerdictAndPositionIntoReferenceWin(
Preference verdict, bool referenceInPositionA, bool expectedReferenceWon) =>
Assert.Equal(expectedReferenceWon, JudgeParsing.Resolve(verdict, referenceInPositionA));

[Fact]
public void ResolveIsNullForIndeterminate() =>
Assert.Null(JudgeParsing.Resolve(Preference.Indeterminate, referenceInPositionA: true));

private static Trial InA(Preference verdict) => new(ReferenceInPositionA: true, verdict);
private static Trial InB(Preference verdict) => new(ReferenceInPositionA: false, verdict);

[Fact]
public void SummarizeCountsWinsAndIndeterminates()
{
var report = JudgeParsing.Summarize([
InA(Preference.A), // reference in A, judge picks A -> reference wins
InA(Preference.B), // reference in A, judge picks B -> other wins
InA(Preference.Indeterminate),
InB(Preference.B), // reference in B, judge picks B -> reference wins
InB(Preference.B)
]);

Assert.Equal(3, report.ReferenceWins);
Assert.Equal(1, report.OtherWins);
Assert.Equal(1, report.Indeterminate);
}

[Fact]
public void JudgeThatAlwaysPicksTheSameSlot_IsFullPositionBias()
{
var report = JudgeParsing.Summarize([
InA(Preference.A), InA(Preference.A), InB(Preference.A), InB(Preference.A)
]);

// Reference wins 100% of the time it sits in A, 0% of the time it sits in B.
Assert.Equal(1.0, report.PositionSwing);
}

[Fact]
public void JudgeThatIsSimplyWrong_IsNotPositionBias()
{
// The judge prefers the weaker candidate every single time, in both slots. That is a bad
// judge, not a position-dependent one - the pre-fix rate reported this as 100% bias
// because it folded the slot away before forming the statistic.
var report = JudgeParsing.Summarize([
InA(Preference.B), InA(Preference.B), InB(Preference.A), InB(Preference.A)
]);

Assert.Equal(4, report.OtherWins);
Assert.Equal(0.0, report.PositionSwing);
}

[Fact]
public void ConsistentJudge_HasNoPositionSwing()
{
var report = JudgeParsing.Summarize([
InA(Preference.A), InA(Preference.A), InB(Preference.B), InB(Preference.B)
]);

Assert.Equal(0.0, report.PositionSwing);
}

[Fact]
public void IdenticalOutcomesInDifferentSlotsProduceDifferentStatistics()
{
// The reference candidate wins every trial in both runs - identical outcomes - but one run
// never left slot A. The pre-fix rate folded the slot away before forming the number and
// so reported the same value for both, making the randomisation do no work at all.
var oneSlot = JudgeParsing.Summarize([InA(Preference.A), InA(Preference.A), InA(Preference.A)]);
var bothSlots = JudgeParsing.Summarize([InA(Preference.A), InB(Preference.B), InA(Preference.A)]);

Assert.Equal(oneSlot.ReferenceWins, bothSlots.ReferenceWins);
Assert.NotEqual(oneSlot.PositionSwing, bothSlots.PositionSwing);
}

[Fact]
public void SwingIsUnmeasurableWhenOnlyOneSlotWasSampled()
{
// Five coin flips land all five trials in one slot 6.25% of the time; Program.cs uses a
// balanced 3/2 shuffle so this cannot happen there, but the statistic still refuses to
// invent a position measurement from a single slot.
var report = JudgeParsing.Summarize([InA(Preference.A), InA(Preference.B), InA(Preference.A)]);

Assert.Null(report.PositionSwing);
}

[Fact]
public void IndeterminateVerdictsAreExcludedFromTheSwing()
{
// Both slots: every determinate verdict is a reference win, so both rates are 1 and the
// swing is 0. The unparseable verdicts are piled onto slot B only - count them in the
// denominator and slot B's rate drops to 1/3, inventing a swing out of noise.
var report = JudgeParsing.Summarize([
InA(Preference.A),
InB(Preference.B), InB(Preference.Indeterminate), InB(Preference.Indeterminate)
]);

Assert.Equal(2, report.Indeterminate);
Assert.Equal(0.0, report.PositionSwing);
}

[Fact]
public void SwingIsUnmeasurableWhenEverythingIsIndeterminate()
{
var report = JudgeParsing.Summarize([
InA(Preference.Indeterminate), InB(Preference.Indeterminate), InA(Preference.Indeterminate)
]);

Assert.Equal(3, report.Indeterminate);
Assert.Null(report.PositionSwing);
}
}

// Drives the real RubricJudgeEvaluator end to end - a scripted judge reply through
// EvaluateAsync - because the defect lived in how the evaluator reads that reply.
public class RubricJudgeEvaluatorTests
{
private static async Task<NumericMetric> JudgeSaysAsync(string judgeReply)
{
var client = new ScriptedChatClient(
new ChatResponse(new ChatMessage(ChatRole.Assistant, judgeReply)));

var result = await new RubricJudgeEvaluator().EvaluateAsync(
[new ChatMessage(ChatRole.User, "What warranty do the laptops come with?")],
new ChatResponse(new ChatMessage(ChatRole.Assistant, "Two years.")),
new ChatConfiguration(client));

return result.Get<NumericMetric>(RubricJudgeEvaluator.RubricScoreMetricName);
}

[Theory]
[InlineData("")] // truncated to nothing
[InlineData("not json")] // prose instead of JSON
[InlineData("{}")] // JSON, but no score
[InlineData("null")] // literal JSON null
[InlineData(" ")] // whitespace only
[InlineData("[1,2,3]")] // JSON of the wrong shape
[InlineData("{\"score\":0,\"justification\":\"x\"}")] // below the rubric floor
[InlineData("{\"score\":9,\"justification\":\"x\"}")] // above the rubric ceiling
public async Task UnreadableVerdictIsIndeterminate_NeverThrows_NeverANumber(string judgeReply)
{
var metric = await JudgeSaysAsync(judgeReply);

// Indeterminate is "no value", not 0: 0 is below the rubric's own floor of 1, so scoring
// an unreadable verdict as a number would rank it worse than the worst possible answer.
Assert.Null(metric.Value);
Assert.Contains("Indeterminate", metric.Reason);
}

[Fact]
public async Task ParseableVerdictKeepsScoreAndJustification()
{
var metric = await JudgeSaysAsync("{\"score\":4,\"justification\":\"Accurate but terse.\"}");

Assert.Equal(4, metric.Value);
Assert.Equal("Accurate but terse.", metric.Reason);
}

[Fact]
public async Task ScoreIsNeverBelowTheRubricFloor()
{
foreach (var reply in new[] { "", "not json", "{}", "null", "{\"score\":-3}" })
Assert.True(await JudgeSaysAsync(reply) is { Value: null or >= 1 });
}
}
Loading
Loading