Added Model Comparison assistant

This commit is contained in:
Dominic Neuburg committed 2026-09-26 21:01:36 +02:00
1 parent 1379ff6aab
commit 80669f98b3
38 files changed
+4272 -1

No files matched your search

@@ -0,0 +1,124 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how the judge's verdicts across a batch are tallied.
/// </summary>
/// <remarks>
/// The point of the tally is that it survives the presentation order changing from run to run: a
/// model which won column A in one run and column B in the next must still be counted as the same
/// winner both times.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonBatchJudgeSummaryTests
{
[Test]
public void AnEmptyBatchScoresNothing()
{
var summary = ModelComparisonBatchJudgeSummary.From([]);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.Zero);
Assert.That(summary.ScoredRunCount, Is.Zero);
Assert.That(summary.FirstModelWins, Is.Zero);
Assert.That(summary.SecondModelWins, Is.Zero);
Assert.That(summary.Ties, Is.Zero);
});
}
[Test]
public void WinsAreCountedByModelNotByColumn()
{
// The first model stands in column A twice and in column B once, and wins every time -- the tally has to follow the model, not the column:
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A),
Entry(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.COLUMN_B),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A),
};
var summary = ModelComparisonBatchJudgeSummary.From(entries);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(3));
Assert.That(summary.ScoredRunCount, Is.EqualTo(3));
Assert.That(summary.FirstModelWins, Is.EqualTo(3));
Assert.That(summary.SecondModelWins, Is.Zero);
Assert.That(summary.Ties, Is.Zero);
});
}
[Test]
public void ATieIsCountedAsATieRegardlessOfOrder()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.TIE),
Entry(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.TIE),
};
var summary = ModelComparisonBatchJudgeSummary.From(entries);
Assert.That(summary.Ties, Is.EqualTo(2));
}
[Test]
public void ARunWithoutACompletedVerdictIsNotScored()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A),
EntryWithoutAVerdict(),
};
var summary = ModelComparisonBatchJudgeSummary.From(entries);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(2), "Both runs happened.");
Assert.That(summary.ScoredRunCount, Is.EqualTo(1), "Only one run had a usable verdict.");
Assert.That(summary.FirstModelWins, Is.EqualTo(1));
});
}
private static ModelComparisonBatchEntry Entry(ModelComparisonPresentationOrder order, ModelComparisonVote judgePreferred) => new()
{
PresentationOrder = order,
Result = new ModelComparisonRunResult
{
First = Answer("First"),
Second = Answer("Second"),
Judge = new ModelComparisonJudgeVerdict
{
Label = "Judge",
Completed = true,
Preferred = judgePreferred,
},
},
};
private static ModelComparisonBatchEntry EntryWithoutAVerdict() => new()
{
PresentationOrder = ModelComparisonPresentationOrder.FIRST_MODEL_FIRST,
Result = new ModelComparisonRunResult
{
First = Answer("First"),
Second = Answer("Second"),
Judge = new ModelComparisonJudgeVerdict
{
Label = "Judge",
Completed = false,
},
},
};
private static ModelComparisonAnswer Answer(string label) => new()
{
Label = label,
Text = $"The answer of {label}.",
Completed = true,
};
}
@@ -0,0 +1,258 @@
using System.Runtime.CompilerServices;
using AIStudio.Assistants.ModelComparison;
using Microsoft.Extensions.Logging.Abstractions;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how a batch makes several independent runs of the same comparison.
/// </summary>
/// <remarks>
/// What is under test is that every run is genuinely independent -- its own request to both models,
/// its own presentation order -- not the single run itself, which <see cref="ModelComparisonRunnerTests"/>
/// already covers.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonBatchRunnerTests
{
private static readonly ModelComparisonRequest REQUEST = new("The document.", "Summarize it.");
[Test]
public async Task TheRequestedNumberOfRunsIsMade()
{
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 5,
CancellationToken.None);
var entries = await Task.WhenAll(tasks);
Assert.That(entries, Has.Length.EqualTo(5));
}
[Test]
public void EveryRunStartsWithoutWaitingForAnyOfThemToFinish()
{
// Nothing here completes on its own; Start must still hand back its tasks right away. The
// cancellation afterward is only cleanup, so the never-ending calls do not outlive the test:
using var cancellation = new CancellationTokenSource();
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, token) => Never(token)),
CreateParticipant("2", (_, token) => Never(token)),
REQUEST,
runCount: 3,
cancellation.Token);
try
{
Assert.That(tasks, Has.Count.EqualTo(3));
}
finally
{
cancellation.Cancel();
}
}
[Test]
public async Task NoMoreRunsThanTheLimitAskAModelAtOnce()
{
// Ten runs, twenty model calls between them; if every run started at once, every one of
// those twenty calls would be concurrent. A short delay gives the batch runner room to queue
// the rest behind the limit, which the peak below has to stay within -- two models per run:
var concurrentCalls = 0;
var peakConcurrentCalls = 0;
var peakLock = new object();
async IAsyncEnumerable<string> Ask(string _, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken token)
{
var current = Interlocked.Increment(ref concurrentCalls);
lock (peakLock)
peakConcurrentCalls = Math.Max(peakConcurrentCalls, current);
await Task.Delay(TimeSpan.FromMilliseconds(30), token);
Interlocked.Decrement(ref concurrentCalls);
yield return "An answer.";
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", Ask),
CreateParticipant("2", Ask),
REQUEST,
runCount: 10,
CancellationToken.None);
await Task.WhenAll(tasks);
var expectedPeak = ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS * 2;
Assert.That(peakConcurrentCalls, Is.LessThanOrEqualTo(expectedPeak), $"At most {ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS} runs at a time, two model calls each.");
}
[Test]
public async Task EachRunAsksBothModelsAfresh()
{
var firstAskedCount = 0;
var secondAskedCount = 0;
IAsyncEnumerable<string> AskFirst(string _, CancellationToken __)
{
Interlocked.Increment(ref firstAskedCount);
return Chunks("An answer.");
}
IAsyncEnumerable<string> AskSecond(string _, CancellationToken __)
{
Interlocked.Increment(ref secondAskedCount);
return Chunks("Another answer.");
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", AskFirst),
CreateParticipant("2", AskSecond),
REQUEST,
runCount: 4,
CancellationToken.None);
await Task.WhenAll(tasks);
Assert.Multiple(() =>
{
Assert.That(firstAskedCount, Is.EqualTo(4));
Assert.That(secondAskedCount, Is.EqualTo(4));
});
}
[Test]
public async Task EveryRunCompletesOnItsOwnEvenWhenOneFails()
{
// Every third run fails; the rest must not be dragged down with it:
var callNumber = 0;
IAsyncEnumerable<string> Flaky(string _, CancellationToken __)
{
var thisCall = Interlocked.Increment(ref callNumber);
return thisCall % 3 == 0 ? Failing() : Chunks("An answer.");
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", Flaky),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 6,
CancellationToken.None);
var entries = await Task.WhenAll(tasks);
Assert.Multiple(() =>
{
Assert.That(entries, Has.Length.EqualTo(6));
Assert.That(entries.Count(entry => entry.Result.BothCompleted), Is.EqualTo(4));
Assert.That(entries.Count(entry => !entry.Result.BothCompleted), Is.EqualTo(2));
});
}
[Test]
public void ARunCountBelowOneIsRefused()
{
var batchRunner = CreateBatchRunner();
Assert.That(() => batchRunner.Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 0,
CancellationToken.None), Throws.TypeOf<ArgumentOutOfRangeException>());
}
[Test]
public async Task BothPresentationOrdersCanComeUp()
{
// With enough runs, drawing only ever one of the two orders would be a bug, not bad luck:
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 60,
CancellationToken.None);
var entries = await Task.WhenAll(tasks);
var orders = entries.Select(entry => entry.PresentationOrder).Distinct().ToArray();
Assert.That(orders, Is.EquivalentTo(Enum.GetValues<ModelComparisonPresentationOrder>()));
}
[Test]
public async Task WithAJudgeEveryRunHasItsOwnVerdict()
{
var judgeAskedCount = 0;
IAsyncEnumerable<string> AskJudge(string _, CancellationToken __)
{
Interlocked.Increment(ref judgeAskedCount);
return Chunks("""{"preferred": "A", "reasoning": "n/a"}""");
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 3,
CancellationToken.None,
judge: CreateParticipant("judge", AskJudge));
var entries = await Task.WhenAll(tasks);
Assert.Multiple(() =>
{
Assert.That(judgeAskedCount, Is.EqualTo(3));
Assert.That(entries.All(entry => entry.Result.Judge is { Completed: true }), Is.True);
});
}
private static ModelComparisonBatchRunner CreateBatchRunner() => new(new ModelComparisonRunner(NullLogger.Instance));
private static ModelComparisonParticipant CreateParticipant(string id, Func<string, CancellationToken, IAsyncEnumerable<string>> ask) => new()
{
Label = $"Model {id}",
Ask = ask,
};
private static async IAsyncEnumerable<string> Chunks(params string[] chunks)
{
foreach (var chunk in chunks)
{
await Task.Yield();
yield return chunk;
}
}
private static async IAsyncEnumerable<string> Never([EnumeratorCancellation] CancellationToken token)
{
try
{
await Task.Delay(Timeout.Infinite, token);
}
catch (OperationCanceledException)
{
// Expected once the test cleans up; there is nothing left to answer with.
}
#pragma warning disable CS0162 // Unreachable code: the yield is what makes this an iterator
yield break;
#pragma warning restore CS0162
}
private static async IAsyncEnumerable<string> Failing()
{
await Task.Yield();
throw new InvalidOperationException("The model is not reachable.");
#pragma warning disable CS0162 // Unreachable code: the yield is what makes this an iterator
yield break;
#pragma warning restore CS0162
}
}
@@ -0,0 +1,118 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how the user's own votes across a batch are tallied.
/// </summary>
/// <remarks>
/// Mirrors <see cref="ModelComparisonBatchJudgeSummaryTests"/>: the tally has to survive the
/// presentation order changing from run to run, and a run without a vote must not be counted.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonBatchVoteSummaryTests
{
[Test]
public void AnEmptyBatchScoresNothing()
{
var summary = ModelComparisonBatchVoteSummary.From([], []);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.Zero);
Assert.That(summary.ScoredRunCount, Is.Zero);
Assert.That(summary.FirstModelWins, Is.Zero);
Assert.That(summary.SecondModelWins, Is.Zero);
Assert.That(summary.Ties, Is.Zero);
});
}
[Test]
public void WinsAreCountedByModelNotByColumn()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
};
// The first model stands in column A twice and in column B once, and the user votes for it every time:
var votes = new ModelComparisonVote?[]
{
ModelComparisonVote.COLUMN_A,
ModelComparisonVote.COLUMN_B,
ModelComparisonVote.COLUMN_A,
};
var summary = ModelComparisonBatchVoteSummary.From(entries, votes);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(3));
Assert.That(summary.ScoredRunCount, Is.EqualTo(3));
Assert.That(summary.FirstModelWins, Is.EqualTo(3));
Assert.That(summary.SecondModelWins, Is.Zero);
});
}
[Test]
public void ARunWithoutAVoteIsNotScored()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
};
var votes = new ModelComparisonVote?[] { ModelComparisonVote.COLUMN_A, null };
var summary = ModelComparisonBatchVoteSummary.From(entries, votes);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(2), "Both runs happened.");
Assert.That(summary.ScoredRunCount, Is.EqualTo(1), "Only one run was voted on.");
Assert.That(summary.FirstModelWins, Is.EqualTo(1));
});
}
[Test]
public void AVoteWithNoMatchingEntryIndexIsNotScored()
{
// Fewer votes than entries -- the batch was left before reaching the rest:
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
};
var votes = new ModelComparisonVote?[] { ModelComparisonVote.TIE };
var summary = ModelComparisonBatchVoteSummary.From(entries, votes);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(2));
Assert.That(summary.ScoredRunCount, Is.EqualTo(1));
Assert.That(summary.Ties, Is.EqualTo(1));
});
}
private static ModelComparisonBatchEntry Entry(ModelComparisonPresentationOrder order) => new()
{
PresentationOrder = order,
Result = new ModelComparisonRunResult
{
First = Answer("First"),
Second = Answer("Second"),
},
};
private static ModelComparisonAnswer Answer(string label) => new()
{
Label = label,
Text = $"The answer of {label}.",
Completed = true,
};
}
@@ -0,0 +1,101 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks the prompt the judge is asked with.
/// </summary>
[TestFixture]
public sealed class ModelComparisonJudgeRequestTests
{
private static readonly ModelComparisonRequest REQUEST = new("The document.", "Summarize it.");
[Test]
public void ThePromptCarriesTheUserInstructionsAndBothAnswers()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, "A good summary keeps every number.", "Answer from the first model.", "Answer from the second model.");
var prompt = judgeRequest.ToPrompt();
Assert.Multiple(() =>
{
Assert.That(prompt, Does.Contain("A good summary keeps every number."));
Assert.That(prompt, Does.Contain("Answer from the first model."));
Assert.That(prompt, Does.Contain("Answer from the second model."));
Assert.That(prompt, Does.Contain(REQUEST.ToPrompt()));
});
}
[Test]
public void WhiteSpaceAroundTheInstructionsIsDropped()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, " Keep it short. \n", "First.", "Second.");
Assert.That(judgeRequest.Instructions, Is.EqualTo("Keep it short."));
}
[Test]
public void WithoutInstructionsThePromptStillCarriesAGenericCriterion()
{
// The judge already sees the request and both answers, so the field is a refinement, not a requirement:
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "Answer from the first model.", "Answer from the second model.");
var prompt = judgeRequest.ToPrompt();
Assert.Multiple(() =>
{
Assert.That(prompt, Does.Contain("better answers the question or better completes the task"));
Assert.That(prompt, Does.Contain("Answer from the first model."));
Assert.That(prompt, Does.Contain("Answer from the second model."));
Assert.That(prompt, Does.Not.Contain("Pay particular attention to this:"));
});
}
[Test]
public void WithoutALanguageNameThePromptAsksForNoParticularOne()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.");
Assert.That(judgeRequest.ToPrompt(), Does.Not.Contain("Write the reasoning in"));
}
[Test]
public void ALanguageNameTellsTheJudgeWhatToWriteIn()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.", "German");
Assert.That(judgeRequest.ToPrompt(), Does.Contain("Write the reasoning in German"));
}
[Test]
public void TheAnswersAreLabeledAAndBNotByModelOrder()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "Answer from the first model.", "Answer from the second model.");
var prompt = judgeRequest.ToPrompt();
Assert.Multiple(() =>
{
Assert.That(prompt, Does.Contain("# Answer A"));
Assert.That(prompt, Does.Contain("# Answer B"));
Assert.That(prompt, Does.Not.Contain("# Answer 1"));
Assert.That(prompt, Does.Not.Contain("# Answer 2"));
});
}
[Test]
public void WithoutACacheBusterThePromptIsUnchanged()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.");
Assert.That(judgeRequest.ToPrompt(), Is.EqualTo(judgeRequest.ToPrompt(string.Empty)));
}
[Test]
public void TwoDifferentCacheBustersBuildDifferentPrompts()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.");
Assert.That(judgeRequest.ToPrompt("one"), Is.Not.EqualTo(judgeRequest.ToPrompt("two")));
}
}
@@ -0,0 +1,79 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how the judge's raw answer is turned into a verdict.
/// </summary>
/// <remarks>
/// The judge is asked for one JSON object and nothing else, but a model rarely holds to that
/// perfectly. Parsing has to look past a stray sentence around the object, and has to end up with a
/// verdict which is not <see cref="ModelComparisonJudgeVerdict.Completed"/> rather than throw when
/// there is nothing usable to read.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonJudgeVerdictTests
{
[Test]
public void APlainJsonObjectIsRead()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, """{"preferred": "A", "reasoning": "It kept every number from the document."}""");
Assert.Multiple(() =>
{
Assert.That(verdict.Completed, Is.True);
Assert.That(verdict.Label, Is.EqualTo("Judge"));
Assert.That(verdict.Preferred, Is.EqualTo(ModelComparisonVote.COLUMN_A));
Assert.That(verdict.Reasoning, Is.EqualTo("It kept every number from the document."));
});
}
[Test]
public void APreferenceIsReadWithoutRegardToCase()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, """{"preferred": "tie", "reasoning": "Both cover the same points."}""");
Assert.That(verdict.Preferred, Is.EqualTo(ModelComparisonVote.TIE));
}
[Test]
public void TextAroundTheJsonObjectIsIgnored()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, """
Sure, here is my verdict:
```json
{"preferred": "B", "reasoning": "It is shorter and just as complete."}
```
""");
Assert.Multiple(() =>
{
Assert.That(verdict.Completed, Is.True);
Assert.That(verdict.Preferred, Is.EqualTo(ModelComparisonVote.COLUMN_B));
});
}
[Test]
public void AJudgeWhichDidNotAnswerIsNotCompleted()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", false, string.Empty);
Assert.Multiple(() =>
{
Assert.That(verdict.Completed, Is.False);
Assert.That(verdict.Preferred, Is.Null);
Assert.That(verdict.Reasoning, Is.Empty);
});
}
[TestCase("Sorry, I cannot help with that.")]
[TestCase("""{"reasoning": "Missing the preference field."}""")]
[TestCase("""{"preferred": "C", "reasoning": "Not one of the three allowed values."}""")]
[TestCase("""{"preferred": "A", "reasoning": }""")]
public void AnUnreadableAnswerIsNotCompleted(string judgeText)
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, judgeText);
Assert.That(verdict.Completed, Is.False);
}
}
@@ -0,0 +1,104 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks which answer stands in which column.
/// </summary>
/// <remarks>
/// A wrong assignment does not crash. It shows the answer of one model under the name of the other
/// once the vote is in, and nothing on the screen looks off.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonPresentationOrderTests
{
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, "The first model", "The second model")]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, "The second model", "The first model")]
public void TheColumnsShowTheAnswersOfTheModelsTheOrderSays(ModelComparisonPresentationOrder order, string expectedInColumnA, string expectedInColumnB)
{
var run = CreateRunResult();
Assert.Multiple(() =>
{
Assert.That(order.InColumnA(run).Label, Is.EqualTo(expectedInColumnA));
Assert.That(order.InColumnB(run).Label, Is.EqualTo(expectedInColumnB));
});
}
[Test]
public void TheTextOfAColumnIsTheTextOfTheModelWhoseNameItCarries()
{
// The name and the text travel together in one answer, so a column can never mix them up:
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
{
Assert.That(order.InColumnA(run).Text, Does.Contain(order.InColumnA(run).Label), $"{order}, column A");
Assert.That(order.InColumnB(run).Text, Does.Contain(order.InColumnB(run).Label), $"{order}, column B");
}
}
[Test]
public void AnswersInTheTwoColumnsAreNeverTheSame()
{
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
Assert.That(order.InColumnA(run).Label, Is.Not.EqualTo(order.InColumnB(run).Label), order.ToString());
}
[Test]
public void EveryOrderIsKnown()
{
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
{
Assert.DoesNotThrow(() => order.InColumnA(run), order.ToString());
Assert.DoesNotThrow(() => order.InColumnB(run), order.ToString());
}
}
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A, ModelComparisonModelChoice.FIRST)]
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_B, ModelComparisonModelChoice.SECOND)]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.COLUMN_A, ModelComparisonModelChoice.SECOND)]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.COLUMN_B, ModelComparisonModelChoice.FIRST)]
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.TIE, ModelComparisonModelChoice.TIE)]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.TIE, ModelComparisonModelChoice.TIE)]
public void AColumnBasedVoteIsTranslatedBackToItsModel(ModelComparisonPresentationOrder order, ModelComparisonVote vote, ModelComparisonModelChoice expected)
{
Assert.That(order.ToModelChoice(vote), Is.EqualTo(expected));
}
[Test]
public void TranslatingAVoteAndBackAgainIsConsistentWithTheColumns()
{
// Whichever model a column names, translating that model's vote back must land on the model actually standing there:
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
{
var modelInColumnA = order.InColumnA(run).Label;
var choiceForColumnA = order.ToModelChoice(ModelComparisonVote.COLUMN_A);
var expectedLabel = choiceForColumnA is ModelComparisonModelChoice.FIRST ? "The first model" : "The second model";
Assert.That(modelInColumnA, Is.EqualTo(expectedLabel), order.ToString());
}
}
private static ModelComparisonRunResult CreateRunResult() => new()
{
First = new ModelComparisonAnswer
{
Label = "The first model",
Text = "The answer of The first model.",
Completed = true,
},
Second = new ModelComparisonAnswer
{
Label = "The second model",
Text = "The answer of The second model.",
Completed = true,
},
};
}
@@ -0,0 +1,144 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks the text both models receive.
/// </summary>
/// <remarks>
/// A difference in this text would be measured as a difference between the models, so what matters
/// is that the shape stays the same, and that a request without a context is not dressed up.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonRequestTests
{
[Test]
public void AQuestionAloneIsSentAsItIs()
{
var request = new ModelComparisonRequest(string.Empty, "What is the capital of France?");
Assert.That(request.ToPrompt(), Is.EqualTo("What is the capital of France?"));
}
[Test]
public void AContextAndAQuestionAreSeparatedByHeadings()
{
var request = new ModelComparisonRequest("The document.", "Summarize it.");
Assert.That(request.ToPrompt(), Is.EqualTo("# Context\nThe document.\n\n# Question or task\nSummarize it.").Or.EqualTo("# Context\r\nThe document.\r\n\r\n# Question or task\r\nSummarize it."));
}
[TestCase("")]
[TestCase(" ")]
[TestCase("\r\n\t\n")]
public void AContextOfWhiteSpaceCountsAsNone(string context)
{
var request = new ModelComparisonRequest(context, "Summarize it.");
Assert.Multiple(() =>
{
Assert.That(request.Context, Is.Empty);
Assert.That(request.ToPrompt(), Is.EqualTo("Summarize it."));
});
}
[Test]
public void WhiteSpaceAroundTheTextsIsDropped()
{
var request = new ModelComparisonRequest("\n\nThe document.\n", " Summarize it. ");
Assert.Multiple(() =>
{
Assert.That(request.Context, Is.EqualTo("The document."));
Assert.That(request.Question, Is.EqualTo("Summarize it."));
});
}
[Test]
public void AShortQuestionIsItsOwnPreview()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.GetQuestionPreview(300), Is.EqualTo("Summarize it."));
}
[Test]
public void ALongQuestionIsCutAndMarked()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize this document in a few short sentences.");
Assert.That(request.GetQuestionPreview(18), Is.EqualTo("Summarize this doc…"));
}
[Test]
public void ACutDoesNotLeaveHalfOfACharacter()
{
// The emoji is two chars long. A cut after the first of them would leave a lone surrogate:
var request = new ModelComparisonRequest(string.Empty, "aaaa😀bbb");
Assert.That(request.GetQuestionPreview(5), Is.EqualTo("aaaa…"));
}
[Test]
public void ThePreviewOfAnEmptyLengthIsRefused()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.Throws<ArgumentOutOfRangeException>(() => request.GetQuestionPreview(0));
}
[Test]
public void TheSameInputAlwaysBuildsTheSamePrompt()
{
var first = new ModelComparisonRequest("The document.", "Summarize it.");
var second = new ModelComparisonRequest("The document.", "Summarize it.");
Assert.That(first.ToPrompt(), Is.EqualTo(second.ToPrompt()));
}
[Test]
public void WithoutALanguageNameThePromptNamesNone()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.ToPrompt(), Does.Not.Contain("Answer in"));
}
[Test]
public void ALanguageNameIsAppendedAsAnInstruction()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.", "German");
Assert.That(request.ToPrompt(), Is.EqualTo("Summarize it.\n\nAnswer in German."));
}
[Test]
public void WithoutACacheBusterThePromptIsUnchanged()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.ToPrompt(), Is.EqualTo(request.ToPrompt(string.Empty)));
}
[Test]
public void ACacheBusterIsAppendedAsAPlainlyLabeledLine()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
var prompt = request.ToPrompt("a1b2c3");
Assert.Multiple(() =>
{
Assert.That(prompt, Does.StartWith("Summarize it."));
Assert.That(prompt, Does.Contain("a1b2c3"));
});
}
[Test]
public void TwoDifferentCacheBustersBuildDifferentPrompts()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.ToPrompt("one"), Is.Not.EqualTo(request.ToPrompt("two")));
}
}
@@ -0,0 +1,394 @@
using System.Collections.Concurrent;
using System.Runtime.CompilerServices;
using AIStudio.Assistants.ModelComparison;
using Microsoft.Extensions.Logging.Abstractions;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how a comparison run treats two models: that they are asked the same, at the same time,
/// and that whatever one of them does cannot spoil the result of the other.
/// </summary>
/// <remarks>
/// The models are made up here. What is under test is the run, not a provider.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonRunnerTests
{
private static readonly ModelComparisonRequest REQUEST = new("The document.", "Summarize it.");
private static readonly TimeSpan PATIENCE = TimeSpan.FromSeconds(5);
[Test]
public async Task BothModelsReceiveTheSamePrompt()
{
// Not REQUEST.ToPrompt() itself: the run adds a cache-busting request ID neither call here supplies:
var prompts = new ConcurrentBag<string>();
IAsyncEnumerable<string> Ask(string prompt, CancellationToken _)
{
prompts.Add(prompt);
return Chunks("An answer.");
}
await CreateRunner().RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(prompts, Has.Count.EqualTo(2));
Assert.That(prompts.Distinct().Count(), Is.EqualTo(1), "Both models must receive the exact same prompt as each other.");
});
}
[Test]
public async Task RepeatedRunsGetDifferentPromptsSoACachingGatewayCannotAnswerFromACache()
{
var prompts = new ConcurrentBag<string>();
IAsyncEnumerable<string> Ask(string prompt, CancellationToken _)
{
prompts.Add(prompt);
return Chunks("An answer.");
}
var runner = CreateRunner();
await runner.RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
await runner.RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
// Both models of the same run still share one prompt, the same as in BothModelsReceiveTheSamePrompt;
// it is the two runs that must differ from each other, so a caching gateway cannot answer a
// repeat run from what it already has cached for the first one:
Assert.That(prompts.Distinct().Count(), Is.EqualTo(2), "Each run must see a prompt of its own, distinct from the other run's.");
}
[Test]
public async Task TheModelsAreAskedAtTheSameTime()
{
var started = 0;
var bothStarted = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously);
//
// Neither model answers before the other one has been asked. Asked one after the other, the
// first would wait in vain, run out of patience and fail, and so would the assertion below.
//
async IAsyncEnumerable<string> Ask(string _, [EnumeratorCancellation] CancellationToken token)
{
if (Interlocked.Increment(ref started) == 2)
bothStarted.SetResult();
await bothStarted.Task.WaitAsync(PATIENCE, token);
yield return "An answer.";
}
var result = await CreateRunner().RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
Assert.That(result.BothCompleted, Is.True);
}
[Test]
public async Task EachAnswerStaysWithTheModelWhichGaveIt()
{
var secondIsDone = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously);
// The first model finishes last, so a result assembled in the order of arrival would swap them:
async IAsyncEnumerable<string> AskFirst(string _, [EnumeratorCancellation] CancellationToken token)
{
await secondIsDone.Task.WaitAsync(PATIENCE, token);
yield return "from the first";
}
async IAsyncEnumerable<string> AskSecond(string _, [EnumeratorCancellation] CancellationToken __)
{
yield return "from the second";
secondIsDone.TrySetResult();
await Task.CompletedTask;
}
var result = await CreateRunner().RunAsync(CreateParticipant("1", AskFirst), CreateParticipant("2", AskSecond), REQUEST, CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Text, Is.EqualTo("from the first"));
Assert.That(result.First.Label, Is.EqualTo("Model 1"));
Assert.That(result.Second.Text, Is.EqualTo("from the second"));
Assert.That(result.Second.Label, Is.EqualTo("Model 2"));
});
}
[Test]
public async Task AnAnswerIsTheChunksJoinedWithoutTheThinking()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("<think>Let me think.", "</think>", "Hello", " world ")),
CreateParticipant("2", (_, _) => Chunks("Hi")),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Text, Is.EqualTo("Hello world"));
Assert.That(result.First.Completed, Is.True);
});
}
[Test]
public async Task TheTimesAreMeasuredOnTheClockOfTheRun()
{
var clock = new ManualTimeProvider();
async IAsyncEnumerable<string> AskFirst(string _, [EnumeratorCancellation] CancellationToken __)
{
clock.Advance(TimeSpan.FromMilliseconds(200));
yield return "Hi";
clock.Advance(TimeSpan.FromMilliseconds(300));
yield return " there";
await Task.CompletedTask;
}
// The second model does not touch the clock, so the times of the first one are the only ones which move:
var result = await CreateRunner(clock).RunAsync(
CreateParticipant("1", AskFirst),
CreateParticipant("2", (_, _) => Failing()),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.FirstTokenTime, Is.EqualTo(TimeSpan.FromMilliseconds(200)));
Assert.That(result.First.TotalTime, Is.EqualTo(TimeSpan.FromMilliseconds(500)));
});
}
[Test]
public async Task AModelWhichFailsDoesNotTakeTheOtherOneDown()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Failing()),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.True);
Assert.That(result.First.Text, Is.EqualTo("An answer."));
Assert.That(result.Second.Completed, Is.False);
Assert.That(result.Second.TotalTime, Is.Null, "A failed request has no total time.");
Assert.That(result.BothCompleted, Is.False);
});
}
[Test]
public async Task AnAnswerWhichSaysNothingCountsAsAFailure()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks()),
CreateParticipant("2", (_, _) => Chunks(" ", "\n")),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.False);
Assert.That(result.Second.Completed, Is.False);
Assert.That(result.BothCompleted, Is.False);
});
}
[Test]
public async Task AnAnswerWhichEndsInsideTheThinkingCountsAsAFailure()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("<think>Still thinking, and out of room")),
CreateParticipant("2", (_, _) => Chunks("An answer.")),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.False);
Assert.That(result.First.Text, Is.Empty);
Assert.That(result.Second.Completed, Is.True);
});
}
[Test]
public async Task ACanceledRunEndsBothModelsWithoutACompletedAnswer()
{
using var cancellation = new CancellationTokenSource();
async IAsyncEnumerable<string> Ask(string _, [EnumeratorCancellation] CancellationToken token)
{
yield return "A first piece";
await Task.Delay(Timeout.Infinite, token);
}
cancellation.CancelAfter(TimeSpan.FromMilliseconds(50));
var result = await CreateRunner().RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, cancellation.Token);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.False);
Assert.That(result.Second.Completed, Is.False);
Assert.That(result.First.TotalTime, Is.Null);
Assert.That(result.BothCompleted, Is.False);
});
}
[Test]
public async Task TheJudgeIsAskedAboutBothCompleteAnswers()
{
string? judgePrompt = null;
IAsyncEnumerable<string> AskJudge(string prompt, CancellationToken _)
{
judgePrompt = prompt;
return Chunks("""{"preferred": "A", "reasoning": "It is more complete."}""");
}
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("Answer from the first model.")),
CreateParticipant("2", (_, _) => Chunks("Answer from the second model.")),
REQUEST,
CancellationToken.None,
judge: CreateParticipant("judge", AskJudge),
judgeInstructions: "Prefer the more complete answer.");
Assert.Multiple(() =>
{
Assert.That(judgePrompt, Does.Contain("Answer from the first model."));
Assert.That(judgePrompt, Does.Contain("Answer from the second model."));
Assert.That(judgePrompt, Does.Contain("Prefer the more complete answer."));
Assert.That(result.Judge, Is.Not.Null);
Assert.That(result.Judge!.Completed, Is.True);
Assert.That(result.Judge.Preferred, Is.EqualTo(ModelComparisonVote.COLUMN_A));
Assert.That(result.Judge.Label, Is.EqualTo("Model judge"));
});
}
[Test]
public async Task TheJudgeIsToldTheAnswersInPresentationOrder()
{
string? judgePrompt = null;
IAsyncEnumerable<string> AskJudge(string prompt, CancellationToken _)
{
judgePrompt = prompt;
return Chunks("""{"preferred": "A", "reasoning": "n/a"}""");
}
// The second model is drawn first, so it is what the judge is told is "Answer A":
await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("Answer from the first model.")),
CreateParticipant("2", (_, _) => Chunks("Answer from the second model.")),
REQUEST,
CancellationToken.None,
ModelComparisonPresentationOrder.SECOND_MODEL_FIRST,
CreateParticipant("judge", AskJudge));
Assert.Multiple(() =>
{
Assert.That(judgePrompt, Does.Match(@"# Answer A\r?\nAnswer from the second model\."));
Assert.That(judgePrompt, Does.Match(@"# Answer B\r?\nAnswer from the first model\."));
});
}
[Test]
public async Task TheJudgeIsNotAskedWhenAModelDidNotAnswer()
{
var judgeWasAsked = false;
IAsyncEnumerable<string> AskJudge(string _, CancellationToken __)
{
judgeWasAsked = true;
return Chunks("""{"preferred": "TIE", "reasoning": "n/a"}""");
}
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Failing()),
REQUEST,
CancellationToken.None,
judge: CreateParticipant("judge", AskJudge));
Assert.Multiple(() =>
{
Assert.That(judgeWasAsked, Is.False, "A judge would only ever see one complete answer, which is nothing to judge.");
Assert.That(result.Judge, Is.Null);
});
}
[Test]
public async Task WithoutAJudgeTheResultHasNoVerdict()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
CancellationToken.None);
Assert.That(result.Judge, Is.Null);
}
[Test]
public async Task AJudgeWhichFailsLeavesAVerdictWhichIsNotCompleted()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
CancellationToken.None,
judge: CreateParticipant("judge", (_, _) => Failing()));
Assert.Multiple(() =>
{
Assert.That(result.Judge, Is.Not.Null);
Assert.That(result.Judge!.Completed, Is.False);
});
}
private static ModelComparisonRunner CreateRunner(TimeProvider? clock = null) => new(NullLogger.Instance, clock);
private static ModelComparisonParticipant CreateParticipant(string id, Func<string, CancellationToken, IAsyncEnumerable<string>> ask) => new()
{
Label = $"Model {id}",
Ask = ask,
};
private static async IAsyncEnumerable<string> Chunks(params string[] chunks)
{
foreach (var chunk in chunks)
{
await Task.Yield();
yield return chunk;
}
}
private static async IAsyncEnumerable<string> Failing()
{
await Task.Yield();
throw new InvalidOperationException("The model is not reachable.");
#pragma warning disable CS0162 // Unreachable code: the yield is what makes this an iterator
yield break;
#pragma warning restore CS0162
}
/// <summary>
/// A clock which only moves when a test says so.
/// </summary>
private sealed class ManualTimeProvider : TimeProvider
{
private long ticks;
// One timestamp is one tick of a TimeSpan, so the elapsed time of the runner is exactly what Advance added:
public override long TimestampFrequency => TimeSpan.TicksPerSecond;
public void Advance(TimeSpan by) => Interlocked.Add(ref this.ticks, by.Ticks);
public override long GetTimestamp() => Interlocked.Read(ref this.ticks);
}
}