This commit is contained in:
Dominic Neuburg 2026-09-26 21:09:56 +02:00 committed by GitHub
commit b030000862
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
38 changed files with 4272 additions and 1 deletions

View File

@ -2134,6 +2134,240 @@ UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::LOGVIEWER::ASSISTANTLOGVIEWER::T469116133
-- Clear
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::LOGVIEWER::ASSISTANTLOGVIEWER::T77955010"] = "Clear"
-- Both models receive this text together with your question, for example the document they are supposed to summarize.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1084467271"] = "Both models receive this text together with your question, for example the document they are supposed to summarize."
-- The judge saw both answers as equally good, matching your own vote.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T110452334"] = "The judge saw both answers as equally good, matching your own vote."
-- You do not know yet which model wrote which answer, and the columns are in random order. Compare them by what they say.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1255084431"] = "You do not know yet which model wrote which answer, and the columns are in random order. Compare them by what they say."
-- Skipping the remaining votes. Waiting for the rest of the batch to finish...
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1308855949"] = "Skipping the remaining votes. Waiting for the rest of the batch to finish..."
-- Setup
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T137864150"] = "Setup"
-- Repeat the same comparison several times to see how consistently the models -- and the judge -- come out the same way. Every run gets its own blind vote.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1382748771"] = "Repeat the same comparison several times to see how consistently the models -- and the judge -- come out the same way. Every run gets its own blind vote."
-- Run {0}: no usable answer, nothing to vote on.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1407798402"] = "Run {0}: no usable answer, nothing to vote on."
-- Answer A
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1420338340"] = "Answer A"
-- 20 (heavy load on the models)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1426001529"] = "20 (heavy load on the models)"
-- Run {0}: the models are done, waiting for the judge.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1451649171"] = "Run {0}: the models are done, waiting for the judge."
-- Answer B
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1470671197"] = "Answer B"
-- Please provide a question or task for both models.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1481551697"] = "Please provide a question or task for both models."
-- Delete this preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T148310729"] = "Delete this preset"
-- The judge preferred {0}, differing from your own vote.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1499258629"] = "The judge preferred {0}, differing from your own vote."
-- All {0} runs are done.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1508408704"] = "All {0} runs are done."
-- Run {0}: queued, waiting for a free slot.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1600973378"] = "Run {0}: queued, waiting for a free slot."
-- Your request
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1607440787"] = "Your request"
-- The preset '{0}' is deleted right away. This cannot be undone.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T163336631"] = "The preset '{0}' is deleted right away. This cannot be undone."
-- Your question or task
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1646098156"] = "Your question or task"
-- First token after {0}, complete after {1}
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1658271318"] = "First token after {0}, complete after {1}"
-- Send the same request to two models and compare their answers side by side. You vote blindly for the better answer, and only after your vote you learn which model wrote which answer.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1766730560"] = "Send the same request to two models and compare their answers side by side. You vote blindly for the better answer, and only after your vote you learn which model wrote which answer."
-- Start a new request
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1767930003"] = "Start a new request"
-- Run {0}: still working.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1838060603"] = "Run {0}: still working."
-- Update preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1860438243"] = "Update preset"
-- Judge verdicts
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1879874010"] = "Judge verdicts"
-- Judge
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2094884932"] = "Judge"
-- Run {0}: the models and the judge are done, your vote is missing.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2115467037"] = "Run {0}: the models and the judge are done, your vote is missing."
-- Save as new preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2148615337"] = "Save as new preset"
-- Skipping the remaining votes. Waiting for the judges to finish...
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2167132827"] = "Skipping the remaining votes. Waiting for the judges to finish..."
-- 1 (single comparison)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2267053576"] = "1 (single comparison)"
-- Which answer is better?
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2388547151"] = "Which answer is better?"
-- The judge preferred {0}, matching your own vote.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2457939612"] = "The judge preferred {0}, matching your own vote."
-- Yes, delete it
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2466176832"] = "Yes, delete it"
-- Equally good
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T248601509"] = "Equally good"
-- Context: {0} characters
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2553774988"] = "Context: {0} characters"
-- Progress Runs
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2638545408"] = "Progress Runs"
-- Finish
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2670325426"] = "Finish"
-- Both models are answering. This can take a while.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2709682697"] = "Both models are answering. This can take a while."
-- Run {0}: your vote was skipped.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2715329048"] = "Run {0}: your vote was skipped."
-- Both are equally good
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2723559248"] = "Both are equally good"
-- {0} of {1} runs have no usable judge verdict.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2725027740"] = "{0} of {1} runs have no usable judge verdict."
-- Preset from {0}
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2841002060"] = "Preset from {0}"
-- Context (optional)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2925041491"] = "Context (optional)"
-- {0} of {1} runs have no usable vote -- never cast, or the run itself never got a usable answer -- and are not counted above.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3101172189"] = "{0} of {1} runs have no usable vote -- never cast, or the run itself never got a usable answer -- and are not counted above."
-- Answer A is better
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3123852190"] = "Answer A is better"
-- Skip remaining votes
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3188579167"] = "Skip remaining votes"
-- The judge saw both answers as equally good, while you preferred one of them.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3189061594"] = "The judge saw both answers as equally good, while you preferred one of them."
-- Skip this run
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3334593867"] = "Skip this run"
-- Load the context from file
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3370653375"] = "Load the context from file"
-- More runs are still being generated in the background.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3475250486"] = "More runs are still being generated in the background."
-- LLM judge (optional)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3486621434"] = "LLM judge (optional)"
-- Next run
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3533713133"] = "Next run"
-- At least one of the models did not answer in this run, so there is nothing to compare.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3561467042"] = "At least one of the models did not answer in this run, so there is nothing to compare."
-- Compare
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3581545044"] = "Compare"
-- What should the judge pay particular attention to? (Optional)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3591351622"] = "What should the judge pay particular attention to? (Optional)"
-- Your votes
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T360017507"] = "Your votes"
-- The judge already sees the request and both answers, and judges by that alone unless you add something more specific here, for example: a good summary keeps every number and names its source. It never learns which model wrote which answer, and its opinion is shown to you only after your own vote.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3613128085"] = "The judge already sees the request and both answers, and judges by that alone unless you add something more specific here, for example: a good summary keeps every number and names its source. It never learns which model wrote which answer, and its opinion is shown to you only after your own vote."
-- Delete this preset?
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3766894914"] = "Delete this preset?"
-- Run {0}: voted on.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3789682551"] = "Run {0}: voted on."
-- Done.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T38691245"] = "Done."
-- Expand
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3889190145"] = "Expand"
-- Presets
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3897629751"] = "Presets"
-- Model Comparison
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3910956815"] = "Model Comparison"
-- Load a saved preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3970228636"] = "Load a saved preset"
-- Preset name
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T40334785"] = "Preset name"
-- First model
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4044223184"] = "First model"
-- No, keep it
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4188329028"] = "No, keep it"
-- Run {0}: the models are done, your vote is missing.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4197671504"] = "Run {0}: the models are done, your vote is missing."
-- The judge did not give a usable verdict.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4278818434"] = "The judge did not give a usable verdict."
-- Answers
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T46578788"] = "Answers"
-- This provider does not meet the confidence requirements of this assistant. Please choose another one.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T512106357"] = "This provider does not meet the confidence requirements of this assistant. Please choose another one."
-- Collapse
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T5885182"] = "Collapse"
-- LLM judge: {0}
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T609232321"] = "LLM judge: {0}"
-- Number of runs
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T752176293"] = "Number of runs"
-- Waiting for the next run...
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T758625750"] = "Waiting for the next run..."
-- Second model
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T867467332"] = "Second model"
-- Answer B is better
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T933724747"] = "Answer B is better"
-- Judge model
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T953975257"] = "Judge model"
-- You can enter text, attach one or more documents, or use both. At least one input is required.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MYTASKS::ASSISTANTMYTASKS::T1442535450"] = "You can enter text, attach one or more documents, or use both. At least one input is required."
@ -9127,12 +9361,18 @@ UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3756213118"] = "Generate an ERI s
-- Use an LLM to find an icon for a given context.
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3881504200"] = "Use an LLM to find an icon for a given context."
-- Model Comparison
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3910956815"] = "Model Comparison"
-- Job Posting
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3930052338"] = "Job Posting"
-- Ask a question about a legal document.
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3970214537"] = "Ask a question about a legal document."
-- Compare two models on the same request and vote blindly for the better answer.
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T401960846"] = "Compare two models on the same request and vote blindly for the better answer."
-- Log Viewer
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T4130241777"] = "Log Viewer"
@ -10804,6 +11044,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::COMPONENTSEXTENSIONS::T555062689"] = "Log View
-- New Chat
UI_TEXT_CONTENT["AISTUDIO::TOOLS::COMPONENTSEXTENSIONS::T826248509"] = "New Chat"
-- Model Comparison Assistant
UI_TEXT_CONTENT["AISTUDIO::TOOLS::COMPONENTSEXTENSIONS::T941908687"] = "Model Comparison Assistant"
-- Trust LLM providers from the USA
UI_TEXT_CONTENT["AISTUDIO::TOOLS::CONFIDENCESCHEMESEXTENSIONS::T1748300640"] = "Trust LLM providers from the USA"

View File

@ -0,0 +1,477 @@
@attribute [Route(Routes.ASSISTANT_MODEL_COMPARISON)]
@inherits AssistantBaseCore<AIStudio.Dialogs.Settings.NoSettingsPanel>
@using AIStudio.Chat
@* Two views: the form to fill in, and the batch in short while it runs and once it has. *@
@if (this.batchEntries.Count == 0 && !this.isComparing)
{
<MudPaper Outlined="true" Class="pa-4 mb-4">
@* Only the icon button toggles, the same as everywhere else in the app a header collapses a section: the rest of the row is decorative, not a second, redundant click target. Save as new preset stays outside the collapse: capturing the form right now should not need opening this box first. *@
<MudStack Row="true" AlignItems="AlignItems.Center" Spacing="1" Class="mb-2" Wrap="Wrap.Wrap">
<MudIcon Icon="@Icons.Material.Filled.Bookmarks" Color="Color.Primary"/>
<MudText Typo="Typo.subtitle1" Class="flex-grow-1">
@T("Presets")
</MudText>
<MudButton Variant="Variant.Outlined" Size="Size.Small" StartIcon="@Icons.Material.Filled.AddCircleOutline" OnClick="@this.SaveAsNewPreset">
@T("Save as new preset")
</MudButton>
<MudIconButton Icon="@(this.presetsSectionExpanded ? Icons.Material.Filled.ExpandLess : Icons.Material.Filled.ExpandMore)" Size="Size.Small" OnClick="@(() => this.presetsSectionExpanded = !this.presetsSectionExpanded)" title="@(this.presetsSectionExpanded ? T("Collapse") : T("Expand"))"/>
</MudStack>
<MudCollapse Expanded="@this.presetsSectionExpanded">
<MudGrid Spacing="2">
<MudItem xs="12" sm="8">
<MudSelect T="string" Value="@this.selectedPresetId" ValueChanged="@this.LoadPreset" Label="@T("Load a saved preset")" Variant="Variant.Outlined" Clearable="true" Margin="Margin.Dense">
@foreach (var preset in this.SettingsManager.ConfigurationData.ModelComparison.Presets)
{
<MudSelectItem Value="@preset.Id">@preset.Name</MudSelectItem>
}
</MudSelect>
</MudItem>
<MudItem xs="12" sm="4">
<MudStack Row="true" Spacing="1" Justify="Justify.FlexEnd" Wrap="Wrap.Wrap">
@if (this.SelectedPreset is not null)
{
<MudButton Variant="Variant.Filled" Color="Color.Primary" StartIcon="@Icons.Material.Filled.Save" OnClick="@this.UpdateSelectedPreset">
@T("Update preset")
</MudButton>
<MudIconButton Icon="@Icons.Material.Filled.Delete" OnClick="@this.DeletePreset" title="@T("Delete this preset")"/>
}
</MudStack>
</MudItem>
</MudGrid>
@if (this.SelectedPreset is not null)
{
<MudTextField T="string" @bind-Text="@this.presetName" OnBlur="@this.PresetNameWasChanged" Label="@T("Preset name")" Variant="Variant.Outlined" Margin="Margin.Dense" Class="mt-2"/>
}
</MudCollapse>
</MudPaper>
<MudPaper Outlined="true" Class="pa-4 mb-4">
<MudStack Row="true" AlignItems="AlignItems.Center" Spacing="1" Class="mb-3">
<MudIcon Icon="@Icons.Material.Filled.CompareArrows" Color="Color.Primary"/>
<MudText Typo="Typo.subtitle1">
@T("Setup")
</MudText>
</MudStack>
<MudGrid Spacing="3" Class="mb-3">
<MudItem xs="12" sm="6">
<MudText Typo="Typo.subtitle2" Class="mb-1">
@T("First model")
</MudText>
<ProviderSelection @bind-ProviderSettings="@this.ProviderSettings" ValidateProvider="@this.ValidatingProvider"/>
</MudItem>
<MudItem xs="12" sm="6">
<MudText Typo="Typo.subtitle2" Class="mb-1">
@T("Second model")
</MudText>
<ProviderSelection @bind-ProviderSettings="@this.secondProvider" ValidateProvider="@this.ValidatingSecondProvider"/>
</MudItem>
</MudGrid>
<MudSelect T="int" @bind-Value="@this.runCount" Label="@T("Number of runs")" HelperText="@T("Repeat the same comparison several times to see how consistently the models -- and the judge -- come out the same way. Every run gets its own blind vote.")" Variant="Variant.Outlined">
<MudSelectItem Value="1">@T("1 (single comparison)")</MudSelectItem>
<MudSelectItem Value="3">3</MudSelectItem>
<MudSelectItem Value="5">5</MudSelectItem>
<MudSelectItem Value="10">10</MudSelectItem>
<MudSelectItem Value="15">15</MudSelectItem>
<MudSelectItem Value="20">@T("20 (heavy load on the models)")</MudSelectItem>
</MudSelect>
</MudPaper>
<MudPaper Outlined="true" Class="pa-4 mb-4">
<MudStack Row="true" AlignItems="AlignItems.Center" Spacing="1" Class="mb-3">
<MudIcon Icon="@Icons.Material.Filled.QuestionAnswer" Color="Color.Primary"/>
<MudText Typo="Typo.subtitle1">
@T("Your request")
</MudText>
</MudStack>
<ReadFileContent Text="@T("Load the context from file")" @bind-FileContent="@this.inputContext" EnableDragDrop="true" CatchAllDocuments="true"/>
<MudTextField T="string" @bind-Text="@this.inputContext" AdornmentIcon="@Icons.Material.Filled.DocumentScanner" Adornment="Adornment.Start" Label="@T("Context (optional)")" HelperText="@T("Both models receive this text together with your question, for example the document they are supposed to summarize.")" Variant="Variant.Outlined" Lines="6" AutoGrow="@true" MaxLines="24" Class="mb-3 mt-3" UserAttributes="@USER_INPUT_ATTRIBUTES"/>
<MudTextField T="string" @bind-Text="@this.inputQuestion" Validation="@this.ValidatingQuestion" AdornmentIcon="@Icons.Material.Filled.QuestionAnswer" Adornment="Adornment.Start" Label="@T("Your question or task")" Variant="Variant.Outlined" Lines="3" AutoGrow="@true" MaxLines="12" UserAttributes="@USER_INPUT_ATTRIBUTES"/>
</MudPaper>
<MudPaper Outlined="true" Class="pa-4 mb-4">
<MudSwitch T="bool" @bind-Value="@this.judgeEnabled" Color="Color.Primary">
<MudStack Row="true" AlignItems="AlignItems.Center" Spacing="1">
<MudIcon Icon="@Icons.Material.Filled.Gavel" Color="Color.Primary"/>
<MudText Typo="Typo.subtitle1">
@T("LLM judge (optional)")
</MudText>
</MudStack>
</MudSwitch>
@if (this.judgeEnabled)
{
<MudText Typo="Typo.subtitle2" Class="mt-3 mb-1">
@T("Judge model")
</MudText>
<ProviderSelection @bind-ProviderSettings="@this.judgeProvider" ValidateProvider="@this.ValidatingJudgeProvider"/>
<MudTextField T="string" @bind-Text="@this.judgeInstructions" AdornmentIcon="@Icons.Material.Filled.Gavel" Adornment="Adornment.Start" Label="@T("What should the judge pay particular attention to? (Optional)")" HelperText="@T("The judge already sees the request and both answers, and judges by that alone unless you add something more specific here, for example: a good summary keeps every number and names its source. It never learns which model wrote which answer, and its opinion is shown to you only after your own vote.")" Variant="Variant.Outlined" Lines="3" AutoGrow="@true" MaxLines="12" Class="mt-3"/>
}
</MudPaper>
}
else
{
@* What was asked, in short: without it, the answers below would have to be judged from memory. *@
var request = this.CurrentRequest;
<MudPaper Outlined="true" Class="pa-3 mb-3">
<MudStack Row="true" AlignItems="AlignItems.Center" Spacing="1" Class="mb-2">
<MudIcon Icon="@Icons.Material.Filled.QuestionAnswer" Color="Color.Primary"/>
<MudText Typo="Typo.subtitle1" Class="flex-grow-1">
@T("Your request")
</MudText>
<MudIconButton Icon="@(this.requestSummaryExpanded ? Icons.Material.Filled.ExpandLess : Icons.Material.Filled.ExpandMore)" Size="Size.Small" OnClick="@(() => this.requestSummaryExpanded = !this.requestSummaryExpanded)" title="@(this.requestSummaryExpanded ? T("Collapse") : T("Expand"))"/>
</MudStack>
<MudCollapse Expanded="@this.requestSummaryExpanded">
<MudText Typo="Typo.body2" Style="white-space: pre-wrap;" Class="mb-1 mt-1">
@request.GetQuestionPreview(QUESTION_PREVIEW_LENGTH)
</MudText>
@if (request.Context.Length > 0)
{
<MudText Typo="Typo.caption" Class="d-block mb-1">
@string.Format(DisplayCulture, T("Context: {0} characters"), request.Context.Length.CompactCount())
</MudText>
}
</MudCollapse>
@if (this.runCount > 1)
{
<MudText Typo="Typo.subtitle2" Class="mt-2 mb-1">
@T("Progress Runs")
</MudText>
<MudStack Row="true" Spacing="2" Wrap="Wrap.Wrap" Class="mb-1">
@for (var i = 0; i < this.runCount; i++)
{
var index = i;
@* A native title attribute, not MudTooltip: while the batch is still running, this row re-renders often enough that MudTooltip's JS-side hover tracking kept losing track of which badge is which, and stopped showing anything at all. A native tooltip lives entirely in the browser, so re-rendering the content underneath it cannot break it. *@
@* The current entry gets a ring around its avatar: with several runs done and unvoted, position alone -- second from the left, say -- is too easy to lose track of. *@
<MudStack @key="index" Row="false" Spacing="0" AlignItems="AlignItems.Center" title="@this.DescribeRunStatus(index)">
<MudBadge Icon="@Icons.Material.Filled.Gavel" Color="Color.Tertiary" Overlap="true" Visible="@(this.judgeEnabled && this.RunHasJudgeVerdict(index))">
<MudAvatar Size="Size.Medium" Color="@(this.RunHasArrived(index) ? Color.Primary : Color.Default)" Variant="@(this.RunIsVoted(index) ? Variant.Filled : Variant.Outlined)"
Style="@(index == this.currentEntryIndex ? "outline: 2px solid var(--mud-palette-primary); outline-offset: 2px;" : null)">
@if (this.RunIsVoted(index))
{
<MudIcon Icon="@Icons.Material.Filled.Check"/>
}
else if (this.RunIsActive(index))
{
@* Only the runs actually in flight get the spinner -- the ones still queued behind the concurrency limit get an hourglass instead, further down: *@
<MudProgressCircular Indeterminate="true" Size="Size.Small" Color="Color.Primary"/>
}
else if (!this.RunHasArrived(index))
{
<MudIcon Icon="@Icons.Material.Filled.HourglassEmpty" Style="opacity: 0.5;"/>
}
else if (this.RunNotCompleted(index) || this.RunVoteWasSkipped(index))
{
<MudIcon Icon="@Icons.Material.Filled.Block"/>
}
else
{
<MudIcon Icon="@Icons.Material.Filled.QuestionMark"/>
}
</MudAvatar>
</MudBadge>
<MudText Typo="Typo.caption" Class="mt-1">
@(index + 1)
</MudText>
</MudStack>
}
</MudStack>
}
</MudPaper>
}
@code {
/// <summary>
/// The card of one answer, with its title above.
/// </summary>
/// <remarks>
/// The same markup for both, so that nothing but the position tells them apart. The two cards
/// are in one row of the grid and fill it, so they are equally high however long the answers
/// are. That is why the names of the models are not part of the card: how long a name is
/// would make the cards differ.
/// </remarks>
/// <param name="title">"Answer A" or "Answer B".</param>
/// <param name="content">The answer.</param>
/// <param name="column">The column this is, as the vote names it.</param>
private RenderFragment AnswerCard(string title, ContentText content, ModelComparisonVote column) => @<MudItem xs="6" Class="d-flex flex-column">
<MudText Typo="Typo.subtitle1">
@title
</MudText>
<ContentBlockComponent Role="ChatRole.AI" Type="ContentType.TEXT" Time="@this.CurrentEntry!.Result.Time" Content="@content" Class="@this.GetAnswerCardClass(column)"/>
</MudItem>;
/// <summary>
/// What is said about the model of an answer once the vote is in: its name, and how long it took.
/// </summary>
/// <remarks>
/// Below the answer, not above it: an answer can be long, and what stands above it is out of
/// sight by the time it has been read.
/// </remarks>
/// <param name="answer">The answer in this column, which knows who wrote it.</param>
/// <param name="column">The column this is, as the vote names it.</param>
private RenderFragment AnswerCaption(ModelComparisonAnswer answer, ModelComparisonVote column) => @<MudItem xs="6">
<MudStack Row="true" AlignItems="AlignItems.Center" Spacing="1">
<MudIcon Icon="@this.GetVerdictIcon(column)" Color="@this.GetVerdictColor(column)" Size="Size.Small"/>
<MudText Typo="Typo.body1" Color="@this.GetVerdictColor(column)" Class="font-weight-bold">
@answer.Label
</MudText>
@if (this.JudgePreferredThisColumn(column))
{
<MudChip T="string" Icon="@Icons.Material.Filled.Gavel" Size="Size.Medium" Variant="Variant.Filled" Color="Color.Tertiary" Class="font-weight-bold">
@T("Judge")
</MudChip>
}
</MudStack>
<MudText Typo="Typo.body2">
@this.DescribeTimes(answer)
</MudText>
</MudItem>;
/// <summary>
/// What the judge said about the two answers, shown once the vote is in -- the same point at
/// which the names of the two compared models are revealed. Never shown before, so it cannot
/// anchor the user's own vote.
/// </summary>
/// <param name="judge">What the judge said, or that it could not be understood.</param>
private RenderFragment JudgeVerdictCard(ModelComparisonJudgeVerdict judge) => @<MudPaper Outlined="true" Class="pa-3 mb-3">
<MudStack Row="true" AlignItems="AlignItems.Center" Spacing="1" Class="mb-1">
<MudIcon Icon="@Icons.Material.Filled.Gavel" Size="Size.Small"/>
<MudText Typo="Typo.subtitle1">
@string.Format(DisplayCulture, T("LLM judge: {0}"), judge.Label)
</MudText>
</MudStack>
@if (judge.Completed)
{
<MudText Typo="Typo.body1" Class="mb-1">
@this.DescribeJudgePreference(judge)
</MudText>
<MudText Typo="Typo.body2" Style="white-space: pre-wrap;">
@judge.Reasoning
</MudText>
}
else
{
<MudText Typo="Typo.body2">
@T("The judge did not give a usable verdict.")
</MudText>
}
</MudPaper>;
/// <summary>
/// Everything which follows the Compare button: the wait, the answers, the vote, and the way to
/// the next run of the batch.
/// </summary>
/// <remarks>
/// Rendered by the base class below its submit row. This is where the button has to stay above
/// the vote: content of the body would come before it.
/// </remarks>
private protected override RenderFragment? BelowSubmitContent => @<div>
@if (this.skippingRemainingVotes && !this.BatchFinished)
{
<MudPaper Outlined="true" Class="pa-4 mb-4">
<MudProgressLinear Color="Color.Primary" Indeterminate="true" Class="mb-3"/>
<MudText Typo="Typo.body1">
@(this.judgeEnabled
? T("Skipping the remaining votes. Waiting for the judges to finish...")
: T("Skipping the remaining votes. Waiting for the rest of the batch to finish..."))
</MudText>
</MudPaper>
}
else if (this.isComparing && this.CurrentEntry is null)
{
<MudPaper Outlined="true" Class="pa-4 mb-4">
@* The same waiting as in the chat: a bar, and lines where the answers will appear. Both columns look alike, so this says nothing about the models. *@
<MudProgressLinear Color="Color.Primary" Indeterminate="true" Class="mb-3"/>
<MudText Typo="Typo.body1" Class="mb-3">
@(this.batchEntries.Count == 0 ? T("Both models are answering. This can take a while.") : T("Waiting for the next run..."))
</MudText>
<MudGrid Spacing="3">
<MudItem xs="6">
<MudText Typo="Typo.subtitle1">
@T("Answer A")
</MudText>
<MudSkeleton Width="30%" Height="42px"/>
<MudSkeleton Width="80%"/>
<MudSkeleton Width="100%"/>
</MudItem>
<MudItem xs="6">
<MudText Typo="Typo.subtitle1">
@T("Answer B")
</MudText>
<MudSkeleton Width="30%" Height="42px"/>
<MudSkeleton Width="80%"/>
<MudSkeleton Width="100%"/>
</MudItem>
</MudGrid>
</MudPaper>
}
else if (this.CurrentEntry is { } entry)
{
@if (this.answerAContent is not null && this.answerBContent is not null)
{
var columnA = entry.PresentationOrder.InColumnA(entry.Result);
var columnB = entry.PresentationOrder.InColumnB(entry.Result);
<MudPaper Outlined="true" Class="pa-4 mb-4">
@* Until the vote is in, nothing on this screen may say which model wrote which answer: no name, no time. A small model is usually faster, so a time would give the model away. *@
<MudText Typo="Typo.h5" Class="mb-3">
@T("Answers")
</MudText>
@if (this.CurrentVote is null)
{
<MudText Typo="Typo.body2" Class="mb-3">
@T("You do not know yet which model wrote which answer, and the columns are in random order. Compare them by what they say.")
</MudText>
}
@* Two columns in every width, and the two cards in the same row: that is what keeps them equally high. *@
<MudGrid Class="mb-3" Spacing="3">
@this.AnswerCard(T("Answer A"), this.answerAContent!, ModelComparisonVote.COLUMN_A)
@this.AnswerCard(T("Answer B"), this.answerBContent!, ModelComparisonVote.COLUMN_B)
@if (this.CurrentVote is not null)
{
@this.AnswerCaption(columnA, ModelComparisonVote.COLUMN_A)
@this.AnswerCaption(columnB, ModelComparisonVote.COLUMN_B)
}
</MudGrid>
@if (this.CurrentVote is null)
{
<MudText Typo="Typo.h6" Class="mb-2">
@T("Which answer is better?")
</MudText>
<MudStack Row="true" Wrap="Wrap.Wrap" Spacing="2">
<MudButton Variant="Variant.Filled" OnClick="@(() => this.Vote(ModelComparisonVote.COLUMN_A))">
@T("Answer A is better")
</MudButton>
<MudButton Variant="Variant.Filled" OnClick="@(() => this.Vote(ModelComparisonVote.COLUMN_B))">
@T("Answer B is better")
</MudButton>
<MudButton Variant="Variant.Outlined" OnClick="@(() => this.Vote(ModelComparisonVote.TIE))">
@T("Both are equally good")
</MudButton>
@if (this.runCount > 1)
{
<MudButton Variant="Variant.Text" OnClick="@this.SkipThisRun">
@T("Skip this run")
</MudButton>
<MudButton Variant="Variant.Text" OnClick="@this.SkipRemainingVotes">
@T("Skip remaining votes")
</MudButton>
}
</MudStack>
}
else
{
<MudStack Row="true" Wrap="Wrap.Wrap" Spacing="2" AlignItems="AlignItems.Center">
<MudButton Variant="Variant.Filled" Color="Color.Primary" StartIcon="@Icons.Material.Filled.ArrowForward" OnClick="@this.GoToNextEntry">
@(this.IsLastEntry ? T("Finish") : T("Next run"))
</MudButton>
@if (this.isComparing)
{
<MudText Typo="Typo.caption">
@T("More runs are still being generated in the background.")
</MudText>
}
</MudStack>
}
</MudPaper>
@if (this.CurrentVote is not null && entry.Result.Judge is { } judge)
{
@this.JudgeVerdictCard(judge)
}
}
else
{
<MudPaper Outlined="true" Class="pa-4 mb-4">
<MudText Typo="Typo.body1" Class="mb-3">
@T("At least one of the models did not answer in this run, so there is nothing to compare.")
</MudText>
<MudButton Variant="Variant.Filled" Color="Color.Primary" StartIcon="@Icons.Material.Filled.ArrowForward" OnClick="@this.GoToNextEntry">
@(this.IsLastEntry ? T("Finish") : T("Next run"))
</MudButton>
</MudPaper>
}
}
else if (this.BatchFinished)
{
<MudPaper Outlined="true" Class="pa-4 mb-4">
<MudText Typo="Typo.h5" Class="mb-3">
@(this.runCount > 1 ? string.Format(DisplayCulture, T("All {0} runs are done."), this.runCount) : T("Done."))
</MudText>
@if (this.batchEntries.Count > 0)
{
var firstLabel = this.batchEntries[0].Result.First.Label;
var secondLabel = this.batchEntries[0].Result.Second.Label;
var voteSummary = ModelComparisonBatchVoteSummary.From(this.batchEntries, this.batchVotes);
var judgeSummary = this.judgeEnabled ? ModelComparisonBatchJudgeSummary.From(this.batchEntries) : null;
<MudSimpleTable Dense="true" Hover="true" Class="mb-1">
<thead>
<tr>
<th></th>
<th>@firstLabel</th>
<th>@secondLabel</th>
<th>@T("Equally good")</th>
</tr>
</thead>
<tbody>
<tr>
<td>@T("Your votes")</td>
<td>@voteSummary.FirstModelWins</td>
<td>@voteSummary.SecondModelWins</td>
<td>@voteSummary.Ties</td>
</tr>
@if (judgeSummary is not null)
{
<tr>
<td>@T("Judge verdicts")</td>
<td>@judgeSummary.FirstModelWins</td>
<td>@judgeSummary.SecondModelWins</td>
<td>@judgeSummary.Ties</td>
</tr>
}
</tbody>
</MudSimpleTable>
var votesMissing = voteSummary.TotalRunCount - voteSummary.ScoredRunCount;
var judgeVerdictsMissing = judgeSummary is { } summary ? summary.TotalRunCount - summary.ScoredRunCount : 0;
@if (votesMissing > 0)
{
<MudText Typo="Typo.caption" Class="d-block">
@string.Format(DisplayCulture, T("{0} of {1} runs have no usable vote -- never cast, or the run itself never got a usable answer -- and are not counted above."), votesMissing, voteSummary.TotalRunCount)
</MudText>
}
@if (judgeSummary is not null && judgeVerdictsMissing > 0 && judgeVerdictsMissing != votesMissing)
{
<MudText Typo="Typo.caption" Class="d-block">
@string.Format(DisplayCulture, T("{0} of {1} runs have no usable judge verdict."), judgeVerdictsMissing, judgeSummary.TotalRunCount)
</MudText>
}
}
</MudPaper>
<MudButton Variant="Variant.Filled" Color="Color.Primary" StartIcon="@Icons.Material.Filled.Refresh" OnClick="@this.StartNewBatch">
@T("Start a new request")
</MudButton>
}
</div>;
}

View File

@ -0,0 +1,912 @@
using AIStudio.Chat;
using AIStudio.Dialogs.Settings;
using AIStudio.Settings.DataModel;
using AIStudio.Tools.AssistantSessions;
using Microsoft.AspNetCore.Components;
namespace AIStudio.Assistants.ModelComparison;
public partial class AssistantModelComparison : AssistantBaseCore<NoSettingsPanel>
{
[Inject]
private IDialogService DialogService { get; init; } = null!;
protected override Tools.Components Component => Tools.Components.MODEL_COMPARISON_ASSISTANT;
protected override string Title => T("Model Comparison");
protected override string Description => T("Send the same request to two models and compare their answers side by side. You vote blindly for the better answer, and only after your vote you learn which model wrote which answer.");
protected override string SystemPrompt => string.Empty;
protected override bool AllowProfiles => false;
protected override bool ShowResult => false;
/// <remarks>
/// The footer of the base class offers to send the result to another assistant, and to copy it.
/// Both work on the one result the base class knows about, and this assistant has none: its
/// answers are shown by the page itself. Each answer has a copy button of its own.
/// </remarks>
protected override bool ShowSendTo => false;
protected override bool ShowCopyResult => false;
protected override string SubmitText => T("Compare");
/// <summary>
/// Once there are results, the request is not editable, so there is nothing to submit. The way
/// to the next comparison is the button below the vote.
/// </summary>
/// <remarks>
/// The button of the base class cannot be hidden, and repurposing it would start a session of
/// the base class just to switch the view, with the waiting animation flashing by.
/// </remarks>
protected override bool SubmitDisabled => this.isComparing || this.batchEntries.Count > 0;
protected override Func<Task> SubmitAction => this.Compare;
/// <remarks>
/// Starts over with the form, both models included: the base class has already cleared the first
/// one by the time this runs.
/// </remarks>
protected override void ResetForm()
{
this.inputContext = string.Empty;
this.inputQuestion = string.Empty;
this.secondProvider = AIStudio.Settings.Provider.NONE;
this.judgeEnabled = true;
this.judgeProvider = AIStudio.Settings.Provider.NONE;
this.judgeInstructions = string.Empty;
this.runCount = 1;
this.hasAttemptedCompare = false;
this.isComparing = false;
this.ClearBatch();
}
/// <summary>
/// How many independent runs to make of the same comparison. One is the plain comparison this
/// assistant always offered; more repeat the same request several times, to see how consistently
/// the two models -- and the judge -- come out the same way.
/// </summary>
private int runCount = 1;
/// <summary>
/// The runs of the current batch, in the order they finished -- not the order they were started
/// in, since every run asks the same two models the same question, and nothing distinguishes
/// "run 3" from "run 7" beyond how long each happened to take. Filled in one at a time as each
/// run completes, so the user can start voting on the first one without waiting for the slowest.
/// </summary>
private readonly List<ModelComparisonBatchEntry> batchEntries = [];
/// <summary>
/// What the user voted for each entry of <see cref="batchEntries"/>, at the same index. Null at
/// an index means that entry has not been voted on yet.
/// </summary>
private readonly List<ModelComparisonVote?> batchVotes = [];
/// <summary>
/// Which entry of the batch the user is currently looking at. Advances once its vote is in, or
/// the user chose to skip it, via <see cref="GoToNextEntry"/>/<see cref="SkipThisRun"/>, so it
/// never skips ahead of an entry which is still blind on its own -- except when
/// <see cref="skippingRemainingVotes"/> jumps it straight past every run left.
/// </summary>
private int currentEntryIndex;
/// <summary>
/// Whether the batch is being asked right now: at least one run is still in flight. Set once the
/// form has been validated, and not when the session starts: the base class marks the session as
/// processing before the request has been checked, and the form must not be taken away before
/// that.
/// </summary>
private bool isComparing;
/// <summary>
/// Whether the user chose to stop voting on the runs of this batch. Every run still resolves
/// itself as it arrives -- the judge, if one is configured, still judges it -- but none of them
/// are shown individually anymore: the summary appears once the whole batch has settled.
/// </summary>
private bool skippingRemainingVotes;
/// <summary>
/// The entry the user is currently looking at, or null once every run has been voted on, or
/// before the first one has arrived.
/// </summary>
private ModelComparisonBatchEntry? CurrentEntry => this.currentEntryIndex < this.batchEntries.Count ? this.batchEntries[this.currentEntryIndex] : null;
/// <summary>
/// What the user voted for <see cref="CurrentEntry"/>, or null while that vote is still to come.
/// </summary>
private ModelComparisonVote? CurrentVote => this.currentEntryIndex < this.batchVotes.Count ? this.batchVotes[this.currentEntryIndex] : null;
/// <summary>
/// Whether every run of the batch has both arrived and been voted on.
/// </summary>
private bool BatchFinished => this.currentEntryIndex >= this.runCount;
/// <summary>
/// Whether <see cref="CurrentEntry"/> is the last run of the batch. Only used to label its "move
/// on" button "Finish" instead of "Next run" -- both call <see cref="GoToNextEntry"/> all the
/// same, which lands on the batch summary once nothing is left to move on to.
/// </summary>
private bool IsLastEntry => this.currentEntryIndex + 1 >= this.runCount;
/// <summary>
/// Whether the run at this index has both models' answers in yet -- and, when a judge is
/// configured, its verdict too, since an entry only exists once everything about it is settled.
/// </summary>
/// <param name="index">A run's position in the batch, zero-based.</param>
private bool RunHasArrived(int index) => index < this.batchEntries.Count;
/// <summary>
/// Whether the run at this index is one of the ones actually in flight right now, as opposed to
/// arrived already or still queued behind <see cref="ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS"/>
/// others.
/// </summary>
/// <remarks>
/// Nothing tracks which run is which once they are all started -- they are interchangeable, and
/// only how many have arrived is known. So this counts positions, not identities: the runs still
/// outstanding are shown as if the ones nearest to arriving were the ones actually running, which
/// is true in substance even when it is not true of that exact run.
/// </remarks>
/// <param name="index">A run's position in the batch, zero-based.</param>
private bool RunIsActive(int index) => !this.RunHasArrived(index) && index < this.batchEntries.Count + ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS;
/// <summary>
/// Whether the run at this index carries a completed judge verdict.
/// </summary>
/// <param name="index">A run's position in the batch, zero-based.</param>
private bool RunHasJudgeVerdict(int index) => index < this.batchEntries.Count && this.batchEntries[index].Result.Judge is { Completed: true };
/// <summary>
/// Whether the user has voted on the run at this index.
/// </summary>
/// <param name="index">A run's position in the batch, zero-based.</param>
private bool RunIsVoted(int index) => index < this.batchVotes.Count && this.batchVotes[index] is not null;
/// <summary>
/// Whether the run at this index arrived without a usable answer from at least one of the two
/// models -- because the batch was stopped before it got there, or because a model failed on its
/// own. Both are shown the same way: there is nothing left to vote on either way.
/// </summary>
/// <param name="index">A run's position in the batch, zero-based.</param>
private bool RunNotCompleted(int index) => this.RunHasArrived(index) && !this.batchEntries[index].Result.BothCompleted;
/// <summary>
/// Whether the run at this index had a usable answer, but the user moved past it -- via
/// <see cref="SkipThisRun"/> or <see cref="SkipRemainingVotes"/> -- without voting on it.
/// </summary>
/// <param name="index">A run's position in the batch, zero-based.</param>
private bool RunVoteWasSkipped(int index) => this.RunHasArrived(index) && !this.RunNotCompleted(index) && !this.RunIsVoted(index) && index < this.currentEntryIndex;
/// <summary>
/// What the progress badge for this run means, for the tooltip over it: the badge alone only
/// shows colour and a small icon, not enough on its own to be sure what each states for.
/// </summary>
/// <param name="index">A run's position in the batch, zero-based.</param>
private string DescribeRunStatus(int index)
{
var runNumber = index + 1;
if (!this.RunHasArrived(index))
return this.RunIsActive(index)
? string.Format(DisplayCulture, T("Run {0}: still working."), runNumber)
: string.Format(DisplayCulture, T("Run {0}: queued, waiting for a free slot."), runNumber);
if (this.RunNotCompleted(index))
return string.Format(DisplayCulture, T("Run {0}: no usable answer, nothing to vote on."), runNumber);
if (this.RunIsVoted(index))
return string.Format(DisplayCulture, T("Run {0}: voted on."), runNumber);
if (this.RunVoteWasSkipped(index))
return string.Format(DisplayCulture, T("Run {0}: your vote was skipped."), runNumber);
if (this.judgeEnabled && !this.RunHasJudgeVerdict(index))
return string.Format(DisplayCulture, T("Run {0}: the models are done, waiting for the judge."), runNumber);
return this.judgeEnabled
? string.Format(DisplayCulture, T("Run {0}: the models and the judge are done, your vote is missing."), runNumber)
: string.Format(DisplayCulture, T("Run {0}: the models are done, your vote is missing."), runNumber);
}
/// <summary>
/// How much of the question the summary above the results shows.
/// </summary>
private const int QUESTION_PREVIEW_LENGTH = 300;
/// <summary>
/// The answers in the order of the columns, as the chat component wants them, for
/// <see cref="CurrentEntry"/>. Null unless both models of that entry answered. They are built
/// once per entry and not on every render: the component compares what it is given to decide
/// whether anything has to be drawn again.
/// </summary>
private ContentText? answerAContent;
private ContentText? answerBContent;
/// <summary>
/// A render already queued on the UI dispatcher, if one is still pending. Reused by
/// <see cref="RequestBackgroundUIRefresh"/> so a burst of runs arriving close together queues at
/// most one more render instead of one per run.
/// </summary>
private Task? pendingBackgroundRefresh;
/// <summary>
/// Asks for a UI refresh the same way <see cref="RefreshAssistantUIAsync"/> does, but skips
/// asking again while one is already queued.
/// </summary>
/// <remarks>
/// Blazor Server processes every queued render, and every click, on the same one dispatcher
/// queue, each in its own turn. With up to <see cref="ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS"/>
/// runs arriving close together, a render per run queues up several of them ahead of whatever the
/// user just clicked -- the vote button above all -- which is exactly the multi-second stall this
/// avoids. A run which arrives while a render is already queued does not need one of its own: the
/// queued render still runs after this run was added to <see cref="batchEntries"/>, so it shows it
/// all the same.
/// </remarks>
private void RequestBackgroundUIRefresh()
{
if (this.pendingBackgroundRefresh is { IsCompleted: false })
return;
this.pendingBackgroundRefresh = this.RefreshAssistantUIAsync();
}
/// <summary>
/// Forgets the current batch, which brings the form back.
/// </summary>
private void ClearBatch()
{
this.batchEntries.Clear();
this.batchVotes.Clear();
this.currentEntryIndex = 0;
this.skippingRemainingVotes = false;
this.RebuildShownAnswers();
}
/// <summary>
/// Builds the answers of the columns from <see cref="CurrentEntry"/>.
/// </summary>
/// <remarks>
/// Called wherever the current entry changes: when it newly arrives, when the user moves on to
/// the next one, after a reset, and after the state has been restored. What is shown is derived
/// from the entry, never stored on its own.
/// </remarks>
private void RebuildShownAnswers()
{
if (this.CurrentEntry is not { Result.BothCompleted: true } entry)
{
this.answerAContent = null;
this.answerBContent = null;
return;
}
this.answerAContent = new ContentText { Text = entry.PresentationOrder.InColumnA(entry.Result).Text };
this.answerBContent = new ContentText { Text = entry.PresentationOrder.InColumnB(entry.Result).Text };
}
private string inputContext = string.Empty;
private string inputQuestion = string.Empty;
/// <summary>
/// What both models are asked, as the user entered it right now. Both models, and the judge with
/// them, are asked to answer in the app's active language, regardless of what language the
/// question itself is written in.
/// </summary>
private ModelComparisonRequest CurrentRequest => new(this.inputContext, this.inputQuestion, ActiveLanguageName);
protected override bool MightPreselectValues() => false;
/// <summary>
/// The first model is <see cref="AssistantLowerBase.ProviderSettings"/>, so that everything the
/// base class does with a provider keeps working for it. This is the second one.
/// </summary>
private AIStudio.Settings.Provider secondProvider = AIStudio.Settings.Provider.NONE;
/// <summary>
/// Whether an optional third model judges the two answers. On by default; the user can still
/// switch it off for a plain two-model comparison.
/// </summary>
private bool judgeEnabled = true;
/// <summary>
/// The judge model, asked only when <see cref="judgeEnabled"/> is set.
/// </summary>
private AIStudio.Settings.Provider judgeProvider = AIStudio.Settings.Provider.NONE;
/// <summary>
/// What the user wants the judge to pay attention to, in their own words, for example what makes
/// a good summary. Sent to the judge together with the request and both answers, but the judge
/// never learns which model wrote which answer.
/// </summary>
private string judgeInstructions = string.Empty;
/// <summary>
/// Whether the user has tried to submit the form at least once. The judge's provider field mounts
/// only once the judge is switched on, which can happen well after the form itself first
/// rendered, and MudForm would otherwise mark it invalid on sight, before the user had any chance
/// to pick one. <see cref="ValidatingJudgeProvider"/> stays silent until this is set, regardless
/// of when MudForm itself happens to call it.
/// </summary>
private bool hasAttemptedCompare;
/// <summary>
/// Whether the presets box is expanded. Collapsed by default: presets are a convenience for
/// repeated testing, not something every visit to this assistant needs to see.
/// </summary>
private bool presetsSectionExpanded;
/// <summary>
/// Whether the summary of the request (the question preview and the context's length) is
/// expanded, while a batch is shown. Collapsed by default, for the same reason as
/// <see cref="presetsSectionExpanded"/>: useful to check back on, not something to stare at while
/// a batch runs.
/// </summary>
private bool requestSummaryExpanded;
/// <summary>
/// The Id of the saved preset currently loaded into the form, or null while nothing is loaded.
/// Not itself persisted: it only drives which preset the rename field and the delete button act
/// on, for the lifetime of this session.
/// </summary>
private string? selectedPresetId;
/// <summary>
/// What the rename field shows for <see cref="SelectedPreset"/>. Kept apart from the preset's own
/// <see cref="ModelComparisonPreset.Name"/> so a half-typed name is not written back on every
/// keystroke, only once the field loses focus.
/// </summary>
private string presetName = string.Empty;
/// <summary>
/// The saved preset <see cref="selectedPresetId"/> points to, or null when nothing is loaded or
/// the preset was deleted from another session in the meantime.
/// </summary>
private ModelComparisonPreset? SelectedPreset => this.SettingsManager.ConfigurationData.ModelComparison.Presets.FirstOrDefault(preset => preset.Id == this.selectedPresetId);
/// <summary>
/// Loads a saved preset's fields into the form, or clears back to a blank form when the picker
/// was cleared.
/// </summary>
private void LoadPreset(string? presetId)
{
this.selectedPresetId = presetId;
var preset = this.SelectedPreset;
if (preset is null)
{
this.presetName = string.Empty;
return;
}
this.ProviderSettings = this.SettingsManager.GetProviderById(preset.FirstProviderId);
this.secondProvider = this.SettingsManager.GetProviderById(preset.SecondProviderId);
this.judgeEnabled = preset.JudgeEnabled;
this.judgeProvider = this.SettingsManager.GetProviderById(preset.JudgeProviderId);
this.judgeInstructions = preset.JudgeInstructions;
this.inputContext = preset.Context;
this.inputQuestion = preset.Question;
this.runCount = preset.RunCount;
this.presetName = preset.Name;
}
/// <summary>
/// Captures the form as it stands right now into a new saved preset, under a placeholder name the
/// user is expected to change with the rename field which appears once it is selected.
/// </summary>
private async Task SaveAsNewPreset()
{
var preset = new ModelComparisonPreset
{
Id = Guid.NewGuid().ToString(),
Name = string.Format(DisplayCulture, T("Preset from {0}"), DateTimeOffset.Now),
FirstProviderId = this.ProviderSettings.Id,
SecondProviderId = this.secondProvider.Id,
JudgeEnabled = this.judgeEnabled,
JudgeProviderId = this.judgeProvider.Id,
JudgeInstructions = this.judgeInstructions,
Context = this.inputContext,
Question = this.inputQuestion,
RunCount = this.runCount,
};
this.SettingsManager.ConfigurationData.ModelComparison.Presets.Add(preset);
this.selectedPresetId = preset.Id;
this.presetName = preset.Name;
// Opens the box so the new preset -- and the rename field to give it a real name -- is
// actually visible, instead of the save appearing to have done nothing:
this.presetsSectionExpanded = true;
await this.SettingsManager.StoreSettings();
}
/// <summary>
/// Captures the form as it stands right now back into <see cref="SelectedPreset"/>, keeping its
/// Id and name, instead of creating a new preset next to it.
/// </summary>
private async Task UpdateSelectedPreset()
{
var preset = this.SelectedPreset;
if (preset is null)
return;
var presets = this.SettingsManager.ConfigurationData.ModelComparison.Presets;
var index = presets.IndexOf(preset);
if (index < 0)
return;
presets[index] = preset with
{
FirstProviderId = this.ProviderSettings.Id,
SecondProviderId = this.secondProvider.Id,
JudgeEnabled = this.judgeEnabled,
JudgeProviderId = this.judgeProvider.Id,
JudgeInstructions = this.judgeInstructions,
Context = this.inputContext,
Question = this.inputQuestion,
RunCount = this.runCount,
};
await this.SettingsManager.StoreSettings();
}
/// <summary>
/// Writes a typed-in name back to <see cref="SelectedPreset"/>, once the field loses focus. An
/// empty name is refused rather than saved: the picker would show nothing to click on.
/// </summary>
private async Task PresetNameWasChanged()
{
var preset = this.SelectedPreset;
if (preset is null || string.IsNullOrWhiteSpace(this.presetName))
return;
preset.Name = this.presetName.Trim();
await this.SettingsManager.StoreSettings();
}
/// <summary>
/// Deletes <see cref="SelectedPreset"/>, after asking: a preset can carry a long context text
/// that took real effort to assemble, and one click by mistake would lose it for good.
/// </summary>
private async Task DeletePreset()
{
var preset = this.SelectedPreset;
if (preset is null)
return;
var discard = await this.DialogService.ShowMessageBox(
T("Delete this preset?"),
string.Format(T("The preset '{0}' is deleted right away. This cannot be undone."), preset.Name),
T("Yes, delete it"),
T("No, keep it"));
if (discard is not true)
return;
this.SettingsManager.ConfigurationData.ModelComparison.Presets.Remove(preset);
this.selectedPresetId = null;
this.presetName = string.Empty;
await this.SettingsManager.StoreSettings();
}
private static readonly AssistantSessionStateKey<string> INPUT_CONTEXT_STATE_KEY = new(nameof(inputContext));
private static readonly AssistantSessionStateKey<string> INPUT_QUESTION_STATE_KEY = new(nameof(inputQuestion));
private static readonly AssistantSessionStateKey<bool> IS_COMPARING_STATE_KEY = new(nameof(isComparing));
private static readonly AssistantSessionStateKey<int> RUN_COUNT_STATE_KEY = new(nameof(runCount));
private static readonly AssistantSessionStateKey<List<ModelComparisonBatchEntry>> BATCH_ENTRIES_STATE_KEY = new(nameof(batchEntries));
private static readonly AssistantSessionStateKey<List<ModelComparisonVote?>> BATCH_VOTES_STATE_KEY = new(nameof(batchVotes));
private static readonly AssistantSessionStateKey<int> CURRENT_ENTRY_INDEX_STATE_KEY = new(nameof(currentEntryIndex));
private static readonly AssistantSessionStateKey<AIStudio.Settings.Provider> SECOND_PROVIDER_STATE_KEY = new(nameof(secondProvider));
private static readonly AssistantSessionStateKey<bool> JUDGE_ENABLED_STATE_KEY = new(nameof(judgeEnabled));
private static readonly AssistantSessionStateKey<AIStudio.Settings.Provider> JUDGE_PROVIDER_STATE_KEY = new(nameof(judgeProvider));
private static readonly AssistantSessionStateKey<string> JUDGE_INSTRUCTIONS_STATE_KEY = new(nameof(judgeInstructions));
/// <inheritdoc />
protected override void CaptureCustomAssistantSessionState(AssistantSessionStateWriter state)
{
state.Set(INPUT_CONTEXT_STATE_KEY, this.inputContext);
state.Set(INPUT_QUESTION_STATE_KEY, this.inputQuestion);
state.Set(IS_COMPARING_STATE_KEY, this.isComparing);
state.Set(RUN_COUNT_STATE_KEY, this.runCount);
state.SetList(BATCH_ENTRIES_STATE_KEY, this.batchEntries);
state.SetList(BATCH_VOTES_STATE_KEY, this.batchVotes);
state.Set(CURRENT_ENTRY_INDEX_STATE_KEY, this.currentEntryIndex);
state.Set(SECOND_PROVIDER_STATE_KEY, this.secondProvider);
state.Set(JUDGE_ENABLED_STATE_KEY, this.judgeEnabled);
state.Set(JUDGE_PROVIDER_STATE_KEY, this.judgeProvider);
state.Set(JUDGE_INSTRUCTIONS_STATE_KEY, this.judgeInstructions);
}
/// <inheritdoc />
protected override void RestoreCustomAssistantSessionState(AssistantSessionStateReader state)
{
state.Restore(INPUT_CONTEXT_STATE_KEY, value => this.inputContext = value);
state.Restore(INPUT_QUESTION_STATE_KEY, value => this.inputQuestion = value);
state.Restore(IS_COMPARING_STATE_KEY, value => this.isComparing = value);
state.Restore(RUN_COUNT_STATE_KEY, value => this.runCount = value);
state.RestoreList(BATCH_ENTRIES_STATE_KEY, this.batchEntries);
state.RestoreList(BATCH_VOTES_STATE_KEY, this.batchVotes);
state.Restore(CURRENT_ENTRY_INDEX_STATE_KEY, value => this.currentEntryIndex = value);
state.Restore(SECOND_PROVIDER_STATE_KEY, value => this.secondProvider = value);
state.Restore(JUDGE_ENABLED_STATE_KEY, value => this.judgeEnabled = value);
state.Restore(JUDGE_PROVIDER_STATE_KEY, value => this.judgeProvider = value);
state.Restore(JUDGE_INSTRUCTIONS_STATE_KEY, value => this.judgeInstructions = value);
this.RebuildShownAnswers();
}
/// <summary>
/// Starts every run of the batch and reveals each one as soon as it is ready.
/// </summary>
/// <remarks>
/// The base class has already started the session, so the token of the stop button is the one
/// which ends every run still in flight. Cancelling it needs no special handling here: every run,
/// in flight or still queued behind <see cref="ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS"/>,
/// resolves into an entry to show either way -- <see cref="ModelComparisonBatchRunner"/> never lets
/// a run end any other way -- so the loop below simply keeps consuming them until none are left.
/// </remarks>
private async Task Compare()
{
//
// Set before validating, so the judge provider field -- silent about being empty until now --
// reports a real error on this very attempt, the same as every other required field does:
//
this.hasAttemptedCompare = true;
//
// The validation has to come first, while the fields of the form are still on the screen: it
// checks the fields which are mounted. A form which has been taken away has nothing left to
// find fault with, and an empty question would go through.
//
await this.Form!.Validate();
if (!this.InputIsValid)
return;
this.ClearBatch();
// From here on, the request is shown as a summary instead of the form:
this.isComparing = true;
await this.CheckpointAssistantSession();
await this.RefreshAssistantUIAsync();
var token = this.CancellationTokenSource?.Token ?? CancellationToken.None;
try
{
var batchRunner = new ModelComparisonBatchRunner(new ModelComparisonRunner(this.Logger));
var judge = this.judgeEnabled
? ModelComparisonProviderAdapter.CreateParticipant(this.judgeProvider, this.SettingsManager, this.Component, this.Title)
: null;
var tasks = batchRunner.Start(
ModelComparisonProviderAdapter.CreateParticipant(this.ProviderSettings, this.SettingsManager, this.Component, this.Title),
ModelComparisonProviderAdapter.CreateParticipant(this.secondProvider, this.SettingsManager, this.Component, this.Title),
this.CurrentRequest,
this.runCount,
token,
judge,
this.judgeInstructions);
//
// ConfigureAwait(false) throughout this loop, deliberately: Blazor Server processes every
// click on this page through the same synchronization context this method would otherwise
// keep resuming on for as long as the batch runs. A click has to wait its turn behind
// whatever is already queued on that context, and a background loop which keeps hopping
// back onto it -- once per run, for a whole batch -- can make that turn a long time coming.
// Its one deliberate touch of that context, the render below, is not awaited either, for
// the same reason -- see the remark right above it.
//
await foreach (var completedTask in Task.WhenEach(tasks).ConfigureAwait(false))
{
var entry = await completedTask.ConfigureAwait(false);
this.batchEntries.Add(entry);
this.batchVotes.Add(null);
// Only a rebuild for the entry the user is actually looking at avoids rebuilding the
// shown answers -- and so the chat component re-parsing them -- for runs which arrived
// while the user was still on an earlier one:
if (this.batchEntries.Count - 1 == this.currentEntryIndex)
this.RebuildShownAnswers();
//
// Coalesced, not one render per run: every render -- this one included -- queues on
// the same one dispatcher a click on this page has to wait its turn on too. Several
// runs arriving close together, up to ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS of
// them, would otherwise queue up that many renders ahead of whatever the user just
// clicked, which is exactly the multi-second stall this avoids.
//
this.RequestBackgroundUIRefresh();
}
await this.CheckpointAssistantSession();
}
finally
{
// Whatever ended the batch, a canceled one as well, the form is back unless there are results:
this.isComparing = false;
// Nothing left to arrive and flip this instead, now that the batch has settled itself.
// A stopped batch goes straight to the summary the same way: there is no point asking
// for votes on runs the user just chose to stop waiting for.
if (this.skippingRemainingVotes || token.IsCancellationRequested)
this.currentEntryIndex = this.runCount;
this.RequestBackgroundUIRefresh();
}
}
/// <summary>
/// Goes back from the results to the request, which is as the user left it.
/// </summary>
/// <remarks>
/// Offered without a question once every run has been voted on, or when a run has nothing to
/// vote on because a model did not answer. Before that, stopping the batch (the button next to
/// its progress) reaches the same screen: whatever has arrived by then stays shown.
/// </remarks>
private async Task StartNewBatch()
{
this.ClearBatch();
await this.CheckpointAssistantSession();
}
/// <summary>
/// Takes the vote for <see cref="CurrentEntry"/>, which ends the blind part: from here on, the
/// names of the models are shown.
/// </summary>
/// <remarks>
/// The vote is final. A second click, or a second one on another button, changes nothing: once
/// the names are on the screen, the answers can no longer be judged without knowing them.
///
/// The reveal is shown before the checkpoint is awaited, not after: a batch snapshot carries the
/// answer text of every run so far, and saving it can take a moment the click must not wait on.
/// </remarks>
private async Task Vote(ModelComparisonVote vote)
{
if (this.CurrentVote is not null || this.CurrentEntry is not { Result.BothCompleted: true })
return;
this.batchVotes[this.currentEntryIndex] = vote;
await this.RefreshAssistantUIAsync();
await this.CheckpointAssistantSession();
}
/// <summary>
/// Moves on to the next run of the batch, once the current one has been voted on -- or right
/// away when there was nothing to vote on, because a model of that run did not answer.
/// </summary>
private Task GoToNextEntry()
{
if (this.CurrentVote is null && this.CurrentEntry is { Result.BothCompleted: true })
return Task.CompletedTask;
return this.AdvanceToNextEntry();
}
/// <summary>
/// Moves on to the next run without voting on this one, even though there was something to vote
/// on -- the user's own choice, unlike the silent skip in <see cref="GoToNextEntry"/> for a run
/// with nothing to vote on.
/// </summary>
private Task SkipThisRun() => this.AdvanceToNextEntry();
/// <summary>
/// Advances past the current entry unconditionally. <see cref="GoToNextEntry"/> and
/// <see cref="SkipThisRun"/> both end up here; only whether a vote is required before doing so
/// tells them apart.
/// </summary>
/// <remarks>
/// The same reveal-before-checkpoint order as <see cref="Vote"/>, for the same reason.
/// </remarks>
private async Task AdvanceToNextEntry()
{
this.currentEntryIndex++;
this.RebuildShownAnswers();
await this.RefreshAssistantUIAsync();
await this.CheckpointAssistantSession();
}
/// <summary>
/// Stops asking for a vote on every remaining run of this batch. A run still resolves itself as
/// it arrives, but none of them are shown individually anymore: the summary appears once the
/// whole batch has settled, which may already be the case right now.
/// </summary>
private async Task SkipRemainingVotes()
{
this.skippingRemainingVotes = true;
// Nothing left in flight to flip this later, so it has to happen here instead:
if (!this.isComparing)
this.currentEntryIndex = this.runCount;
this.RebuildShownAnswers();
await this.RefreshAssistantUIAsync();
await this.CheckpointAssistantSession();
}
/// <summary>
/// How a model did in the vote, as a colour: green when its answer was preferred, red when the
/// other one was, orange when the user found them equally good.
/// </summary>
/// <param name="column">The column the model's answer stood in.</param>
private Color GetVerdictColor(ModelComparisonVote column) => this.CurrentVote switch
{
ModelComparisonVote.TIE => Color.Warning,
{ } vote when vote == column => Color.Success,
_ => Color.Error,
};
/// <summary>
/// The classes of the card around an answer: it fills the height of its column, and once the
/// vote is in, it has a border in the colour which goes with <see cref="GetVerdictColor"/>.
/// </summary>
/// <remarks>
/// The classes are the ones of MudBlazor: a border two pixels wide, in the colour of the outcome.
/// They override the thin grey line of the card, which is why they are told to win. Before the
/// vote, the answers stand in the plain frame of the chat.
/// </remarks>
/// <param name="column">The column the model's answer stood in.</param>
private string GetAnswerCardClass(ModelComparisonVote column)
{
const string FILL_THE_COLUMN = "flex-grow-1";
if (this.CurrentVote is null)
return FILL_THE_COLUMN;
var colour = this.GetVerdictColor(column) switch
{
Color.Success => "mud-border-success",
Color.Warning => "mud-border-warning",
_ => "mud-border-error",
};
return $"{FILL_THE_COLUMN} border-2 border-solid {colour}";
}
/// <summary>
/// The icon which goes with <see cref="GetVerdictColor"/>. The colour alone would not tell
/// apart who is colour blind, so every outcome has a shape of its own as well.
/// </summary>
/// <param name="column">The column the model's answer stood in.</param>
private string GetVerdictIcon(ModelComparisonVote column) => this.CurrentVote switch
{
ModelComparisonVote.TIE => Icons.Material.Filled.DragHandle,
{ } vote when vote == column => Icons.Material.Filled.CheckCircle,
_ => Icons.Material.Filled.Cancel,
};
/// <summary>
/// Names what the judge preferred, by the model's revealed name, and says whether that matches
/// the user's own vote.
/// </summary>
/// <remarks>
/// Only called once the vote is in, so the model names of <see cref="CurrentEntry"/> are the ones
/// the user is meant to see by now.
/// </remarks>
private string DescribeJudgePreference(ModelComparisonJudgeVerdict judge)
{
if (judge.Preferred is not { } preferred || this.CurrentEntry is not { } entry)
return string.Empty;
var agreesWithVote = this.CurrentVote == preferred;
if (preferred is ModelComparisonVote.TIE)
return agreesWithVote
? T("The judge saw both answers as equally good, matching your own vote.")
: T("The judge saw both answers as equally good, while you preferred one of them.");
var preferredAnswer = preferred is ModelComparisonVote.COLUMN_A
? entry.PresentationOrder.InColumnA(entry.Result)
: entry.PresentationOrder.InColumnB(entry.Result);
return agreesWithVote
? string.Format(DisplayCulture, T("The judge preferred {0}, matching your own vote."), preferredAnswer.Label)
: string.Format(DisplayCulture, T("The judge preferred {0}, differing from your own vote."), preferredAnswer.Label);
}
/// <summary>
/// Whether the judge's badge belongs on the answer in this column: either it is the one the
/// judge preferred, or the judge saw a tie, in which case both columns carry the badge.
/// </summary>
/// <remarks>
/// Shown next to <see cref="AnswerCaption"/>, which already gates on the vote being in, so this
/// never has to check that itself.
/// </remarks>
private bool JudgePreferredThisColumn(ModelComparisonVote column)
{
if (this.CurrentEntry is not { Result.Judge: { Completed: true, Preferred: { } preferred } })
return false;
return preferred is ModelComparisonVote.TIE || preferred == column;
}
/// <summary>
/// Describes how long an answer took: until the first piece arrived, and until it was complete.
/// Shown together with the name of the model, once the vote is in.
/// </summary>
private string DescribeTimes(ModelComparisonAnswer answer) => string.Format(
DisplayCulture,
T("First token after {0}, complete after {1}"),
FormatDuration(answer.FirstTokenTime),
FormatDuration(answer.TotalTime));
private static string FormatDuration(TimeSpan? duration) => duration is { } value
? $"{value.TotalSeconds.ToString("0.0", DisplayCulture)} s"
: "-";
/// <summary>
/// The culture of the active language. Written out in full because the namespace of the
/// localization assistant, AIStudio.Assistants.I18N, hides the class I18N from every assistant.
/// </summary>
private static System.Globalization.CultureInfo DisplayCulture => AIStudio.Tools.PluginSystem.I18N.I.Culture;
/// <summary>
/// The active language's own English name, without a region ("German", not "German (Germany)"),
/// for a prompt instruction the models and the judge understand regardless of their own default
/// language. <see cref="DisplayCulture"/> carries a region, so its neutral parent is what names
/// the language alone -- except for the invariant culture, which has no parent to ask.
/// </summary>
private static string ActiveLanguageName => DisplayCulture.IsNeutralCulture || DisplayCulture.Equals(System.Globalization.CultureInfo.InvariantCulture)
? DisplayCulture.EnglishName
: DisplayCulture.Parent.EnglishName;
/// <summary>
/// Only the question is required. A comparison without a context is a comparison of how two
/// models answer a plain question, which is a fair thing to want to know.
/// </summary>
private string? ValidatingQuestion(string question)
{
if (string.IsNullOrWhiteSpace(question))
return T("Please provide a question or task for both models.");
return null;
}
/// <summary>
/// Checks the second model. The selection only offers providers which meet the confidence
/// requirements, but it keeps what was chosen earlier: a provider selected before the
/// requirements were raised must not be sent a request all the same.
/// </summary>
/// <remarks>
/// <see cref="ModelComparisonProviderAdapter"/> refuses such a provider as well, but only once
/// the run has started, and then the model shows up as one which did not answer. Here the
/// user is told before that.
/// </remarks>
private string? ValidatingSecondProvider(AIStudio.Settings.Provider provider)
{
var selectionIssue = this.ValidatingProvider(provider);
if (selectionIssue is not null)
return selectionIssue;
if (!this.SettingsManager.IsProviderConfident(provider, this.Component))
return T("This provider does not meet the confidence requirements of this assistant. Please choose another one.");
return null;
}
/// <summary>
/// Checks the judge model. It may be the same provider as either compared model, the same as
/// the two compared models may be the same provider as each other: nothing here rules out a
/// strong model judging one it would also have been a fair pick for, or a model being compared
/// against itself to see how consistent it is with its own answers.
/// </summary>
/// <remarks>
/// Silent until <see cref="hasAttemptedCompare"/>: this field mounts only once the judge is
/// switched on, often well after the form first rendered, and MudForm would otherwise flag it as
/// soon as it exists, before the user had any chance to fill it in.
/// </remarks>
private string? ValidatingJudgeProvider(AIStudio.Settings.Provider provider)
{
if (!this.hasAttemptedCompare)
return null;
var selectionIssue = this.ValidatingProvider(provider);
if (selectionIssue is not null)
return selectionIssue;
if (!this.SettingsManager.IsProviderConfident(provider, this.Component))
return T("This provider does not meet the confidence requirements of this assistant. Please choose another one.");
return null;
}
}

View File

@ -0,0 +1,37 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// What one model answered, together with how the request went.
/// </summary>
public sealed record ModelComparisonAnswer
{
/// <summary>
/// The name to show for the model. It says who wrote the answer, and is not shown until the
/// user has voted.
/// </summary>
public required string Label { get; init; }
/// <summary>
/// The answer without the thinking of the model. Only an answer which is
/// <see cref="Completed"/> is complete: after a failure or a cancellation, this is whatever
/// arrived before, and it must not be shown as a result.
/// </summary>
public required string Text { get; init; }
/// <summary>
/// Whether the model answered. False when the request failed, when the user canceled it, and
/// when the model ended without saying anything.
/// </summary>
public required bool Completed { get; init; }
/// <summary>
/// The time from sending the request until the first piece of the answer arrived, or null when
/// nothing arrived.
/// </summary>
public TimeSpan? FirstTokenTime { get; init; }
/// <summary>
/// The time from sending the request until the answer was complete, or null when it never was.
/// </summary>
public TimeSpan? TotalTime { get; init; }
}

View File

@ -0,0 +1,16 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// One run within a batch: its own random presentation order, and what came of it.
/// </summary>
/// <remarks>
/// The presentation order travels with its result because it takes both to translate a
/// <see cref="ModelComparisonVote"/> -- the user's own vote, or the judge's -- back into a model: the
/// same column means a different model in every run.
/// </remarks>
public sealed record ModelComparisonBatchEntry
{
public required ModelComparisonPresentationOrder PresentationOrder { get; init; }
public required ModelComparisonRunResult Result { get; init; }
}

View File

@ -0,0 +1,65 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// How often the judge came out on each side across a batch of runs.
/// </summary>
/// <remarks>
/// Counted in First/Second terms, never in columns: the column a model stood in was drawn fresh for
/// every run, so only the model identity stays comparable across runs. A run whose judge did not
/// answer, or whose verdict could not be read, is left out of every count -- <see cref="ScoredRunCount"/>
/// says how many were actually counted, which is why it can be smaller than <see cref="TotalRunCount"/>.
/// </remarks>
public sealed record ModelComparisonBatchJudgeSummary
{
public required int TotalRunCount { get; init; }
public required int ScoredRunCount { get; init; }
public required int FirstModelWins { get; init; }
public required int SecondModelWins { get; init; }
public required int Ties { get; init; }
/// <summary>
/// Tallies the judge's verdicts across a batch. Never throws: a run whose judge did not answer,
/// or whose verdict could not be understood, is simply left out of the count.
/// </summary>
/// <param name="entries">The runs of the batch, in any order.</param>
public static ModelComparisonBatchJudgeSummary From(IReadOnlyList<ModelComparisonBatchEntry> entries)
{
var firstModelWins = 0;
var secondModelWins = 0;
var ties = 0;
foreach (var entry in entries)
{
if (entry.Result.Judge is not { Completed: true, Preferred: { } preferred })
continue;
switch (entry.PresentationOrder.ToModelChoice(preferred))
{
case ModelComparisonModelChoice.FIRST:
firstModelWins++;
break;
case ModelComparisonModelChoice.SECOND:
secondModelWins++;
break;
case ModelComparisonModelChoice.TIE:
ties++;
break;
}
}
return new ModelComparisonBatchJudgeSummary
{
TotalRunCount = entries.Count,
ScoredRunCount = firstModelWins + secondModelWins + ties,
FirstModelWins = firstModelWins,
SecondModelWins = secondModelWins,
Ties = ties,
};
}
}

View File

@ -0,0 +1,127 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// Starts the same comparison several times in a row, to see how consistently the two models -- and
/// an optional judge -- come out the same way.
/// </summary>
/// <remarks>
/// Each run is independent: its own pair of answers, its own random presentation order, and its own
/// judge verdict, because that variation between runs is exactly what a batch is meant to surface. A
/// single, shared draw would defeat the point.
///
/// This starts every run at once and hands back one task per run, rather than awaiting them itself:
/// the caller decides whether to wait for all of them together (<c>Task.WhenAll</c>) or to react to
/// each one as it finishes (<c>Task.WhenEach</c>), for example to reveal runs to the user as they
/// become ready instead of only once the slowest of them has answered. "Started" only means the task
/// exists and is queued, though: <see cref="MAX_CONCURRENT_RUNS"/> of them are actually asking a
/// model at any one time, never more, because a whole batch's worth of requests landing on a model at
/// once is exactly what queues up and slows every one of them down on a provider with limited
/// concurrent capacity -- a self-hosted one most of all. A model comparison is not something run many
/// times a day, so trading wall-clock time for going easier on the model is the right default.
///
/// <see cref="Start"/> is called from a Blazor Server page, on its own synchronisation context, so
/// <see cref="RunOneAsync"/> uses <c>ConfigureAwait(false)</c> throughout, the same as
/// <see cref="ModelComparisonRunner"/> does further down: none of this is UI work, and resuming on
/// that context regardless would only queue up behind whatever the page itself is waiting to do,
/// a click included.
/// </remarks>
/// <param name="runner">Runs one comparison. Reused for every run of a batch, since the two compared
/// models and the judge are stateless requests, not a resource a run could use up.</param>
public sealed class ModelComparisonBatchRunner(ModelComparisonRunner runner)
{
/// <summary>
/// How many runs may be actually asking a model at once. Fixed, not a setting: this exists to
/// protect whatever the batch is running against, not to be tuned per comparison. Public so the
/// UI can show which of the runs still outstanding are actually in flight right now, rather than
/// still queued behind this same limit.
/// </summary>
public const int MAX_CONCURRENT_RUNS = 3;
private readonly SemaphoreSlim concurrencyLimit = new(MAX_CONCURRENT_RUNS, MAX_CONCURRENT_RUNS);
/// <param name="first">The first model, asked fresh in every run.</param>
/// <param name="second">The second model, asked fresh in every run.</param>
/// <param name="request">What both models are asked, the same in every run.</param>
/// <param name="runCount">How many independent runs to start. At least one.</param>
/// <param name="token">Cancels every run of the batch, in flight or still queued.</param>
/// <param name="judge">The optional judge, asked in every run once its two answers are in.</param>
/// <param name="judgeInstructions">What the user wants the judge to pay attention to. Ignored without a <paramref name="judge"/>.</param>
/// <returns>One task per run, already started, in no particular order of completion.</returns>
public IReadOnlyList<Task<ModelComparisonBatchEntry>> Start(
ModelComparisonParticipant first,
ModelComparisonParticipant second,
ModelComparisonRequest request,
int runCount,
CancellationToken token,
ModelComparisonParticipant? judge = null,
string judgeInstructions = "")
{
ArgumentOutOfRangeException.ThrowIfLessThan(runCount, 1);
var tasks = new Task<ModelComparisonBatchEntry>[runCount];
for (var i = 0; i < runCount; i++)
tasks[i] = this.RunOneAsync(first, second, request, token, judge, judgeInstructions);
return tasks;
}
/// <summary>
/// Waits for a free slot among <see cref="MAX_CONCURRENT_RUNS"/>, then runs one comparison.
/// </summary>
/// <remarks>
/// Never throws, a queued run included: cancellation while still waiting for a slot is reported
/// the same way <see cref="ModelComparisonRunner"/> reports a model which did not answer, not as
/// a fault on the task -- every run in a batch is meant to end up as an entry to show, whatever
/// happened to it.
/// </remarks>
private async Task<ModelComparisonBatchEntry> RunOneAsync(
ModelComparisonParticipant first,
ModelComparisonParticipant second,
ModelComparisonRequest request,
CancellationToken token,
ModelComparisonParticipant? judge,
string judgeInstructions)
{
var presentationOrder = Random.Shared.Next(2) is 0
? ModelComparisonPresentationOrder.FIRST_MODEL_FIRST
: ModelComparisonPresentationOrder.SECOND_MODEL_FIRST;
try
{
await this.concurrencyLimit.WaitAsync(token).ConfigureAwait(false);
}
catch (OperationCanceledException) when (token.IsCancellationRequested)
{
return new ModelComparisonBatchEntry
{
PresentationOrder = presentationOrder,
Result = new ModelComparisonRunResult
{
First = NotAnswered(first.Label),
Second = NotAnswered(second.Label),
},
};
}
try
{
var result = await runner.RunAsync(first, second, request, token, presentationOrder, judge, judgeInstructions).ConfigureAwait(false);
return new ModelComparisonBatchEntry
{
PresentationOrder = presentationOrder,
Result = result,
};
}
finally
{
this.concurrencyLimit.Release();
}
}
private static ModelComparisonAnswer NotAnswered(string label) => new()
{
Label = label,
Text = string.Empty,
Completed = false,
};
}

View File

@ -0,0 +1,66 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// How often the user's own vote came out on each side across a batch of runs.
/// </summary>
/// <remarks>
/// Counted in First/Second terms, never in columns, for the same reason as
/// <see cref="ModelComparisonBatchJudgeSummary"/>: the column a model stood in was drawn fresh for
/// every run. A run the user never voted on -- the batch was left before reaching it -- is left out
/// of every count, the same way an unanswered judge is left out of the judge tally.
/// </remarks>
public sealed record ModelComparisonBatchVoteSummary
{
public required int TotalRunCount { get; init; }
public required int ScoredRunCount { get; init; }
public required int FirstModelWins { get; init; }
public required int SecondModelWins { get; init; }
public required int Ties { get; init; }
/// <summary>
/// Tallies the user's own votes across a batch, alongside its entries. Never throws: a run with
/// no vote at its index is simply left out of the count.
/// </summary>
/// <param name="entries">The runs of the batch, each with its own presentation order.</param>
/// <param name="votes">The user's vote for each entry, at the same index. Null means not voted on.</param>
public static ModelComparisonBatchVoteSummary From(IReadOnlyList<ModelComparisonBatchEntry> entries, IReadOnlyList<ModelComparisonVote?> votes)
{
var firstModelWins = 0;
var secondModelWins = 0;
var ties = 0;
for (var index = 0; index < entries.Count; index++)
{
if (index >= votes.Count || votes[index] is not { } vote)
continue;
switch (entries[index].PresentationOrder.ToModelChoice(vote))
{
case ModelComparisonModelChoice.FIRST:
firstModelWins++;
break;
case ModelComparisonModelChoice.SECOND:
secondModelWins++;
break;
case ModelComparisonModelChoice.TIE:
ties++;
break;
}
}
return new ModelComparisonBatchVoteSummary
{
TotalRunCount = entries.Count,
ScoredRunCount = firstModelWins + secondModelWins + ties,
FirstModelWins = firstModelWins,
SecondModelWins = secondModelWins,
Ties = ties,
};
}
}

View File

@ -0,0 +1,102 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// What the judge is asked: the same request the two models received, their two answers, and
/// optionally what the user wants the judge to pay particular attention to.
/// </summary>
/// <remarks>
/// The judge is asked once both models have answered, and is kept as blind to which model wrote
/// which answer as the user is: the answers go in as plain text, labeled "Answer A" and "Answer B",
/// never by model name. These are the same letters the user sees, in the same order, so the judge's
/// own words about "Answer A"/"Answer B" line up with what is on screen -- the caller has to settle
/// the presentation order before building this, not after. <see cref="ModelComparisonVote"/> is what
/// the judge answers with, the same column terms the user's own vote uses.
/// </remarks>
public sealed record ModelComparisonJudgeRequest
{
/// <param name="request">The request the two models were asked.</param>
/// <param name="instructions">
/// What the user wants the judge to pay particular attention to, in their own words, or an empty
/// text to leave the judge with nothing but the generic criterion: which answer better answers
/// the question or completes the task. The judge already sees the request and both answers, so
/// this is a refinement, not a requirement.
/// </param>
/// <param name="answerA">The answer standing in column A.</param>
/// <param name="answerB">The answer standing in column B.</param>
/// <param name="languageName">
/// The language the judge is asked to write its reasoning in, by its English name (for example
/// "German"), or an empty text to leave the language to the judge itself.
/// </param>
public ModelComparisonJudgeRequest(ModelComparisonRequest request, string instructions, string answerA, string answerB, string languageName = "")
{
this.Request = request;
this.Instructions = instructions.Trim();
this.AnswerA = answerA;
this.AnswerB = answerB;
this.LanguageName = languageName.Trim();
}
public ModelComparisonRequest Request { get; }
public string Instructions { get; }
public string AnswerA { get; }
public string AnswerB { get; }
public string LanguageName { get; }
/// <summary>
/// Tells the judge which language to answer in, or nothing when no language was given.
/// </summary>
private string LanguageInstruction => this.LanguageName.Length == 0
? string.Empty
: $" Write the reasoning in {this.LanguageName}, regardless of the language of the request or the answers.";
/// <summary>
/// What the judge is told to look for: the generic criterion alone, or that criterion refined by
/// what the user wrote, when they wrote anything.
/// </summary>
private string Criterion => this.Instructions.Length == 0
? "Judge which answer better answers the question or better completes the task in the request below."
: $"Judge which answer better answers the question or better completes the task in the request below. Pay particular attention to this: {this.Instructions}";
/// <summary>
/// Builds the prompt for the judge model, asking for one JSON object and nothing else.
/// </summary>
/// <param name="cacheBuster">
/// An identifier to append as a plainly labeled, ignorable line, or an empty text to leave the
/// prompt exactly as asked. See <see cref="ModelComparisonRequest.ToPrompt"/> for why: the judge
/// is asked fresh for every run of a batch, same as the two models, and a caching gateway cannot
/// tell that apart from one identical request repeated unless the prompt says so itself -- most
/// of all here, where two similar answers can easily make for a near-identical judge prompt too.
/// </param>
/// <returns>The user prompt the judge model receives.</returns>
public string ToPrompt(string cacheBuster = "")
{
var prompt = $$"""
You judge two AI answers to the same request. You do not know, and must not guess, which model wrote which answer, so judge only by what the answers say.
{{this.Criterion}}
# Request
{{this.Request.ToPrompt()}}
# Answer A
{{this.AnswerA}}
# Answer B
{{this.AnswerB}}
Decide which answer is better, or whether they are equally good. Respond with exactly one JSON object and nothing else, in this shape:
{"preferred": "A", "reasoning": "A short explanation, a few sentences at most."}
The value of "preferred" must be exactly one of "A", "B", or "TIE". In the reasoning, refer to each answer only by its letter, A or B, never by a model name -- translated into whatever language you write the reasoning in, not forced into English.{{this.LanguageInstruction}}
""";
return cacheBuster.Length == 0
? prompt
: $"{prompt}\n\n(Request ID: {cacheBuster}. This has no bearing on your verdict; it is only here so repeated requests are not answered from a cache.)";
}
}

View File

@ -0,0 +1,110 @@
using System.Text.Json;
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// What the judge said about the two answers, once it could be understood.
/// </summary>
/// <remarks>
/// Built from whatever the judge model answered, the same way <see cref="ModelComparisonAnswer"/>
/// is built from what the two compared models answered: a judge which did not answer, or whose
/// answer could not be read as the JSON it was asked for, ends up with <see cref="Completed"/>
/// false and nothing else to show.
/// </remarks>
public sealed record ModelComparisonJudgeVerdict
{
/// <summary>
/// The name to show for the judge model. Shown together with its verdict, once the user has
/// voted -- the same point at which the names of the two compared models are revealed.
/// </summary>
public required string Label { get; init; }
/// <summary>
/// Whether the judge could be understood: it answered, and that answer held a preference in the
/// expected shape.
/// </summary>
public required bool Completed { get; init; }
/// <summary>
/// Which column the judge preferred, or null when <see cref="Completed"/> is false. The same
/// column terms the user's own vote uses, since the judge is asked about "Answer A"/"Answer B",
/// never about a model.
/// </summary>
public ModelComparisonVote? Preferred { get; init; }
/// <summary>
/// Why the judge decided as it did, or an empty text when <see cref="Completed"/> is false.
/// </summary>
public string Reasoning { get; init; } = string.Empty;
private static readonly JsonSerializerOptions JSON_SERIALIZER_OPTIONS = new()
{
PropertyNamingPolicy = JsonNamingPolicy.SnakeCaseLower,
};
/// <summary>
/// Reads the judge's answer. Never throws: whatever cannot be understood becomes a verdict
/// which is not <see cref="Completed"/>, the same as a judge which did not answer at all.
/// </summary>
/// <param name="label">The name of the judge model.</param>
/// <param name="judgeAnswered">Whether the judge model answered at all, before its text is even looked at.</param>
/// <param name="judgeText">What the judge model answered.</param>
public static ModelComparisonJudgeVerdict Parse(string label, bool judgeAnswered, string judgeText)
{
var notCompleted = new ModelComparisonJudgeVerdict { Label = label, Completed = false };
if (!judgeAnswered)
return notCompleted;
var json = ExtractJsonObject(judgeText);
if (json.Length == 0)
return notCompleted;
try
{
var parsed = JsonSerializer.Deserialize<JudgeResponse>(json, JSON_SERIALIZER_OPTIONS);
if (parsed is null || ParsePreference(parsed.Preferred) is not { } preference)
return notCompleted;
return new ModelComparisonJudgeVerdict
{
Label = label,
Completed = true,
Preferred = preference,
Reasoning = parsed.Reasoning?.Trim() ?? string.Empty,
};
}
catch (JsonException)
{
return notCompleted;
}
}
/// <summary>
/// Reads the column the judge named. "A" and "B" rather than the enum's own member names, since
/// those are what the judge was asked to answer with.
/// </summary>
private static ModelComparisonVote? ParsePreference(string? preferred) => preferred?.Trim().ToUpperInvariant() switch
{
"A" => ModelComparisonVote.COLUMN_A,
"B" => ModelComparisonVote.COLUMN_B,
"TIE" => ModelComparisonVote.TIE,
_ => null,
};
/// <summary>
/// Cuts out the JSON object the judge was asked for, from the first '{' to the last '}'. The
/// judge is asked for nothing but that object, but models add a sentence around it all the same.
/// </summary>
private static string ExtractJsonObject(string text)
{
var start = text.IndexOf('{');
var end = text.LastIndexOf('}');
if (start < 0 || end <= start)
return string.Empty;
return text[start..(end + 1)];
}
private sealed record JudgeResponse(string Preferred, string? Reasoning);
}

View File

@ -0,0 +1,17 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// Which model a column-based choice -- a vote or a judge verdict -- actually means, once translated
/// out of the run's own presentation order.
/// </summary>
/// <remarks>
/// First/Second names the models the way the user picked them, the one thing that stays comparable
/// across every run of a batch: the column a model stood in is drawn fresh for every run, so a tally
/// kept in column terms would count nothing meaningful.
/// </remarks>
public enum ModelComparisonModelChoice
{
FIRST,
SECOND,
TIE,
}

View File

@ -0,0 +1,22 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// One of the two models in a comparison, and the way to ask it.
/// </summary>
/// <remarks>
/// The way to ask is a function rather than a provider, so the comparison itself does not depend on
/// anything of the app. <see cref="ModelComparisonProviderAdapter"/> builds the function for a real
/// provider.
/// </remarks>
public sealed record ModelComparisonParticipant
{
/// <summary>
/// The name to show for the model. It is copied into the answer, see <see cref="ModelComparisonAnswer.Label"/>.
/// </summary>
public required string Label { get; init; }
/// <summary>
/// Sends a prompt to the model and streams the answer back, chunk by chunk.
/// </summary>
public required Func<string, CancellationToken, IAsyncEnumerable<string>> Ask { get; init; }
}

View File

@ -0,0 +1,14 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// Which of the two models is shown first, that is, in column A.
/// </summary>
/// <remarks>
/// The order is drawn at random for every comparison, so that the position of an answer says
/// nothing about the model. The first model is the one the user picked first.
/// </remarks>
public enum ModelComparisonPresentationOrder
{
FIRST_MODEL_FIRST,
SECOND_MODEL_FIRST,
}

View File

@ -0,0 +1,48 @@
namespace AIStudio.Assistants.ModelComparison;
public static class ModelComparisonPresentationOrderExtensions
{
/// <summary>
/// Gets the answer which is shown in column A.
/// </summary>
/// <param name="order">Which model is shown in column A.</param>
/// <param name="run">The answers of both models.</param>
/// <returns>The answer for column A.</returns>
public static ModelComparisonAnswer InColumnA(this ModelComparisonPresentationOrder order, ModelComparisonRunResult run) => order switch
{
ModelComparisonPresentationOrder.FIRST_MODEL_FIRST => run.First,
ModelComparisonPresentationOrder.SECOND_MODEL_FIRST => run.Second,
_ => throw new ArgumentOutOfRangeException(nameof(order), order, $"The presentation order '{order}' is not known."),
};
/// <summary>
/// Gets the answer which is shown in column B.
/// </summary>
/// <param name="order">Which model is shown in column A.</param>
/// <param name="run">The answers of both models.</param>
/// <returns>The answer for column B.</returns>
public static ModelComparisonAnswer InColumnB(this ModelComparisonPresentationOrder order, ModelComparisonRunResult run) => order switch
{
ModelComparisonPresentationOrder.FIRST_MODEL_FIRST => run.Second,
ModelComparisonPresentationOrder.SECOND_MODEL_FIRST => run.First,
_ => throw new ArgumentOutOfRangeException(nameof(order), order, $"The presentation order '{order}' is not known."),
};
/// <summary>
/// Turns a column-based choice back into the model it means. The reverse of
/// <see cref="InColumnA"/>/<see cref="InColumnB"/>: those turn a known model into its column,
/// this turns a known column back into its model.
/// </summary>
/// <param name="order">Which model stood in column A for this choice.</param>
/// <param name="vote">The column-based choice, from the user's own vote or from a judge verdict.</param>
public static ModelComparisonModelChoice ToModelChoice(this ModelComparisonPresentationOrder order, ModelComparisonVote vote) => vote switch
{
ModelComparisonVote.TIE => ModelComparisonModelChoice.TIE,
ModelComparisonVote.COLUMN_A => order is ModelComparisonPresentationOrder.FIRST_MODEL_FIRST ? ModelComparisonModelChoice.FIRST : ModelComparisonModelChoice.SECOND,
ModelComparisonVote.COLUMN_B => order is ModelComparisonPresentationOrder.FIRST_MODEL_FIRST ? ModelComparisonModelChoice.SECOND : ModelComparisonModelChoice.FIRST,
_ => throw new ArgumentOutOfRangeException(nameof(vote), vote, $"The vote '{vote}' is not known."),
};
}

View File

@ -0,0 +1,101 @@
using System.Runtime.CompilerServices;
using AIStudio.Chat;
using AIStudio.Provider;
using AIStudio.Settings;
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// Connects a provider of the app to the model comparison.
/// </summary>
public static class ModelComparisonProviderAdapter
{
/// <summary>
/// Creates the participant for a provider.
/// </summary>
/// <param name="providerSettings">The provider, with the model to ask.</param>
/// <param name="settingsManager">The settings the provider needs for its request.</param>
/// <param name="component">The component whose confidence requirements the provider has to meet.</param>
/// <param name="title">The title of the assistant, as the name of the throwaway chat.</param>
/// <returns>The participant.</returns>
/// <remarks>
/// The provider is built once here, not once per run: it holds nothing but its own
/// <c>HttpClient</c>, so every run of a batch reuses the same connection instead of paying for a
/// fresh one every time.
/// </remarks>
public static ModelComparisonParticipant CreateParticipant(AIStudio.Settings.Provider providerSettings, SettingsManager settingsManager, Tools.Components component, string title)
{
var provider = providerSettings.CreateProvider();
return new()
{
Label = providerSettings.ToString(),
Ask = (prompt, token) => StreamAnswer(provider, providerSettings, settingsManager, component, title, prompt, token),
};
}
/// <summary>
/// Asks the model straight through the provider, without <c>ContentText.CreateFromProviderAsync</c>.
/// </summary>
/// <remarks>
/// Two reasons. The stream of chunks is what the first-token time is measured on, and the
/// streaming event of a <c>ContentText</c> only fires every few seconds in the energy saving
/// mode. And that method turns a provider which is not allowed into an empty answer without a
/// word, which would let the model lose every vote. Here it is an error, so the run reports it
/// as a failure of that model.
///
/// The request is what the model comparison promises: no profile, no system prompt of its own,
/// no tools and no data sources, only the prompt.
/// </remarks>
private static async IAsyncEnumerable<string> StreamAnswer(
IProvider provider,
AIStudio.Settings.Provider providerSettings,
SettingsManager settingsManager,
Tools.Components component,
string title,
string prompt,
[EnumeratorCancellation] CancellationToken token)
{
if (!settingsManager.IsProviderConfident(providerSettings, component))
throw new InvalidOperationException($"The provider '{providerSettings.InstanceName}' does not meet the confidence requirements of the model comparison.");
var chatThread = new ChatThread
{
IncludeDateTime = false,
SelectedProvider = providerSettings.Id,
SelectedProfile = Profile.NO_PROFILE.Id,
SelectedToolIds = [],
SystemPrompt = string.Empty,
WorkspaceId = Guid.Empty,
ChatId = Guid.NewGuid(),
Name = title,
Blocks = [],
RuntimeComponent = component,
RuntimeSelectedToolIds = [],
RuntimeToolsAreAssistantManaged = true,
};
chatThread.Blocks.Add(new ContentBlock
{
Time = DateTimeOffset.Now,
ContentType = ContentType.TEXT,
Role = ChatRole.USER,
Content = new ContentText
{
Text = prompt,
},
});
// The provider writes tool traces into the last block of the AI, so the thread has one:
chatThread.Blocks.Add(new ContentBlock
{
Time = DateTimeOffset.Now,
ContentType = ContentType.TEXT,
Role = ChatRole.AI,
Content = new ContentText(),
});
await foreach (var chunk in provider.StreamChatCompletion(providerSettings.Model, chatThread, settingsManager, token).ConfigureAwait(false))
yield return chunk.Content;
}
}

View File

@ -0,0 +1,85 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// What both models are asked: an optional context, such as a document, and the question about it.
/// </summary>
/// <remarks>
/// Both models get exactly the text <see cref="ToPrompt"/> returns, in the same shape, and nothing
/// else. That is what makes the answers comparable: a difference in the wording of a prompt would
/// count as a difference between the models.
/// </remarks>
public sealed record ModelComparisonRequest
{
/// <summary>
/// Creates the request from what the user entered. Leading and trailing white space is dropped,
/// so a stray line break at the end of a pasted document does not count as text.
/// </summary>
/// <param name="context">The context, or an empty text when there is none.</param>
/// <param name="question">The question or task.</param>
/// <param name="languageName">
/// The language both models are asked to answer in, by its English name (for example
/// "German"), or an empty text to leave the language to the models themselves.
/// </param>
public ModelComparisonRequest(string context, string question, string languageName = "")
{
this.Context = context.Trim();
this.Question = question.Trim();
this.LanguageName = languageName.Trim();
}
public string Context { get; }
public string Question { get; }
public string LanguageName { get; }
/// <summary>
/// Shortens the question to something which fits into a summary, and marks the cut.
/// </summary>
/// <param name="maxCharacters">The number of characters of the question to keep, at most.</param>
/// <returns>The question, or its beginning followed by an ellipsis.</returns>
public string GetQuestionPreview(int maxCharacters)
{
ArgumentOutOfRangeException.ThrowIfNegativeOrZero(maxCharacters);
if (this.Question.Length <= maxCharacters)
return this.Question;
// Cutting between the two halves of a character outside the basic plane, an emoji for example, would leave half of it:
var length = char.IsHighSurrogate(this.Question[maxCharacters - 1]) ? maxCharacters - 1 : maxCharacters;
return this.Question[..length].TrimEnd() + "…";
}
/// <summary>
/// Builds the prompt. Without a context, it is the question alone: headings around a lone
/// question would only give the models something to comment on.
/// </summary>
/// <param name="cacheBuster">
/// An identifier to append as a plainly labeled, ignorable line, or an empty text to leave the
/// prompt exactly as asked. Comparing a model against itself relies on asking it more than once,
/// and a byte-identical prompt is exactly what a caching gateway between here and the model --
/// LiteLLM among them -- is built to recognize and answer from its cache instead of asking again.
/// A fixed request would silently stop being independent samples without this.
/// </param>
/// <returns>The user prompt both models receive.</returns>
public string ToPrompt(string cacheBuster = "")
{
var body = this.Context.Length == 0
? this.Question
: $"""
# Context
{this.Context}
# Question or task
{this.Question}
""";
var withLanguage = this.LanguageName.Length == 0
? body
: $"{body}\n\nAnswer in {this.LanguageName}.";
return cacheBuster.Length == 0
? withLanguage
: $"{withLanguage}\n\n(Request ID: {cacheBuster}. This has no bearing on your answer; it is only here so repeated requests are not answered from a cache.)";
}
}

View File

@ -0,0 +1,29 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// What a comparison run brought back: the answers of the two models.
/// </summary>
public sealed record ModelComparisonRunResult
{
/// <summary>
/// When the run ended. Both answers are shown with this one time: a time of their own would
/// tell the faster model from the slower one.
/// </summary>
public DateTimeOffset Time { get; init; } = DateTimeOffset.Now;
public required ModelComparisonAnswer First { get; init; }
public required ModelComparisonAnswer Second { get; init; }
/// <summary>
/// What an optional judge said about the two answers, or null when no judge was asked, or when
/// there was nothing for it to judge because at least one model did not answer.
/// </summary>
public ModelComparisonJudgeVerdict? Judge { get; init; }
/// <summary>
/// Whether both models answered. Only then there is anything to vote on: a vote between an
/// answer and a failure would say nothing about the models.
/// </summary>
public bool BothCompleted => this.First.Completed && this.Second.Completed;
}

View File

@ -0,0 +1,154 @@
using System.Text;
using AIStudio.Chat;
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// Sends one request to two models at the same time and measures how each of them does. An optional
/// third model, the judge, is asked about the two answers once both are in.
/// </summary>
/// <remarks>
/// The two requests do not depend on each other. A model which fails, or takes very long, costs
/// the other one nothing, and only a cancellation of the token ends both. Nothing here throws for a
/// model which did not answer: that is a result to report, and the caller has to be able to show it
/// next to the answer of the other model. The judge is different: it needs both answers, so it runs
/// only after them, not alongside them.
///
/// Every await here uses <c>ConfigureAwait(false)</c>: this class is invoked from a Blazor Server
/// page (<c>ModelComparisonBatchRunner</c>, in turn from the assistant itself), and the first of
/// them would otherwise still be running on that page's own synchronisation context -- caught before
/// its own first suspension, the same way <c>ModelComparisonBatchRunner.RunOneAsync</c> is. Without
/// this, every chunk streamed from a model, the judge included, would try to resume on that same
/// context, the very thing a click on the page also has to wait its turn on.
/// </remarks>
/// <param name="logger">Where to say why a model did not answer. The users see no reason: the text of an error may name a provider.</param>
/// <param name="timeProvider">The clock which measures the times. The system clock, unless a test brings its own.</param>
public sealed class ModelComparisonRunner(ILogger logger, TimeProvider? timeProvider = null)
{
private readonly TimeProvider time = timeProvider ?? TimeProvider.System;
/// <param name="first">The first model.</param>
/// <param name="second">The second model.</param>
/// <param name="request">What both models are asked.</param>
/// <param name="token">Cancels the run, first model, second model, and judge alike.</param>
/// <param name="presentationOrder">
/// Which of the two models will stand in column A once the answers are shown. The judge is told
/// the answers as "Answer A" and "Answer B", the same terms the user sees, so this has to be
/// settled before it is asked rather than after. Ignored without a <paramref name="judge"/>.
/// </param>
/// <param name="judge">The optional judge, asked once both models have answered.</param>
/// <param name="judgeInstructions">What the user wants the judge to pay attention to. Ignored without a <paramref name="judge"/>.</param>
public async Task<ModelComparisonRunResult> RunAsync(ModelComparisonParticipant first, ModelComparisonParticipant second, ModelComparisonRequest request, CancellationToken token, ModelComparisonPresentationOrder presentationOrder = ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonParticipant? judge = null, string judgeInstructions = "")
{
//
// A fresh ID for every run, the same one for both models: it stops a caching gateway between
// here and the models from answering a repeat run out of its cache instead of asking again,
// without making the two models' prompts differ from each other.
//
var prompt = request.ToPrompt(Guid.NewGuid().ToString("N"));
//
// Task.Run on purpose: a provider may do work before its first await, and without it the
// second model would only be asked once the first one had got that far.
//
var firstTask = Task.Run(() => this.AskAsync(first, prompt, token), CancellationToken.None);
var secondTask = Task.Run(() => this.AskAsync(second, prompt, token), CancellationToken.None);
await Task.WhenAll(firstTask, secondTask).ConfigureAwait(false);
var firstAnswer = await firstTask.ConfigureAwait(false);
var secondAnswer = await secondTask.ConfigureAwait(false);
return new ModelComparisonRunResult
{
First = firstAnswer,
Second = secondAnswer,
Judge = await this.AskJudgeAsync(judge, judgeInstructions, request, presentationOrder, firstAnswer, secondAnswer, token).ConfigureAwait(false),
};
}
/// <summary>
/// Asks the judge once both models have answered.
/// </summary>
/// <remarks>
/// The judge needs both complete answers to judge, so it is asked after them, never alongside
/// them. Nothing here throws: a judge which fails is a verdict which is not completed, the same
/// as a model which did not answer in <see cref="AskAsync"/>.
/// </remarks>
private async Task<ModelComparisonJudgeVerdict?> AskJudgeAsync(ModelComparisonParticipant? judge, string judgeInstructions, ModelComparisonRequest request, ModelComparisonPresentationOrder presentationOrder, ModelComparisonAnswer first, ModelComparisonAnswer second, CancellationToken token)
{
if (judge is null || !first.Completed || !second.Completed || token.IsCancellationRequested)
return null;
var (answerA, answerB) = presentationOrder is ModelComparisonPresentationOrder.FIRST_MODEL_FIRST
? (first.Text, second.Text)
: (second.Text, first.Text);
var judgeRequest = new ModelComparisonJudgeRequest(request, judgeInstructions, answerA, answerB, request.LanguageName);
var judgeAnswer = await this.AskAsync(judge, judgeRequest.ToPrompt(Guid.NewGuid().ToString("N")), token).ConfigureAwait(false);
return ModelComparisonJudgeVerdict.Parse(judgeAnswer.Label, judgeAnswer.Completed, judgeAnswer.Text);
}
/// <summary>
/// Asks one model. Never throws: whatever goes wrong leaves the answer not completed.
/// </summary>
private async Task<ModelComparisonAnswer> AskAsync(ModelComparisonParticipant participant, string prompt, CancellationToken token)
{
var text = new StringBuilder();
var completed = true;
TimeSpan? firstTokenTime = null;
var start = this.time.GetTimestamp();
try
{
await foreach (var chunk in participant.Ask(prompt, token).WithCancellation(token).ConfigureAwait(false))
{
// A model which is asked to stop may still hand over a last chunk:
if (token.IsCancellationRequested)
break;
if (chunk.Length == 0)
continue;
firstTokenTime ??= this.time.GetElapsedTime(start);
text.Append(chunk);
}
if (token.IsCancellationRequested)
completed = false;
}
catch (OperationCanceledException) when (token.IsCancellationRequested)
{
completed = false;
}
catch (Exception exception)
{
logger.LogWarning(exception, "The model '{Label}' did not answer in the model comparison.", participant.Label);
completed = false;
}
var totalTime = this.time.GetElapsedTime(start);
var answer = text.ToString().RemoveThinkTags().Trim();
//
// A stream which ends cleanly without saying anything is a failure, too. So is one which
// ends inside the thinking of the model: nothing of what is left is an answer. Counted
// as answers, both would go into a vote as an empty text -- and always lose it.
//
if (completed && answer.Length == 0)
{
logger.LogWarning("The model '{Label}' ended without an answer in the model comparison.", participant.Label);
completed = false;
}
return new ModelComparisonAnswer
{
Label = participant.Label,
Text = answer,
Completed = completed,
FirstTokenTime = firstTokenTime,
TotalTime = completed ? totalTime : null,
};
}
}

View File

@ -0,0 +1,15 @@
namespace AIStudio.Assistants.ModelComparison;
/// <summary>
/// What the voter chose while the answers were still shown without the names of the models.
/// </summary>
/// <remarks>
/// This names a column on the screen, not a model. Which model stood in a column is what
/// <see cref="ModelComparisonPresentationOrder"/> says.
/// </remarks>
public enum ModelComparisonVote
{
COLUMN_A,
COLUMN_B,
TIE,
}

View File

@ -19,6 +19,7 @@
<AssistantBlock TSettings="SettingsDialogRewrite" Component="Components.REWRITE_ASSISTANT" Name="@T("Rewrite & Improve")" Description="@T("Rewrite and improve a given text for a chosen style.")" Icon="@Icons.Material.Filled.Edit" Link="@Routes.ASSISTANT_REWRITE"/>
<AssistantBlock TSettings="SettingsDialogPromptOptimizer" Component="Components.PROMPT_OPTIMIZER_ASSISTANT" Name="@T("Prompt Optimizer")" Description="@T("Optimize your prompt using a structured guideline.")" Icon="@Icons.Material.Filled.AutoFixHigh" Link="@Routes.ASSISTANT_PROMPT_OPTIMIZER"/>
<AssistantBlock TSettings="SettingsDialogSynonyms" Component="Components.SYNONYMS_ASSISTANT" Name="@T("Synonyms")" Description="@T("Find synonyms for a given word or phrase.")" Icon="@Icons.Material.Filled.Spellcheck" Link="@Routes.ASSISTANT_SYNONYMS"/>
<AssistantBlock TSettings="NoSettingsPanel" Component="Components.MODEL_COMPARISON_ASSISTANT" Name="@T("Model Comparison")" Description="@T("Compare two models on the same request and vote blindly for the better answer.")" Icon="@Icons.Material.Filled.Compare" Link="@Routes.ASSISTANT_MODEL_COMPARISON"/>
<AssistantBlock TSettings="NoSettingsPanel" Component="Components.META_ASSISTANT" RequiredPreviewFeature="PreviewFeatures.PRE_META_ASSISTANT_V1" Name="@T("Assistant Builder")" Description="@T("Generate your own assistants.")" Icon="@Icons.Material.Filled.AutoMode" Link="@Routes.ASSISTANT_META_ASSISTANT"/>
</AssistantCategoryBlock>

View File

@ -598,7 +598,7 @@ CONFIG["SETTINGS"] = {}
-- LEGAL_CHECK_ASSISTANT, SYNONYMS_ASSISTANT, MY_TASKS_ASSISTANT,
-- JOB_POSTING_ASSISTANT, BIAS_DAY_ASSISTANT, ERI_ASSISTANT,
-- DOCUMENT_ANALYSIS_ASSISTANT, BATCH_PROCESSING_ASSISTANT, SLIDE_BUILDER_ASSISTANT,
-- VISUAL_BRIEFING_ASSISTANT, I18N_ASSISTANT,
-- VISUAL_BRIEFING_ASSISTANT, MODEL_COMPARISON_ASSISTANT, I18N_ASSISTANT,
-- LOG_VIEWER_ASSISTANT
--
-- Replaces, does not merge: a configuration with a higher priority replaces this list

View File

@ -2136,6 +2136,240 @@ UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::LOGVIEWER::ASSISTANTLOGVIEWER::T469116133
-- Clear
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::LOGVIEWER::ASSISTANTLOGVIEWER::T77955010"] = "Löschen"
-- Beide Modelle erhalten diesen Text zusammen mit Ihrer Frage, zum Beispiel das Dokument, das sie zusammenfassen sollen.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1084467271"] = "Beide Modelle erhalten diesen Text zusammen mit Ihrer Frage, zum Beispiel das Dokument, das sie zusammenfassen sollen."
-- Der Beurteiler hielt beide Antworten für gleich gut, was Ihrer eigenen Stimme entsprach.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T110452334"] = "Der Beurteiler hielt beide Antworten für gleich gut, was Ihrer eigenen Stimme entsprach."
-- Sie wissen noch nicht, welches Modell welche Antwort geschrieben hat, und die Spalten stehen in zufälliger Reihenfolge. Vergleichen Sie anhand des Inhalts.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1255084431"] = "Sie wissen noch nicht, welches Modell welche Antwort geschrieben hat, und die Spalten stehen in zufälliger Reihenfolge. Vergleichen Sie anhand des Inhalts."
-- Skipping the remaining votes. Waiting for the rest of the batch to finish...
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1308855949"] = "Übrige Abstimmungen werden übersprungen. Warte, bis der Rest des Batches fertig ist …"
-- Setup
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T137864150"] = "Einrichtung"
-- Repeat the same comparison several times to see how consistently the models -- and the judge -- come out the same way. Every run gets its own blind vote.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1382748771"] = "Führen Sie denselben Vergleich mehrmals durch, um zu sehen, wie konsistent die Modelle – und der Beurteiler – zu denselben Ergebnissen kommen. Jeder Durchlauf erhält eine eigene blinde Bewertung."
-- Run {0}: no usable answer, nothing to vote on.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1407798402"] = "Durchlauf {0}: Keine verwertbare Antwort, nichts zum Abstimmen."
-- Antwort A
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1420338340"] = "Antwort A"
-- 20 (heavy load on the models)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1426001529"] = "20 (hohe Auslastung der Modelle)"
-- Run {0}: the models are done, waiting for the judge.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1451649171"] = "Durchlauf {0}: Die Modelle sind fertig – Warten auf den Beurteiler."
-- Antwort B
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1470671197"] = "Antwort B"
-- Bitte formulieren Sie eine Frage oder Aufgabe für beide Modelle.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1481551697"] = "Bitte formulieren Sie eine Frage oder Aufgabe für beide Modelle."
-- Delete this preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T148310729"] = "Diese Vorlage löschen"
-- Der Beurteiler bevorzugte {0}, was von Ihrer eigenen Abstimmung abwich.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1499258629"] = "Der Beurteiler bevorzugte {0}, was von Ihrer eigenen Abstimmung abwich."
-- All {0} runs are done.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1508408704"] = "Alle {0} Durchläufe sind abgeschlossen."
-- Run {0}: queued, waiting for a free slot.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1600973378"] = "Durchlauf {0}: in der Warteschlange, warte auf einen freien Platz."
-- Ihre Anfrage
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1607440787"] = "Ihre Anfrage"
-- The preset '{0}' is deleted right away. This cannot be undone.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T163336631"] = "Die Vorlage '{0}' wird sofort gelöscht. Diese Aktion kann nicht rückgängig gemacht werden."
-- Ihre Frage oder Aufgabe
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1646098156"] = "Ihre Frage oder Aufgabe"
-- Erstes Token nach {0}, vollständig nach {1}
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1658271318"] = "Erstes Token nach {0}, vollständig nach {1}"
-- Senden Sie dieselbe Anfrage an zwei Modelle und vergleichen Sie deren Antworten nebeneinander. Sie stimmen blind für die bessere Antwort ab und erfahren erst nach Ihrer Abstimmung, welches Modell welche Antwort verfasst hat.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1766730560"] = "Senden Sie dieselbe Anfrage an zwei Modelle und vergleichen Sie deren Antworten nebeneinander. Sie stimmen blind für die bessere Antwort ab und erfahren erst nach Ihrer Abstimmung, welches Modell welche Antwort verfasst hat."
-- Start a new request
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1767930003"] = "Neue Anfrage starten"
-- Run {0}: still working.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1838060603"] = "Durchlauf {0}: Läuft noch."
-- Update preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1860438243"] = "Vorlage aktualisieren"
-- Judge verdicts
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T1879874010"] = "Beurteiler-Entscheidungen"
-- Beurteiler
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2094884932"] = "Beurteiler"
-- Run {0}: the models and the judge are done, your vote is missing.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2115467037"] = "Durchlauf {0}: Die Modelle und der Beurteiler sind fertig, deine Stimme fehlt."
-- Save as new preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2148615337"] = "Als neue Vorlage speichern"
-- Skipping the remaining votes. Waiting for the judges to finish...
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2167132827"] = "Restliche Abstimmungen werden übersprungen. Warte, bis die Beurteiler fertig sind…"
-- 1 (single comparison)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2267053576"] = "1 (einzelner Vergleich)"
-- Welche Antwort ist besser?
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2388547151"] = "Welche Antwort ist besser?"
-- Der Beurteiler hat {0} favorisiert, was mit Ihrer eigenen Stimme übereinstimmt.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2457939612"] = "Der Beurteiler hat {0} favorisiert, was mit Ihrer eigenen Stimme übereinstimmt."
-- Yes, delete it
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2466176832"] = "Ja, löschen"
-- Equally good
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T248601509"] = "Gleich gut"
-- Kontext: {0} Zeichen
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2553774988"] = "Kontext: {0} Zeichen"
-- Progress Runs
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2638545408"] = "Fortschritt der Durchläufe"
-- Finish
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2670325426"] = "Fertig"
-- Beide Modelle antworten gerade. Das kann etwas dauern.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2709682697"] = "Beide Modelle antworten gerade. Das kann etwas dauern."
-- Run {0}: your vote was skipped.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2715329048"] = "Durchlauf {0}: Ihre Stimme wurde übersprungen."
-- Beide sind gleich gut
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2723559248"] = "Beide sind gleich gut"
-- {0} of {1} runs have no usable judge verdict.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2725027740"] = "{0} von {1} Durchläufen haben kein verwertbares Beurteiler-Ergebnis."
-- Preset from {0}
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2841002060"] = "Vorlage von {0}"
-- Kontext (optional)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T2925041491"] = "Kontext (optional)"
-- {0} of {1} runs have no usable vote -- never cast, or the run itself never got a usable answer -- and are not counted above.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3101172189"] = "Bei {0} von {1} Durchläufen liegt keine verwertbare Bewertung vor – sei es, weil nie abgestimmt wurde oder der Durchlauf selbst keine verwertbare Antwort geliefert hat – und diese werden oben nicht mitgezählt."
-- Antwort A ist besser
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3123852190"] = "Antwort A ist besser"
-- Skip remaining votes
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3188579167"] = "Verbleibende Stimmen überspringen"
-- Der Beurteiler sah beide Antworten als gleichwertig, während Sie eine von ihnen bevorzugten.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3189061594"] = "Der Beurteiler sah beide Antworten als gleichwertig, während Sie eine von ihnen bevorzugten."
-- Skip this run
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3334593867"] = "Diesen Durchlauf überspringen"
-- Kontext aus Datei laden
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3370653375"] = "Kontext aus Datei laden"
-- More runs are still being generated in the background.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3475250486"] = "Im Hintergrund werden noch weitere Durchläufe generiert."
-- LLM judge (optional)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3486621434"] = "LLM-Beurteiler (optional)"
-- Next run
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3533713133"] = "Nächster Durchlauf"
-- At least one of the models did not answer in this run, so there is nothing to compare.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3561467042"] = "Mindestens eines der Modelle hat in diesem Durchlauf nicht geantwortet, daher gibt es nichts zu vergleichen."
-- Vergleichen
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3581545044"] = "Vergleichen"
-- Was sollte der Beurteiler besonders beachten? (optional)
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3591351622"] = "Was sollte der Beurteiler besonders beachten? (optional)"
-- Your votes
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T360017507"] = "Ihre Stimmen"
-- Der Beurteiler sieht bereits die Anfrage und beide Antworten und urteilt allein auf dieser Grundlage, es sei denn, Sie fügen hier etwas spezifischeres hinzu, beispielsweise: eine gute Zusammenfassung behält alle Zahlen bei und nennt seine Quelle. Er weiß nicht, welches Modell welche Antwort verfasst hat, und seine Meinung wird Ihnen erst nach Ihrem eigenen Votum angezeigt.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3613128085"] = "Der Beurteiler sieht bereits die Anfrage und beide Antworten und urteilt allein auf dieser Grundlage, es sei denn, Sie fügen hier etwas spezifischeres hinzu, beispielsweise: eine gute Zusammenfassung behält alle Zahlen bei und nennt seine Quelle. Er weiß nicht, welches Modell welche Antwort verfasst hat, und seine Meinung wird Ihnen erst nach Ihrem eigenen Votum angezeigt."
-- Delete this preset?
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3766894914"] = "Diese Vorlage löschen?"
-- Run {0}: voted on.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3789682551"] = "Durchlauf {0}: abgestimmt."
-- Done.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T38691245"] = "Fertig."
-- Expand
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3889190145"] = "Erweitern"
-- Presets
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3897629751"] = "Vorlagen"
-- Modellvergleich
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3910956815"] = "Modellvergleich"
-- Load a saved preset
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T3970228636"] = "Gespeicherte Vorlage laden"
-- Preset name
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T40334785"] = "Vorlagen-Name"
-- Erstes Modell
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4044223184"] = "Erstes Modell"
-- No, keep it
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4188329028"] = "Nein, so lassen"
-- Run {0}: the models are done, your vote is missing.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4197671504"] = "Durchgang {0}: Die Modelle sind fertig – Ihre Stimme fehlt."
-- Der Beurteiler gab kein nutzbares Urteil ab.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T4278818434"] = "Der Beurteiler gab kein nutzbares Urteil ab."
-- Antworten
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T46578788"] = "Antworten"
-- Dieser Anbieter erfüllt die Vertrauensanforderungen dieses Assistenten nicht. Bitte wählen Sie einen anderen aus.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T512106357"] = "Dieser Anbieter erfüllt die Vertrauensanforderungen dieses Assistenten nicht. Bitte wählen Sie einen anderen aus."
-- Collapse
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T5885182"] = "Einklappen"
-- LLM Urteil: {0}
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T609232321"] = "LLM Urteil: {0}"
-- Number of runs
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T752176293"] = "Anzahl der Durchläufe"
-- Waiting for the next run...
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T758625750"] = "Warten auf den nächsten Lauf…"
-- Zweites Modell
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T867467332"] = "Zweites Modell"
-- Antwort B ist besser
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T933724747"] = "Antwort B ist besser"
-- Urteilsmodell
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MODELCOMPARISON::ASSISTANTMODELCOMPARISON::T953975257"] = "Urteilsmodell"
-- You can enter text, attach one or more documents, or use both. At least one input is required.
UI_TEXT_CONTENT["AISTUDIO::ASSISTANTS::MYTASKS::ASSISTANTMYTASKS::T1442535450"] = "Sie können Text eingeben, ein oder mehrere Dokumente anhängen oder beides verwenden. Mindestens eine Eingabe ist erforderlich."
@ -9129,12 +9363,18 @@ UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3756213118"] = "Erstellen Sie ein
-- Use an LLM to find an icon for a given context.
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3881504200"] = "Verwenden Sie ein LLM, um ein Icon für einen bestimmten Kontext zu finden."
-- Modellvergleich
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3910956815"] = "Modellvergleich"
-- Job Posting
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3930052338"] = "Stellenanzeige"
-- Ask a question about a legal document.
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T3970214537"] = "Stellen Sie Fragen zu einem juristischen Dokument."
-- Vergleichen Sie zwei Modelle anhand derselben Anfrage und stimmen Sie blind für die bessere Antwort ab.
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T401960846"] = "Vergleichen Sie zwei Modelle anhand derselben Anfrage und stimmen Sie blind für die bessere Antwort ab."
-- Log Viewer
UI_TEXT_CONTENT["AISTUDIO::PAGES::ASSISTANTS::T4130241777"] = "Protokollanzeige"
@ -10806,6 +11046,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::COMPONENTSEXTENSIONS::T555062689"] = "Assisten
-- New Chat
UI_TEXT_CONTENT["AISTUDIO::TOOLS::COMPONENTSEXTENSIONS::T826248509"] = "Neuer Chat"
-- Modellvergleichs-Assistent
UI_TEXT_CONTENT["AISTUDIO::TOOLS::COMPONENTSEXTENSIONS::T941908687"] = "Modellvergleichs-Assistent"
-- Trust LLM providers from the USA
UI_TEXT_CONTENT["AISTUDIO::TOOLS::CONFIDENCESCHEMESEXTENSIONS::T1748300640"] = "LLM-Anbietern aus den USA vertrauen"

View File

@ -33,6 +33,7 @@ public sealed partial class Routes
public const string ASSISTANT_AI_STUDIO_I18N = "/assistant/ai-studio/i18n";
public const string ASSISTANT_DOCUMENT_ANALYSIS = "/assistant/document-analysis";
public const string ASSISTANT_BATCH_PROCESSING = "/assistant/batch-processing";
public const string ASSISTANT_MODEL_COMPARISON = "/assistant/model-comparison";
public const string ASSISTANT_DYNAMIC = "/assistant/dynamic";
public const string ASSISTANT_META_ASSISTANT = "/assistant/builder";
public const string ASSISTANT_LOG_VIEWER = "/assistant/log-viewer";

View File

@ -28,6 +28,7 @@ public enum ConfigurableAssistant
LOG_VIEWER_ASSISTANT,
VISUAL_BRIEFING_ASSISTANT,
BATCH_PROCESSING_ASSISTANT,
MODEL_COMPARISON_ASSISTANT,
// ReSharper disable InconsistentNaming
I18N_ASSISTANT,

View File

@ -149,6 +149,8 @@ public sealed class Data
public DataDocumentAnalysis DocumentAnalysis { get; init; } = new();
public DataModelComparison ModelComparison { get; init; } = new();
/// <summary>
/// Gets the managed Batch Processing Assistant defaults.
/// </summary>

View File

@ -0,0 +1,9 @@
namespace AIStudio.Settings.DataModel;
public sealed class DataModelComparison
{
/// <summary>
/// The user's saved presets for the model comparison assistant.
/// </summary>
public List<ModelComparisonPreset> Presets { get; set; } = [];
}

View File

@ -0,0 +1,48 @@
namespace AIStudio.Settings.DataModel;
/// <summary>
/// A saved set of model comparison inputs, so a familiar test case does not have to be typed and
/// picked by hand again every time.
/// </summary>
/// <remarks>
/// Purely local to this user: unlike <see cref="DataDocumentAnalysisPolicy"/>, there is no Lua import
/// path and no enterprise ownership, because nothing here needs to be rolled out centrally.
/// </remarks>
public sealed record ModelComparisonPreset
{
public required string Id { get; init; }
/// <summary>
/// The name shown in the preset picker. Mutable, unlike the rest of a saved preset: renaming is
/// an edit to a saved entry, while every other field is only ever set once, at the moment the
/// current form is captured.
/// </summary>
public required string Name { get; set; }
/// <summary>
/// The first model's provider ID, or empty when none was selected while saving. Resolved back to
/// a provider through <see cref="SettingsManager.GetProviderById"/>, which already answers
/// <see cref="Provider.NONE"/> for an ID that no longer exists.
/// </summary>
public string FirstProviderId { get; init; } = string.Empty;
/// <summary>
/// The second model's provider ID. See <see cref="FirstProviderId"/>.
/// </summary>
public string SecondProviderId { get; init; } = string.Empty;
public bool JudgeEnabled { get; init; }
/// <summary>
/// The judge's provider ID. See <see cref="FirstProviderId"/>.
/// </summary>
public string JudgeProviderId { get; init; } = string.Empty;
public string JudgeInstructions { get; init; } = string.Empty;
public string Context { get; init; } = string.Empty;
public string Question { get; init; } = string.Empty;
public int RunCount { get; init; } = 1;
}

View File

@ -67,6 +67,7 @@ public static class AssistantVisibilityExtensions
Components.ERI_ASSISTANT => ConfigurableAssistant.ERI_ASSISTANT,
Components.DOCUMENT_ANALYSIS_ASSISTANT => ConfigurableAssistant.DOCUMENT_ANALYSIS_ASSISTANT,
Components.BATCH_PROCESSING_ASSISTANT => ConfigurableAssistant.BATCH_PROCESSING_ASSISTANT,
Components.MODEL_COMPARISON_ASSISTANT => ConfigurableAssistant.MODEL_COMPARISON_ASSISTANT,
Components.SLIDE_BUILDER_ASSISTANT => ConfigurableAssistant.SLIDE_BUILDER_ASSISTANT,
Components.VISUAL_BRIEFING_ASSISTANT => ConfigurableAssistant.VISUAL_BRIEFING_ASSISTANT,
Components.I18N_ASSISTANT => ConfigurableAssistant.I18N_ASSISTANT,

View File

@ -42,4 +42,5 @@ public enum Components
LOG_VIEWER_ASSISTANT,
VISUAL_BRIEFING_ASSISTANT,
BATCH_PROCESSING_ASSISTANT,
MODEL_COMPARISON_ASSISTANT,
}

View File

@ -67,6 +67,7 @@ public static class ComponentsExtensions
Components.I18N_ASSISTANT => false,
Components.DOCUMENT_ANALYSIS_ASSISTANT => false,
Components.BATCH_PROCESSING_ASSISTANT => false,
Components.MODEL_COMPARISON_ASSISTANT => false,
Components.LOG_VIEWER_ASSISTANT => false,
Components.APP_SETTINGS => false,
@ -100,6 +101,7 @@ public static class ComponentsExtensions
Components.I18N_ASSISTANT => TB("Localization Assistant"),
Components.DOCUMENT_ANALYSIS_ASSISTANT => TB("Document Analysis Assistant"),
Components.BATCH_PROCESSING_ASSISTANT => TB("Batch Processing Assistant"),
Components.MODEL_COMPARISON_ASSISTANT => TB("Model Comparison Assistant"),
Components.SLIDE_BUILDER_ASSISTANT => TB("Slide Planner Assistant"),
Components.VISUAL_BRIEFING_ASSISTANT => TB("Visual Briefing Assistant"),
Components.META_ASSISTANT => TB("Assistant Builder"),

View File

@ -0,0 +1,124 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how the judge's verdicts across a batch are tallied.
/// </summary>
/// <remarks>
/// The point of the tally is that it survives the presentation order changing from run to run: a
/// model which won column A in one run and column B in the next must still be counted as the same
/// winner both times.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonBatchJudgeSummaryTests
{
[Test]
public void AnEmptyBatchScoresNothing()
{
var summary = ModelComparisonBatchJudgeSummary.From([]);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.Zero);
Assert.That(summary.ScoredRunCount, Is.Zero);
Assert.That(summary.FirstModelWins, Is.Zero);
Assert.That(summary.SecondModelWins, Is.Zero);
Assert.That(summary.Ties, Is.Zero);
});
}
[Test]
public void WinsAreCountedByModelNotByColumn()
{
// The first model stands in column A twice and in column B once, and wins every time -- the tally has to follow the model, not the column:
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A),
Entry(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.COLUMN_B),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A),
};
var summary = ModelComparisonBatchJudgeSummary.From(entries);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(3));
Assert.That(summary.ScoredRunCount, Is.EqualTo(3));
Assert.That(summary.FirstModelWins, Is.EqualTo(3));
Assert.That(summary.SecondModelWins, Is.Zero);
Assert.That(summary.Ties, Is.Zero);
});
}
[Test]
public void ATieIsCountedAsATieRegardlessOfOrder()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.TIE),
Entry(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.TIE),
};
var summary = ModelComparisonBatchJudgeSummary.From(entries);
Assert.That(summary.Ties, Is.EqualTo(2));
}
[Test]
public void ARunWithoutACompletedVerdictIsNotScored()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A),
EntryWithoutAVerdict(),
};
var summary = ModelComparisonBatchJudgeSummary.From(entries);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(2), "Both runs happened.");
Assert.That(summary.ScoredRunCount, Is.EqualTo(1), "Only one run had a usable verdict.");
Assert.That(summary.FirstModelWins, Is.EqualTo(1));
});
}
private static ModelComparisonBatchEntry Entry(ModelComparisonPresentationOrder order, ModelComparisonVote judgePreferred) => new()
{
PresentationOrder = order,
Result = new ModelComparisonRunResult
{
First = Answer("First"),
Second = Answer("Second"),
Judge = new ModelComparisonJudgeVerdict
{
Label = "Judge",
Completed = true,
Preferred = judgePreferred,
},
},
};
private static ModelComparisonBatchEntry EntryWithoutAVerdict() => new()
{
PresentationOrder = ModelComparisonPresentationOrder.FIRST_MODEL_FIRST,
Result = new ModelComparisonRunResult
{
First = Answer("First"),
Second = Answer("Second"),
Judge = new ModelComparisonJudgeVerdict
{
Label = "Judge",
Completed = false,
},
},
};
private static ModelComparisonAnswer Answer(string label) => new()
{
Label = label,
Text = $"The answer of {label}.",
Completed = true,
};
}

View File

@ -0,0 +1,258 @@
using System.Runtime.CompilerServices;
using AIStudio.Assistants.ModelComparison;
using Microsoft.Extensions.Logging.Abstractions;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how a batch makes several independent runs of the same comparison.
/// </summary>
/// <remarks>
/// What is under test is that every run is genuinely independent -- its own request to both models,
/// its own presentation order -- not the single run itself, which <see cref="ModelComparisonRunnerTests"/>
/// already covers.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonBatchRunnerTests
{
private static readonly ModelComparisonRequest REQUEST = new("The document.", "Summarize it.");
[Test]
public async Task TheRequestedNumberOfRunsIsMade()
{
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 5,
CancellationToken.None);
var entries = await Task.WhenAll(tasks);
Assert.That(entries, Has.Length.EqualTo(5));
}
[Test]
public void EveryRunStartsWithoutWaitingForAnyOfThemToFinish()
{
// Nothing here completes on its own; Start must still hand back its tasks right away. The
// cancellation afterward is only cleanup, so the never-ending calls do not outlive the test:
using var cancellation = new CancellationTokenSource();
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, token) => Never(token)),
CreateParticipant("2", (_, token) => Never(token)),
REQUEST,
runCount: 3,
cancellation.Token);
try
{
Assert.That(tasks, Has.Count.EqualTo(3));
}
finally
{
cancellation.Cancel();
}
}
[Test]
public async Task NoMoreRunsThanTheLimitAskAModelAtOnce()
{
// Ten runs, twenty model calls between them; if every run started at once, every one of
// those twenty calls would be concurrent. A short delay gives the batch runner room to queue
// the rest behind the limit, which the peak below has to stay within -- two models per run:
var concurrentCalls = 0;
var peakConcurrentCalls = 0;
var peakLock = new object();
async IAsyncEnumerable<string> Ask(string _, [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken token)
{
var current = Interlocked.Increment(ref concurrentCalls);
lock (peakLock)
peakConcurrentCalls = Math.Max(peakConcurrentCalls, current);
await Task.Delay(TimeSpan.FromMilliseconds(30), token);
Interlocked.Decrement(ref concurrentCalls);
yield return "An answer.";
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", Ask),
CreateParticipant("2", Ask),
REQUEST,
runCount: 10,
CancellationToken.None);
await Task.WhenAll(tasks);
var expectedPeak = ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS * 2;
Assert.That(peakConcurrentCalls, Is.LessThanOrEqualTo(expectedPeak), $"At most {ModelComparisonBatchRunner.MAX_CONCURRENT_RUNS} runs at a time, two model calls each.");
}
[Test]
public async Task EachRunAsksBothModelsAfresh()
{
var firstAskedCount = 0;
var secondAskedCount = 0;
IAsyncEnumerable<string> AskFirst(string _, CancellationToken __)
{
Interlocked.Increment(ref firstAskedCount);
return Chunks("An answer.");
}
IAsyncEnumerable<string> AskSecond(string _, CancellationToken __)
{
Interlocked.Increment(ref secondAskedCount);
return Chunks("Another answer.");
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", AskFirst),
CreateParticipant("2", AskSecond),
REQUEST,
runCount: 4,
CancellationToken.None);
await Task.WhenAll(tasks);
Assert.Multiple(() =>
{
Assert.That(firstAskedCount, Is.EqualTo(4));
Assert.That(secondAskedCount, Is.EqualTo(4));
});
}
[Test]
public async Task EveryRunCompletesOnItsOwnEvenWhenOneFails()
{
// Every third run fails; the rest must not be dragged down with it:
var callNumber = 0;
IAsyncEnumerable<string> Flaky(string _, CancellationToken __)
{
var thisCall = Interlocked.Increment(ref callNumber);
return thisCall % 3 == 0 ? Failing() : Chunks("An answer.");
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", Flaky),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 6,
CancellationToken.None);
var entries = await Task.WhenAll(tasks);
Assert.Multiple(() =>
{
Assert.That(entries, Has.Length.EqualTo(6));
Assert.That(entries.Count(entry => entry.Result.BothCompleted), Is.EqualTo(4));
Assert.That(entries.Count(entry => !entry.Result.BothCompleted), Is.EqualTo(2));
});
}
[Test]
public void ARunCountBelowOneIsRefused()
{
var batchRunner = CreateBatchRunner();
Assert.That(() => batchRunner.Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 0,
CancellationToken.None), Throws.TypeOf<ArgumentOutOfRangeException>());
}
[Test]
public async Task BothPresentationOrdersCanComeUp()
{
// With enough runs, drawing only ever one of the two orders would be a bug, not bad luck:
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 60,
CancellationToken.None);
var entries = await Task.WhenAll(tasks);
var orders = entries.Select(entry => entry.PresentationOrder).Distinct().ToArray();
Assert.That(orders, Is.EquivalentTo(Enum.GetValues<ModelComparisonPresentationOrder>()));
}
[Test]
public async Task WithAJudgeEveryRunHasItsOwnVerdict()
{
var judgeAskedCount = 0;
IAsyncEnumerable<string> AskJudge(string _, CancellationToken __)
{
Interlocked.Increment(ref judgeAskedCount);
return Chunks("""{"preferred": "A", "reasoning": "n/a"}""");
}
var tasks = CreateBatchRunner().Start(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
runCount: 3,
CancellationToken.None,
judge: CreateParticipant("judge", AskJudge));
var entries = await Task.WhenAll(tasks);
Assert.Multiple(() =>
{
Assert.That(judgeAskedCount, Is.EqualTo(3));
Assert.That(entries.All(entry => entry.Result.Judge is { Completed: true }), Is.True);
});
}
private static ModelComparisonBatchRunner CreateBatchRunner() => new(new ModelComparisonRunner(NullLogger.Instance));
private static ModelComparisonParticipant CreateParticipant(string id, Func<string, CancellationToken, IAsyncEnumerable<string>> ask) => new()
{
Label = $"Model {id}",
Ask = ask,
};
private static async IAsyncEnumerable<string> Chunks(params string[] chunks)
{
foreach (var chunk in chunks)
{
await Task.Yield();
yield return chunk;
}
}
private static async IAsyncEnumerable<string> Never([EnumeratorCancellation] CancellationToken token)
{
try
{
await Task.Delay(Timeout.Infinite, token);
}
catch (OperationCanceledException)
{
// Expected once the test cleans up; there is nothing left to answer with.
}
#pragma warning disable CS0162 // Unreachable code: the yield is what makes this an iterator
yield break;
#pragma warning restore CS0162
}
private static async IAsyncEnumerable<string> Failing()
{
await Task.Yield();
throw new InvalidOperationException("The model is not reachable.");
#pragma warning disable CS0162 // Unreachable code: the yield is what makes this an iterator
yield break;
#pragma warning restore CS0162
}
}

View File

@ -0,0 +1,118 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how the user's own votes across a batch are tallied.
/// </summary>
/// <remarks>
/// Mirrors <see cref="ModelComparisonBatchJudgeSummaryTests"/>: the tally has to survive the
/// presentation order changing from run to run, and a run without a vote must not be counted.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonBatchVoteSummaryTests
{
[Test]
public void AnEmptyBatchScoresNothing()
{
var summary = ModelComparisonBatchVoteSummary.From([], []);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.Zero);
Assert.That(summary.ScoredRunCount, Is.Zero);
Assert.That(summary.FirstModelWins, Is.Zero);
Assert.That(summary.SecondModelWins, Is.Zero);
Assert.That(summary.Ties, Is.Zero);
});
}
[Test]
public void WinsAreCountedByModelNotByColumn()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
};
// The first model stands in column A twice and in column B once, and the user votes for it every time:
var votes = new ModelComparisonVote?[]
{
ModelComparisonVote.COLUMN_A,
ModelComparisonVote.COLUMN_B,
ModelComparisonVote.COLUMN_A,
};
var summary = ModelComparisonBatchVoteSummary.From(entries, votes);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(3));
Assert.That(summary.ScoredRunCount, Is.EqualTo(3));
Assert.That(summary.FirstModelWins, Is.EqualTo(3));
Assert.That(summary.SecondModelWins, Is.Zero);
});
}
[Test]
public void ARunWithoutAVoteIsNotScored()
{
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
};
var votes = new ModelComparisonVote?[] { ModelComparisonVote.COLUMN_A, null };
var summary = ModelComparisonBatchVoteSummary.From(entries, votes);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(2), "Both runs happened.");
Assert.That(summary.ScoredRunCount, Is.EqualTo(1), "Only one run was voted on.");
Assert.That(summary.FirstModelWins, Is.EqualTo(1));
});
}
[Test]
public void AVoteWithNoMatchingEntryIndexIsNotScored()
{
// Fewer votes than entries -- the batch was left before reaching the rest:
var entries = new[]
{
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
Entry(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST),
};
var votes = new ModelComparisonVote?[] { ModelComparisonVote.TIE };
var summary = ModelComparisonBatchVoteSummary.From(entries, votes);
Assert.Multiple(() =>
{
Assert.That(summary.TotalRunCount, Is.EqualTo(2));
Assert.That(summary.ScoredRunCount, Is.EqualTo(1));
Assert.That(summary.Ties, Is.EqualTo(1));
});
}
private static ModelComparisonBatchEntry Entry(ModelComparisonPresentationOrder order) => new()
{
PresentationOrder = order,
Result = new ModelComparisonRunResult
{
First = Answer("First"),
Second = Answer("Second"),
},
};
private static ModelComparisonAnswer Answer(string label) => new()
{
Label = label,
Text = $"The answer of {label}.",
Completed = true,
};
}

View File

@ -0,0 +1,101 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks the prompt the judge is asked with.
/// </summary>
[TestFixture]
public sealed class ModelComparisonJudgeRequestTests
{
private static readonly ModelComparisonRequest REQUEST = new("The document.", "Summarize it.");
[Test]
public void ThePromptCarriesTheUserInstructionsAndBothAnswers()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, "A good summary keeps every number.", "Answer from the first model.", "Answer from the second model.");
var prompt = judgeRequest.ToPrompt();
Assert.Multiple(() =>
{
Assert.That(prompt, Does.Contain("A good summary keeps every number."));
Assert.That(prompt, Does.Contain("Answer from the first model."));
Assert.That(prompt, Does.Contain("Answer from the second model."));
Assert.That(prompt, Does.Contain(REQUEST.ToPrompt()));
});
}
[Test]
public void WhiteSpaceAroundTheInstructionsIsDropped()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, " Keep it short. \n", "First.", "Second.");
Assert.That(judgeRequest.Instructions, Is.EqualTo("Keep it short."));
}
[Test]
public void WithoutInstructionsThePromptStillCarriesAGenericCriterion()
{
// The judge already sees the request and both answers, so the field is a refinement, not a requirement:
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "Answer from the first model.", "Answer from the second model.");
var prompt = judgeRequest.ToPrompt();
Assert.Multiple(() =>
{
Assert.That(prompt, Does.Contain("better answers the question or better completes the task"));
Assert.That(prompt, Does.Contain("Answer from the first model."));
Assert.That(prompt, Does.Contain("Answer from the second model."));
Assert.That(prompt, Does.Not.Contain("Pay particular attention to this:"));
});
}
[Test]
public void WithoutALanguageNameThePromptAsksForNoParticularOne()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.");
Assert.That(judgeRequest.ToPrompt(), Does.Not.Contain("Write the reasoning in"));
}
[Test]
public void ALanguageNameTellsTheJudgeWhatToWriteIn()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.", "German");
Assert.That(judgeRequest.ToPrompt(), Does.Contain("Write the reasoning in German"));
}
[Test]
public void TheAnswersAreLabeledAAndBNotByModelOrder()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "Answer from the first model.", "Answer from the second model.");
var prompt = judgeRequest.ToPrompt();
Assert.Multiple(() =>
{
Assert.That(prompt, Does.Contain("# Answer A"));
Assert.That(prompt, Does.Contain("# Answer B"));
Assert.That(prompt, Does.Not.Contain("# Answer 1"));
Assert.That(prompt, Does.Not.Contain("# Answer 2"));
});
}
[Test]
public void WithoutACacheBusterThePromptIsUnchanged()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.");
Assert.That(judgeRequest.ToPrompt(), Is.EqualTo(judgeRequest.ToPrompt(string.Empty)));
}
[Test]
public void TwoDifferentCacheBustersBuildDifferentPrompts()
{
var judgeRequest = new ModelComparisonJudgeRequest(REQUEST, string.Empty, "First.", "Second.");
Assert.That(judgeRequest.ToPrompt("one"), Is.Not.EqualTo(judgeRequest.ToPrompt("two")));
}
}

View File

@ -0,0 +1,79 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how the judge's raw answer is turned into a verdict.
/// </summary>
/// <remarks>
/// The judge is asked for one JSON object and nothing else, but a model rarely holds to that
/// perfectly. Parsing has to look past a stray sentence around the object, and has to end up with a
/// verdict which is not <see cref="ModelComparisonJudgeVerdict.Completed"/> rather than throw when
/// there is nothing usable to read.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonJudgeVerdictTests
{
[Test]
public void APlainJsonObjectIsRead()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, """{"preferred": "A", "reasoning": "It kept every number from the document."}""");
Assert.Multiple(() =>
{
Assert.That(verdict.Completed, Is.True);
Assert.That(verdict.Label, Is.EqualTo("Judge"));
Assert.That(verdict.Preferred, Is.EqualTo(ModelComparisonVote.COLUMN_A));
Assert.That(verdict.Reasoning, Is.EqualTo("It kept every number from the document."));
});
}
[Test]
public void APreferenceIsReadWithoutRegardToCase()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, """{"preferred": "tie", "reasoning": "Both cover the same points."}""");
Assert.That(verdict.Preferred, Is.EqualTo(ModelComparisonVote.TIE));
}
[Test]
public void TextAroundTheJsonObjectIsIgnored()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, """
Sure, here is my verdict:
```json
{"preferred": "B", "reasoning": "It is shorter and just as complete."}
```
""");
Assert.Multiple(() =>
{
Assert.That(verdict.Completed, Is.True);
Assert.That(verdict.Preferred, Is.EqualTo(ModelComparisonVote.COLUMN_B));
});
}
[Test]
public void AJudgeWhichDidNotAnswerIsNotCompleted()
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", false, string.Empty);
Assert.Multiple(() =>
{
Assert.That(verdict.Completed, Is.False);
Assert.That(verdict.Preferred, Is.Null);
Assert.That(verdict.Reasoning, Is.Empty);
});
}
[TestCase("Sorry, I cannot help with that.")]
[TestCase("""{"reasoning": "Missing the preference field."}""")]
[TestCase("""{"preferred": "C", "reasoning": "Not one of the three allowed values."}""")]
[TestCase("""{"preferred": "A", "reasoning": }""")]
public void AnUnreadableAnswerIsNotCompleted(string judgeText)
{
var verdict = ModelComparisonJudgeVerdict.Parse("Judge", true, judgeText);
Assert.That(verdict.Completed, Is.False);
}
}

View File

@ -0,0 +1,104 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks which answer stands in which column.
/// </summary>
/// <remarks>
/// A wrong assignment does not crash. It shows the answer of one model under the name of the other
/// once the vote is in, and nothing on the screen looks off.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonPresentationOrderTests
{
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, "The first model", "The second model")]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, "The second model", "The first model")]
public void TheColumnsShowTheAnswersOfTheModelsTheOrderSays(ModelComparisonPresentationOrder order, string expectedInColumnA, string expectedInColumnB)
{
var run = CreateRunResult();
Assert.Multiple(() =>
{
Assert.That(order.InColumnA(run).Label, Is.EqualTo(expectedInColumnA));
Assert.That(order.InColumnB(run).Label, Is.EqualTo(expectedInColumnB));
});
}
[Test]
public void TheTextOfAColumnIsTheTextOfTheModelWhoseNameItCarries()
{
// The name and the text travel together in one answer, so a column can never mix them up:
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
{
Assert.That(order.InColumnA(run).Text, Does.Contain(order.InColumnA(run).Label), $"{order}, column A");
Assert.That(order.InColumnB(run).Text, Does.Contain(order.InColumnB(run).Label), $"{order}, column B");
}
}
[Test]
public void AnswersInTheTwoColumnsAreNeverTheSame()
{
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
Assert.That(order.InColumnA(run).Label, Is.Not.EqualTo(order.InColumnB(run).Label), order.ToString());
}
[Test]
public void EveryOrderIsKnown()
{
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
{
Assert.DoesNotThrow(() => order.InColumnA(run), order.ToString());
Assert.DoesNotThrow(() => order.InColumnB(run), order.ToString());
}
}
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_A, ModelComparisonModelChoice.FIRST)]
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.COLUMN_B, ModelComparisonModelChoice.SECOND)]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.COLUMN_A, ModelComparisonModelChoice.SECOND)]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.COLUMN_B, ModelComparisonModelChoice.FIRST)]
[TestCase(ModelComparisonPresentationOrder.FIRST_MODEL_FIRST, ModelComparisonVote.TIE, ModelComparisonModelChoice.TIE)]
[TestCase(ModelComparisonPresentationOrder.SECOND_MODEL_FIRST, ModelComparisonVote.TIE, ModelComparisonModelChoice.TIE)]
public void AColumnBasedVoteIsTranslatedBackToItsModel(ModelComparisonPresentationOrder order, ModelComparisonVote vote, ModelComparisonModelChoice expected)
{
Assert.That(order.ToModelChoice(vote), Is.EqualTo(expected));
}
[Test]
public void TranslatingAVoteAndBackAgainIsConsistentWithTheColumns()
{
// Whichever model a column names, translating that model's vote back must land on the model actually standing there:
var run = CreateRunResult();
foreach (var order in Enum.GetValues<ModelComparisonPresentationOrder>())
{
var modelInColumnA = order.InColumnA(run).Label;
var choiceForColumnA = order.ToModelChoice(ModelComparisonVote.COLUMN_A);
var expectedLabel = choiceForColumnA is ModelComparisonModelChoice.FIRST ? "The first model" : "The second model";
Assert.That(modelInColumnA, Is.EqualTo(expectedLabel), order.ToString());
}
}
private static ModelComparisonRunResult CreateRunResult() => new()
{
First = new ModelComparisonAnswer
{
Label = "The first model",
Text = "The answer of The first model.",
Completed = true,
},
Second = new ModelComparisonAnswer
{
Label = "The second model",
Text = "The answer of The second model.",
Completed = true,
},
};
}

View File

@ -0,0 +1,144 @@
using AIStudio.Assistants.ModelComparison;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks the text both models receive.
/// </summary>
/// <remarks>
/// A difference in this text would be measured as a difference between the models, so what matters
/// is that the shape stays the same, and that a request without a context is not dressed up.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonRequestTests
{
[Test]
public void AQuestionAloneIsSentAsItIs()
{
var request = new ModelComparisonRequest(string.Empty, "What is the capital of France?");
Assert.That(request.ToPrompt(), Is.EqualTo("What is the capital of France?"));
}
[Test]
public void AContextAndAQuestionAreSeparatedByHeadings()
{
var request = new ModelComparisonRequest("The document.", "Summarize it.");
Assert.That(request.ToPrompt(), Is.EqualTo("# Context\nThe document.\n\n# Question or task\nSummarize it.").Or.EqualTo("# Context\r\nThe document.\r\n\r\n# Question or task\r\nSummarize it."));
}
[TestCase("")]
[TestCase(" ")]
[TestCase("\r\n\t\n")]
public void AContextOfWhiteSpaceCountsAsNone(string context)
{
var request = new ModelComparisonRequest(context, "Summarize it.");
Assert.Multiple(() =>
{
Assert.That(request.Context, Is.Empty);
Assert.That(request.ToPrompt(), Is.EqualTo("Summarize it."));
});
}
[Test]
public void WhiteSpaceAroundTheTextsIsDropped()
{
var request = new ModelComparisonRequest("\n\nThe document.\n", " Summarize it. ");
Assert.Multiple(() =>
{
Assert.That(request.Context, Is.EqualTo("The document."));
Assert.That(request.Question, Is.EqualTo("Summarize it."));
});
}
[Test]
public void AShortQuestionIsItsOwnPreview()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.GetQuestionPreview(300), Is.EqualTo("Summarize it."));
}
[Test]
public void ALongQuestionIsCutAndMarked()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize this document in a few short sentences.");
Assert.That(request.GetQuestionPreview(18), Is.EqualTo("Summarize this doc…"));
}
[Test]
public void ACutDoesNotLeaveHalfOfACharacter()
{
// The emoji is two chars long. A cut after the first of them would leave a lone surrogate:
var request = new ModelComparisonRequest(string.Empty, "aaaa😀bbb");
Assert.That(request.GetQuestionPreview(5), Is.EqualTo("aaaa…"));
}
[Test]
public void ThePreviewOfAnEmptyLengthIsRefused()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.Throws<ArgumentOutOfRangeException>(() => request.GetQuestionPreview(0));
}
[Test]
public void TheSameInputAlwaysBuildsTheSamePrompt()
{
var first = new ModelComparisonRequest("The document.", "Summarize it.");
var second = new ModelComparisonRequest("The document.", "Summarize it.");
Assert.That(first.ToPrompt(), Is.EqualTo(second.ToPrompt()));
}
[Test]
public void WithoutALanguageNameThePromptNamesNone()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.ToPrompt(), Does.Not.Contain("Answer in"));
}
[Test]
public void ALanguageNameIsAppendedAsAnInstruction()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.", "German");
Assert.That(request.ToPrompt(), Is.EqualTo("Summarize it.\n\nAnswer in German."));
}
[Test]
public void WithoutACacheBusterThePromptIsUnchanged()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.ToPrompt(), Is.EqualTo(request.ToPrompt(string.Empty)));
}
[Test]
public void ACacheBusterIsAppendedAsAPlainlyLabeledLine()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
var prompt = request.ToPrompt("a1b2c3");
Assert.Multiple(() =>
{
Assert.That(prompt, Does.StartWith("Summarize it."));
Assert.That(prompt, Does.Contain("a1b2c3"));
});
}
[Test]
public void TwoDifferentCacheBustersBuildDifferentPrompts()
{
var request = new ModelComparisonRequest(string.Empty, "Summarize it.");
Assert.That(request.ToPrompt("one"), Is.Not.EqualTo(request.ToPrompt("two")));
}
}

View File

@ -0,0 +1,394 @@
using System.Collections.Concurrent;
using System.Runtime.CompilerServices;
using AIStudio.Assistants.ModelComparison;
using Microsoft.Extensions.Logging.Abstractions;
namespace AIStudio.Tests.Assistants.ModelComparison;
/// <summary>
/// Checks how a comparison run treats two models: that they are asked the same, at the same time,
/// and that whatever one of them does cannot spoil the result of the other.
/// </summary>
/// <remarks>
/// The models are made up here. What is under test is the run, not a provider.
/// </remarks>
[TestFixture]
public sealed class ModelComparisonRunnerTests
{
private static readonly ModelComparisonRequest REQUEST = new("The document.", "Summarize it.");
private static readonly TimeSpan PATIENCE = TimeSpan.FromSeconds(5);
[Test]
public async Task BothModelsReceiveTheSamePrompt()
{
// Not REQUEST.ToPrompt() itself: the run adds a cache-busting request ID neither call here supplies:
var prompts = new ConcurrentBag<string>();
IAsyncEnumerable<string> Ask(string prompt, CancellationToken _)
{
prompts.Add(prompt);
return Chunks("An answer.");
}
await CreateRunner().RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(prompts, Has.Count.EqualTo(2));
Assert.That(prompts.Distinct().Count(), Is.EqualTo(1), "Both models must receive the exact same prompt as each other.");
});
}
[Test]
public async Task RepeatedRunsGetDifferentPromptsSoACachingGatewayCannotAnswerFromACache()
{
var prompts = new ConcurrentBag<string>();
IAsyncEnumerable<string> Ask(string prompt, CancellationToken _)
{
prompts.Add(prompt);
return Chunks("An answer.");
}
var runner = CreateRunner();
await runner.RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
await runner.RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
// Both models of the same run still share one prompt, the same as in BothModelsReceiveTheSamePrompt;
// it is the two runs that must differ from each other, so a caching gateway cannot answer a
// repeat run from what it already has cached for the first one:
Assert.That(prompts.Distinct().Count(), Is.EqualTo(2), "Each run must see a prompt of its own, distinct from the other run's.");
}
[Test]
public async Task TheModelsAreAskedAtTheSameTime()
{
var started = 0;
var bothStarted = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously);
//
// Neither model answers before the other one has been asked. Asked one after the other, the
// first would wait in vain, run out of patience and fail, and so would the assertion below.
//
async IAsyncEnumerable<string> Ask(string _, [EnumeratorCancellation] CancellationToken token)
{
if (Interlocked.Increment(ref started) == 2)
bothStarted.SetResult();
await bothStarted.Task.WaitAsync(PATIENCE, token);
yield return "An answer.";
}
var result = await CreateRunner().RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, CancellationToken.None);
Assert.That(result.BothCompleted, Is.True);
}
[Test]
public async Task EachAnswerStaysWithTheModelWhichGaveIt()
{
var secondIsDone = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously);
// The first model finishes last, so a result assembled in the order of arrival would swap them:
async IAsyncEnumerable<string> AskFirst(string _, [EnumeratorCancellation] CancellationToken token)
{
await secondIsDone.Task.WaitAsync(PATIENCE, token);
yield return "from the first";
}
async IAsyncEnumerable<string> AskSecond(string _, [EnumeratorCancellation] CancellationToken __)
{
yield return "from the second";
secondIsDone.TrySetResult();
await Task.CompletedTask;
}
var result = await CreateRunner().RunAsync(CreateParticipant("1", AskFirst), CreateParticipant("2", AskSecond), REQUEST, CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Text, Is.EqualTo("from the first"));
Assert.That(result.First.Label, Is.EqualTo("Model 1"));
Assert.That(result.Second.Text, Is.EqualTo("from the second"));
Assert.That(result.Second.Label, Is.EqualTo("Model 2"));
});
}
[Test]
public async Task AnAnswerIsTheChunksJoinedWithoutTheThinking()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("<think>Let me think.", "</think>", "Hello", " world ")),
CreateParticipant("2", (_, _) => Chunks("Hi")),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Text, Is.EqualTo("Hello world"));
Assert.That(result.First.Completed, Is.True);
});
}
[Test]
public async Task TheTimesAreMeasuredOnTheClockOfTheRun()
{
var clock = new ManualTimeProvider();
async IAsyncEnumerable<string> AskFirst(string _, [EnumeratorCancellation] CancellationToken __)
{
clock.Advance(TimeSpan.FromMilliseconds(200));
yield return "Hi";
clock.Advance(TimeSpan.FromMilliseconds(300));
yield return " there";
await Task.CompletedTask;
}
// The second model does not touch the clock, so the times of the first one are the only ones which move:
var result = await CreateRunner(clock).RunAsync(
CreateParticipant("1", AskFirst),
CreateParticipant("2", (_, _) => Failing()),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.FirstTokenTime, Is.EqualTo(TimeSpan.FromMilliseconds(200)));
Assert.That(result.First.TotalTime, Is.EqualTo(TimeSpan.FromMilliseconds(500)));
});
}
[Test]
public async Task AModelWhichFailsDoesNotTakeTheOtherOneDown()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Failing()),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.True);
Assert.That(result.First.Text, Is.EqualTo("An answer."));
Assert.That(result.Second.Completed, Is.False);
Assert.That(result.Second.TotalTime, Is.Null, "A failed request has no total time.");
Assert.That(result.BothCompleted, Is.False);
});
}
[Test]
public async Task AnAnswerWhichSaysNothingCountsAsAFailure()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks()),
CreateParticipant("2", (_, _) => Chunks(" ", "\n")),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.False);
Assert.That(result.Second.Completed, Is.False);
Assert.That(result.BothCompleted, Is.False);
});
}
[Test]
public async Task AnAnswerWhichEndsInsideTheThinkingCountsAsAFailure()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("<think>Still thinking, and out of room")),
CreateParticipant("2", (_, _) => Chunks("An answer.")),
REQUEST,
CancellationToken.None);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.False);
Assert.That(result.First.Text, Is.Empty);
Assert.That(result.Second.Completed, Is.True);
});
}
[Test]
public async Task ACanceledRunEndsBothModelsWithoutACompletedAnswer()
{
using var cancellation = new CancellationTokenSource();
async IAsyncEnumerable<string> Ask(string _, [EnumeratorCancellation] CancellationToken token)
{
yield return "A first piece";
await Task.Delay(Timeout.Infinite, token);
}
cancellation.CancelAfter(TimeSpan.FromMilliseconds(50));
var result = await CreateRunner().RunAsync(CreateParticipant("1", Ask), CreateParticipant("2", Ask), REQUEST, cancellation.Token);
Assert.Multiple(() =>
{
Assert.That(result.First.Completed, Is.False);
Assert.That(result.Second.Completed, Is.False);
Assert.That(result.First.TotalTime, Is.Null);
Assert.That(result.BothCompleted, Is.False);
});
}
[Test]
public async Task TheJudgeIsAskedAboutBothCompleteAnswers()
{
string? judgePrompt = null;
IAsyncEnumerable<string> AskJudge(string prompt, CancellationToken _)
{
judgePrompt = prompt;
return Chunks("""{"preferred": "A", "reasoning": "It is more complete."}""");
}
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("Answer from the first model.")),
CreateParticipant("2", (_, _) => Chunks("Answer from the second model.")),
REQUEST,
CancellationToken.None,
judge: CreateParticipant("judge", AskJudge),
judgeInstructions: "Prefer the more complete answer.");
Assert.Multiple(() =>
{
Assert.That(judgePrompt, Does.Contain("Answer from the first model."));
Assert.That(judgePrompt, Does.Contain("Answer from the second model."));
Assert.That(judgePrompt, Does.Contain("Prefer the more complete answer."));
Assert.That(result.Judge, Is.Not.Null);
Assert.That(result.Judge!.Completed, Is.True);
Assert.That(result.Judge.Preferred, Is.EqualTo(ModelComparisonVote.COLUMN_A));
Assert.That(result.Judge.Label, Is.EqualTo("Model judge"));
});
}
[Test]
public async Task TheJudgeIsToldTheAnswersInPresentationOrder()
{
string? judgePrompt = null;
IAsyncEnumerable<string> AskJudge(string prompt, CancellationToken _)
{
judgePrompt = prompt;
return Chunks("""{"preferred": "A", "reasoning": "n/a"}""");
}
// The second model is drawn first, so it is what the judge is told is "Answer A":
await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("Answer from the first model.")),
CreateParticipant("2", (_, _) => Chunks("Answer from the second model.")),
REQUEST,
CancellationToken.None,
ModelComparisonPresentationOrder.SECOND_MODEL_FIRST,
CreateParticipant("judge", AskJudge));
Assert.Multiple(() =>
{
Assert.That(judgePrompt, Does.Match(@"# Answer A\r?\nAnswer from the second model\."));
Assert.That(judgePrompt, Does.Match(@"# Answer B\r?\nAnswer from the first model\."));
});
}
[Test]
public async Task TheJudgeIsNotAskedWhenAModelDidNotAnswer()
{
var judgeWasAsked = false;
IAsyncEnumerable<string> AskJudge(string _, CancellationToken __)
{
judgeWasAsked = true;
return Chunks("""{"preferred": "TIE", "reasoning": "n/a"}""");
}
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Failing()),
REQUEST,
CancellationToken.None,
judge: CreateParticipant("judge", AskJudge));
Assert.Multiple(() =>
{
Assert.That(judgeWasAsked, Is.False, "A judge would only ever see one complete answer, which is nothing to judge.");
Assert.That(result.Judge, Is.Null);
});
}
[Test]
public async Task WithoutAJudgeTheResultHasNoVerdict()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
CancellationToken.None);
Assert.That(result.Judge, Is.Null);
}
[Test]
public async Task AJudgeWhichFailsLeavesAVerdictWhichIsNotCompleted()
{
var result = await CreateRunner().RunAsync(
CreateParticipant("1", (_, _) => Chunks("An answer.")),
CreateParticipant("2", (_, _) => Chunks("Another answer.")),
REQUEST,
CancellationToken.None,
judge: CreateParticipant("judge", (_, _) => Failing()));
Assert.Multiple(() =>
{
Assert.That(result.Judge, Is.Not.Null);
Assert.That(result.Judge!.Completed, Is.False);
});
}
private static ModelComparisonRunner CreateRunner(TimeProvider? clock = null) => new(NullLogger.Instance, clock);
private static ModelComparisonParticipant CreateParticipant(string id, Func<string, CancellationToken, IAsyncEnumerable<string>> ask) => new()
{
Label = $"Model {id}",
Ask = ask,
};
private static async IAsyncEnumerable<string> Chunks(params string[] chunks)
{
foreach (var chunk in chunks)
{
await Task.Yield();
yield return chunk;
}
}
private static async IAsyncEnumerable<string> Failing()
{
await Task.Yield();
throw new InvalidOperationException("The model is not reachable.");
#pragma warning disable CS0162 // Unreachable code: the yield is what makes this an iterator
yield break;
#pragma warning restore CS0162
}
/// <summary>
/// A clock which only moves when a test says so.
/// </summary>
private sealed class ManualTimeProvider : TimeProvider
{
private long ticks;
// One timestamp is one tick of a TimeSpan, so the elapsed time of the runner is exactly what Advance added:
public override long TimestampFrequency => TimeSpan.TicksPerSecond;
public void Advance(TimeSpan by) => Interlocked.Add(ref this.ticks, by.Ticks);
public override long GetTimestamp() => Interlocked.Read(ref this.ticks);
}
}