@attribute [Route(Routes.ASSISTANT_MODEL_COMPARISON)] @inherits AssistantBaseCore @using AIStudio.Chat @* Two views: the form to fill in, and the batch in short while it runs and once it has. *@ @if (this.batchEntries.Count == 0 && !this.isComparing) { @* Only the icon button toggles, the same as everywhere else in the app a header collapses a section: the rest of the row is decorative, not a second, redundant click target. Save as new preset stays outside the collapse: capturing the form right now should not need opening this box first. *@ @T("Presets") @T("Save as new preset") @foreach (var preset in this.SettingsManager.ConfigurationData.ModelComparison.Presets) { @preset.Name } @if (this.SelectedPreset is not null) { @T("Update preset") } @if (this.SelectedPreset is not null) { } @T("Setup") @T("First model") @T("Second model") @T("1 (single comparison)") 3 5 10 15 @T("20 (heavy load on the models)") @T("Your request") @T("LLM judge (optional)") @if (this.judgeEnabled) { @T("Judge model") } } else { @* What was asked, in short: without it, the answers below would have to be judged from memory. *@ var request = this.CurrentRequest; @T("Your request") @request.GetQuestionPreview(QUESTION_PREVIEW_LENGTH) @if (request.Context.Length > 0) { @string.Format(DisplayCulture, T("Context: {0} characters"), request.Context.Length.CompactCount()) } @if (this.runCount > 1) { @T("Progress Runs") @for (var i = 0; i < this.runCount; i++) { var index = i; @* A native title attribute, not MudTooltip: while the batch is still running, this row re-renders often enough that MudTooltip's JS-side hover tracking kept losing track of which badge is which, and stopped showing anything at all. A native tooltip lives entirely in the browser, so re-rendering the content underneath it cannot break it. *@ @* The current entry gets a ring around its avatar: with several runs done and unvoted, position alone -- second from the left, say -- is too easy to lose track of. *@ @if (this.RunIsVoted(index)) { } else if (this.RunIsActive(index)) { @* Only the runs actually in flight get the spinner -- the ones still queued behind the concurrency limit get an hourglass instead, further down: *@ } else if (!this.RunHasArrived(index)) { } else if (this.RunNotCompleted(index) || this.RunVoteWasSkipped(index)) { } else { } @(index + 1) } } } @code { /// /// The card of one answer, with its title above. /// /// /// The same markup for both, so that nothing but the position tells them apart. The two cards /// are in one row of the grid and fill it, so they are equally high however long the answers /// are. That is why the names of the models are not part of the card: how long a name is /// would make the cards differ. /// /// "Answer A" or "Answer B". /// The answer. /// The column this is, as the vote names it. private RenderFragment AnswerCard(string title, ContentText content, ModelComparisonVote column) => @ @title ; /// /// What is said about the model of an answer once the vote is in: its name, and how long it took. /// /// /// Below the answer, not above it: an answer can be long, and what stands above it is out of /// sight by the time it has been read. /// /// The answer in this column, which knows who wrote it. /// The column this is, as the vote names it. private RenderFragment AnswerCaption(ModelComparisonAnswer answer, ModelComparisonVote column) => @ @answer.Label @if (this.JudgePreferredThisColumn(column)) { @T("Judge") } @this.DescribeTimes(answer) ; /// /// What the judge said about the two answers, shown once the vote is in -- the same point at /// which the names of the two compared models are revealed. Never shown before, so it cannot /// anchor the user's own vote. /// /// What the judge said, or that it could not be understood. private RenderFragment JudgeVerdictCard(ModelComparisonJudgeVerdict judge) => @ @string.Format(DisplayCulture, T("LLM judge: {0}"), judge.Label) @if (judge.Completed) { @this.DescribeJudgePreference(judge) @judge.Reasoning } else { @T("The judge did not give a usable verdict.") } ; /// /// Everything which follows the Compare button: the wait, the answers, the vote, and the way to /// the next run of the batch. /// /// /// Rendered by the base class below its submit row. This is where the button has to stay above /// the vote: content of the body would come before it. /// private protected override RenderFragment? BelowSubmitContent => @
@if (this.skippingRemainingVotes && !this.BatchFinished) { @(this.judgeEnabled ? T("Skipping the remaining votes. Waiting for the judges to finish...") : T("Skipping the remaining votes. Waiting for the rest of the batch to finish...")) } else if (this.isComparing && this.CurrentEntry is null) { @* The same waiting as in the chat: a bar, and lines where the answers will appear. Both columns look alike, so this says nothing about the models. *@ @(this.batchEntries.Count == 0 ? T("Both models are answering. This can take a while.") : T("Waiting for the next run...")) @T("Answer A") @T("Answer B") } else if (this.CurrentEntry is { } entry) { @if (this.answerAContent is not null && this.answerBContent is not null) { var columnA = entry.PresentationOrder.InColumnA(entry.Result); var columnB = entry.PresentationOrder.InColumnB(entry.Result); @* Until the vote is in, nothing on this screen may say which model wrote which answer: no name, no time. A small model is usually faster, so a time would give the model away. *@ @T("Answers") @if (this.CurrentVote is null) { @T("You do not know yet which model wrote which answer, and the columns are in random order. Compare them by what they say.") } @* Two columns in every width, and the two cards in the same row: that is what keeps them equally high. *@ @this.AnswerCard(T("Answer A"), this.answerAContent!, ModelComparisonVote.COLUMN_A) @this.AnswerCard(T("Answer B"), this.answerBContent!, ModelComparisonVote.COLUMN_B) @if (this.CurrentVote is not null) { @this.AnswerCaption(columnA, ModelComparisonVote.COLUMN_A) @this.AnswerCaption(columnB, ModelComparisonVote.COLUMN_B) } @if (this.CurrentVote is null) { @T("Which answer is better?") @T("Answer A is better") @T("Answer B is better") @T("Both are equally good") @if (this.runCount > 1) { @T("Skip this run") @T("Skip remaining votes") } } else { @(this.IsLastEntry ? T("Finish") : T("Next run")) @if (this.isComparing) { @T("More runs are still being generated in the background.") } } @if (this.CurrentVote is not null && entry.Result.Judge is { } judge) { @this.JudgeVerdictCard(judge) } } else { @T("At least one of the models did not answer in this run, so there is nothing to compare.") @(this.IsLastEntry ? T("Finish") : T("Next run")) } } else if (this.BatchFinished) { @(this.runCount > 1 ? string.Format(DisplayCulture, T("All {0} runs are done."), this.runCount) : T("Done.")) @if (this.batchEntries.Count > 0) { var firstLabel = this.batchEntries[0].Result.First.Label; var secondLabel = this.batchEntries[0].Result.Second.Label; var voteSummary = ModelComparisonBatchVoteSummary.From(this.batchEntries, this.batchVotes); var judgeSummary = this.judgeEnabled ? ModelComparisonBatchJudgeSummary.From(this.batchEntries) : null; @firstLabel @secondLabel @T("Equally good") @T("Your votes") @voteSummary.FirstModelWins @voteSummary.SecondModelWins @voteSummary.Ties @if (judgeSummary is not null) { @T("Judge verdicts") @judgeSummary.FirstModelWins @judgeSummary.SecondModelWins @judgeSummary.Ties } var votesMissing = voteSummary.TotalRunCount - voteSummary.ScoredRunCount; var judgeVerdictsMissing = judgeSummary is { } summary ? summary.TotalRunCount - summary.ScoredRunCount : 0; @if (votesMissing > 0) { @string.Format(DisplayCulture, T("{0} of {1} runs have no usable vote -- never cast, or the run itself never got a usable answer -- and are not counted above."), votesMissing, voteSummary.TotalRunCount) } @if (judgeSummary is not null && judgeVerdictsMissing > 0 && judgeVerdictsMissing != votesMissing) { @string.Format(DisplayCulture, T("{0} of {1} runs have no usable judge verdict."), judgeVerdictsMissing, judgeSummary.TotalRunCount) } } @T("Start a new request") }
; }