mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-04 18:09:40 +00:00
Rebuilt how AI Studio knows what a model can do (#960)
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
This commit is contained in:
1 parent
d21e09dd1e
commit
d85b4e71b6
287 files changed
+18341
-3677
No files matched your search
@@ -0,0 +1,129 @@
|
||||
using System.Text;
|
||||
|
||||
using AIStudio.Tests.Models.Corpus;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Holds the current capability rules to their word, model by model.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// These tests state nothing about what is right. They state what the code answers today, so that
|
||||
/// rebuilding the capability system cannot change an answer by accident: every difference shows up
|
||||
/// here and has to be either a porting mistake or a decision somebody wrote down.
|
||||
///
|
||||
/// When a diff appears, read it before touching anything. If every line of it is wanted, run the
|
||||
/// snapshot writer and commit the new file together with the change that caused it.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class CapabilityCharacterizationTests
|
||||
{
|
||||
/// <summary>
|
||||
/// How many differing lines the failure message shows before it stops.
|
||||
/// </summary>
|
||||
private const int LINES_SHOWN = 25;
|
||||
|
||||
/// <summary>
|
||||
/// How many columns a snapshot line carries after the model ID.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The capabilities, the kind, the context window, the image limit, and the tokenizer. Adding a
|
||||
/// column to the snapshot means raising this, and forgetting to would split a line inside its
|
||||
/// last column instead of in front of it -- which makes every model look changed at once.
|
||||
/// </remarks>
|
||||
private const int TRAILING_COLUMNS = 5;
|
||||
|
||||
[Test]
|
||||
public void TheCorpusStillGetsTheAnswersTheSnapshotRecorded()
|
||||
{
|
||||
var recorded = CapabilitySnapshot.Read();
|
||||
var current = CapabilitySnapshot.Render(ModelCorpus.ENTRIES);
|
||||
|
||||
if (recorded is null)
|
||||
{
|
||||
File.WriteAllText(CapabilitySnapshot.FILE_PATH, current);
|
||||
Assert.Fail($"There was no snapshot yet, so one was written to {CapabilitySnapshot.FILE_PATH}. Read it line by line and commit it, then this test turns green.");
|
||||
return;
|
||||
}
|
||||
|
||||
if (recorded == current)
|
||||
{
|
||||
//
|
||||
// A leftover file from an earlier failure would otherwise sit in the working tree and
|
||||
// get committed by somebody who did not notice it:
|
||||
//
|
||||
File.Delete(CapabilitySnapshot.ACTUAL_FILE_PATH);
|
||||
return;
|
||||
}
|
||||
|
||||
File.WriteAllText(CapabilitySnapshot.ACTUAL_FILE_PATH, current);
|
||||
Assert.Fail($"The capabilities of {DescribeDifference(recorded, current)}{Environment.NewLine}{Environment.NewLine}The full result was written to {CapabilitySnapshot.ACTUAL_FILE_PATH}.");
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Describes how two snapshots differ, in the words of the lines that differ.
|
||||
/// </summary>
|
||||
/// <param name="recorded">The snapshot as it was recorded.</param>
|
||||
/// <param name="current">The snapshot as the code answers now.</param>
|
||||
/// <returns>A description naming the changed, added, and removed lines.</returns>
|
||||
private static string DescribeDifference(string recorded, string current)
|
||||
{
|
||||
var recordedLines = ModelLinesOf(recorded);
|
||||
var currentLines = ModelLinesOf(current);
|
||||
|
||||
var changed = recordedLines.Keys.Intersect(currentLines.Keys).Where(model => recordedLines[model] != currentLines[model]).ToList();
|
||||
var added = currentLines.Keys.Except(recordedLines.Keys).ToList();
|
||||
var removed = recordedLines.Keys.Except(currentLines.Keys).ToList();
|
||||
|
||||
var message = new StringBuilder($"{changed.Count} model(s) changed, {added.Count} came into the corpus, {removed.Count} left it:").Append(Environment.NewLine);
|
||||
foreach (var model in changed.Take(LINES_SHOWN))
|
||||
{
|
||||
message.Append(Environment.NewLine).Append(" ").Append(model);
|
||||
message.Append(Environment.NewLine).Append(" was: ").Append(recordedLines[model]);
|
||||
message.Append(Environment.NewLine).Append(" now: ").Append(currentLines[model]);
|
||||
}
|
||||
|
||||
foreach (var model in added.Take(LINES_SHOWN))
|
||||
message.Append(Environment.NewLine).Append(" + ").Append(model).Append(": ").Append(currentLines[model]);
|
||||
|
||||
foreach (var model in removed.Take(LINES_SHOWN))
|
||||
message.Append(Environment.NewLine).Append(" - ").Append(model).Append(": ").Append(recordedLines[model]);
|
||||
|
||||
return message.ToString();
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Splits a snapshot into what each line says about which model.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The provider and the model ID make up everything before the trailing columns, and those are
|
||||
/// the one place a split is safe: a model ID may contain anything, while the capability list,
|
||||
/// the kind and the context window may not.
|
||||
/// </remarks>
|
||||
/// <param name="snapshot">The snapshot text.</param>
|
||||
/// <returns>What every line says about a model, keyed by provider and model.</returns>
|
||||
private static Dictionary<string, string> ModelLinesOf(string snapshot)
|
||||
{
|
||||
var lines = new Dictionary<string, string>(StringComparer.Ordinal);
|
||||
foreach (var line in snapshot.Split('\n'))
|
||||
{
|
||||
if (line.Length is 0 || line.StartsWith('#'))
|
||||
continue;
|
||||
|
||||
var separatorIndex = line.Length;
|
||||
for (var column = 0; column < TRAILING_COLUMNS; column++)
|
||||
{
|
||||
separatorIndex = line.LastIndexOf(" | ", separatorIndex - 1, StringComparison.Ordinal);
|
||||
if (separatorIndex is -1)
|
||||
break;
|
||||
}
|
||||
|
||||
if (separatorIndex is -1)
|
||||
continue;
|
||||
|
||||
lines[line[..separatorIndex]] = line[(separatorIndex + 3)..];
|
||||
}
|
||||
|
||||
return lines;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the two things about the capability enum which the rest of the app relies on.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Capabilities became a set carried in one value, which only works while each member owns a bit of
|
||||
/// its own. And the names are what an override written years ago addresses, so a member which is no
|
||||
/// longer handed out still has to answer to its name.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class CapabilityTests
|
||||
{
|
||||
/// <summary>
|
||||
/// Every capability the app has ever written into a configuration.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Deliberately spelled out instead of read from the enum: a test which asks the enum about
|
||||
/// itself would agree with any change made to it, including a member being deleted. Removing
|
||||
/// one of these names silently drops the override an organization wrote for it.
|
||||
/// </remarks>
|
||||
private static readonly string[] NAMES_THAT_MUST_KEEP_WORKING =
|
||||
[
|
||||
"NONE", "UNKNOWN",
|
||||
"TEXT_INPUT", "AUDIO_INPUT", "SINGLE_IMAGE_INPUT", "MULTIPLE_IMAGE_INPUT", "SPEECH_INPUT", "VIDEO_INPUT",
|
||||
"TEXT_OUTPUT", "AUDIO_OUTPUT", "IMAGE_OUTPUT", "SPEECH_OUTPUT", "VIDEO_OUTPUT",
|
||||
"OPTIONAL_REASONING", "ALWAYS_REASONING", "REASONING_BY_DEFAULT",
|
||||
"EMBEDDING", "REALTIME", "FUNCTION_CALLING", "WEB_SEARCH",
|
||||
"CHAT_COMPLETION_API", "RESPONSES_API",
|
||||
];
|
||||
|
||||
[Test]
|
||||
public void EveryCapabilityOwnsOneBitOfItsOwn()
|
||||
{
|
||||
var bits = new Dictionary<ulong, Capability>();
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
foreach (var capability in Enum.GetValues<Capability>())
|
||||
{
|
||||
if (capability is Capability.NONE)
|
||||
continue;
|
||||
|
||||
var value = (ulong) capability;
|
||||
Assert.That(ulong.IsPow2(value), Is.True, $"{capability} is not a single bit, so it cannot be part of a set.");
|
||||
|
||||
if (bits.TryGetValue(value, out var other))
|
||||
Assert.Fail($"{capability} and {other} share a bit, so the app cannot tell them apart.");
|
||||
|
||||
bits[value] = capability;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void NoCapabilityLostItsName() => Assert.That(Enum.GetNames<Capability>(), Is.SupersetOf(NAMES_THAT_MUST_KEEP_WORKING));
|
||||
|
||||
[Test]
|
||||
public void TheReasoningVocabularyIsExactlyTheThreeReasoningMembers()
|
||||
{
|
||||
const Capability THE_THREE = Capability.OPTIONAL_REASONING | Capability.ALWAYS_REASONING | Capability.REASONING_BY_DEFAULT;
|
||||
|
||||
Assert.That(ModelProfile.REASONING_VOCABULARY, Is.EqualTo(THE_THREE));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,107 @@
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Settings;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the context windows the rules state, where a number was read from a vendor's page.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The snapshot already records every one of these numbers, so this fixture is not here to catch a
|
||||
/// changed answer. It is here for the handful of cases where the number is easy to get wrong by
|
||||
/// writing a rule the obvious way: a generation which inherits the window of the one before it
|
||||
/// although the vendor raised it, a variant which must not inherit a window at all, and the
|
||||
/// question of which of two numbers a vendor states is the one a conversation is measured against.
|
||||
///
|
||||
/// Every number below is one somebody can check against the source its family names. A number
|
||||
/// nobody could check does not belong in the rules in the first place.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ContextWindowRuleTests
|
||||
{
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5", 400_000, Description = "OpenAI states the whole window, input and output together.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5.1", 400_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5.2", 400_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5.4", 1_050_000, Description = "Where the window grows in this line.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5.5", 1_050_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5.6", 1_050_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-6-astra", 1_050_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "o1", 200_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "o3", 200_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "o4-mini", 200_000, Description = "The o3 generation under another number, window included.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-4o", 128_000)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-4", 8_192)]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-4-turbo", 128_000)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-3-5-sonnet-latest", 200_000)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-sonnet-4-0", 200_000)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-haiku-4-5-20251001", 200_000)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-opus-5", 1_000_000)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-sonnet-5", 1_000_000)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-fable-5-1", 1_000_000)]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-2.5-pro", 1_048_576, Description = "Google's input limit, which is what a conversation is measured against.")]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-2.5-flash-lite", 1_048_576)]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-3-pro", 1_000_000)]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-flash-latest", 1_000_000)]
|
||||
[TestCase(LLMProviders.X, "grok-4.20-0309-reasoning", 1_000_000)]
|
||||
[TestCase(LLMProviders.X, "grok-build-0.1", 256_000)]
|
||||
[TestCase(LLMProviders.MISTRAL, "mistral-large-2512", 256_000)]
|
||||
[TestCase(LLMProviders.MISTRAL, "pixtral-large-2411", 128_000)]
|
||||
public void TheWindowOfAModelIsTheOneItsVendorStates(LLMProviders provider, string modelId, int tokens)
|
||||
{
|
||||
var window = provider.GetModelProfile(new Model(modelId, null)).Context;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(window.IsKnown, Is.True);
|
||||
Assert.That(window.DefaultTokens, Is.EqualTo(tokens));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AGenerationNobodyDocumentsInheritsNoWindowFromTheOneBeforeIt()
|
||||
{
|
||||
//
|
||||
// OpenAI has no model page for a 5.3, so the rule for it exists only to keep such a model
|
||||
// answering like the rest of its line if one ever appears. Taking 5.1's window along would
|
||||
// turn "nobody has looked this up" into a number on somebody's screen.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.OPEN_AI, "gpt-5.3");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Context.IsKnown, Is.False);
|
||||
Assert.That(profile.Has(Capability.FUNCTION_CALLING), Is.True, "Everything else it does inherit.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheSameModelThroughAGatewayKeepsItsWindow()
|
||||
{
|
||||
//
|
||||
// A gateway cuts what the transport cannot carry, which is about APIs. How much the model
|
||||
// reads is a property of the model and survives the trip.
|
||||
//
|
||||
var directly = ModelRegistry.Shared.Profile(LLMProviders.OPEN_AI, "gpt-5.1");
|
||||
var throughAGateway = ModelRegistry.Shared.Profile(LLMProviders.OPEN_ROUTER, "openai/gpt-5.1");
|
||||
|
||||
Assert.That(throughAGateway.Context, Is.EqualTo(directly.Context));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModelNobodyStatedAWindowForSaysSoRatherThanGuessing()
|
||||
{
|
||||
//
|
||||
// The honest answer, and the common one: most models of the open-weights world are served
|
||||
// at whatever their operator configured, so the rules state nothing and the app shows a
|
||||
// person what their conversation uses without inventing a limit for it.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.SELF_HOSTED, "some-model-nobody-wrote-a-rule-for");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Context.IsKnown, Is.False);
|
||||
Assert.That(profile.Context.DefaultTokens, Is.Zero, "And the number next to it is meaningless, which is why nothing may read it without asking first.");
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,221 @@
|
||||
using System.Globalization;
|
||||
using System.Runtime.CompilerServices;
|
||||
using System.Text;
|
||||
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Settings;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// Renders the corpus and its capabilities as one text, and says where that text is kept.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The snapshot is compared as text rather than parsed back into entries. A model ID may be
|
||||
/// anything a provider chooses to answer with, empty strings and separator characters included, and
|
||||
/// a parser would have to be right about all of it to be worth anything. Comparing the rendered text
|
||||
/// cannot be wrong about a name, and a diff of it reads the same in the test output as in the IDE.
|
||||
/// </remarks>
|
||||
public static class CapabilitySnapshot
|
||||
{
|
||||
/// <summary>
|
||||
/// What a model with no capabilities at all is written as.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// An empty column would be an invisible statement. Providers do answer with nothing: an empty
|
||||
/// model ID and the "no provider" entry both do, and both are in the corpus.
|
||||
/// </remarks>
|
||||
private const string NOTHING = "(nothing)";
|
||||
|
||||
/// <summary>
|
||||
/// What a model nobody stated a context window for is written as.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Deliberately not a zero. A window nobody has looked up is a different statement from a
|
||||
/// window of no tokens, and reading the two as the same is the mistake this whole rebuild set
|
||||
/// out to stop making.
|
||||
/// </remarks>
|
||||
private const string NO_WINDOW = "(unknown)";
|
||||
|
||||
/// <summary>
|
||||
/// What a model nobody stated an image limit for is written as.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The same reasoning as the window, and the same warning against reading it as a zero: a model
|
||||
/// whose vendor says nothing takes as many images as it takes, and the app treats it that way.
|
||||
/// </remarks>
|
||||
private const string NO_IMAGE_LIMIT = "(unknown)";
|
||||
|
||||
/// <summary>
|
||||
/// What a model nobody named a tokenizer for is written as.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Which is what the app already does for all of them: it counts with the tokenizer it ships
|
||||
/// and says that the number is an estimate. Naming one changes nothing about the counting yet;
|
||||
/// it tells a person which file to look for, and for two vendors that there is none.
|
||||
/// </remarks>
|
||||
private const string NO_TOKENIZER = "(unknown)";
|
||||
|
||||
private const string HEADER =
|
||||
"""
|
||||
# What the rules answer, for every model of the corpus.
|
||||
#
|
||||
# Generated. Do not edit by hand: run the SnapshotWriter test to write it anew, then read
|
||||
# the diff. Every line of it is a statement about a model which somebody has to agree with.
|
||||
#
|
||||
# Columns are provider, model ID as the provider reports it, the capabilities sorted by
|
||||
# name, what the model is made for, its context window in tokens, how many images it
|
||||
# accepts, and which tokenizer it uses. The ID stands here unchanged, so a line may well
|
||||
# carry leading or trailing spaces.
|
||||
#
|
||||
# A window or an image limit written as "(unknown)" is one nobody has stated a source for.
|
||||
# That is a gap, not a claim: the app then shows a person how many tokens their conversation
|
||||
# uses without telling them what it may grow to, and it stops nobody from attaching a
|
||||
# hundred pictures to a model which may well take them.
|
||||
#
|
||||
# Every model of the corpus stands here, the ones the audit found a wrong answer for
|
||||
# included. While the old rules still stood those were kept out, so that a known-wrong
|
||||
# answer could not be frozen into this file. The old rules are gone and their answers are
|
||||
# corrected, so keeping them out only hid four of their columns: ExpectedChanges.cs states
|
||||
# what each of them must answer, but it states capabilities alone.
|
||||
#
|
||||
|
||||
""";
|
||||
|
||||
/// <summary>
|
||||
/// The directory this source file lives in, filled in by the compiler.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The snapshot is read from the source tree, not from the build output. It is a file somebody
|
||||
/// reviews and commits, so the test has to fail against the file in the working copy rather
|
||||
/// than against a stale copy next to the assembly.
|
||||
/// </remarks>
|
||||
private static readonly string DIRECTORY = ResolveDirectory();
|
||||
|
||||
/// <summary>
|
||||
/// Where the snapshot is kept.
|
||||
/// </summary>
|
||||
public static readonly string FILE_PATH = Path.Combine(DIRECTORY, "CapabilitySnapshot.txt");
|
||||
|
||||
/// <summary>
|
||||
/// Where a mismatching snapshot is written for comparison in the IDE.
|
||||
/// </summary>
|
||||
public static readonly string ACTUAL_FILE_PATH = Path.Combine(DIRECTORY, "CapabilitySnapshot.actual.txt");
|
||||
|
||||
/// <summary>
|
||||
/// Renders the given entries and the capabilities the current rules answer with.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The text ends with the last model rather than with a line break, which is how this repository
|
||||
/// keeps its files. A generator disagreeing with that by one byte makes the test fail the next
|
||||
/// time an editor tidies the file up, and the failure says that nothing changed -- which is both
|
||||
/// true and useless.
|
||||
/// </remarks>
|
||||
/// <param name="entries">The entries to render.</param>
|
||||
/// <returns>The snapshot text, without a trailing newline and without carriage returns.</returns>
|
||||
public static string Render(IEnumerable<CorpusEntry> entries)
|
||||
{
|
||||
var lines = entries
|
||||
.OrderBy(entry => entry.Provider.ToString(), StringComparer.Ordinal)
|
||||
.ThenBy(entry => entry.ModelId, StringComparer.Ordinal)
|
||||
.Select(Line);
|
||||
|
||||
return new StringBuilder(HEADER).AppendJoin('\n', lines).ToString();
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Writes one corpus entry as a snapshot line.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The kind stands next to the capabilities rather than among them: the two answer different
|
||||
/// questions, and a model changing from a chat model into an embedding one is a different kind
|
||||
/// of news than a model gaining image input.
|
||||
/// </remarks>
|
||||
/// <param name="entry">The entry to write.</param>
|
||||
/// <returns>The line.</returns>
|
||||
private static string Line(CorpusEntry entry)
|
||||
{
|
||||
var profile = entry.Provider.GetModelProfile(new Model(entry.ModelId, null));
|
||||
return $"{entry.Provider} | {entry.ModelId} | {Describe(RebuiltRules.AsCapabilities(profile))} | {profile.Kind} | {Describe(profile.Context)} | {Describe(profile.Images)} | {Describe(profile.Tokenizer)}";
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Writes a tokenizer reference the way a snapshot line does.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The kind travels with the name, because the name alone would be a riddle: "o200k_base" is
|
||||
/// not a repository somebody can open, and "/v1/messages/count_tokens" is not a file somebody
|
||||
/// can download. What sort of thing it is decides what a person can do with it.
|
||||
/// </remarks>
|
||||
/// <param name="tokenizer">The reference to write.</param>
|
||||
/// <returns>The reference, or a marker when nobody named one.</returns>
|
||||
public static string Describe(TokenizerRef tokenizer) => tokenizer.IsKnown ? $"{tokenizer.Kind} {tokenizer.Id}" : NO_TOKENIZER;
|
||||
|
||||
/// <summary>
|
||||
/// Writes an image limit the way a snapshot line does.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Both numbers are named where both are known, because they answer different questions and a
|
||||
/// vendor may state either alone. Naming the one which happens to be smaller would turn two
|
||||
/// statements into one and lose which of them was actually read from a page.
|
||||
/// </remarks>
|
||||
/// <param name="limits">The limits to write.</param>
|
||||
/// <returns>The limits, or a marker when nobody stated any.</returns>
|
||||
public static string Describe(ImageLimits limits)
|
||||
{
|
||||
if (!limits.IsKnown)
|
||||
return NO_IMAGE_LIMIT;
|
||||
|
||||
var parts = new List<string>(2);
|
||||
if (limits.MaxPerMessage is { } perMessage)
|
||||
parts.Add($"{perMessage.ToString(CultureInfo.InvariantCulture)} per message");
|
||||
|
||||
if (limits.MaxPerRequest is { } perRequest)
|
||||
parts.Add($"{perRequest.ToString(CultureInfo.InvariantCulture)} per request");
|
||||
|
||||
return string.Join(", ", parts);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Writes a context window the way a snapshot line does.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Plain digits rather than thousands separators: the number is read by whoever reviews the
|
||||
/// diff, and a separator would make the file depend on which machine generated it.
|
||||
/// </remarks>
|
||||
/// <param name="window">The window to write.</param>
|
||||
/// <returns>The window, or a marker when nobody stated one.</returns>
|
||||
public static string Describe(ContextWindow window)
|
||||
{
|
||||
if (!window.IsKnown)
|
||||
return NO_WINDOW;
|
||||
|
||||
return window.RaisableToTokens is { } raisable
|
||||
? $"{window.DefaultTokens} up to {raisable}"
|
||||
: window.DefaultTokens.ToString(CultureInfo.InvariantCulture);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Writes capabilities the way a snapshot line does.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Sorted by name, and duplicates are kept rather than folded away: a capability appearing twice
|
||||
/// is something to see, not something to hide.
|
||||
/// </remarks>
|
||||
/// <param name="capabilities">The capabilities to write.</param>
|
||||
/// <returns>The capability names, or a marker when there are none.</returns>
|
||||
public static string Describe(IEnumerable<Capability> capabilities)
|
||||
{
|
||||
var names = capabilities.Select(capability => capability.ToString()).Order(StringComparer.Ordinal).ToList();
|
||||
return names.Count is 0 ? NOTHING : string.Join(", ", names);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reads the snapshot as it stands in the source tree.
|
||||
/// </summary>
|
||||
/// <returns>The snapshot text with its line endings normalized, or null when there is none yet.</returns>
|
||||
public static string? Read() => File.Exists(FILE_PATH) ? File.ReadAllText(FILE_PATH).Replace("\r\n", "\n") : null;
|
||||
|
||||
private static string ResolveDirectory([CallerFilePath] string sourceFilePath = "") => Path.GetDirectoryName(sourceFilePath)!;
|
||||
}
|
||||
@@ -0,0 +1,299 @@
|
||||
# What the rules answer, for every model of the corpus.
|
||||
#
|
||||
# Generated. Do not edit by hand: run the SnapshotWriter test to write it anew, then read
|
||||
# the diff. Every line of it is a statement about a model which somebody has to agree with.
|
||||
#
|
||||
# Columns are provider, model ID as the provider reports it, the capabilities sorted by
|
||||
# name, what the model is made for, its context window in tokens, how many images it
|
||||
# accepts, and which tokenizer it uses. The ID stands here unchanged, so a line may well
|
||||
# carry leading or trailing spaces.
|
||||
#
|
||||
# A window or an image limit written as "(unknown)" is one nobody has stated a source for.
|
||||
# That is a gap, not a claim: the app then shows a person how many tokens their conversation
|
||||
# uses without telling them what it may grow to, and it stops nobody from attaching a
|
||||
# hundred pictures to a model which may well take them.
|
||||
#
|
||||
# Every model of the corpus stands here, the ones the audit found a wrong answer for
|
||||
# included. While the old rules still stood those were kept out, so that a known-wrong
|
||||
# answer could not be frozen into this file. The old rules are gone and their answers are
|
||||
# corrected, so keeping them out only hid four of their columns: ExpectedChanges.cs states
|
||||
# what each of them must answer, but it states capabilities alone.
|
||||
#
|
||||
ALIBABA_CLOUD | qvq-max | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen-max-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen-mt-plus | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen-plus-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen-turbo-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen-vl-max | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen2.5-14b-instruct-1m | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen2.5-72b-instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen2.5-omni-7b | AUDIO_INPUT, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, SPEECH_OUTPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen2.5-vl-72b-instruct | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3-235b-a22b | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3-omni-flash | AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, SPEECH_OUTPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3-vl-plus | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.5-plus | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.6-max | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.7-max | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.7-max-2026-05-17 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.7-max-2026-06-08 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.7-max-preview | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.8-27b | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.8-flash | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwen3.8-max | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1000000 | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwq-32b | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | qwq-plus | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
ALIBABA_CLOUD | text-embedding-v3 | EMBEDDING, TEXT_INPUT | EMBEDDING | (unknown) | (unknown) | (unknown)
|
||||
ANTHROPIC | claude-3-5-haiku-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-3-5-sonnet-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-3-7-sonnet-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-3-opus-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-7-sonnet | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-fable-5-1 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | 600 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-haiku-4-5-20251001 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-mythos-5 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | 600 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-opus-4-0 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-opus-5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | 600 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-sonnet-4-0 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | 100 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
ANTHROPIC | claude-sonnet-5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | 600 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
DEEP_SEEK | deepseek-chat | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
DEEP_SEEK | deepseek-reasoner | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
DEEP_SEEK | deepseek-v3.2-exp | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
DEEP_SEEK | deepseek-v4 | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | (unknown) | (unknown)
|
||||
DEEP_SEEK | deepseek-v4-vision | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | (unknown) | (unknown)
|
||||
FIREWORKS | accounts/fireworks/models/deepseek-v3 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
FIREWORKS | accounts/fireworks/models/llama-v3p1-405b-instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 131072 | (unknown) | (unknown)
|
||||
FIREWORKS | accounts/fireworks/models/qwen3-235b-a22b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
FIREWORKS | whisper-v3 | SPEECH_INPUT, TEXT_OUTPUT | TRANSCRIPTION | (unknown) | (unknown) | (unknown)
|
||||
GOOGLE | gemini-1.0-pro-vision | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GOOGLE | gemini-2.0-flash | AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | gemini-2.0-flash-live-001 | AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, SPEECH_INPUT, SPEECH_OUTPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GOOGLE | gemini-2.5-flash | ALWAYS_REASONING, AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1048576 | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | gemini-2.5-flash-image | CHAT_COMPLETION_API, IMAGE_OUTPUT, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | IMAGE_GENERATION | (unknown) | (unknown) | (unknown)
|
||||
GOOGLE | gemini-2.5-flash-lite | AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1048576 | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | gemini-2.5-pro | ALWAYS_REASONING, AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1048576 | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | gemini-3-flash | ALWAYS_REASONING, AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1000000 | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | gemini-3-pro | ALWAYS_REASONING, AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1000000 | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | gemini-3-pro-image | ALWAYS_REASONING, CHAT_COMPLETION_API, IMAGE_OUTPUT, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | IMAGE_GENERATION | (unknown) | (unknown) | (unknown)
|
||||
GOOGLE | gemini-3.1-flash-image | ALWAYS_REASONING, CHAT_COMPLETION_API, IMAGE_OUTPUT, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | IMAGE_GENERATION | (unknown) | (unknown) | (unknown)
|
||||
GOOGLE | gemini-flash-latest | ALWAYS_REASONING, AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1000000 | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | gemini-pro-latest | ALWAYS_REASONING, AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1000000 | 3600 per request | PROVIDER_API countTokens
|
||||
GOOGLE | imagen-4.0-generate-001 | IMAGE_OUTPUT, TEXT_INPUT | IMAGE_GENERATION | (unknown) | (unknown) | (unknown)
|
||||
GOOGLE | text-embedding-004 | EMBEDDING, TEXT_INPUT | EMBEDDING | (unknown) | (unknown) | (unknown)
|
||||
GROQ | llama-3.3-70b-versatile | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 131072 | (unknown) | (unknown)
|
||||
GROQ | moonshotai/kimi-k2-instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GROQ | openai/gpt-oss-120b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 131072 | (unknown) | (unknown)
|
||||
GROQ | qwen/qwen3-32b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GROQ | whisper-large-v3-turbo | SPEECH_INPUT, TEXT_OUTPUT | TRANSCRIPTION | (unknown) | (unknown) | (unknown)
|
||||
GWDG | claude-sonnet-5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | 600 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
GWDG | deepseek-r1 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GWDG | e5-mistral-7b-instruct | EMBEDDING, TEXT_INPUT | EMBEDDING | (unknown) | (unknown) | (unknown)
|
||||
GWDG | gemma-3-27b-it | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GWDG | gpt-5.5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | TIKTOKEN o200k_base
|
||||
GWDG | internvl2.5-8b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GWDG | meta-llama-3.1-8b-instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 131072 | (unknown) | (unknown)
|
||||
GWDG | qwen3-235b-a22b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
GWDG | whisper-large-v2 | SPEECH_INPUT, TEXT_OUTPUT | TRANSCRIPTION | (unknown) | (unknown) | (unknown)
|
||||
HELMHOLTZ | 01 - GPT-5.5 - great overall performance | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | TIKTOKEN o200k_base
|
||||
HELMHOLTZ | 1 - Llama3 405 the best general model | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HELMHOLTZ | 10 - Muse Glimmer 30b - the newest META model | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HELMHOLTZ | Qwen 3.8-27B with DFlash on haicluster | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HELMHOLTZ | alias-qwen38-27b | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HETZNER | gpt-oss-120b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 131072 | (unknown) | (unknown)
|
||||
HETZNER | qwen3-coder-30b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HUGGINGFACE | HuggingFaceTB/SmolLM3-3B | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HUGGINGFACE | Qwen/Qwen3.8-27B | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HUGGINGFACE | deepseek-ai/DeepSeek-R1-Distill-Qwen-32B | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HUGGINGFACE | google/gemma-4-31B-it:novita | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HUGGINGFACE | meta-llama/Llama-4-Scout-17B-16E-Instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HUGGINGFACE | meta-llama/Meta-Llama-3.1-405B-Instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 131072 | (unknown) | (unknown)
|
||||
HUGGINGFACE | mistralai/Magistral-Small-2509 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
HUGGINGFACE | openai/gpt-oss-120b:fireworks-ai | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 131072 | (unknown) | (unknown)
|
||||
IONOS | meta-llama/Llama-3.3-70B-Instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 131072 | (unknown) | (unknown)
|
||||
IONOS | mistralai/Mistral-Small-24B-Instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
LITE_LLM | anthropic/claude-sonnet-5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | 600 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
LITE_LLM | azure/gpt-5.6 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | TIKTOKEN o200k_base
|
||||
LITE_LLM | bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
LITE_LLM | the-fast-one | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | codestral-2508 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | magistral-medium-2506 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | ministral-14b-2512 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | ministral-3b-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | ministral-8b-2410 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | mistral-large-2411 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-large-2512 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-large-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-medium-2505 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-medium-2508 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-medium-2604 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-medium-3-5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-medium-3.5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-medium-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
MISTRAL | mistral-saba-2502 | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | mistral-small-2501 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | mistral-small-2503 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | mistral-small-2603 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | mistral-small-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | open-mistral-nemo | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
MISTRAL | pixtral-12b-2409 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 128000 | (unknown) | (unknown)
|
||||
MISTRAL | pixtral-large-latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 128000 | (unknown) | (unknown)
|
||||
MISTRAL | voxtral-small-2507 | CHAT_COMPLETION_API, FUNCTION_CALLING, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT | TRANSCRIPTION | (unknown) | (unknown) | (unknown)
|
||||
NONE | gpt-5.6 | (nothing) | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_AI | | (nothing) | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_AI | gpt-3.5-turbo | RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | TIKTOKEN cl100k_base
|
||||
OPEN_AI | gpt-3.5-turbo-16k | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | TIKTOKEN cl100k_base
|
||||
OPEN_AI | gpt-4 | RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | 8192 | (unknown) | TIKTOKEN cl100k_base
|
||||
OPEN_AI | gpt-4-0613 | RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | 8192 | (unknown) | TIKTOKEN cl100k_base
|
||||
OPEN_AI | gpt-4-turbo | FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | 128000 | (unknown) | TIKTOKEN cl100k_base
|
||||
OPEN_AI | gpt-4o | FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 128000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-4o-audio-preview | FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | SPEECH_SYNTHESIS | 128000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-4o-mini | FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 128000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-4o-mini-search-preview | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 128000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-4o-search-preview | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 128000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5 | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 400000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5-chat-latest | FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 400000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5-mini | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 400000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5-nano | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 400000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5.1 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 400000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5.1-codex | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 400000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5.2 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 400000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5.3 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | (unknown) | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5.4 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5.5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-5.6 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | gpt-6-astra | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | (unknown)
|
||||
OPEN_AI | gpt-6-astra-mini | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | (unknown)
|
||||
OPEN_AI | o1 | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | o1-mini | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | o1-pro | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | 200000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | o3 | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 200000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | o3-mini | ALWAYS_REASONING, FUNCTION_CALLING, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | o3-pro | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 200000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | o4-mini | ALWAYS_REASONING, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, RESPONSES_API, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 200000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_AI | text-embedding-3-large | EMBEDDING, TEXT_INPUT | EMBEDDING | (unknown) | (unknown) | TIKTOKEN cl100k_base
|
||||
OPEN_AI | whisper-1 | SPEECH_INPUT, TEXT_OUTPUT | TRANSCRIPTION | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | anthropic/claude-opus-5 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | 600 per request | PROVIDER_API /v1/messages/count_tokens
|
||||
OPEN_ROUTER | deepseek/deepseek-chat-v3.1 | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | deepseek/deepseek-r1-distill-llama-70b | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | google/gemini-3.7-flash | ALWAYS_REASONING, AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | 1000000 | 3600 per request | PROVIDER_API countTokens
|
||||
OPEN_ROUTER | google/gemma-4-31b-it | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | meta-llama/llama-4-maverick | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | minimax/minimax-m2 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | mistralai/mistral-large-3 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
OPEN_ROUTER | moonshotai/kimi-k2-thinking | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | nvidia/nemotron-3-49b | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | openai/gpt-5.6 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 1050000 | (unknown) | TIKTOKEN o200k_base
|
||||
OPEN_ROUTER | openai/gpt-oss-120b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 131072 | (unknown) | (unknown)
|
||||
OPEN_ROUTER | perplexity/sonar-reasoning | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | qwen/qwen3.8-flash-next | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
OPEN_ROUTER | z-ai/glm-5.3 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
PERPLEXITY | sonar | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | (unknown) | (unknown) | (unknown)
|
||||
PERPLEXITY | sonar-deep-research | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | (unknown) | (unknown) | (unknown)
|
||||
PERPLEXITY | sonar-pro | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | (unknown) | (unknown) | (unknown)
|
||||
PERPLEXITY | sonar-reasoning | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | (unknown) | (unknown) | (unknown)
|
||||
PERPLEXITY | sonar-reasoning-pro | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | | (nothing) | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | --- | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | 01-ai/yi-large | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | a-model-nobody-has-heard-of | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | apertus-1.5-8b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | apriel-1.5-15b-thinker | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | apriel-1.6-15b-thinker | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | aya-expanse:8b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | aya-vision:8b | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | command-a-plus | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | command-a-reasoning | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | command-a-vision | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | command-a:111b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | command-r7b:7b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | deepseek-r1-distill-llama-70b | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | deepseek-r1:32b | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | deepseek-v2.5 | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | deepseek-v3.1:671b | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | ernie-4.5-21b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | ernie-4.5-vl-28b | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | ernie-x1.1-thinking | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | eurollm-9b-instruct | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | falcon-h1-1.5b-tool-calling | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | falcon-h1:7b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | falcon3:10b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | gemma2:9b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | gemma3:1b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | gemma3:27b | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | gemma3n:e4b | AUDIO_INPUT, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | gemma4:31b | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | gemma4:e2b | AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | glm-4-9b-chat | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | glm-4.5v | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | glm-4.6:latest | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | glm-5-2 | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | glm-5.3-flash-nvfp4 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | gpt-oss:20b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT, WEB_SEARCH | CHAT | 131072 | (unknown) | (unknown)
|
||||
SELF_HOSTED | granite-embedding:278m | EMBEDDING, TEXT_INPUT | EMBEDDING | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | granite3.2-vision:2b | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | granite3.3:8b | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | granite4.2:8b | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | hunyuan:7b | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | inclusionai/ling-mini-2.0 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | internlm3:8b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | internvl3-8b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | kimi-k2.7-code | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | kimi-k2:1t | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | kimi-k3:latest | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT, VIDEO_INPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | kimi-vl:16b | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | ling-1t | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | llama-3.1-405b-base | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | 131072 | (unknown) | (unknown)
|
||||
SELF_HOSTED | llama-3.3-nemotron-super-49b | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | llama2:13b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | llama3.2-vision:11b | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | llama3.2:3b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | 131072 | (unknown) | (unknown)
|
||||
SELF_HOSTED | magistral:24b | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | minimax-m2:latest | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | minimax-text-01 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | ministral-8b-instruct-2410 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | mistral-nemo:12b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | mistral-small-3.1-24b-instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | mistral-small3.2:24b | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | muse-glimmer-30b | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | nemotron-3-49b | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | nomic-embed-text:latest | EMBEDDING, TEXT_INPUT | EMBEDDING | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | nvidia-nemotron-3.5-lightning-30b-a3b-nvfp4 | CHAT_COMPLETION_API, FUNCTION_CALLING, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | occiglot-7b-eu5 | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | olmo-3-32b-think | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | olmo2:13b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | olmo3:7b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | phi-4-mini-reasoning | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | phi-4-multimodal-instruct | AUDIO_INPUT, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | phi-4-reasoning-vision | ALWAYS_REASONING, CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | phi3:14b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | phi4-mini:latest | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | phi4:14b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen2.5-vl-7b-instruct | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen3-coder:30b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen3.5:32b | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen3.6:32b | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen3.8-2.4t-a95b | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen3.8:27b | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen3.8:27b-mlx | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwen3.8:latest | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, REASONING_BY_DEFAULT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | qwq:32b | ALWAYS_REASONING, CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | ring-1t | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | salamandra-7b-instruct | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | salamandra-7b-instruct-tools | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | seed-oss:36b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | smollm2:1.7b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | smollm3:3b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | starling-lm:7b | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | tencent/hy3 | CHAT_COMPLETION_API, FUNCTION_CALLING, OPTIONAL_REASONING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | teuken-7b-instruct | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | voxtral-mini-3b | CHAT_COMPLETION_API, FUNCTION_CALLING, SPEECH_INPUT, TEXT_INPUT, TEXT_OUTPUT | TRANSCRIPTION | (unknown) | (unknown) | (unknown)
|
||||
SELF_HOSTED | yi-1.5:9b | CHAT_COMPLETION_API, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
X | grok-2-vision-1212 | CHAT_COMPLETION_API, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
X | grok-3 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
X | grok-3-mini | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
X | grok-4 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
X | grok-4-fast-reasoning | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
X | grok-4.20 | ALWAYS_REASONING, CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | (unknown) | (unknown)
|
||||
X | grok-4.20-non-reasoning | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 1000000 | (unknown) | (unknown)
|
||||
X | grok-5 | CHAT_COMPLETION_API, FUNCTION_CALLING, TEXT_INPUT, TEXT_OUTPUT | CHAT | (unknown) | (unknown) | (unknown)
|
||||
X | grok-build-0.1 | CHAT_COMPLETION_API, FUNCTION_CALLING, MULTIPLE_IMAGE_INPUT, TEXT_INPUT, TEXT_OUTPUT | CHAT | 256000 | (unknown) | (unknown)
|
||||
@@ -0,0 +1,11 @@
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// One model of the corpus, written the way one provider writes it.
|
||||
/// </summary>
|
||||
/// <param name="Provider">The provider the model is reached through.</param>
|
||||
/// <param name="ModelId">The model ID exactly as that provider reports it, before any normalization.</param>
|
||||
/// <param name="Origin">Where this spelling comes from.</param>
|
||||
public sealed record CorpusEntry(LLMProviders Provider, string ModelId, CorpusOrigin Origin);
|
||||
@@ -0,0 +1,42 @@
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// Says where a spelling in the corpus comes from.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A corpus is only worth as much as the names in it. Anybody can invent a model ID which makes a
|
||||
/// rule look right, so every entry has to say who writes the name that way. The values below are
|
||||
/// ordered by how easy the claim is to check: the first three point at something in this repository,
|
||||
/// the last one does not and is the reason the fallback needs testing at all.
|
||||
/// </remarks>
|
||||
public enum CorpusOrigin
|
||||
{
|
||||
/// <summary>
|
||||
/// A rule in the current capability code names this spelling literally.
|
||||
/// </summary>
|
||||
NAMED_BY_A_RULE,
|
||||
|
||||
/// <summary>
|
||||
/// The app carries this model in a built-in list, such as the one Alibaba Cloud models are
|
||||
/// picked from when the provider serves no catalog.
|
||||
/// </summary>
|
||||
BUILT_INTO_THE_APP,
|
||||
|
||||
/// <summary>
|
||||
/// A comment in the current capability code quotes this spelling as an example of how some
|
||||
/// host writes model names: an Ollama tag, a Fireworks path, a hub prefix, a Blablador
|
||||
/// sentence.
|
||||
/// </summary>
|
||||
QUOTED_AS_A_NAME_SHAPE,
|
||||
|
||||
/// <summary>
|
||||
/// The manual verification list of the rebuild plan asks for this model.
|
||||
/// </summary>
|
||||
ON_THE_MANUAL_TEST_LIST,
|
||||
|
||||
/// <summary>
|
||||
/// A name the provider serves which no rule literal mentions. These are the entries which say
|
||||
/// what happens to everything the rules were not written for.
|
||||
/// </summary>
|
||||
NAMED_BY_NO_RULE,
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// A corpus entry whose current answer the audit showed to be wrong.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// While the old rules still stood, these were kept out of the snapshot: it says "this must not
|
||||
/// change", and writing a known-wrong answer into it would have turned the rebuild into a copy of
|
||||
/// the mistake. That has been over since the old rules were deleted, and keeping them out had a
|
||||
/// price nobody had counted -- an entry here states capabilities and nothing else, so the kind, the
|
||||
/// context window, the image limit and the tokenizer of these models were reviewed nowhere at all.
|
||||
/// Five embedding models sat in that blind spot. They are in the snapshot now like everything else,
|
||||
/// and what this file still does is the part no snapshot can: saying what the answer has to be,
|
||||
/// rather than only noticing that it changed.
|
||||
/// </remarks>
|
||||
/// <param name="Provider">The provider the model is reached through.</param>
|
||||
/// <param name="ModelId">The model ID, exactly as it appears in the corpus.</param>
|
||||
/// <param name="AnswerToday">What the rules being replaced answered. History now: the code that produced it is gone, so nothing checks this any more. It stays because an entry saying only what is right leaves the reader wondering what was wrong.</param>
|
||||
/// <param name="AnswerWanted">What the rebuilt rules have to answer.</param>
|
||||
/// <param name="Reason">Why the current answer is wrong, in one sentence.</param>
|
||||
/// <param name="Source">Where that can be checked.</param>
|
||||
public sealed record ExpectedChange(LLMProviders Provider, string ModelId, IReadOnlyList<Capability> AnswerToday, IReadOnlyList<Capability> AnswerWanted, string Reason, string Source);
|
||||
@@ -0,0 +1,181 @@
|
||||
using static AIStudio.Provider.Capability;
|
||||
using static AIStudio.Provider.LLMProviders;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// The answers the rebuild has to change, one entry per model.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every entry here was found by running the corpus against the rules as they stand and reading
|
||||
/// what came back. They are kept out of the snapshot so that the rebuild does not copy them: a
|
||||
/// snapshot says "do not change this", and a wrong answer is the one thing that must change.
|
||||
///
|
||||
/// Three kinds of mistake are collected below, and they are the three the new architecture is meant
|
||||
/// to make impossible rather than fix one by one:
|
||||
///
|
||||
/// - A model which is not a chat model at all is answered as if it were one. The app already knows
|
||||
/// better: it asks its providers for embedding and transcription models through methods of their
|
||||
/// own. The capability rules never hear about that and hand out tool calling and image input.
|
||||
/// - The same model gets two different answers depending on which spelling it arrives in. That is
|
||||
/// the routing graph leaking into the rules, and it is what the explicit hosts are for.
|
||||
/// - A prefix rule swallows a variant whose name says the opposite. That is priority written by
|
||||
/// hand, and it is what computed specificity is for.
|
||||
/// </remarks>
|
||||
public static class ExpectedChanges
|
||||
{
|
||||
/// <summary>
|
||||
/// Where the app itself states that a model is not a chat model.
|
||||
/// </summary>
|
||||
private const string THE_APP_LISTS_IT_AS_AN_EMBEDDING_MODEL = "The app asks every provider for its embedding models separately, through IProvider.GetEmbeddingModels.";
|
||||
|
||||
/// <summary>
|
||||
/// Every model whose answer has to change.
|
||||
/// </summary>
|
||||
public static readonly IReadOnlyList<ExpectedChange> ENTRIES =
|
||||
[
|
||||
//
|
||||
// Embedding models. They turn text into a vector; there is nothing for them to call a
|
||||
// function with and no image for them to look at. What they need stated is that they embed,
|
||||
// which the capability vocabulary has a word for and the rules never use.
|
||||
//
|
||||
new(OPEN_AI, "text-embedding-3-large",
|
||||
AnswerToday: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, RESPONSES_API, WEB_SEARCH],
|
||||
AnswerWanted: [TEXT_INPUT, EMBEDDING],
|
||||
Reason: "An embedding model is answered with the OpenAI chat default, tool calling and image input included.",
|
||||
Source: THE_APP_LISTS_IT_AS_AN_EMBEDDING_MODEL),
|
||||
|
||||
new(GOOGLE, "text-embedding-004",
|
||||
AnswerToday: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, EMBEDDING],
|
||||
Reason: "An embedding model is answered with the Google default for everything which is not a Gemini.",
|
||||
Source: THE_APP_LISTS_IT_AS_AN_EMBEDDING_MODEL),
|
||||
|
||||
new(ALIBABA_CLOUD, "text-embedding-v3",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, EMBEDDING],
|
||||
Reason: "An embedding model is answered with the Alibaba default, because its name starts with none of the Qwen prefixes.",
|
||||
Source: "Provider/AlibabaCloud/ProviderAlibabaCloud.cs adds it in GetEmbeddingModels and filters the catalog by the prefix \"text-embedding-\"."),
|
||||
|
||||
new(SELF_HOSTED, "nomic-embed-text:latest",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, EMBEDDING],
|
||||
Reason: "An embedding model reaches the global fallback, which assumes an instruction-tuned model that calls functions.",
|
||||
Source: THE_APP_LISTS_IT_AS_AN_EMBEDDING_MODEL),
|
||||
|
||||
new(SELF_HOSTED, "granite-embedding:278m",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, EMBEDDING],
|
||||
Reason: "The Granite block answers about a checkpoint which embeds, and it hands out tool calling for it.",
|
||||
Source: THE_APP_LISTS_IT_AS_AN_EMBEDDING_MODEL),
|
||||
|
||||
new(GWDG, "e5-mistral-7b-instruct",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, EMBEDDING],
|
||||
Reason: "An embedding model is judged by the Mistral rules, because its name carries the word.",
|
||||
Source: THE_APP_LISTS_IT_AS_AN_EMBEDDING_MODEL),
|
||||
|
||||
//
|
||||
// Transcription models. They take speech and write it down. Three of the four are in the
|
||||
// app's own list of transcription models, with the provider's documentation next to them.
|
||||
//
|
||||
new(OPEN_AI, "whisper-1",
|
||||
AnswerToday: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, RESPONSES_API, WEB_SEARCH],
|
||||
AnswerWanted: [SPEECH_INPUT, TEXT_OUTPUT],
|
||||
Reason: "A transcription model is answered with the OpenAI chat default, web search and image input included.",
|
||||
Source: "The app asks every provider for its transcription models separately, through IProvider.GetTranscriptionModels."),
|
||||
|
||||
new(FIREWORKS, "whisper-v3",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [SPEECH_INPUT, TEXT_OUTPUT],
|
||||
Reason: "A transcription model reaches the global fallback and is told it calls functions.",
|
||||
Source: "Provider/Fireworks/ProviderFireworks.cs returns it from GetTranscriptionModels."),
|
||||
|
||||
new(GWDG, "whisper-large-v2",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [SPEECH_INPUT, TEXT_OUTPUT],
|
||||
Reason: "A transcription model reaches the global fallback and is told it calls functions.",
|
||||
Source: "Provider/GWDG/ProviderGWDG.cs returns it from GetTranscriptionModels."),
|
||||
|
||||
new(GROQ, "whisper-large-v3-turbo",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [SPEECH_INPUT, TEXT_OUTPUT],
|
||||
Reason: "A transcription model reaches the global fallback and is told it calls functions.",
|
||||
Source: "Same model family as the Whisper entries the app lists for Fireworks and GWDG."),
|
||||
|
||||
//
|
||||
// An image generation model. It draws a picture from a description; there is no
|
||||
// conversation in it and nothing to call a function with.
|
||||
//
|
||||
new(GOOGLE, "imagen-4.0-generate-001",
|
||||
AnswerToday: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, IMAGE_OUTPUT],
|
||||
Reason: "An image generation model is answered with the Google default for everything which is not a Gemini: it is told it reads images, writes text, and calls functions, and the one thing it does is not said at all.",
|
||||
Source: "Provider/Google/ProviderGoogle.cs keeps only names beginning with \"gemini-\" in its chat model list, so this model is never a chat model to begin with; Provider/ModelKindExtensions.cs classifies image generation separately."),
|
||||
|
||||
//
|
||||
// One model, two spellings, two answers.
|
||||
//
|
||||
new(LITE_LLM, "bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
Reason: "Claude 3.5 Sonnet loses its image input when it arrives under the Bedrock spelling: the vendor sits behind a dot rather than a slash, so neither the gateway detection nor the reseller check finds it.",
|
||||
Source: "The same model as \"anthropic/claude-sonnet-4-0\" and the other Claude entries of this corpus, which all report image input."),
|
||||
|
||||
new(SELF_HOSTED, "mistral-small-3.1-24b-instruct",
|
||||
AnswerToday: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, OPTIONAL_REASONING, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
Reason: "Mistral Small 3.1 is told that it thinks, and the very same model is told the opposite when it arrives through Mistral's own API. The rules for the open weights answer for the whole 3 and 4 range in one line, and reasoning arrived with 4.",
|
||||
Source: "The corpus entry \"mistral-small-2503\" is this model at Mistral and reports no reasoning; Mistral names Magistral as the thinking model of that generation."),
|
||||
|
||||
new(HELMHOLTZ, "01 - GPT-5.5 - great overall performance",
|
||||
AnswerToday: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, WEB_SEARCH, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, REASONING_BY_DEFAULT, WEB_SEARCH, CHAT_COMPLETION_API],
|
||||
Reason: "The descriptive name is recognized as a GPT model and then placed nowhere: every version rule matches the beginning of the name, which here is the list number. The model loses the reasoning it is known for.",
|
||||
Source: "The GWDG entry \"gpt-5.5\" of this corpus is the same model and does report reasoning by default."),
|
||||
|
||||
//
|
||||
// A rule written for one spelling of a name, while the engine people actually run writes
|
||||
// another. The rule is right about the model and never fires.
|
||||
//
|
||||
new(SELF_HOSTED, "granite4.2:8b",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, TEXT_OUTPUT, REASONING_BY_DEFAULT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
Reason: "Granite 4.2 thinks unless the request says otherwise, and there is a rule which says so -- for \"granite-4.2\". Ollama glues the version to the family name, so the rule never sees the models anybody runs locally.",
|
||||
Source: "IBM documents thinking on by default from Granite 4.2; the Ollama library lists the same checkpoint as \"granite4.2\"."),
|
||||
|
||||
new(SELF_HOSTED, "granite3.3:8b",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, TEXT_OUTPUT, OPTIONAL_REASONING, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
Reason: "The same spelling problem one generation earlier: Granite 3.3 has a thinking toggle, and the rule for it is written as \"granite-3.3\".",
|
||||
Source: "IBM documents the thinking toggle for Granite 3.2 and 3.3; the Ollama library lists the checkpoint as \"granite3.3\"."),
|
||||
|
||||
//
|
||||
// One vendor's block swallowing another vendor's model, for no reason but where the two
|
||||
// blocks stand in the file.
|
||||
//
|
||||
new(SELF_HOSTED, "llama-3.3-nemotron-super-49b",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, TEXT_OUTPUT, OPTIONAL_REASONING, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
Reason: "An NVIDIA model is answered by the Llama rules because it carries the name of the weights it was built from, and the Llama block stands above the Nemotron one. It loses the thinking switch, which is one of the two things NVIDIA changed about those weights.",
|
||||
Source: "The corpus entry \"nemotron-3-49b\" is the generation after it and does report thinking; NVIDIA documents the detailed thinking switch for the Llama-Nemotron models."),
|
||||
|
||||
//
|
||||
// A prefix rule swallowing the variant which says the opposite.
|
||||
//
|
||||
new(OPEN_AI, "gpt-5-chat-latest",
|
||||
AnswerToday: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, ALWAYS_REASONING, WEB_SEARCH, RESPONSES_API],
|
||||
AnswerWanted: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, WEB_SEARCH, RESPONSES_API],
|
||||
Reason: "The alias for the non-reasoning GPT-5 is claimed by the \"gpt-5-\" prefix rule and is told it always reasons, which is the one thing its name rules out.",
|
||||
Source: "OpenAI names this alias as the non-reasoning model of the GPT-5 line; the corpus entry \"gpt-5\" next to it is the reasoning one."),
|
||||
|
||||
//
|
||||
// A model nobody had written a rule for yet, found while testing the switch-over.
|
||||
//
|
||||
new(X, "grok-build-0.1",
|
||||
AnswerToday: [TEXT_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
AnswerWanted: [TEXT_INPUT, MULTIPLE_IMAGE_INPUT, TEXT_OUTPUT, FUNCTION_CALLING, CHAT_COMPLETION_API],
|
||||
Reason: "The agentic coding model of the Grok line reads pictures, and the family fallback it reaches says text only.",
|
||||
Source: "https://x.ai/news/grok-build-0-1 states text and image input, tool calling, and a 256K context window."),
|
||||
];
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
using static AIStudio.Provider.LLMProviders;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// The models of the corpus no rule answers for, and why each of them is all right that way.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Hugging Face carries more than a hundred thousand models. Writing a rule for each is not a goal
|
||||
/// anybody could reach, so the question was never whether models fall through to the default but
|
||||
/// which ones may. This list is that decision, written down: every model here was looked at once,
|
||||
/// and leaving it to the default was the answer.
|
||||
///
|
||||
/// It exists because the alternative is silence. A family nobody got round to and a family nobody
|
||||
/// wanted look exactly the same from the outside -- both are simply missing -- and the difference
|
||||
/// only survives if somebody writes it down. The verification run reads this list, and a test holds
|
||||
/// it against the rules from both sides: nothing falls through unlisted, and nothing stays listed
|
||||
/// once a rule does answer for it.
|
||||
///
|
||||
/// What the default says is that a model reads and writes text, speaks the chat completion API, and
|
||||
/// calls functions. The last part is a guess, and the one that matters: the models where it goes
|
||||
/// the wrong way are named in WithoutToolCallingFamily instead of being left here.
|
||||
/// </remarks>
|
||||
public static class LeftToTheDefault
|
||||
{
|
||||
/// <summary>
|
||||
/// A model whose answer the default already gets right, word for word.
|
||||
/// </summary>
|
||||
private const string THE_DEFAULT_SAYS_THE_SAME = "The default answers exactly what the rules for it answer today: text in, text out, and tool calling.";
|
||||
|
||||
/// <summary>
|
||||
/// A model which keeps what it needs and loses what was extra.
|
||||
/// </summary>
|
||||
private const string THE_DEFAULT_KEEPS_WHAT_MATTERS = "A family we decided not to write down. The default keeps the chat and the tool calling; what it drops is the thinking, which a person turns back on in the expert settings and an organization states in a model plugin.";
|
||||
|
||||
/// <summary>
|
||||
/// A model which reads more than text, and is told it does not.
|
||||
/// </summary>
|
||||
private const string THE_DEFAULT_DROPS_THE_MODALITIES = "A family we decided not to write down. The default cannot know what it reads besides text, so images have to be turned on by hand -- in the expert settings, or for everybody through a model plugin.";
|
||||
|
||||
/// <summary>
|
||||
/// Something a provider answered with which was never the name of a model.
|
||||
/// </summary>
|
||||
private const string NOT_A_MODEL_AT_ALL = "Not a model name. It is in the corpus because providers really answer with it, and the rules have to stay quiet rather than invent something.";
|
||||
|
||||
/// <summary>
|
||||
/// Every model which reaches the global default on purpose.
|
||||
/// </summary>
|
||||
public static readonly IReadOnlyList<ModelLeftToTheDefault> ENTRIES =
|
||||
[
|
||||
//
|
||||
// Families whose answer the default already is. Writing them down would add a file and
|
||||
// change nothing about a single answer.
|
||||
//
|
||||
new(SELF_HOSTED, "olmo3:7b", THE_DEFAULT_SAYS_THE_SAME),
|
||||
new(SELF_HOSTED, "falcon3:10b", THE_DEFAULT_SAYS_THE_SAME),
|
||||
new(SELF_HOSTED, "falcon-h1-1.5b-tool-calling", THE_DEFAULT_SAYS_THE_SAME),
|
||||
new(SELF_HOSTED, "salamandra-7b-instruct-tools", THE_DEFAULT_SAYS_THE_SAME),
|
||||
new(SELF_HOSTED, "ling-1t", THE_DEFAULT_SAYS_THE_SAME),
|
||||
new(SELF_HOSTED, "inclusionai/ling-mini-2.0", THE_DEFAULT_SAYS_THE_SAME),
|
||||
new(SELF_HOSTED, "starling-lm:7b", THE_DEFAULT_SAYS_THE_SAME),
|
||||
new(SELF_HOSTED, "phi3:14b", "The Phi rules were written for the fourth generation and the ones before it already reached the default, which answers them the same as it does today."),
|
||||
|
||||
//
|
||||
// Families which lose their thinking to the default. It is the ability a person misses
|
||||
// least: the model still answers, and the answer still carries the thinking -- it is only
|
||||
// not announced, so the thinking settings stay hidden.
|
||||
//
|
||||
new(SELF_HOSTED, "olmo-3-32b-think", THE_DEFAULT_KEEPS_WHAT_MATTERS),
|
||||
new(SELF_HOSTED, "seed-oss:36b", THE_DEFAULT_KEEPS_WHAT_MATTERS),
|
||||
new(SELF_HOSTED, "ring-1t", THE_DEFAULT_KEEPS_WHAT_MATTERS),
|
||||
new(SELF_HOSTED, "smollm3:3b", THE_DEFAULT_KEEPS_WHAT_MATTERS),
|
||||
new(HUGGINGFACE, "HuggingFaceTB/SmolLM3-3B", THE_DEFAULT_KEEPS_WHAT_MATTERS),
|
||||
new(SELF_HOSTED, "internlm3:8b", THE_DEFAULT_KEEPS_WHAT_MATTERS),
|
||||
|
||||
//
|
||||
// Families which read more than text. These are the ones the decision costs something:
|
||||
// until somebody says otherwise, the chat will not offer to send them a picture.
|
||||
//
|
||||
new(SELF_HOSTED, "internvl3-8b", THE_DEFAULT_DROPS_THE_MODALITIES),
|
||||
new(GWDG, "internvl2.5-8b", THE_DEFAULT_DROPS_THE_MODALITIES),
|
||||
new(SELF_HOSTED, "apertus-1.5-8b", "A family we decided not to write down, and the one which loses the most by it: it reads images and listens to audio, and the default knows about neither."),
|
||||
|
||||
//
|
||||
// Names which were never models.
|
||||
//
|
||||
new(OPEN_AI, "", NOT_A_MODEL_AT_ALL),
|
||||
new(SELF_HOSTED, " ", NOT_A_MODEL_AT_ALL),
|
||||
new(SELF_HOSTED, "---", NOT_A_MODEL_AT_ALL),
|
||||
new(NONE, "gpt-5.6", "A model without a provider. There is no way to reach it, so there is nothing to say about how it could be used."),
|
||||
new(LITE_LLM, "the-fast-one", "A freely chosen LiteLLM alias. Nothing in the name says what is behind it, which is what the default exists for."),
|
||||
new(SELF_HOSTED, "a-model-nobody-has-heard-of", "The corpus entry for the default itself. It has to reach it, or the default would never be measured."),
|
||||
|
||||
//
|
||||
// Still to do rather than decided.
|
||||
//
|
||||
new(LITE_LLM, "bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0", "Not a decision: the LiteLLM host cannot take the Bedrock spelling apart yet, because the vendor sits behind a dot rather than a slash. ExpectedChanges holds the answer it has to arrive at."),
|
||||
];
|
||||
}
|
||||
@@ -0,0 +1,427 @@
|
||||
using static AIStudio.Provider.LLMProviders;
|
||||
using static AIStudio.Tests.Models.Corpus.CorpusOrigin;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// The model IDs the capability rules are measured against.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// This list is the ruler for rebuilding the capability system. Every entry is a name some provider
|
||||
/// really answers with, together with the provider it arrives from, because the same model gets a
|
||||
/// different answer depending on who serves it: an ID travels through the rules of its host before
|
||||
/// it reaches the rules of its family.
|
||||
///
|
||||
/// Two things make an entry worth having. Either it is the only name that reaches a particular
|
||||
/// rule, so removing it would let that rule rot unnoticed. Or it is a name no rule was written for,
|
||||
/// which is what the fallback exists for and what nobody looks at otherwise. Names that merely vary
|
||||
/// a size or a date are left out; they exercise the same rule twice and only make the snapshot
|
||||
/// longer.
|
||||
///
|
||||
/// Sizes, dates, and quantization suffixes appear where they change the answer, and only there.
|
||||
/// </remarks>
|
||||
public static class ModelCorpus
|
||||
{
|
||||
/// <summary>
|
||||
/// OpenAI, reached directly. Its rules are the only ones that hand out the Responses API.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] OPEN_AI_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "gpt-6-astra", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-6-astra-mini", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5.6", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5.5", ON_THE_MANUAL_TEST_LIST),
|
||||
new(OPEN_AI, "gpt-5.4", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5.3", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5.2", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5.1", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5.1-codex", NAMED_BY_NO_RULE),
|
||||
new(OPEN_AI, "gpt-5", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5-mini", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5-nano", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-5-chat-latest", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-4o", NAMED_BY_NO_RULE),
|
||||
new(OPEN_AI, "gpt-4o-mini", NAMED_BY_NO_RULE),
|
||||
new(OPEN_AI, "gpt-4o-search-preview", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-4o-mini-search-preview", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-4o-audio-preview", NAMED_BY_NO_RULE),
|
||||
new(OPEN_AI, "gpt-4-turbo", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-4", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-4-0613", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-3.5-turbo", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "gpt-3.5-turbo-16k", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "o1", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "o1-pro", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "o1-mini", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "o3", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "o3-pro", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "o3-mini", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "o4-mini", NAMED_BY_A_RULE),
|
||||
new(OPEN_AI, "text-embedding-3-large", NAMED_BY_NO_RULE),
|
||||
new(OPEN_AI, "whisper-1", NAMED_BY_NO_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Anthropic, reached directly. The six dated aliases come from the list the app falls back to.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] ANTHROPIC_ENTRIES =
|
||||
[
|
||||
new(ANTHROPIC, "claude-mythos-5", NAMED_BY_A_RULE),
|
||||
new(ANTHROPIC, "claude-fable-5-1", NAMED_BY_A_RULE),
|
||||
new(ANTHROPIC, "claude-opus-5", NAMED_BY_A_RULE),
|
||||
new(ANTHROPIC, "claude-sonnet-5", ON_THE_MANUAL_TEST_LIST),
|
||||
new(ANTHROPIC, "claude-haiku-4-5-20251001", NAMED_BY_A_RULE),
|
||||
new(ANTHROPIC, "claude-opus-4-0", BUILT_INTO_THE_APP),
|
||||
new(ANTHROPIC, "claude-sonnet-4-0", BUILT_INTO_THE_APP),
|
||||
new(ANTHROPIC, "claude-3-7-sonnet-latest", BUILT_INTO_THE_APP),
|
||||
new(ANTHROPIC, "claude-3-5-sonnet-latest", BUILT_INTO_THE_APP),
|
||||
new(ANTHROPIC, "claude-3-5-haiku-latest", BUILT_INTO_THE_APP),
|
||||
new(ANTHROPIC, "claude-3-opus-latest", BUILT_INTO_THE_APP),
|
||||
new(ANTHROPIC, "claude-7-sonnet", NAMED_BY_NO_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Google, reached directly. Everything hangs on whether the name carries "gemini-" at all.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] GOOGLE_ENTRIES =
|
||||
[
|
||||
new(GOOGLE, "gemini-3-pro", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-3-flash", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-3-pro-image", ON_THE_MANUAL_TEST_LIST),
|
||||
new(GOOGLE, "gemini-3.1-flash-image", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-flash-latest", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-pro-latest", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-2.5-pro", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-2.5-flash", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-2.5-flash-lite", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-2.5-flash-image", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-2.0-flash", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-2.0-flash-live-001", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "gemini-1.0-pro-vision", NAMED_BY_A_RULE),
|
||||
new(GOOGLE, "text-embedding-004", NAMED_BY_NO_RULE),
|
||||
new(GOOGLE, "imagen-4.0-generate-001", NAMED_BY_NO_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Mistral, reached directly. The family is versioned by release date, so the dated names are
|
||||
/// what the rules really read; the marketing names are a table on the side.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] MISTRAL_ENTRIES =
|
||||
[
|
||||
new(MISTRAL, "mistral-large-latest", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-large-2512", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-large-2411", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-medium-latest", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-medium-2604", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-medium-2508", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-medium-2505", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-medium-3.5", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-medium-3-5", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-small-latest", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-small-2603", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-small-2503", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-small-2501", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "ministral-3b-latest", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "ministral-8b-2410", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "ministral-14b-2512", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(MISTRAL, "pixtral-large-latest", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "pixtral-12b-2409", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "mistral-saba-2502", NAMED_BY_A_RULE),
|
||||
new(MISTRAL, "magistral-medium-2506", NAMED_BY_NO_RULE),
|
||||
new(MISTRAL, "voxtral-small-2507", NAMED_BY_NO_RULE),
|
||||
new(MISTRAL, "codestral-2508", NAMED_BY_NO_RULE),
|
||||
new(MISTRAL, "open-mistral-nemo", NAMED_BY_NO_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Alibaba Cloud. Everything below the two dozen models the app carries is one Qwen tier per
|
||||
/// entry, because each tier answers differently about thinking and vision.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] ALIBABA_ENTRIES =
|
||||
[
|
||||
new(ALIBABA_CLOUD, "qwq-plus", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen-max-latest", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen-plus-latest", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen-turbo-latest", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qvq-max", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen-vl-max", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen-mt-plus", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen2.5-72b-instruct", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen2.5-14b-instruct-1m", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen2.5-omni-7b", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen2.5-vl-72b-instruct", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "text-embedding-v3", BUILT_INTO_THE_APP),
|
||||
new(ALIBABA_CLOUD, "qwen3-omni-flash", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3-vl-plus", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3-235b-a22b", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.5-plus", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.6-max", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.7-max", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.7-max-preview", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.7-max-2026-05-17", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.7-max-2026-06-08", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.8-flash", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.8-max", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwen3.8-27b", NAMED_BY_A_RULE),
|
||||
new(ALIBABA_CLOUD, "qwq-32b", ON_THE_MANUAL_TEST_LIST),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The DeepSeek platform. Two of its names are aliases of their own; everything else is the open
|
||||
/// weights under their published name, which is why the rules hand those on.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] DEEP_SEEK_ENTRIES =
|
||||
[
|
||||
new(DEEP_SEEK, "deepseek-chat", NAMED_BY_A_RULE),
|
||||
new(DEEP_SEEK, "deepseek-reasoner", NAMED_BY_A_RULE),
|
||||
new(DEEP_SEEK, "deepseek-v3.2-exp", NAMED_BY_A_RULE),
|
||||
new(DEEP_SEEK, "deepseek-v4", NAMED_BY_A_RULE),
|
||||
new(DEEP_SEEK, "deepseek-v4-vision", NAMED_BY_A_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Perplexity. One rule separates the thinking Sonar models from the rest.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] PERPLEXITY_ENTRIES =
|
||||
[
|
||||
new(PERPLEXITY, "sonar", NAMED_BY_NO_RULE),
|
||||
new(PERPLEXITY, "sonar-pro", NAMED_BY_NO_RULE),
|
||||
new(PERPLEXITY, "sonar-reasoning", NAMED_BY_A_RULE),
|
||||
new(PERPLEXITY, "sonar-reasoning-pro", NAMED_BY_A_RULE),
|
||||
new(PERPLEXITY, "sonar-deep-research", NAMED_BY_A_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// xAI. It is served by the rules for open weights, which is where the Grok block lives.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] XAI_ENTRIES =
|
||||
[
|
||||
new(X, "grok-4", NAMED_BY_A_RULE),
|
||||
new(X, "grok-4-fast-reasoning", NAMED_BY_A_RULE),
|
||||
new(X, "grok-4.20", NAMED_BY_A_RULE),
|
||||
new(X, "grok-4.20-non-reasoning", NAMED_BY_A_RULE),
|
||||
new(X, "grok-3", NAMED_BY_A_RULE),
|
||||
new(X, "grok-3-mini", NAMED_BY_A_RULE),
|
||||
new(X, "grok-2-vision-1212", NAMED_BY_A_RULE),
|
||||
new(X, "grok-5", NAMED_BY_NO_RULE),
|
||||
new(X, "grok-build-0.1", NAMED_BY_A_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The gateways, which name a model "vendor/model" and serve everything through the chat
|
||||
/// completion API. Each entry picks a different branch of the vendor detection.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] GATEWAY_ENTRIES =
|
||||
[
|
||||
new(OPEN_ROUTER, "openai/gpt-5.6", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(OPEN_ROUTER, "openai/gpt-oss-120b", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "anthropic/claude-opus-5", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(OPEN_ROUTER, "google/gemini-3.7-flash", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(OPEN_ROUTER, "google/gemma-4-31b-it", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "mistralai/mistral-large-3", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "perplexity/sonar-reasoning", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "qwen/qwen3.8-flash-next", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(OPEN_ROUTER, "deepseek/deepseek-r1-distill-llama-70b", ON_THE_MANUAL_TEST_LIST),
|
||||
new(OPEN_ROUTER, "deepseek/deepseek-chat-v3.1", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "moonshotai/kimi-k2-thinking", ON_THE_MANUAL_TEST_LIST),
|
||||
new(OPEN_ROUTER, "z-ai/glm-5.3", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "meta-llama/llama-4-maverick", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "nvidia/nemotron-3-49b", NAMED_BY_A_RULE),
|
||||
new(OPEN_ROUTER, "minimax/minimax-m2", NAMED_BY_A_RULE),
|
||||
new(LITE_LLM, "anthropic/claude-sonnet-5", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(LITE_LLM, "azure/gpt-5.6", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(LITE_LLM, "bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0", NAMED_BY_NO_RULE),
|
||||
new(LITE_LLM, "the-fast-one", NAMED_BY_NO_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Hugging Face. Two things have to come off before a name says anything: the routing suffix,
|
||||
/// which names the inference provider, and the organization in front of the slash.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] HUGGING_FACE_ENTRIES =
|
||||
[
|
||||
new(HUGGINGFACE, "google/gemma-4-31B-it:novita", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(HUGGINGFACE, "meta-llama/Llama-4-Scout-17B-16E-Instruct", NAMED_BY_A_RULE),
|
||||
new(HUGGINGFACE, "meta-llama/Meta-Llama-3.1-405B-Instruct", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(HUGGINGFACE, "Qwen/Qwen3.8-27B", NAMED_BY_A_RULE),
|
||||
new(HUGGINGFACE, "deepseek-ai/DeepSeek-R1-Distill-Qwen-32B", NAMED_BY_A_RULE),
|
||||
new(HUGGINGFACE, "openai/gpt-oss-120b:fireworks-ai", NAMED_BY_A_RULE),
|
||||
new(HUGGINGFACE, "mistralai/Magistral-Small-2509", NAMED_BY_A_RULE),
|
||||
new(HUGGINGFACE, "HuggingFaceTB/SmolLM3-3B", NAMED_BY_A_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Providers that serve other vendors' models under their plain names, without a prefix. GWDG
|
||||
/// is the case that brought this up: next to open weights it resells Claude and GPT models.
|
||||
/// Blablador answers with a whole sentence instead of an ID.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] RESELLER_ENTRIES =
|
||||
[
|
||||
new(GWDG, "claude-sonnet-5", ON_THE_MANUAL_TEST_LIST),
|
||||
new(GWDG, "gpt-5.5", ON_THE_MANUAL_TEST_LIST),
|
||||
new(GWDG, "meta-llama-3.1-8b-instruct", NAMED_BY_A_RULE),
|
||||
new(GWDG, "qwen3-235b-a22b", NAMED_BY_A_RULE),
|
||||
new(GWDG, "deepseek-r1", NAMED_BY_A_RULE),
|
||||
new(GWDG, "gemma-3-27b-it", NAMED_BY_A_RULE),
|
||||
new(GWDG, "internvl2.5-8b", NAMED_BY_A_RULE),
|
||||
new(GWDG, "e5-mistral-7b-instruct", NAMED_BY_NO_RULE),
|
||||
new(GWDG, "whisper-large-v2", BUILT_INTO_THE_APP),
|
||||
new(HELMHOLTZ, "1 - Llama3 405 the best general model", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(HELMHOLTZ, "01 - GPT-5.5 - great overall performance", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(HELMHOLTZ, "10 - Muse Glimmer 30b - the newest META model", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(HELMHOLTZ, "Qwen 3.8-27B with DFlash on haicluster", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(HELMHOLTZ, "alias-qwen38-27b", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(GROQ, "llama-3.3-70b-versatile", NAMED_BY_A_RULE),
|
||||
new(GROQ, "openai/gpt-oss-120b", NAMED_BY_A_RULE),
|
||||
new(GROQ, "moonshotai/kimi-k2-instruct", NAMED_BY_A_RULE),
|
||||
new(GROQ, "qwen/qwen3-32b", NAMED_BY_A_RULE),
|
||||
new(GROQ, "whisper-large-v3-turbo", NAMED_BY_NO_RULE),
|
||||
new(FIREWORKS, "accounts/fireworks/models/llama-v3p1-405b-instruct", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(FIREWORKS, "accounts/fireworks/models/deepseek-v3", NAMED_BY_A_RULE),
|
||||
new(FIREWORKS, "accounts/fireworks/models/qwen3-235b-a22b", NAMED_BY_A_RULE),
|
||||
new(FIREWORKS, "whisper-v3", BUILT_INTO_THE_APP),
|
||||
new(HETZNER, "gpt-oss-120b", NAMED_BY_A_RULE),
|
||||
new(HETZNER, "qwen3-coder-30b", NAMED_BY_A_RULE),
|
||||
new(IONOS, "meta-llama/Llama-3.3-70B-Instruct", NAMED_BY_A_RULE),
|
||||
new(IONOS, "mistralai/Mistral-Small-24B-Instruct", NAMED_BY_A_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Self-hosted engines. Ollama writes the variant behind a colon, which normalization turns
|
||||
/// into a hyphen, so a rolling tag such as "qwen3.8:latest" carries no size at all. This is the
|
||||
/// longest section on purpose: it is where the open-weight families arrive.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] SELF_HOSTED_ENTRIES =
|
||||
[
|
||||
new(SELF_HOSTED, "qwen3.8:latest", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(SELF_HOSTED, "qwen3.8-2.4t-a95b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "qwen3.8:27b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "qwen3.5:32b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "qwen3.6:32b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "qwen3-coder:30b", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "qwen2.5-vl-7b-instruct", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "qwen3.8:27b-mlx", ON_THE_MANUAL_TEST_LIST),
|
||||
new(SELF_HOSTED, "qwq:32b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "deepseek-r1:32b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "deepseek-r1-distill-llama-70b", ON_THE_MANUAL_TEST_LIST),
|
||||
new(SELF_HOSTED, "deepseek-v3.1:671b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "deepseek-v2.5", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "llama3.2:3b", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(SELF_HOSTED, "llama3.2-vision:11b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "llama2:13b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "llama-3.1-405b-base", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "muse-glimmer-30b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "gemma4:e2b", ON_THE_MANUAL_TEST_LIST),
|
||||
new(SELF_HOSTED, "gemma4:31b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "gemma3:1b", ON_THE_MANUAL_TEST_LIST),
|
||||
new(SELF_HOSTED, "gemma3:27b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "gemma3n:e4b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "gemma2:9b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "gpt-oss:20b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "mistral-small3.2:24b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "mistral-small-3.1-24b-instruct", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "mistral-nemo:12b", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "magistral:24b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "voxtral-mini-3b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "ministral-8b-instruct-2410", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "glm-5.3-flash-nvfp4", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(SELF_HOSTED, "glm-5-2", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(SELF_HOSTED, "glm-4.5v", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "glm-4-9b-chat", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "glm-4.6:latest", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "kimi-k3:latest", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "kimi-k2.7-code", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "kimi-vl:16b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "kimi-k2:1t", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "hunyuan:7b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "tencent/hy3", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(SELF_HOSTED, "nemotron-3-49b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "nvidia-nemotron-3.5-lightning-30b-a3b-nvfp4", QUOTED_AS_A_NAME_SHAPE),
|
||||
new(SELF_HOSTED, "llama-3.3-nemotron-super-49b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "granite4.2:8b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "granite3.3:8b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "granite3.2-vision:2b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "granite-embedding:278m", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "command-a:111b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "command-a-plus", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "command-a-vision", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "command-a-reasoning", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "command-r7b:7b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "aya-expanse:8b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "aya-vision:8b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "olmo3:7b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "olmo-3-32b-think", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "olmo2:13b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "seed-oss:36b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "falcon-h1:7b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "falcon-h1-1.5b-tool-calling", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "falcon3:10b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "ling-1t", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "ring-1t", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "inclusionai/ling-mini-2.0", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "starling-lm:7b", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "ernie-4.5-vl-28b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "ernie-x1.1-thinking", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "ernie-4.5-21b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "smollm3:3b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "smollm2:1.7b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "apriel-1.5-15b-thinker", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "apriel-1.6-15b-thinker", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "internvl3-8b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "internlm3:8b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "apertus-1.5-8b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "phi4-mini:latest", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "phi-4-multimodal-instruct", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "phi-4-mini-reasoning", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "phi-4-reasoning-vision", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "phi4:14b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "phi3:14b", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "minimax-m2:latest", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "minimax-text-01", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "teuken-7b-instruct", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "eurollm-9b-instruct", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "occiglot-7b-eu5", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "salamandra-7b-instruct", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "salamandra-7b-instruct-tools", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "yi-1.5:9b", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "01-ai/yi-large", NAMED_BY_A_RULE),
|
||||
new(SELF_HOSTED, "nomic-embed-text:latest", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "a-model-nobody-has-heard-of", NAMED_BY_NO_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Names that say nothing, and one provider which answers about nothing. They are here because
|
||||
/// a rebuild is exactly where a fresh crash on an empty string gets introduced.
|
||||
/// </summary>
|
||||
private static readonly CorpusEntry[] EDGE_CASE_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, " ", NAMED_BY_NO_RULE),
|
||||
new(SELF_HOSTED, "---", NAMED_BY_NO_RULE),
|
||||
new(NONE, "gpt-5.6", NAMED_BY_NO_RULE),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Every entry of the corpus, in the order the sections above are written.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// This has to stand below the sections it reads: static fields are initialized top to bottom,
|
||||
/// and a field which is not initialized yet is null rather than an error.
|
||||
/// </remarks>
|
||||
public static readonly IReadOnlyList<CorpusEntry> ENTRIES =
|
||||
[
|
||||
..OPEN_AI_ENTRIES,
|
||||
..ANTHROPIC_ENTRIES,
|
||||
..GOOGLE_ENTRIES,
|
||||
..MISTRAL_ENTRIES,
|
||||
..ALIBABA_ENTRIES,
|
||||
..DEEP_SEEK_ENTRIES,
|
||||
..PERPLEXITY_ENTRIES,
|
||||
..XAI_ENTRIES,
|
||||
..GATEWAY_ENTRIES,
|
||||
..HUGGING_FACE_ENTRIES,
|
||||
..RESELLER_ENTRIES,
|
||||
..SELF_HOSTED_ENTRIES,
|
||||
..EDGE_CASE_ENTRIES,
|
||||
];
|
||||
}
|
||||
@@ -0,0 +1,210 @@
|
||||
using static AIStudio.Provider.LLMProviders;
|
||||
using static AIStudio.Provider.ModelKind;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// The names which say what a model is made for.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A list of its own, next to the corpus the capability rules are measured against. The two answer
|
||||
/// different questions and are made of different names: the capability corpus is full of chat models,
|
||||
/// because everything else has no capabilities worth stating, while every name here is one nobody
|
||||
/// should be able to start a conversation with.
|
||||
///
|
||||
/// Each entry says what the model is for. Where the markers being replaced answer something else,
|
||||
/// the entry says that too, with the reason -- so that the port can be held to changing nothing
|
||||
/// except where somebody decided it should.
|
||||
/// </remarks>
|
||||
public static class ModelKindCorpus
|
||||
{
|
||||
/// <summary>
|
||||
/// The models which turn text into a vector.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] EMBEDDING_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "text-embedding-3-small", EMBEDDING),
|
||||
new(SELF_HOSTED, "mxbai-embed-large:latest", EMBEDDING),
|
||||
new(SELF_HOSTED, "bge-m3:567m", EMBEDDING),
|
||||
new(SELF_HOSTED, "multilingual-e5-large", EMBEDDING),
|
||||
new(SELF_HOSTED, "gte-multilingual-base", EMBEDDING),
|
||||
new(SELF_HOSTED, "paraphrase-multilingual-mpnet-base-v2", EMBEDDING),
|
||||
new(SELF_HOSTED, "gritlm-7b", EMBEDDING),
|
||||
|
||||
// The one name whose only marker used to be the organization it was published under:
|
||||
new(SELF_HOSTED, "sentence-transformers/all-MiniLM-L6-v2", EMBEDDING),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which put search results back into order, each named after an embedding model.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] RERANKING_ENTRIES =
|
||||
[
|
||||
new(SELF_HOSTED, "bge-reranker-v2-m3", RERANKING),
|
||||
new(SELF_HOSTED, "gte-multilingual-reranker-base", RERANKING),
|
||||
new(SELF_HOSTED, "qwen3-reranker-8b", RERANKING),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which draw.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] IMAGE_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "gpt-image-1", IMAGE_GENERATION),
|
||||
new(OPEN_AI, "dall-e-3", IMAGE_GENERATION),
|
||||
new(SELF_HOSTED, "flux.1-schnell", IMAGE_GENERATION),
|
||||
new(SELF_HOSTED, "stable-diffusion-3.5-large", IMAGE_GENERATION),
|
||||
new(GOOGLE, "gemini-3-pro-image", IMAGE_GENERATION),
|
||||
|
||||
new(GOOGLE, "imagen-4.0-generate-001", IMAGE_GENERATION, AnsweredTodayAs: CHAT, Reason: "The markers never knew the name; the family ported in the Google step states it. Nobody noticed because the Google provider shows only names beginning with gemini."),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which make video.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] VIDEO_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "sora-2", VIDEO_GENERATION),
|
||||
new(GOOGLE, "veo-3.0-generate-001", VIDEO_GENERATION),
|
||||
new(SELF_HOSTED, "kling-video-v2", VIDEO_GENERATION),
|
||||
|
||||
new(X, "grok-imagine-video", VIDEO_GENERATION, AnsweredTodayAs: CHAT, Reason: "Found in the chat list while testing. No marker knew the name, and the xAI provider only kept models whose name lacks \"-image\" -- which \"-imagine\" does."),
|
||||
new(X, "grok-imagine-video-1.5", VIDEO_GENERATION, AnsweredTodayAs: CHAT, Reason: "The same, one version on."),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which listen and write down what they heard.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] TRANSCRIPTION_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "gpt-4o-transcribe", TRANSCRIPTION),
|
||||
new(SELF_HOSTED, "faster-whisper-large-v3", TRANSCRIPTION),
|
||||
new(SELF_HOSTED, "parakeet-tdt-0.6b-v2", TRANSCRIPTION),
|
||||
new(SELF_HOSTED, "wav2vec2-large-xlsr-53", TRANSCRIPTION),
|
||||
new(MISTRAL, "voxtral-mini-latest", TRANSCRIPTION),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which speak, and the ones which answer in audio.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] SPEECH_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "tts-1-hd", SPEECH_SYNTHESIS),
|
||||
new(OPEN_AI, "gpt-4o-mini-tts", SPEECH_SYNTHESIS),
|
||||
new(OPEN_AI, "gpt-audio", SPEECH_SYNTHESIS),
|
||||
new(OPEN_AI, "gpt-4o-audio-preview", SPEECH_SYNTHESIS),
|
||||
|
||||
// The one name which glues the word to something else, and the reason the three words are
|
||||
// not loosened into substrings:
|
||||
new(SELF_HOSTED, "xtts-v2", SPEECH_SYNTHESIS),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which want a connection of their own.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] REALTIME_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "gpt-realtime", REALTIME),
|
||||
new(OPEN_AI, "gpt-4o-realtime-preview", REALTIME),
|
||||
|
||||
// The name the marker file names as the reason for asking this question before the others:
|
||||
new(OPEN_AI, "gpt-realtime-whisper", REALTIME),
|
||||
|
||||
new(OPEN_AI, "gpt-live-1", REALTIME, AnsweredTodayAs: CHAT, Reason: "Found in the chat list while testing. The line which succeeds the realtime models dropped the word, and it is even less of a chat partner: it listens and speaks at once and leaves the thinking to a text model behind it."),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which work a screen.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] COMPUTER_USE_ENTRIES =
|
||||
[
|
||||
new(GOOGLE, "gemini-2.5-computer-use-preview-10-2025", COMPUTER_USE, AnsweredTodayAs: CHAT, Reason: "Found in the chat list while testing. Its API refuses every request which does not carry the computer use tool, so a conversation with it cannot even begin."),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models from before chat completions existed.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] TEXT_COMPLETION_ENTRIES =
|
||||
[
|
||||
new(HELMHOLTZ, "text-davinci-003", TEXT_COMPLETION),
|
||||
new(OPEN_AI, "babbage-002", TEXT_COMPLETION),
|
||||
new(OPEN_AI, "gpt-3.5-turbo-instruct", TEXT_COMPLETION),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which read text off a page.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] OCR_ENTRIES =
|
||||
[
|
||||
new(MISTRAL, "mistral-ocr-latest", OCR),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The models which judge content instead of writing it.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] MODERATION_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "omni-moderation-latest", MODERATION),
|
||||
new(SELF_HOSTED, "llama-guard-3-8b", MODERATION),
|
||||
|
||||
// Written without a separator, which is why the word is looked for as a plain substring:
|
||||
new(SELF_HOSTED, "Qwen3Guard-Gen-8B", MODERATION),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The entries which are no models at all.
|
||||
/// </summary>
|
||||
private static readonly ModelKindExample[] NOT_A_MODEL_ENTRIES =
|
||||
[
|
||||
new(OPEN_AI, "container", OTHER),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// The names which carry a word of one of the kinds above without being one.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// These are the reason several of the words are looked for as whole name parts. A model sorted
|
||||
/// into the wrong kind disappears from the user's list, and a fine-tune losing its place because
|
||||
/// somebody named it after Star Trek is exactly the kind of defect nobody goes looking for.
|
||||
/// </remarks>
|
||||
private static readonly ModelKindExample[] STILL_CHAT_MODELS =
|
||||
[
|
||||
new(SELF_HOSTED, "llama-2-7b-chat-klingon", CHAT),
|
||||
new(SELF_HOSTED, "llama3.3:70b", CHAT),
|
||||
new(OPEN_AI, "gpt-5.1", CHAT),
|
||||
|
||||
//
|
||||
// Three which were questioned while testing and stay all the same. Grok Build is the coding
|
||||
// model behind the xAI CLI and answers like any other Grok. The Groq compound systems are
|
||||
// models with tools already built in, reached through the ordinary chat completion API. And
|
||||
// Gemini Robotics ER answers in text; it is built for pointing at things in a picture rather
|
||||
// than for conversation, but a conversation with it works, and a model which works belongs
|
||||
// in the list.
|
||||
//
|
||||
new(X, "grok-build-0.1", CHAT),
|
||||
new(GROQ, "groq/compound", CHAT),
|
||||
new(GROQ, "groq/compound-mini", CHAT),
|
||||
new(GOOGLE, "gemini-robotics-er-1.5-preview", CHAT),
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Every example, in the order the kinds are written above.
|
||||
/// </summary>
|
||||
public static readonly IReadOnlyList<ModelKindExample> ENTRIES =
|
||||
[
|
||||
..EMBEDDING_ENTRIES,
|
||||
..RERANKING_ENTRIES,
|
||||
..IMAGE_ENTRIES,
|
||||
..VIDEO_ENTRIES,
|
||||
..TRANSCRIPTION_ENTRIES,
|
||||
..SPEECH_ENTRIES,
|
||||
..REALTIME_ENTRIES,
|
||||
..COMPUTER_USE_ENTRIES,
|
||||
..TEXT_COMPLETION_ENTRIES,
|
||||
..OCR_ENTRIES,
|
||||
..MODERATION_ENTRIES,
|
||||
..NOT_A_MODEL_ENTRIES,
|
||||
..STILL_CHAT_MODELS,
|
||||
];
|
||||
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// One model name together with what the app has to make of it.
|
||||
/// </summary>
|
||||
/// <param name="Provider">The provider the model is reached through.</param>
|
||||
/// <param name="ModelId">The model ID exactly as that provider reports it, before any normalization.</param>
|
||||
/// <param name="Kind">What the model is made for.</param>
|
||||
/// <param name="AnsweredTodayAs">What the markers that used to answer this question said, where they said something else. History now: the code that said it is gone, so nothing checks this any more. It stays because a decision without the thing it decided against reads like an arbitrary statement.</param>
|
||||
/// <param name="Reason">Why the two differ, which is only filled in when they do.</param>
|
||||
public sealed record ModelKindExample(LLMProviders Provider, string ModelId, ModelKind Kind, ModelKind? AnsweredTodayAs = null, string Reason = "");
|
||||
@@ -0,0 +1,11 @@
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// One model of the corpus which no rule answers for, together with why that is all right.
|
||||
/// </summary>
|
||||
/// <param name="Provider">The provider the model is reached through.</param>
|
||||
/// <param name="ModelId">The model ID exactly as that provider reports it.</param>
|
||||
/// <param name="Reason">Why this model is left to the global default.</param>
|
||||
public sealed record ModelLeftToTheDefault(LLMProviders Provider, string ModelId, string Reason);
|
||||
@@ -0,0 +1,60 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Corpus;
|
||||
|
||||
/// <summary>
|
||||
/// Asks the rebuilt rules about a corpus entry, in the words the old ones answered in.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The two systems say the same things in different shapes: the old one hands out a list of
|
||||
/// capabilities, the new one a profile whose reasoning is a field of its own rather than one of
|
||||
/// three flags. Comparing them at all needs one of the two translated, and translating the new one
|
||||
/// into the old vocabulary is the direction which loses nothing -- the profile knows more, and
|
||||
/// everything the old answer could say has a place in it.
|
||||
/// </remarks>
|
||||
public static class RebuiltRules
|
||||
{
|
||||
/// <summary>
|
||||
/// Asks the rebuilt rules about one corpus entry.
|
||||
/// </summary>
|
||||
/// <param name="entry">The entry to ask about.</param>
|
||||
/// <returns>The capabilities, in the vocabulary the old rules answered in.</returns>
|
||||
public static IReadOnlyList<Capability> Ask(CorpusEntry entry) => AsCapabilities(ModelRegistry.Shared.Profile(entry.Provider, entry.ModelId));
|
||||
|
||||
/// <summary>
|
||||
/// Writes a profile as the list of capabilities the old rules would have answered with.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The reasoning field turns back into the flag which stands for it. That mapping is the whole
|
||||
/// reason the flags stay in the vocabulary: a person writing an override still says
|
||||
/// ALWAYS_REASONING, and the expert dialog still shows those five choices.
|
||||
/// </remarks>
|
||||
/// <param name="profile">The profile to write out.</param>
|
||||
/// <returns>The capabilities.</returns>
|
||||
public static IReadOnlyList<Capability> AsCapabilities(in ModelProfile profile)
|
||||
{
|
||||
// A profile handed in by reference cannot be reached from inside a query, and copying one
|
||||
// costs nothing:
|
||||
var answered = profile;
|
||||
var stated = Enum.GetValues<Capability>()
|
||||
.Where(capability => capability is not Capability.NONE && answered.Has(capability))
|
||||
.ToList();
|
||||
|
||||
var reasoning = ReasoningAsCapability(profile.Reasoning);
|
||||
if (reasoning is not Capability.NONE)
|
||||
stated.Add(reasoning);
|
||||
|
||||
return stated;
|
||||
}
|
||||
|
||||
private static Capability ReasoningAsCapability(ReasoningSupport reasoning) => reasoning switch
|
||||
{
|
||||
ReasoningSupport.OPTIONAL => Capability.OPTIONAL_REASONING,
|
||||
ReasoningSupport.ON_BY_DEFAULT => Capability.REASONING_BY_DEFAULT,
|
||||
ReasoningSupport.ALWAYS => Capability.ALWAYS_REASONING,
|
||||
|
||||
_ => Capability.NONE,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Tests.Models.Corpus;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the corpus itself, before it is used to judge anything.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A ruler has to be straight before it can measure. A duplicate entry would silently outvote
|
||||
/// itself in the snapshot, and an entry which no longer belongs to any corpus model would make the
|
||||
/// list of known-wrong answers point at nothing.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class CorpusTests
|
||||
{
|
||||
[Test]
|
||||
public void NoModelAppearsTwiceForTheSameProvider()
|
||||
{
|
||||
var duplicates = ModelCorpus.ENTRIES
|
||||
.GroupBy(entry => (entry.Provider, entry.ModelId))
|
||||
.Where(group => group.Count() > 1)
|
||||
.Select(group => $"{group.Key.Provider} {group.Key.ModelId}")
|
||||
.ToList();
|
||||
|
||||
Assert.That(duplicates, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryKnownWrongAnswerBelongsToAModelOfTheCorpus()
|
||||
{
|
||||
var corpus = ModelCorpus.ENTRIES.Select(entry => (entry.Provider, entry.ModelId)).ToHashSet();
|
||||
var orphans = ExpectedChanges.ENTRIES
|
||||
.Where(change => !corpus.Contains((change.Provider, change.ModelId)))
|
||||
.Select(change => $"{change.Provider} {change.ModelId}")
|
||||
.ToList();
|
||||
|
||||
Assert.That(orphans, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryKnownWrongAnswerSaysWhyAndWhereThatCanBeChecked()
|
||||
{
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
foreach (var change in ExpectedChanges.ENTRIES)
|
||||
{
|
||||
Assert.That(change.Reason, Is.Not.Empty, $"{change.Provider} {change.ModelId} does not say why the current answer is wrong.");
|
||||
Assert.That(change.Source, Is.Not.Empty, $"{change.Provider} {change.ModelId} does not say where that can be checked.");
|
||||
Assert.That(change.AnswerWanted, Is.Not.Empty, $"{change.Provider} {change.ModelId} does not say what the answer should be.");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryProviderTheAppSupportsIsRepresented()
|
||||
{
|
||||
//
|
||||
// A provider missing from the corpus is a whole branch of the dispatch nobody measures.
|
||||
// That includes the ones without rules of their own: which rules they borrow, and what
|
||||
// happens to the answer on the way back, is exactly the part a rebuild gets wrong.
|
||||
//
|
||||
var covered = ModelCorpus.ENTRIES.Select(entry => entry.Provider).ToHashSet();
|
||||
var missing = Enum.GetValues<LLMProviders>().Where(provider => !covered.Contains(provider)).ToList();
|
||||
|
||||
Assert.That(missing, Is.Empty);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
using Microsoft.CodeAnalysis;
|
||||
using Microsoft.CodeAnalysis.CSharp;
|
||||
using Microsoft.CodeAnalysis.Diagnostics;
|
||||
|
||||
namespace AIStudio.Tests.Models.Generation;
|
||||
|
||||
/// <summary>
|
||||
/// Compiles a snippet in memory so that a generator or an analyzer can be asked what it makes of it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Both of them are code which runs while the app is being built, and both fail quietly when they
|
||||
/// are wrong: a generator which finds nothing produces an empty registry, and an analyzer which
|
||||
/// recognizes nothing reports nothing. Neither shows up as a broken build, so neither can be
|
||||
/// checked by building the app. It has to be done here, against source written for the purpose.
|
||||
/// </remarks>
|
||||
public static class CompilationHarness
|
||||
{
|
||||
/// <summary>
|
||||
/// Everything the test process itself was loaded with, which includes the app assembly.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Gathered once. Reading a couple of hundred assemblies off disk per test case would make
|
||||
/// these tests slow enough that somebody stops running them.
|
||||
/// </remarks>
|
||||
private static readonly Lazy<MetadataReference[]> REFERENCES = new(GatherReferences);
|
||||
|
||||
/// <summary>
|
||||
/// Compiles a snippet against the same assemblies the app is built against.
|
||||
/// </summary>
|
||||
/// <param name="source">The C# source to compile.</param>
|
||||
/// <returns>The compilation.</returns>
|
||||
public static CSharpCompilation Compile(string source)
|
||||
{
|
||||
var tree = CSharpSyntaxTree.ParseText(source, new CSharpParseOptions(LanguageVersion.Latest));
|
||||
var options = new CSharpCompilationOptions(OutputKind.DynamicallyLinkedLibrary, nullableContextOptions: NullableContextOptions.Enable);
|
||||
|
||||
return CSharpCompilation.Create("SnippetUnderTest", [tree], REFERENCES.Value, options);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Compiles a snippet and reports what it does not even parse or bind.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Worth asking before believing a generator found nothing: a snippet with a typo in it also
|
||||
/// produces an empty result, and the two look exactly alike from the outside.
|
||||
/// </remarks>
|
||||
/// <param name="compilation">The compilation to check.</param>
|
||||
/// <returns>The errors, each on its own line, or an empty string.</returns>
|
||||
public static string ErrorsOf(Compilation compilation)
|
||||
{
|
||||
var errors = compilation.GetDiagnostics()
|
||||
.Where(diagnostic => diagnostic.Severity is DiagnosticSeverity.Error)
|
||||
.Select(diagnostic => diagnostic.ToString());
|
||||
|
||||
return string.Join(Environment.NewLine, errors);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Runs one analyzer over a snippet.
|
||||
/// </summary>
|
||||
/// <param name="source">The C# source to analyze.</param>
|
||||
/// <param name="analyzer">The analyzer to run.</param>
|
||||
/// <returns>What the analyzer reported.</returns>
|
||||
public static async Task<IReadOnlyList<Diagnostic>> AnalyzeAsync(string source, DiagnosticAnalyzer analyzer)
|
||||
{
|
||||
var compilation = Compile(source);
|
||||
Assert.That(ErrorsOf(compilation), Is.Empty, "The snippet has to compile, or the analyzer is being asked about code which does not exist.");
|
||||
|
||||
var reported = await compilation.WithAnalyzers([analyzer]).GetAnalyzerDiagnosticsAsync();
|
||||
return reported;
|
||||
}
|
||||
|
||||
private static MetadataReference[] GatherReferences()
|
||||
{
|
||||
//
|
||||
// The set the runtime resolves types from, which is exactly what this test assembly was
|
||||
// built against: the framework, the NuGet packages, and the app itself.
|
||||
//
|
||||
if (AppContext.GetData("TRUSTED_PLATFORM_ASSEMBLIES") is not string assemblyPaths)
|
||||
throw new InvalidOperationException("The test host did not say which assemblies it trusts, so no compilation can be built against them.");
|
||||
|
||||
return assemblyPaths
|
||||
.Split(Path.PathSeparator)
|
||||
.Where(path => path.EndsWith(".dll", StringComparison.OrdinalIgnoreCase) && File.Exists(path))
|
||||
.Select(path => (MetadataReference) MetadataReference.CreateFromFile(path))
|
||||
.ToArray();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
using AIStudio.Models.Matching;
|
||||
|
||||
using Microsoft.CodeAnalysis;
|
||||
|
||||
using SourceCodeRules.UsageAnalyzers;
|
||||
|
||||
namespace AIStudio.Tests.Models.Generation;
|
||||
|
||||
/// <summary>
|
||||
/// Checks that a pattern which can never match is refused while compiling.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A pattern carrying a capital letter, an underscore, or a space matches no model name, because
|
||||
/// names are normalized before any rule sees them. At runtime that looks like nothing: the family
|
||||
/// answers for nobody and its models quietly take the global default. MWAIS0013 turns it into a
|
||||
/// build error, and these tests are what says that it actually recognizes the calls it is meant to.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelPatternLiteralAnalyzerTests
|
||||
{
|
||||
/// <summary>
|
||||
/// Patterns and whether the app considers them normalized, checked from both ends.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The analyzer carries its own copy of the normalization, because it cannot reference the app.
|
||||
/// This is the table which keeps the two honest: whatever MatchPattern.IsNormalized says at
|
||||
/// runtime, the compile time rule has to say the same.
|
||||
/// </remarks>
|
||||
private static readonly string[] PATTERNS_TO_AGREE_ON =
|
||||
[
|
||||
"gpt-5.1", "qwen3.8-27b", "deepseek-r1", "yi", "01",
|
||||
"GPT-5.1", "gpt_5", "gpt 5", "gpt--5", "-gpt-5", "gpt-5-", "Qwen3.8:27B", "___",
|
||||
];
|
||||
|
||||
[Test]
|
||||
public async Task APatternWrittenTheWayNamesArriveIsAccepted()
|
||||
{
|
||||
var reported = await AnalyzeAsync("""builder.Rule("gpt-5.1").AsPrefix();""");
|
||||
|
||||
Assert.That(reported, Is.Empty);
|
||||
}
|
||||
|
||||
[TestCase("""builder.Rule("GPT-5.1");""", "gpt-5.1")]
|
||||
[TestCase("""builder.Rule("gpt_5");""", "gpt-5")]
|
||||
[TestCase("""builder.Rule("gpt 5");""", "gpt-5")]
|
||||
[TestCase("""builder.Modifier("BASE");""", "base")]
|
||||
[TestCase("""builder.Rule("gpt-5").AlsoContains("Codex");""", "codex")]
|
||||
[TestCase("""builder.Rule("gpt-5").NotContains("Chat");""", "chat")]
|
||||
[TestCase("""builder.Rule("gpt-5"); builder.Rule("gpt-5-mini").InheritsFrom("GPT-5");""", "gpt-5")]
|
||||
public async Task APatternWhichCanNeverMatchIsRefusedAndTheRightSpellingIsNamed(string statements, string expectedSpelling)
|
||||
{
|
||||
var reported = await AnalyzeAsync(statements);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(reported, Has.Count.EqualTo(1));
|
||||
Assert.That(reported.FirstOrDefault()?.Id, Is.EqualTo("MWAIS0013"));
|
||||
Assert.That(reported.FirstOrDefault()?.GetMessage(), Does.Contain($"write it as \"{expectedSpelling}\""));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task APatternOfWhichNothingSurvivesSaysThatInsteadOfSuggestingAnEmptyOne()
|
||||
{
|
||||
var reported = await AnalyzeAsync("""builder.Rule("___");""");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(reported, Has.Count.EqualTo(1));
|
||||
Assert.That(reported.FirstOrDefault()?.GetMessage(), Does.Contain("nothing of it survives"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task APatternWrittenOnceAsAConstantIsCheckedToo()
|
||||
{
|
||||
var reported = await AnalyzeAsync("""const string THE_PATTERN = "GPT-5"; builder.Rule(THE_PATTERN);""");
|
||||
|
||||
Assert.That(reported, Has.Count.EqualTo(1));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task TextWhichIsNotAPatternIsLeftAlone()
|
||||
{
|
||||
//
|
||||
// A tokenizer is named the way its vendor names it, and o200k_base carries an underscore
|
||||
// because OpenAI writes it that way. An analyzer which cannot tell the two kinds of string
|
||||
// apart would make it impossible to state the truth.
|
||||
//
|
||||
var reported = await AnalyzeAsync("""builder.Rule("gpt-5.1").Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");""");
|
||||
|
||||
Assert.That(reported, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task TheCompileTimeRuleAndTheRuntimeCheckNeverDisagree()
|
||||
{
|
||||
foreach (var pattern in PATTERNS_TO_AGREE_ON)
|
||||
{
|
||||
var reported = await AnalyzeAsync($"""builder.Rule("{pattern}");""");
|
||||
var acceptedWhileCompiling = reported.Count is 0;
|
||||
|
||||
Assert.That(acceptedWhileCompiling, Is.EqualTo(MatchPattern.IsNormalized(pattern)), $"The two normalizations disagree about \"{pattern}\".");
|
||||
}
|
||||
}
|
||||
|
||||
private static async Task<IReadOnlyList<Diagnostic>> AnalyzeAsync(string statements)
|
||||
{
|
||||
var source =
|
||||
$$"""
|
||||
using System;
|
||||
|
||||
using AIStudio.Models;
|
||||
|
||||
namespace Sample;
|
||||
|
||||
public sealed class SampleFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.OPEN_AI;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid", new DateOnly(2026, 9, 11), "a note");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder)
|
||||
{
|
||||
{{statements}}
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
return await CompilationHarness.AnalyzeAsync(source, new ModelPatternLiteralAnalyzer());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,191 @@
|
||||
using AIStudio.Models.Registry;
|
||||
|
||||
using Microsoft.CodeAnalysis;
|
||||
using Microsoft.CodeAnalysis.CSharp;
|
||||
|
||||
using SourceGeneratedMappings;
|
||||
|
||||
namespace AIStudio.Tests.Models.Generation;
|
||||
|
||||
/// <summary>
|
||||
/// Checks that adding a family is one action, and that nothing else is needed to make it count.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The whole point of generating the registry is that nobody has to remember a list. If the
|
||||
/// generator misses a family, the family answers for nothing, its models fall into the global
|
||||
/// default, and they look unremarkable rather than broken -- which is the hardest kind of defect to
|
||||
/// notice. So the generator is asked directly, against source written for the purpose.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelRegistryGeneratorTests
|
||||
{
|
||||
/// <summary>
|
||||
/// Two families, one of them two levels down, one host, and an abstract class in between.
|
||||
/// </summary>
|
||||
private const string TWO_FAMILIES_AND_A_HOST =
|
||||
"""
|
||||
using System;
|
||||
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Hosting;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace Sample;
|
||||
|
||||
public abstract class HalfAFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.OPEN_AI;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/half", new DateOnly(2026, 9, 11), "a note");
|
||||
}
|
||||
|
||||
public sealed class SecondFamily : HalfAFamily
|
||||
{
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("second");
|
||||
}
|
||||
|
||||
public sealed class FirstFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.ANTHROPIC;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/first", new DateOnly(2026, 9, 11), "a note");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("first");
|
||||
}
|
||||
|
||||
public sealed class SampleHost : IModelHost
|
||||
{
|
||||
public LLMProviders Provider => LLMProviders.NONE;
|
||||
|
||||
public ModelSource Source => new("https://example.invalid/host", new DateOnly(2026, 9, 11), "a note");
|
||||
|
||||
public bool TryUnwrap(in ModelId id, out ModelId inner, out ModelVendor? declaredVendor)
|
||||
{
|
||||
inner = id;
|
||||
declaredVendor = null;
|
||||
return false;
|
||||
}
|
||||
|
||||
public ModelProfile ApplyTransport(in ModelProfile profile) => profile;
|
||||
}
|
||||
""";
|
||||
|
||||
/// <summary>
|
||||
/// A family the registry cannot create, because it asks for something to be handed in.
|
||||
/// </summary>
|
||||
private const string A_FAMILY_NEEDING_AN_ARGUMENT =
|
||||
"""
|
||||
using System;
|
||||
|
||||
using AIStudio.Models;
|
||||
|
||||
namespace Sample;
|
||||
|
||||
public sealed class DemandingFamily(int somethingItNeeds) : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.OPEN_AI;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/demanding", new DateOnly(2026, 9, 11), $"needs {somethingItNeeds}");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("demanding");
|
||||
}
|
||||
""";
|
||||
|
||||
[Test]
|
||||
public void EveryFamilyIsFoundWithoutBeingAddedToAnything()
|
||||
{
|
||||
var generated = Generate(TWO_FAMILIES_AND_A_HOST, out _, out _);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(generated, Does.Contain("new global::Sample.FirstFamily()"));
|
||||
Assert.That(generated, Does.Contain("new global::Sample.SecondFamily()"), "A family which inherits through another class is still a family.");
|
||||
Assert.That(generated, Does.Contain("new global::Sample.SampleHost()"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AClassWhichCannotBeAFamilyOnItsOwnIsNotRegistered()
|
||||
{
|
||||
var generated = Generate(TWO_FAMILIES_AND_A_HOST, out _, out _);
|
||||
|
||||
Assert.That(generated, Does.Not.Contain("HalfAFamily"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheRegistryIsWrittenInTheSameOrderEveryTime()
|
||||
{
|
||||
//
|
||||
// The order syntax nodes are visited in is not something a shipped file may depend on: the
|
||||
// same sources have to produce the same bytes, or a rebuild shows up as a change.
|
||||
//
|
||||
var generated = Generate(TWO_FAMILIES_AND_A_HOST, out _, out _);
|
||||
|
||||
Assert.That(generated.IndexOf("Sample.FirstFamily", StringComparison.Ordinal), Is.LessThan(generated.IndexOf("Sample.SecondFamily", StringComparison.Ordinal)));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void WhatIsGeneratedCompiles()
|
||||
{
|
||||
Generate(TWO_FAMILIES_AND_A_HOST, out var updated, out _);
|
||||
|
||||
Assert.That(CompilationHarness.ErrorsOf(updated), Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnAssemblyWithoutAnyFamiliesStillGetsARegistry()
|
||||
{
|
||||
//
|
||||
// Otherwise the registry would fail to compile in exactly the situation where somebody is
|
||||
// about to write their first family.
|
||||
//
|
||||
var generated = Generate("namespace Sample;\n\npublic sealed class NothingToDoWithModels;", out var updated, out _);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(generated, Does.Contain("public static class ModelRegistrations"));
|
||||
Assert.That(CompilationHarness.ErrorsOf(updated), Is.Empty);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AFamilyTheRegistryCannotCreateIsReportedRatherThanSkippedQuietly()
|
||||
{
|
||||
var generated = Generate(A_FAMILY_NEEDING_AN_ARGUMENT, out _, out var diagnostics);
|
||||
var reported = diagnostics.Where(diagnostic => diagnostic.Id is "MDR001").ToList();
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(reported, Has.Count.EqualTo(1));
|
||||
Assert.That(reported.FirstOrDefault()?.GetMessage(), Does.Contain("DemandingFamily"));
|
||||
Assert.That(generated, Does.Not.Contain("DemandingFamily"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheAppItselfHasARegistryTheGeneratorWrote()
|
||||
{
|
||||
//
|
||||
// The tests above run the generator by hand. This one asks whether it also ran while the app
|
||||
// was built, which is a different question and the one that actually matters.
|
||||
//
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(ModelRegistrations.CreateFamilies(), Is.Not.Null);
|
||||
Assert.That(ModelRegistrations.CreateHosts(), Is.Not.Null);
|
||||
});
|
||||
}
|
||||
|
||||
private static string Generate(string source, out Compilation updated, out IReadOnlyList<Diagnostic> diagnostics)
|
||||
{
|
||||
var compilation = CompilationHarness.Compile(source);
|
||||
Assert.That(CompilationHarness.ErrorsOf(compilation), Is.Empty, "The snippet has to compile, or the generator is being asked about code which does not exist.");
|
||||
|
||||
var driver = CSharpGeneratorDriver.Create(new ModelRegistryGenerator().AsSourceGenerator());
|
||||
var afterwards = driver.RunGeneratorsAndUpdateCompilation(compilation, out updated, out var reported);
|
||||
|
||||
diagnostics = reported;
|
||||
return afterwards.GetRunResult().Results.Single().GeneratedSources.Single().SourceText.ToString();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Hosting;
|
||||
using AIStudio.Models.Matching;
|
||||
|
||||
namespace AIStudio.Tests.Models.Hosting;
|
||||
|
||||
/// <summary>
|
||||
/// Checks how one wrapping is taken off a name.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// All of this works on the name as the provider reported it, never on the normalized one, and
|
||||
/// that is the point worth testing: normalizing writes the slash, the colon, and the spaces all as
|
||||
/// hyphens, so afterwards there is nothing left to recognize a wrapping by.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class HostNamingTests
|
||||
{
|
||||
[Test]
|
||||
public void TheOrganizationComesOffAndSaysWhoBuiltTheModel()
|
||||
{
|
||||
var taken = HostNaming.TrySplitOrganization(new ModelId("anthropic/claude-opus-5"), out var inner, out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.True);
|
||||
Assert.That(inner.Original, Is.EqualTo("claude-opus-5"));
|
||||
Assert.That(vendor, Is.EqualTo(ModelVendor.ANTHROPIC));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnOrganizationNobodyRecognizesStatesNoVendorRatherThanAnUnknownOne()
|
||||
{
|
||||
//
|
||||
// "azure" is where the model is running, not who built it. Saying "unknown" here would be a
|
||||
// statement, and it would stop the rules from working out the vendor from the name itself.
|
||||
//
|
||||
var taken = HostNaming.TrySplitOrganization(new ModelId("azure/gpt-5.6"), out var inner, out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.True);
|
||||
Assert.That(inner.Original, Is.EqualTo("gpt-5.6"));
|
||||
Assert.That(vendor, Is.Null);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnOrganizationIsRecognizedWhicheverWayTheHostSpellsIt()
|
||||
{
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(HostNaming.VendorOfOrganization("meta-llama"), Is.EqualTo(ModelVendor.META));
|
||||
Assert.That(HostNaming.VendorOfOrganization("Qwen"), Is.EqualTo(ModelVendor.ALIBABA));
|
||||
Assert.That(HostNaming.VendorOfOrganization("deepseek-ai"), Is.EqualTo(ModelVendor.DEEP_SEEK));
|
||||
Assert.That(HostNaming.VendorOfOrganization("HuggingFaceTB"), Is.EqualTo(ModelVendor.HUGGING_FACE));
|
||||
Assert.That(HostNaming.VendorOfOrganization("somebody-else"), Is.EqualTo(ModelVendor.UNKNOWN));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANameWithoutAnOrganizationIsLeftAlone()
|
||||
{
|
||||
var taken = HostNaming.TrySplitOrganization(new ModelId("llama-3.3-70b-versatile"), out var inner, out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.False);
|
||||
Assert.That(inner.Original, Is.EqualTo("llama-3.3-70b-versatile"));
|
||||
Assert.That(vendor, Is.Null);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void OnlyOneSegmentComesOffAtATime()
|
||||
{
|
||||
//
|
||||
// The account path Fireworks puts in front is three segments deep. Nothing here counts
|
||||
// them: the walk asks again, which is also what covers the two wrappings of Hugging Face.
|
||||
//
|
||||
var taken = HostNaming.TrySplitOrganization(new ModelId("accounts/fireworks/models/llama-v3p1-405b-instruct"), out var inner, out _);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.True);
|
||||
Assert.That(inner.Original, Is.EqualTo("fireworks/models/llama-v3p1-405b-instruct"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnOrganizationWithNothingBehindItIsNotAWrapping()
|
||||
{
|
||||
var taken = HostNaming.TrySplitOrganization(new ModelId("openai/"), out var inner, out _);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.False);
|
||||
Assert.That(inner.Original, Is.EqualTo("openai/"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheRoutingSuffixComesOffAndTheModelStaysWhatItWas()
|
||||
{
|
||||
var taken = HostNaming.TryStripRoutingSuffix(new ModelId("google/gemma-4-31B-it:novita"), out var inner);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.True);
|
||||
Assert.That(inner.Original, Is.EqualTo("google/gemma-4-31B-it"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AMenuPositionComesOff()
|
||||
{
|
||||
var taken = HostNaming.TryStripMenuPosition(new ModelId("10 - Muse Glimmer 30b - the newest META model"), out var inner);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.True);
|
||||
Assert.That(inner.Original, Is.EqualTo("Muse Glimmer 30b - the newest META model"));
|
||||
});
|
||||
}
|
||||
|
||||
[TestCase("70b-instruct", TestName = "A number the model is named after is not a menu position")]
|
||||
[TestCase("3-mini", TestName = "A number followed straight by a hyphen is not a menu position")]
|
||||
[TestCase("alias-qwen38-27b", TestName = "A name not starting with a number is not a menu position")]
|
||||
[TestCase("Qwen 3.8-27B with DFlash on haicluster", TestName = "A sentence without a leading number is not a menu position")]
|
||||
public void WhatOnlyLooksLikeAMenuPositionIsLeftAlone(string modelId)
|
||||
{
|
||||
var taken = HostNaming.TryStripMenuPosition(new ModelId(modelId), out var inner);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(taken, Is.False);
|
||||
Assert.That(inner.Original, Is.EqualTo(modelId));
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,187 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Hosting;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Hosting;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the walk which takes a name apart, and what happens when nobody wrote a host.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// How deep a wrapping goes is the host's business, not the caller's: Hugging Face has two, Fireworks
|
||||
/// has three, most have none. Asking over and over until the host says no is what covers all of
|
||||
/// them, and what has to be bounded so that a host which never says no cannot hang the app.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelHostIndexTests
|
||||
{
|
||||
[Test]
|
||||
public void TheHostsAreKeptInTheOrderOfTheProvidersTheyAnswerFor()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new SplittingHost(), new StubbornHost()]);
|
||||
|
||||
Assert.That(index.Hosts.Select(host => host.Provider), Is.EqualTo(new[] { LLMProviders.OPEN_ROUTER, LLMProviders.LITE_LLM }));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TwoHostsForOneProviderIsRefused()
|
||||
{
|
||||
var refused = Assert.Throws<InvalidOperationException>(() => ModelHostIndex.Build([new SplittingHost(), new SecondHostForTheSameProvider()]));
|
||||
|
||||
Assert.That(refused?.Message, Does.Contain("OPEN_ROUTER"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AHostAnsweringForNoProviderIsRefused()
|
||||
{
|
||||
//
|
||||
// The default value of the provider enum is NONE, so a host which gets this wrong gets it
|
||||
// wrong quietly: it would sit in the index answering for a provider nobody can configure.
|
||||
//
|
||||
var refused = Assert.Throws<InvalidOperationException>(() => ModelHostIndex.Build([new HostForNobody()]));
|
||||
|
||||
Assert.That(refused?.Message, Does.Contain(nameof(HostForNobody)));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ProvidersNobodyWroteAHostForAreNamed()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new SplittingHost()]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(index.ProvidersWithoutAHost, Does.Contain(LLMProviders.ANTHROPIC));
|
||||
Assert.That(index.ProvidersWithoutAHost, Does.Not.Contain(LLMProviders.OPEN_ROUTER));
|
||||
Assert.That(index.ProvidersWithoutAHost, Does.Not.Contain(LLMProviders.NONE), "Nobody can configure it, so nobody has to write a host for it.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AProviderWithoutAHostGetsItsNameBackUntouched()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new SplittingHost()]);
|
||||
var unwrapped = index.Unwrap(new ModelId("anthropic/claude-opus-5"), LLMProviders.ANTHROPIC, out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(index.Of(LLMProviders.ANTHROPIC), Is.Null);
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("anthropic/claude-opus-5"));
|
||||
Assert.That(vendor, Is.Null);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AProviderWithoutAHostStillLosesTheResponsesApi()
|
||||
{
|
||||
//
|
||||
// The safe direction: claiming an API which is not there turns into a failed request, while
|
||||
// not claiming one only means the app does not use it.
|
||||
//
|
||||
var index = ModelHostIndex.Build([new SplittingHost()]);
|
||||
var profile = new ModelProfile { Capabilities = Capability.TEXT_INPUT | Capability.RESPONSES_API };
|
||||
var throughTheProvider = index.ApplyTransport(profile, LLMProviders.ANTHROPIC);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(throughTheProvider.Has(Capability.RESPONSES_API), Is.False);
|
||||
Assert.That(throughTheProvider.Has(Capability.CHAT_COMPLETION_API), Is.True);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheWalkKeepsAskingUntilTheHostSaysNo()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new SplittingHost()]);
|
||||
var unwrapped = index.Unwrap(new ModelId("accounts/fireworks/models/llama-v3p1-405b-instruct"), LLMProviders.OPEN_ROUTER, out _);
|
||||
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("llama-v3p1-405b-instruct"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheInnermostWrappingIsTheOneWhichSaysWhoBuiltTheModel()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new SplittingHost()]);
|
||||
index.Unwrap(new ModelId("anthropic/openai/gpt-5"), LLMProviders.OPEN_ROUTER, out var vendor);
|
||||
|
||||
Assert.That(vendor, Is.EqualTo(ModelVendor.OPEN_AI), "A wrapping closer to the model knows more about it than one further out.");
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AWrappingWhichSaysNothingDoesNotEraseWhatAnOuterOneSaid()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new SplittingHost()]);
|
||||
index.Unwrap(new ModelId("anthropic/somebody-else/claude-opus-5"), LLMProviders.OPEN_ROUTER, out var vendor);
|
||||
|
||||
Assert.That(vendor, Is.EqualTo(ModelVendor.ANTHROPIC));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AHostHandingBackWhatItWasGivenIsNotAskedAgain()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new StubbornHost()]);
|
||||
var unwrapped = index.Unwrap(new ModelId("the-fast-one"), LLMProviders.LITE_LLM, out _);
|
||||
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("the-fast-one"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AHostWhichNeverSaysNoIsStoppedRatherThanFollowedForever()
|
||||
{
|
||||
var index = ModelHostIndex.Build([new GrowingHost()]);
|
||||
var unwrapped = index.Unwrap(new ModelId("thing"), LLMProviders.GROQ, out _);
|
||||
|
||||
Assert.That(unwrapped.Original.Split("-more"), Has.Length.EqualTo(ModelHostIndex.MAX_UNWRAPPING_STEPS + 1));
|
||||
}
|
||||
|
||||
private sealed class SplittingHost : ModelHost
|
||||
{
|
||||
public override LLMProviders Provider => LLMProviders.OPEN_ROUTER;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/splitting", new DateOnly(2026, 9, 11), "A host taking off one organization at a time.");
|
||||
|
||||
public override bool TryUnwrap(in ModelId id, out ModelId inner, out ModelVendor? declaredVendor) => HostNaming.TrySplitOrganization(id, out inner, out declaredVendor);
|
||||
}
|
||||
|
||||
private sealed class SecondHostForTheSameProvider : ModelHost
|
||||
{
|
||||
public override LLMProviders Provider => LLMProviders.OPEN_ROUTER;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/second", new DateOnly(2026, 9, 11), "A second host claiming a provider which already has one.");
|
||||
}
|
||||
|
||||
private sealed class HostForNobody : ModelHost
|
||||
{
|
||||
public override LLMProviders Provider => LLMProviders.NONE;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/nobody", new DateOnly(2026, 9, 11), "A host which names no provider.");
|
||||
}
|
||||
|
||||
private sealed class StubbornHost : ModelHost
|
||||
{
|
||||
public override LLMProviders Provider => LLMProviders.LITE_LLM;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/stubborn", new DateOnly(2026, 9, 11), "A host saying it unwrapped something without shortening anything.");
|
||||
|
||||
public override bool TryUnwrap(in ModelId id, out ModelId inner, out ModelVendor? declaredVendor)
|
||||
{
|
||||
inner = id;
|
||||
declaredVendor = null;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class GrowingHost : ModelHost
|
||||
{
|
||||
public override LLMProviders Provider => LLMProviders.GROQ;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/growing", new DateOnly(2026, 9, 11), "A host handing back a longer name every time it is asked.");
|
||||
|
||||
public override bool TryUnwrap(in ModelId id, out ModelId inner, out ModelVendor? declaredVendor)
|
||||
{
|
||||
inner = new($"{id.Original}-more");
|
||||
declaredVendor = null;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Hosting;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Hosting;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the hosts the app actually ships, against the names the providers actually answer with.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The names in here are the ones from the corpus, which came out of the provider lists and the
|
||||
/// audit rather than out of somebody's head. What is being asked is the routing question only --
|
||||
/// what is left of a name once the way it arrived has been accounted for, and which APIs survive
|
||||
/// the trip. Which model it then is remains a question for the rules.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelHostTests
|
||||
{
|
||||
/// <summary>
|
||||
/// The hosts as the app has them, found by the generator rather than listed here.
|
||||
/// </summary>
|
||||
private static readonly ModelHostIndex INDEX = ModelHostIndex.Build(ModelRegistrations.CreateHosts());
|
||||
|
||||
[Test]
|
||||
public void EveryProviderAPersonCanConfigureHasAHost()
|
||||
{
|
||||
//
|
||||
// This is the one which fails when somebody adds a provider to the app and stops there. It
|
||||
// is not a runtime error -- names would simply be taken as they arrive -- so nothing else
|
||||
// would ever point it out.
|
||||
//
|
||||
Assert.That(INDEX.ProvidersWithoutAHost, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryHostSaysWhereItsBehaviourCanBeCheckedAndWhen()
|
||||
{
|
||||
var unstated = INDEX.Hosts.Where(host => !host.Source.IsStated).Select(host => host.GetType().Name);
|
||||
|
||||
Assert.That(unstated, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AGatewayNameFallsApartIntoTheModelAndWhoBuiltIt()
|
||||
{
|
||||
var unwrapped = Unwrap(LLMProviders.OPEN_ROUTER, "anthropic/claude-opus-5", out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("claude-opus-5"));
|
||||
Assert.That(vendor, Is.EqualTo(ModelVendor.ANTHROPIC));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheHuggingFaceRouterTakesOffTheRouteFirstAndTheOrganizationSecond()
|
||||
{
|
||||
var unwrapped = Unwrap(LLMProviders.HUGGINGFACE, "openai/gpt-oss-120b:fireworks-ai", out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("gpt-oss-120b"));
|
||||
Assert.That(vendor, Is.EqualTo(ModelVendor.OPEN_AI), "OpenAI published the weights, whoever is serving them today.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AHuggingFaceNameWithoutARouteIsStillTakenApart()
|
||||
{
|
||||
var unwrapped = Unwrap(LLMProviders.HUGGINGFACE, "deepseek-ai/DeepSeek-R1-Distill-Qwen-32B", out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("DeepSeek-R1-Distill-Qwen-32B"));
|
||||
Assert.That(vendor, Is.EqualTo(ModelVendor.DEEP_SEEK));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheFireworksAccountPathComesOffWholeWithoutAnybodyCountingItsSegments()
|
||||
{
|
||||
var unwrapped = Unwrap(LLMProviders.FIREWORKS, "accounts/fireworks/models/llama-v3p1-405b-instruct", out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("llama-v3p1-405b-instruct"));
|
||||
Assert.That(vendor, Is.Null, "None of the three path segments names a vendor.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void BlabladorLosesItsPlaceInTheMenu()
|
||||
{
|
||||
var unwrapped = Unwrap(LLMProviders.HELMHOLTZ, "1 - Llama3 405 the best general model", out _);
|
||||
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("Llama3 405 the best general model"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnEngineServingAHubRepositoryHasItReadAsOne()
|
||||
{
|
||||
var unwrapped = Unwrap(LLMProviders.SELF_HOSTED, "meta-llama/Llama-3.3-70B-Instruct", out var vendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("Llama-3.3-70B-Instruct"));
|
||||
Assert.That(vendor, Is.EqualTo(ModelVendor.META));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheVariantOllamaWritesAfterAColonSurvives()
|
||||
{
|
||||
//
|
||||
// The colon means two different things at two different hosts. On the router it says where
|
||||
// the request goes; on Ollama it says which build is running, and taking it off would leave
|
||||
// a name which no longer identifies the model.
|
||||
//
|
||||
var unwrapped = Unwrap(LLMProviders.SELF_HOSTED, "qwen3.8:27b-mlx", out _);
|
||||
|
||||
Assert.That(unwrapped.Original, Is.EqualTo("qwen3.8:27b-mlx"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AResellerLeavesTheNameAloneAndOnlyTakesTheApiAway()
|
||||
{
|
||||
//
|
||||
// This is the GWDG case: it offers Claude and GPT under the names their vendors use, so the
|
||||
// rules recognize them and answer with everything those models can do. Everything except
|
||||
// the API -- the request goes to Göttingen, and the Responses API is not served there.
|
||||
//
|
||||
var atItsVendor = new ModelProfile { Capabilities = Capability.TEXT_INPUT | Capability.FUNCTION_CALLING | Capability.RESPONSES_API };
|
||||
var throughTheReseller = Transport(LLMProviders.GWDG, atItsVendor);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(Unwrap(LLMProviders.GWDG, "claude-sonnet-5", out _).Original, Is.EqualTo("claude-sonnet-5"));
|
||||
Assert.That(throughTheReseller.Has(Capability.FUNCTION_CALLING), Is.True);
|
||||
Assert.That(throughTheReseller.Has(Capability.RESPONSES_API), Is.False);
|
||||
Assert.That(throughTheReseller.Has(Capability.CHAT_COMPLETION_API), Is.True);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void OnlyOpenAIsOwnCloudKeepsTheResponsesApi()
|
||||
{
|
||||
var withBothApis = new ModelProfile { Capabilities = Capability.RESPONSES_API | Capability.CHAT_COMPLETION_API };
|
||||
var elsewhere = INDEX.Hosts
|
||||
.Where(host => host.Provider is not LLMProviders.OPEN_AI)
|
||||
.Where(host => host.ApplyTransport(withBothApis).Has(Capability.RESPONSES_API))
|
||||
.Select(host => host.GetType().Name);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(Transport(LLMProviders.OPEN_AI, withBothApis).Has(Capability.RESPONSES_API), Is.True);
|
||||
Assert.That(elsewhere, Is.Empty, "The app sends a Responses API request from exactly one place.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModelReachedThroughNeitherApiIsNotGivenOne()
|
||||
{
|
||||
//
|
||||
// An embedding model is reached through neither of the two. Answering that it speaks the
|
||||
// chat completion API would be a claim nobody made.
|
||||
//
|
||||
var embedding = new ModelProfile { Capabilities = Capability.EMBEDDING };
|
||||
var throughAGateway = Transport(LLMProviders.OPEN_ROUTER, embedding);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(throughAGateway.Has(Capability.EMBEDDING), Is.True);
|
||||
Assert.That(throughAGateway.HasAny(Capability.CHAT_COMPLETION_API | Capability.RESPONSES_API), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANameWithoutAWrappingComesBackAsItWas()
|
||||
{
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(Unwrap(LLMProviders.OPEN_AI, "gpt-5.6", out _).Original, Is.EqualTo("gpt-5.6"));
|
||||
Assert.That(Unwrap(LLMProviders.GROQ, "llama-3.3-70b-versatile", out _).Original, Is.EqualTo("llama-3.3-70b-versatile"));
|
||||
Assert.That(Unwrap(LLMProviders.LITE_LLM, "the-fast-one", out _).Original, Is.EqualTo("the-fast-one"));
|
||||
});
|
||||
}
|
||||
|
||||
private static ModelId Unwrap(LLMProviders provider, string modelId, out ModelVendor? declaredVendor) => INDEX.Unwrap(new ModelId(modelId), provider, out declaredVendor);
|
||||
|
||||
private static ModelProfile Transport(LLMProviders provider, in ModelProfile profile) => INDEX.ApplyTransport(profile, provider);
|
||||
}
|
||||
@@ -0,0 +1,128 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Settings;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks how many images the rules say a model takes, where a vendor stated a number.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Two vendors state one at all. Anthropic gives a rule rather than a number -- it reads the limit
|
||||
/// off the context window -- and Google gives one number for the whole family. Everybody else either
|
||||
/// says nothing or limits something other than the count: OpenAI caps the image patches of a request
|
||||
/// instead of the images, which is not a number of pictures and is not written down as one here.
|
||||
///
|
||||
/// What is worth a test is therefore not the arithmetic but the two places where writing the rules
|
||||
/// the obvious way gets it wrong: a Claude whose window grew must get the larger image limit without
|
||||
/// anybody saying so, and a model nobody documented must keep answering "as many as it takes"
|
||||
/// instead of inheriting somebody else's ceiling.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ImageLimitRuleTests
|
||||
{
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-3-5-sonnet-latest", 100, Description = "A 200k window, so the smaller limit.")]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-sonnet-4-0", 100)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-haiku-4-5-20251001", 100)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-opus-5", 600, Description = "A million tokens, so Anthropic's limit for every other model.")]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-sonnet-5", 600)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-opus-4-6", 600)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-sonnet-4-6", 600)]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-fable-5-1", 600)]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-2.5-pro", 3_600, Description = "Google states one number for all of Gemini.")]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-3-pro", 3_600)]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-flash-latest", 3_600)]
|
||||
public void TheImageLimitOfAModelIsTheOneItsVendorStates(LLMProviders provider, string modelId, int perRequest)
|
||||
{
|
||||
var limits = provider.GetModelProfile(new Model(modelId, null)).Images;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(limits.IsKnown, Is.True);
|
||||
Assert.That(limits.MaxPerRequest, Is.EqualTo(perRequest));
|
||||
Assert.That(limits.MaxPerMessage, Is.Null, "Neither vendor states a per-message limit, and inventing one would be a ceiling nobody wrote.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AClaudeWhoseWindowGrewGetsTheLargerImageLimitWithoutSayingSo()
|
||||
{
|
||||
//
|
||||
// This is the whole reason the limit is worked out instead of written down: the rule for
|
||||
// Opus 4.6 states its larger window and nothing else, and Anthropic's own page says the
|
||||
// image limit follows from exactly that. Two numbers written by hand would have drifted the
|
||||
// first time somebody added a model and thought of only one of them.
|
||||
//
|
||||
var smallWindow = ModelRegistry.Shared.Profile(LLMProviders.ANTHROPIC, "claude-opus-4-1");
|
||||
var largeWindow = ModelRegistry.Shared.Profile(LLMProviders.ANTHROPIC, "claude-opus-4-6");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(smallWindow.Context.DefaultTokens, Is.EqualTo(200_000));
|
||||
Assert.That(smallWindow.Images.MaxPerRequest, Is.EqualTo(100));
|
||||
Assert.That(largeWindow.Context.DefaultTokens, Is.EqualTo(1_000_000));
|
||||
Assert.That(largeWindow.Images.MaxPerRequest, Is.EqualTo(600));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModelNobodyStatedALimitForTakesAsManyAsItTakes()
|
||||
{
|
||||
//
|
||||
// The common case, and the one which must not become a hidden ceiling. A self-hosted model
|
||||
// is served at whatever its operator configured, and an app which refused the seventh
|
||||
// picture because six is a nice number would be taking something away that works today.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.SELF_HOSTED, "some-model-nobody-wrote-a-rule-for");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Images.IsKnown, Is.False);
|
||||
Assert.That(profile.Images.MaxInOneMessage, Is.Null);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void OpenAIStatesNoNumberOfImagesAndSoNeitherDoWe()
|
||||
{
|
||||
//
|
||||
// Their guide caps a request at 30,000 image patches, which is a budget rather than a count:
|
||||
// how many pictures fit into it depends on how large each of them is. Writing any number of
|
||||
// images here would be our arithmetic presented as their statement.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.OPEN_AI, "gpt-5.1");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.True, "It does read several images.");
|
||||
Assert.That(profile.Images.IsKnown, Is.False, "How many, nobody said.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheSameModelThroughAGatewayKeepsItsImageLimit()
|
||||
{
|
||||
//
|
||||
// A gateway cuts what its transport cannot carry, which is about APIs. How many images the
|
||||
// model reads is a property of the model and survives the trip.
|
||||
//
|
||||
var directly = ModelRegistry.Shared.Profile(LLMProviders.ANTHROPIC, "claude-opus-5");
|
||||
var throughAGateway = ModelRegistry.Shared.Profile(LLMProviders.OPEN_ROUTER, "anthropic/claude-opus-5");
|
||||
|
||||
Assert.That(throughAGateway.Images, Is.EqualTo(directly.Images));
|
||||
}
|
||||
|
||||
[TestCase(null, null, null, Description = "Nobody stated either, so there is nothing to go by.")]
|
||||
[TestCase(8, null, 8)]
|
||||
[TestCase(null, 100, 100)]
|
||||
[TestCase(8, 100, 8, Description = "A message is part of a request, so the smaller of the two decides.")]
|
||||
[TestCase(100, 8, 8)]
|
||||
[TestCase(0, null, 0, Description = "Zero is a real answer: an operator can configure an engine to take no images at all.")]
|
||||
public void WhatMayTravelInOneMessageIsTheSmallerOfWhatIsKnown(int? perMessage, int? perRequest, int? expected)
|
||||
{
|
||||
var limits = new ImageLimits(perMessage, perRequest);
|
||||
|
||||
Assert.That(limits.MaxInOneMessage, Is.EqualTo(expected));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Live;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Live;
|
||||
|
||||
/// <summary>
|
||||
/// Checks what a running installation is allowed to say about the models it serves.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every test here builds its own store rather than using the shared one. What a provider reported
|
||||
/// is state which outlives a single question, and a test leaving some of it behind would decide
|
||||
/// what the next test sees.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ListedModelsTests
|
||||
{
|
||||
private const string ONE_MACHINE = "11111111-1111-1111-1111-111111111111";
|
||||
private const string ANOTHER_MACHINE = "22222222-2222-2222-2222-222222222222";
|
||||
private const string MODEL = "qwen3-32b";
|
||||
|
||||
/// <summary>
|
||||
/// A model the rules have something to say about, so that a report has something to contradict.
|
||||
/// </summary>
|
||||
private static readonly ModelProfile WHAT_THE_RULES_SAY = new()
|
||||
{
|
||||
Capabilities = Capability.TEXT_INPUT | Capability.TEXT_OUTPUT | Capability.MULTIPLE_IMAGE_INPUT,
|
||||
Kind = ModelKind.CHAT,
|
||||
Context = ContextWindow.Of(131_072, 262_144),
|
||||
Images = new(null, 20),
|
||||
};
|
||||
|
||||
[Test]
|
||||
public void AMachineWhichWasNeverAskedSaysNothing()
|
||||
{
|
||||
var listed = new ListedModels();
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(listed.Of(ONE_MACHINE, MODEL), Is.EqualTo(ModelListing.NOTHING));
|
||||
Assert.That(listed.Of(ONE_MACHINE, MODEL).IsKnown, Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void WhatOneMachineSaysIsNotWhatAnotherSays()
|
||||
{
|
||||
//
|
||||
// The same weights behind two engines, each started by somebody who decided for themselves.
|
||||
// This is the whole reason these numbers are kept per configured instance.
|
||||
//
|
||||
var listed = new ListedModels();
|
||||
listed.Report(ONE_MACHINE, [new(MODEL, ContextWindow.Of(32_768))]);
|
||||
listed.Report(ANOTHER_MACHINE, [new(MODEL, ContextWindow.Of(8_192))]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(listed.Of(ONE_MACHINE, MODEL).Context.DefaultTokens, Is.EqualTo(32_768));
|
||||
Assert.That(listed.Of(ANOTHER_MACHINE, MODEL).Context.DefaultTokens, Is.EqualTo(8_192));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void WhatAMachineNoLongerServesStopsAnswering()
|
||||
{
|
||||
var listed = new ListedModels();
|
||||
listed.Report(ONE_MACHINE, [new(MODEL, ContextWindow.Of(32_768)), new("gemma3-27b", ContextWindow.Of(16_384))]);
|
||||
listed.Report(ONE_MACHINE, [new("gemma3-27b", ContextWindow.Of(16_384))]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(listed.Of(ONE_MACHINE, MODEL), Is.EqualTo(ModelListing.NOTHING), "The engine was restarted without it, so nothing is known about it any more.");
|
||||
Assert.That(listed.Of(ONE_MACHINE, "gemma3-27b").Context.DefaultTokens, Is.EqualTo(16_384));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AMachineWhichHalvedItsWindowIsBelievedTheSecondTimeToo()
|
||||
{
|
||||
var listed = new ListedModels();
|
||||
listed.Report(ONE_MACHINE, [new(MODEL, ContextWindow.Of(32_768))]);
|
||||
listed.Report(ONE_MACHINE, [new(MODEL, ContextWindow.Of(16_384))]);
|
||||
|
||||
Assert.That(listed.Of(ONE_MACHINE, MODEL).Context.DefaultTokens, Is.EqualTo(16_384));
|
||||
}
|
||||
|
||||
[TestCase("Qwen3-32B")]
|
||||
[TestCase("qwen3-32b")]
|
||||
public void AModelSomebodyTypedIsStillTheSameModel(string asConfigured)
|
||||
{
|
||||
//
|
||||
// An organization writes the model of a provider into its configuration plugin by hand,
|
||||
// and the availability check already treats such a name as the same model whatever case it
|
||||
// was typed in. Being stricter here would leave exactly those people without the numbers.
|
||||
//
|
||||
var listed = new ListedModels();
|
||||
listed.Report(ONE_MACHINE, [new("qwen3-32b", ContextWindow.Of(32_768))]);
|
||||
|
||||
Assert.That(listed.Of(ONE_MACHINE, asConfigured).Context.DefaultTokens, Is.EqualTo(32_768));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnInstanceWithoutAnIdIsNothingToRemember()
|
||||
{
|
||||
var listed = new ListedModels();
|
||||
listed.Report(string.Empty, [new(MODEL, ContextWindow.Of(32_768))]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(listed.Of(string.Empty, MODEL), Is.EqualTo(ModelListing.NOTHING));
|
||||
Assert.That(listed.Of(ONE_MACHINE, MODEL), Is.EqualTo(ModelListing.NOTHING), "One nameless report does not become every machine's answer.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModelTheMachineSaidNothingAboutIsNotStored()
|
||||
{
|
||||
var listed = new ListedModels();
|
||||
listed.Report(ONE_MACHINE, [new(MODEL, ContextWindow.UNKNOWN), new(string.Empty, ContextWindow.Of(32_768))]);
|
||||
|
||||
Assert.That(listed.Of(ONE_MACHINE, MODEL), Is.EqualTo(ModelListing.NOTHING));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AReportedWindowReplacesTheWholeWindow()
|
||||
{
|
||||
var after = new ModelListing(MODEL, ContextWindow.Of(32_768)).ApplyTo(WHAT_THE_RULES_SAY);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(after.Context.DefaultTokens, Is.EqualTo(32_768));
|
||||
Assert.That(after.Context.RaisableToTokens, Is.Null, "What the weights could be raised to is not a number anybody reaches without restarting this engine.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AWindowSaysNothingAboutAnythingElse()
|
||||
{
|
||||
var after = new ModelListing(MODEL, ContextWindow.Of(32_768)).ApplyTo(WHAT_THE_RULES_SAY);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(after.Capabilities, Is.EqualTo(WHAT_THE_RULES_SAY.Capabilities));
|
||||
Assert.That(after.Images, Is.EqualTo(WHAT_THE_RULES_SAY.Images));
|
||||
Assert.That(after.Kind, Is.EqualTo(WHAT_THE_RULES_SAY.Kind));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void SayingNothingKeepsEverythingTheRulesWorkedOut()
|
||||
{
|
||||
Assert.That(ModelListing.NOTHING.ApplyTo(WHAT_THE_RULES_SAY), Is.EqualTo(WHAT_THE_RULES_SAY));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AWindowAProviderStatedIsTakenAsItIs()
|
||||
{
|
||||
Assert.That(ModelListing.For(MODEL, 32_768).Context.DefaultTokens, Is.EqualTo(32_768));
|
||||
}
|
||||
|
||||
[TestCase(0, TestName = "A window of no tokens")]
|
||||
[TestCase(-1, TestName = "A window of negative tokens")]
|
||||
[TestCase(null, TestName = "No window at all")]
|
||||
public void AWindowWhichIsNoWidthIsDroppedRatherThanRepaired(int? tokens)
|
||||
{
|
||||
//
|
||||
// Every dialect comes through this one factory, so a provider answering with something
|
||||
// nobody can interpret falls back to what the rules say -- and does so the same way for
|
||||
// all of them, rather than once per provider and slightly differently each time.
|
||||
//
|
||||
Assert.That(ModelListing.For(MODEL, tokens), Is.EqualTo(ModelListing.NOTHING));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnEntryWithoutANameIsNoListing()
|
||||
{
|
||||
Assert.That(ModelListing.For(string.Empty, 32_768), Is.EqualTo(ModelListing.NOTHING));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Matching;
|
||||
|
||||
/// <summary>
|
||||
/// Checks what a single pattern claims, before anything compares two of them.
|
||||
/// </summary>
|
||||
[TestFixture]
|
||||
public sealed class MatchPatternTests
|
||||
{
|
||||
[Test]
|
||||
public void APatternBoundToAProviderStaysSilentEverywhereElse()
|
||||
{
|
||||
//
|
||||
// On Alibaba, "qwq" is qwq-plus, a commercial model. Everywhere else it is the open weights
|
||||
// built on Qwen 2.5. Two different models, one name, and the binding is what tells them
|
||||
// apart without anybody writing an order.
|
||||
//
|
||||
var onAlibaba = new MatchPattern { Kind = MatchKind.SEGMENT, Text = "qwq", OnlyOn = LLMProviders.ALIBABA_CLOUD };
|
||||
var name = new ModelId("qwq-32b");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(onAlibaba.Matches(name, LLMProviders.ALIBABA_CLOUD, ModelVendor.UNKNOWN), Is.True);
|
||||
Assert.That(onAlibaba.Matches(name, LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void APatternBoundToAVendorStaysSilentWhenSomebodyElseBuiltTheModel()
|
||||
{
|
||||
var fromAnthropic = new MatchPattern { Kind = MatchKind.SEGMENT, Text = "claude", OnlyFrom = ModelVendor.ANTHROPIC };
|
||||
var name = new ModelId("claude-sonnet-4-0");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(fromAnthropic.Matches(name, LLMProviders.LITE_LLM, ModelVendor.ANTHROPIC), Is.True);
|
||||
Assert.That(fromAnthropic.Matches(name, LLMProviders.LITE_LLM, ModelVendor.UNKNOWN), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnExtraConditionHasToBeAWholeNamePartToo()
|
||||
{
|
||||
var withVision = new MatchPattern { Kind = MatchKind.SEGMENT, Text = "qwen3.8", AlsoContains = ["vl"] };
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(withVision.Matches(new ModelId("qwen3.8-27b-vl"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN), Is.True);
|
||||
Assert.That(withVision.Matches(new ModelId("qwen3.8-27b"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN), Is.False);
|
||||
Assert.That(withVision.Matches(new ModelId("qwen3.8-27b-vllm"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN), Is.False, "\"vl\" inside another name part is not the vision variant.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AForbiddenNamePartRulesAPatternOut()
|
||||
{
|
||||
//
|
||||
// Salamandra does not call functions, except for the variant which was built for it.
|
||||
//
|
||||
var withoutTools = new MatchPattern { Kind = MatchKind.SEGMENT, Text = "salamandra", NotContains = ["tools"] };
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(withoutTools.Matches(new ModelId("salamandra-7b-instruct"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN), Is.True);
|
||||
Assert.That(withoutTools.Matches(new ModelId("salamandra-7b-instruct-tools"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[TestCase("gpt-5.1", true)]
|
||||
[TestCase("qwen3.8-27b", true)]
|
||||
[TestCase("GPT-5.1", false)]
|
||||
[TestCase("gpt_5", false)]
|
||||
[TestCase("gpt 5", false)]
|
||||
[TestCase("-gpt-5", false)]
|
||||
[TestCase("gpt--5", false)]
|
||||
[TestCase("", false)]
|
||||
public void APatternHasToBeWrittenTheWayANameArrives(string text, bool expected) => Assert.That(MatchPattern.IsNormalized(text), Is.EqualTo(expected));
|
||||
|
||||
[Test]
|
||||
public void APatternWhichCannotMatchAnythingSaysSo()
|
||||
{
|
||||
var malformed = new MatchPattern { Kind = MatchKind.SEGMENT, Text = "gpt-5", AlsoContains = ["Codex"] };
|
||||
|
||||
Assert.That(malformed.IsWellFormed, Is.False);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TwoPatternsSayingTheSameThingInADifferentOrderHaveTheSameSignature()
|
||||
{
|
||||
var one = new MatchPattern { Kind = MatchKind.SEGMENT, Text = "llama", AlsoContains = ["instruct", "70b"] };
|
||||
var other = new MatchPattern { Kind = MatchKind.SEGMENT, Text = "llama", AlsoContains = ["70b", "instruct"] };
|
||||
|
||||
Assert.That(one.Signature(), Is.EqualTo(other.Signature()));
|
||||
}
|
||||
|
||||
[TestCase(MatchKind.EXACT, "deepseek-r1", "deepseek")]
|
||||
[TestCase(MatchKind.PREFIX, "gpt-5.1", "gpt")]
|
||||
[TestCase(MatchKind.SEGMENT, "qwen3.8", "qwen3.8")]
|
||||
[TestCase(MatchKind.SUBSTRING, "3.8", "")]
|
||||
public void ThePatternTellsTheIndexWhichNamePartToFileItUnder(MatchKind kind, string text, string expected)
|
||||
{
|
||||
var pattern = new MatchPattern { Kind = kind, Text = text };
|
||||
|
||||
Assert.That(pattern.IndexKey().ToString(), Is.EqualTo(expected));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,240 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Matching;
|
||||
|
||||
/// <summary>
|
||||
/// Checks that the index answers with the rule which says the most, whatever order it heard them in.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The cases below are the ones the old rules got wrong, or only got right because somebody kept
|
||||
/// the blocks in the right order by hand. There are no model families yet: the rules here are
|
||||
/// written out in the test, because what is being checked is the engine and not what it is fed.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelFamilyIndexTests
|
||||
{
|
||||
private const LLMProviders ANY_PROVIDER = LLMProviders.SELF_HOSTED;
|
||||
|
||||
[Test]
|
||||
public void TheRuleSayingMoreAboutANameWinsWithoutAnybodyOrderingTheRules()
|
||||
{
|
||||
//
|
||||
// This is the mistake the old rules made: the Llama block stood above the DeepSeek one, so
|
||||
// it answered for the R1 distills, which are Llama checkpoints fine-tuned on R1 answers and
|
||||
// reason where a plain Llama does not. Here neither rule knows about the other.
|
||||
//
|
||||
var llama = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT | Capability.FUNCTION_CALLING });
|
||||
var distill = Selector("deepseek-r1", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT | Capability.FUNCTION_CALLING, Reasoning = ReasoningSupport.ALWAYS });
|
||||
var name = new ModelId("deepseek-r1-distill-llama-70b");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(ModelFamilyIndex.Build([llama, distill]).Explain(name, ANY_PROVIDER, ModelVendor.UNKNOWN).Selector, Is.SameAs(distill));
|
||||
Assert.That(ModelFamilyIndex.Build([distill, llama]).Explain(name, ANY_PROVIDER, ModelVendor.UNKNOWN).Selector, Is.SameAs(distill), "The order the rules arrive in must not change the answer.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AVariantIsNotSwallowedByThePrefixItBeginsWith()
|
||||
{
|
||||
//
|
||||
// "gpt-5-chat-latest" is the alias for the GPT-5 which does not reason, and the old rules
|
||||
// told it that it always does, because the "gpt-5-" prefix claimed it first.
|
||||
//
|
||||
var reasoning = Selector("gpt-5", MatchKind.PREFIX, new() { Adds = Capability.TEXT_INPUT, Reasoning = ReasoningSupport.ALWAYS });
|
||||
var chat = Selector("gpt-5-chat", MatchKind.PREFIX, new() { Adds = Capability.TEXT_INPUT, Reasoning = ReasoningSupport.NONE });
|
||||
var index = ModelFamilyIndex.Build([reasoning, chat]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(index.Resolve(new ModelId("gpt-5-chat-latest"), ANY_PROVIDER, ModelVendor.UNKNOWN).Reasoning, Is.EqualTo(ReasoningSupport.NONE));
|
||||
Assert.That(index.Resolve(new ModelId("gpt-5-pro"), ANY_PROVIDER, ModelVendor.UNKNOWN).Reasoning, Is.EqualTo(ReasoningSupport.ALWAYS));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void APrefixDoesNotReachAcrossAVersionDot()
|
||||
{
|
||||
//
|
||||
// gpt-5 and gpt-5.1 are two models, and a rule written for one of them must not answer for
|
||||
// the other. Without this, every new point release would silently inherit the old answer.
|
||||
//
|
||||
var index = ModelFamilyIndex.Build([Selector("gpt-5", MatchKind.PREFIX, new() { Adds = Capability.TEXT_INPUT })]);
|
||||
|
||||
Assert.That(index.Explain(new ModelId("gpt-5.1"), ANY_PROVIDER, ModelVendor.UNKNOWN).IsKnown, Is.False);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheRuleWrittenForOneProviderWinsOnThatProviderOnly()
|
||||
{
|
||||
var openWeights = Selector("qwq", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT });
|
||||
var commercial = new ModelRule(
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "qwq", OnlyOn = LLMProviders.ALIBABA_CLOUD },
|
||||
ModelRuleKind.SELECTOR,
|
||||
new() { Adds = Capability.TEXT_INPUT | Capability.FUNCTION_CALLING },
|
||||
"test");
|
||||
|
||||
var index = ModelFamilyIndex.Build([openWeights, commercial]);
|
||||
var name = new ModelId("qwq-32b");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(index.Explain(name, LLMProviders.ALIBABA_CLOUD, ModelVendor.UNKNOWN).Selector, Is.SameAs(commercial));
|
||||
Assert.That(index.Explain(name, LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN).Selector, Is.SameAs(openWeights));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModifierAdjustsWhateverTheSelectorChose()
|
||||
{
|
||||
//
|
||||
// A base checkpoint was never instruction tuned, whatever family it comes from. In the old
|
||||
// rules that had to stand above everything else, which is why nothing below it could state
|
||||
// an exception.
|
||||
//
|
||||
var family = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT | Capability.FUNCTION_CALLING });
|
||||
var baseCheckpoint = Modifier("base", MatchKind.SEGMENT, new() { Removes = Capability.FUNCTION_CALLING });
|
||||
var index = ModelFamilyIndex.Build([family, baseCheckpoint]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(index.Resolve(new ModelId("llama-3.3-70b"), ANY_PROVIDER, ModelVendor.UNKNOWN).Has(Capability.FUNCTION_CALLING), Is.True);
|
||||
Assert.That(index.Resolve(new ModelId("llama-3.3-70b-base"), ANY_PROVIDER, ModelVendor.UNKNOWN).Has(Capability.FUNCTION_CALLING), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheModifierSayingMoreHasTheLastWord()
|
||||
{
|
||||
var family = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT });
|
||||
var broad = Modifier("instruct", MatchKind.SEGMENT, new() { Adds = Capability.FUNCTION_CALLING });
|
||||
var narrow = Modifier("instruct-nano", MatchKind.SEGMENT, new() { Removes = Capability.FUNCTION_CALLING });
|
||||
var resolution = ModelFamilyIndex.Build([narrow, broad, family]).Explain(new ModelId("llama-3.3-instruct-nano"), ANY_PROVIDER, ModelVendor.UNKNOWN);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(resolution.Modifiers.Select(modifier => modifier.Pattern.Text), Is.EqualTo(new[] { "instruct", "instruct-nano" }));
|
||||
Assert.That(resolution.Profile.Has(Capability.FUNCTION_CALLING), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModifierAppliesOnceEvenWhenTheNameRepeatsThePartItWasFoundUnder()
|
||||
{
|
||||
var family = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT });
|
||||
var modifier = Modifier("llama", MatchKind.SEGMENT, new() { Adds = Capability.FUNCTION_CALLING });
|
||||
var resolution = ModelFamilyIndex.Build([family, modifier]).Explain(new ModelId("meta-llama/llama-3.3-70b"), ANY_PROVIDER, ModelVendor.UNKNOWN);
|
||||
|
||||
Assert.That(resolution.Modifiers, Has.Count.EqualTo(1));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ARuleWhichCannotBeFiledUnderANamePartIsStillAsked()
|
||||
{
|
||||
//
|
||||
// A substring pattern may begin in the middle of a name part, so the index cannot narrow it
|
||||
// down and has to check it against every name. Getting that wrong would make such a rule
|
||||
// silently never fire.
|
||||
//
|
||||
var version = Selector("3.8", MatchKind.SUBSTRING, new() { Adds = Capability.TEXT_INPUT });
|
||||
var index = ModelFamilyIndex.Build([version]);
|
||||
|
||||
Assert.That(index.Explain(new ModelId("qwen3.8-27b"), ANY_PROVIDER, ModelVendor.UNKNOWN).Selector, Is.SameAs(version));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TwoRulesClaimingANameWithTheSameRightAreReportedAndStillAnsweredTheSameWay()
|
||||
{
|
||||
var one = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT }, "family-a");
|
||||
var other = Selector("qwen3", MatchKind.SEGMENT, new() { Adds = Capability.WEB_SEARCH }, "family-b");
|
||||
var name = new ModelId("llama-qwen3-merge");
|
||||
|
||||
var oneWay = ModelFamilyIndex.Build([one, other]).Explain(name, ANY_PROVIDER, ModelVendor.UNKNOWN);
|
||||
var otherWay = ModelFamilyIndex.Build([other, one]).Explain(name, ANY_PROVIDER, ModelVendor.UNKNOWN);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(oneWay.IsAmbiguous, Is.True);
|
||||
Assert.That(otherWay.IsAmbiguous, Is.True);
|
||||
Assert.That(oneWay.Selector, Is.SameAs(otherWay.Selector), "Which of the two answers must not depend on the order they arrived in.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TwoRulesClaimingExactlyTheSameNamesAreFoundWhenTheIndexIsBuilt()
|
||||
{
|
||||
var one = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT }, "family-a");
|
||||
var other = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.WEB_SEARCH }, "family-b");
|
||||
|
||||
Assert.That(ModelFamilyIndex.Build([one, other]).Ambiguities, Has.Count.EqualTo(1));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModifierMayShareItsPatternWithASelector()
|
||||
{
|
||||
//
|
||||
// Only selectors compete for a name; a modifier saying something about the same names is
|
||||
// the normal case and must not be reported as a conflict.
|
||||
//
|
||||
var selector = Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT });
|
||||
var modifier = Modifier("llama", MatchKind.SEGMENT, new() { Adds = Capability.FUNCTION_CALLING });
|
||||
|
||||
Assert.That(ModelFamilyIndex.Build([selector, modifier]).Ambiguities, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANameNoRuleKnowsIsAnsweredWithNothingKnown()
|
||||
{
|
||||
var index = ModelFamilyIndex.Build([Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT })]);
|
||||
var resolution = index.Explain(new ModelId("something-nobody-wrote-a-rule-for"), ANY_PROVIDER, ModelVendor.UNKNOWN);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(resolution.IsKnown, Is.False);
|
||||
Assert.That(resolution.Profile, Is.EqualTo(ModelProfile.UNKNOWN));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANameWhichIsNothingIsNotEvenAsked()
|
||||
{
|
||||
var index = ModelFamilyIndex.Build([Selector("llama", MatchKind.SEGMENT, new() { Adds = Capability.TEXT_INPUT })]);
|
||||
|
||||
Assert.That(index.Explain(new ModelId(" "), ANY_PROVIDER, ModelVendor.UNKNOWN), Is.SameAs(ModelResolution.NOTHING));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AnIndexWithoutAnyRulesAnswersInsteadOfFailing()
|
||||
{
|
||||
//
|
||||
// An index over no rules has no comparer to look name parts up with. It has nothing to look
|
||||
// up either, so it has to say so rather than throw on the first question.
|
||||
//
|
||||
var index = ModelFamilyIndex.Build([]);
|
||||
|
||||
Assert.That(index.Explain(new ModelId("gpt-5.1"), ANY_PROVIDER, ModelVendor.UNKNOWN).IsKnown, Is.False);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ARuleNotWrittenInTheFormANameArrivesInIsVisibleToWhoeverAsks()
|
||||
{
|
||||
//
|
||||
// A name is lowercased on its way in, so a pattern carrying a capital letter can never
|
||||
// match anything. That is a mistake, not a rule which happens to stay quiet, and it has to
|
||||
// be findable by reading the rules rather than by noticing a model behaving oddly.
|
||||
//
|
||||
var index = ModelFamilyIndex.Build([Selector("GPT-5", MatchKind.PREFIX, new() { Adds = Capability.TEXT_INPUT })]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(index.Rules.Where(rule => !rule.Pattern.IsWellFormed), Is.Not.Empty);
|
||||
Assert.That(index.Explain(new ModelId("gpt-5.1"), ANY_PROVIDER, ModelVendor.UNKNOWN).IsKnown, Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
private static ModelRule Selector(string text, MatchKind kind, ModelProfileChange change, string origin = "test") => new(new() { Kind = kind, Text = text }, ModelRuleKind.SELECTOR, change, origin);
|
||||
|
||||
private static ModelRule Modifier(string text, MatchKind kind, ModelProfileChange change, string origin = "test") => new(new() { Kind = kind, Text = text }, ModelRuleKind.MODIFIER, change, origin);
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
using AIStudio.Models.Matching;
|
||||
|
||||
namespace AIStudio.Tests.Models.Matching;
|
||||
|
||||
/// <summary>
|
||||
/// Checks that every provider's way of writing a name arrives in the one form the rules are in.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The spellings below are not invented. They are the ones the old rules had to spell out over and
|
||||
/// over, and the ones its comments quote: an Ollama tag, a Fireworks path, a hub prefix, and the
|
||||
/// whole sentence Blablador answers with.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelIdTests
|
||||
{
|
||||
[TestCase("gpt-5.1", "gpt-5.1")]
|
||||
[TestCase("GPT-5.1", "gpt-5.1")]
|
||||
[TestCase("qwen3.8:27b-mlx", "qwen3.8-27b-mlx")]
|
||||
[TestCase("accounts/fireworks/models/llama-v3p1-405b-instruct", "accounts-fireworks-models-llama-v3p1-405b-instruct")]
|
||||
[TestCase("meta-llama/Llama-3.3-70B-Instruct", "meta-llama-llama-3.3-70b-instruct")]
|
||||
[TestCase("10 - Muse Glimmer 30b - the newest META model", "10-muse-glimmer-30b-the-newest-meta-model")]
|
||||
[TestCase("anthropic.claude-3-5-sonnet-20241022-v2:0", "anthropic.claude-3-5-sonnet-20241022-v2-0")]
|
||||
public void ANameArrivesInTheFormTheRulesAreWrittenIn(string reported, string expected) => Assert.That(new ModelId(reported).Normalized, Is.EqualTo(expected));
|
||||
|
||||
[TestCase("")]
|
||||
[TestCase(" ")]
|
||||
[TestCase("---")]
|
||||
[TestCase(" / : - ")]
|
||||
public void ANameWhichIsNothingButSeparatorsIsEmpty(string reported)
|
||||
{
|
||||
var id = new ModelId(reported);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(id.IsEmpty, Is.True);
|
||||
Assert.That(id.Normalized, Is.Empty);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANameKeepsTheSpellingAPersonSees()
|
||||
{
|
||||
var id = new ModelId("Qwen3.8:27B-MLX");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(id.Original, Is.EqualTo("Qwen3.8:27B-MLX"));
|
||||
Assert.That(id.ToString(), Is.EqualTo("Qwen3.8:27B-MLX"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ADefaultModelIdIsEmptyRatherThanBroken()
|
||||
{
|
||||
ModelId untouched = default;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(untouched.IsEmpty, Is.True);
|
||||
Assert.That(untouched.Original, Is.Empty);
|
||||
Assert.That(untouched.Normalized, Is.Empty);
|
||||
Assert.That(untouched.Segments.GetEnumerator().MoveNext(), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TwoNamesWrittenDifferentlyAreTheSameName()
|
||||
{
|
||||
var fromOllama = new ModelId("Qwen3.8:27b");
|
||||
var fromHub = new ModelId("qwen3.8-27b");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(fromOllama, Is.EqualTo(fromHub));
|
||||
Assert.That(fromOllama.GetHashCode(), Is.EqualTo(fromHub.GetHashCode()));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANameIsWalkedOneNamePartAtATime()
|
||||
{
|
||||
var parts = new List<string>();
|
||||
foreach (var part in new ModelId("deepseek-r1-distill-llama-70b").Segments)
|
||||
parts.Add(part.ToString());
|
||||
|
||||
Assert.That(parts, Is.EqualTo(new[] { "deepseek", "r1", "distill", "llama", "70b" }));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AVersionDotDoesNotStartANewNamePart()
|
||||
{
|
||||
//
|
||||
// llama3 and llama3.1 are different models and only the latter calls functions, so the dot
|
||||
// has to stay inside the part rather than cut it in two.
|
||||
//
|
||||
var parts = new List<string>();
|
||||
foreach (var part in new ModelId("qwen3.8:27b").Segments)
|
||||
parts.Add(part.ToString());
|
||||
|
||||
Assert.That(parts, Is.EqualTo(new[] { "qwen3.8", "27b" }));
|
||||
}
|
||||
|
||||
[TestCase("gpt-5-chat-latest", "gpt-5", true)]
|
||||
[TestCase("gpt-55-turbo", "gpt-5", false)]
|
||||
[TestCase("gpt-5.1", "gpt-5", false)]
|
||||
[TestCase("gpt-5", "gpt-5", true)]
|
||||
[TestCase("gpt-5.1-codex", "gpt-5.1", true)]
|
||||
public void ANameBeginsWithATextOnlyWhenANamePartEndsThere(string name, string text, bool expected) => Assert.That(new ModelId(name).StartsWithSegments(text), Is.EqualTo(expected));
|
||||
|
||||
[TestCase("deepseek-r1-distill-llama-70b", "llama", true)]
|
||||
[TestCase("deepseek-r1-distill-llama-70b", "deepseek-r1", true)]
|
||||
[TestCase("meta-llama-llama-3.3-70b-instruct", "llama", true)]
|
||||
[TestCase("yi-34b-chat", "yi", true)]
|
||||
[TestCase("granite-embedding-278m", "yi", false)]
|
||||
[TestCase("qwen3.8-27b", "qwen3", false)]
|
||||
[TestCase("nvidia-nemotron-3.5-lightning-30b-a3b-nvfp4", "v", false)]
|
||||
public void ATextIsFoundInANameOnlyBetweenTwoNamePartBoundaries(string name, string text, bool expected) => Assert.That(new ModelId(name).ContainsSegments(text), Is.EqualTo(expected));
|
||||
|
||||
[Test]
|
||||
public void ATextIsFoundAtALaterBoundaryWhenTheFirstOccurrenceSitsInsideANamePart()
|
||||
{
|
||||
//
|
||||
// The first "llama" here sits inside "meta-llama"; the rule still has to find the one which
|
||||
// stands on its own.
|
||||
//
|
||||
Assert.That(new ModelId("metallama/llama-3.3-70b").ContainsSegments("llama"), Is.True);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ATextIsFoundAnywhereWhenTheRuleAsksForThat() => Assert.That(new ModelId("qwen3.8-27b").ContainsText("3.8"), Is.True);
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Matching;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the order in which the criteria are weighed against each other.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Each test below changes exactly one criterion and leaves the others equal, which is the only way
|
||||
/// to state what beats what. The order itself is the decision this whole rebuild rests on, so it is
|
||||
/// written down here rather than left to be inferred from how the rules happen to behave.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class RuleSpecificityTests
|
||||
{
|
||||
[Test]
|
||||
public void NamingTheWholeModelBeatsNamingHowItsNameBegins() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.EXACT, Text = "gpt-5" },
|
||||
new() { Kind = MatchKind.PREFIX, Text = "gpt-5" });
|
||||
|
||||
[Test]
|
||||
public void NamingHowANameBeginsBeatsNamingAPartOfIt() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.PREFIX, Text = "gpt-5" },
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "gpt-5" });
|
||||
|
||||
[Test]
|
||||
public void NamingAWholeNamePartBeatsAppearingSomewhereInside() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "gpt-5" },
|
||||
new() { Kind = MatchKind.SUBSTRING, Text = "gpt-5" });
|
||||
|
||||
[Test]
|
||||
public void SpellingOutMoreOfTheNameBeatsSpellingOutLess() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "deepseek-r1" },
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "llama" });
|
||||
|
||||
[Test]
|
||||
public void RequiringAFurtherNamePartBeatsNotRequiringOne() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "llama", AlsoContains = ["vision"] },
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "llama" });
|
||||
|
||||
[Test]
|
||||
public void BeingWrittenForOneProviderBeatsHoldingEverywhere() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "qwq", OnlyOn = LLMProviders.ALIBABA_CLOUD },
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "qwq" });
|
||||
|
||||
[Test]
|
||||
public void BeingWrittenForBothAProviderAndAVendorBeatsEitherAlone() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "qwq", OnlyOn = LLMProviders.ALIBABA_CLOUD, OnlyFrom = ModelVendor.ALIBABA },
|
||||
new() { Kind = MatchKind.SEGMENT, Text = "qwq", OnlyOn = LLMProviders.ALIBABA_CLOUD });
|
||||
|
||||
[Test]
|
||||
public void AHandWrittenRankOverrulesEverythingTheComputationWouldSay()
|
||||
{
|
||||
//
|
||||
// The emergency exit has to leave the building. A rank which the length of some other
|
||||
// pattern can overrule would not rescue the case it was written for, so it is weighed
|
||||
// before every computed criterion rather than after them.
|
||||
//
|
||||
AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.SUBSTRING, Text = "r1", ExplicitRank = 1 },
|
||||
new() { Kind = MatchKind.EXACT, Text = "deepseek-r1-distill-llama-70b" });
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANegativeRankPushesARuleBehindEverythingElse() => AssertMoreSpecific(
|
||||
new() { Kind = MatchKind.SUBSTRING, Text = "r1" },
|
||||
new() { Kind = MatchKind.EXACT, Text = "deepseek-r1", ExplicitRank = -1 });
|
||||
|
||||
[Test]
|
||||
public void TwoRulesSayingTheSameAmountAreEqual()
|
||||
{
|
||||
var one = RuleSpecificity.Of(new() { Kind = MatchKind.SEGMENT, Text = "llama" });
|
||||
var other = RuleSpecificity.Of(new() { Kind = MatchKind.SEGMENT, Text = "qwen3" });
|
||||
|
||||
Assert.That(one.CompareTo(other), Is.Zero);
|
||||
}
|
||||
|
||||
private static void AssertMoreSpecific(MatchPattern expectedWinner, MatchPattern expectedLoser)
|
||||
{
|
||||
var winner = RuleSpecificity.Of(expectedWinner);
|
||||
var loser = RuleSpecificity.Of(expectedLoser);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(winner.CompareTo(loser), Is.GreaterThan(0));
|
||||
Assert.That(loser.CompareTo(winner), Is.LessThan(0), "The comparison has to say the same thing in both directions.");
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Models.Mistral;
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Mistral;
|
||||
|
||||
/// <summary>
|
||||
/// Checks how a Mistral release is read out of a model name.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// This is the one place in the rebuilt rules where a capability is calculated rather than stated.
|
||||
/// Mistral names its models after the month they came out, and the same family name stands for
|
||||
/// models which can and cannot see -- so nothing a pattern could match on tells them apart.
|
||||
///
|
||||
/// What makes it worth its own tests is that model names are full of four-digit numbers which are
|
||||
/// not dates: parameter counts, context sizes, versions. Reading one of those as a release would
|
||||
/// silently promise image input for a model which has none.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class MistralReleasesTests
|
||||
{
|
||||
/// <summary>
|
||||
/// A release far enough in the future that no name in these tests reaches it by accident.
|
||||
/// </summary>
|
||||
private const int SOME_LATEST_RELEASE = 2604;
|
||||
|
||||
[TestCase("mistral-large-2512", ExpectedResult = 2512, TestName = "The release is read from the end of the name")]
|
||||
[TestCase("ministral-14b-2512", ExpectedResult = 2512, TestName = "The size of a model is not its release")]
|
||||
[TestCase("ministral-8b-2410", ExpectedResult = 2410, TestName = "A one digit size next to the release is not part of it")]
|
||||
[TestCase("something-25120", ExpectedResult = MistralReleases.UNKNOWN, TestName = "A five digit block is not a release")]
|
||||
[TestCase("something-125120", ExpectedResult = MistralReleases.UNKNOWN, TestName = "A six digit block is not a release either")]
|
||||
[TestCase("something-1912", ExpectedResult = MistralReleases.UNKNOWN, TestName = "A year before Mistral named models after dates is not a release")]
|
||||
[TestCase("something-2513", ExpectedResult = MistralReleases.UNKNOWN, TestName = "A thirteenth month is not a release")]
|
||||
[TestCase("something-2500", ExpectedResult = MistralReleases.UNKNOWN, TestName = "A zeroth month is not a release")]
|
||||
[TestCase("open-mistral-nemo", ExpectedResult = MistralReleases.UNKNOWN, TestName = "A name without any number carries no release")]
|
||||
public int TheReleaseIsReadOnlyWhereThereIsOne(string modelId) => MistralReleases.Of(new ModelId(modelId), SOME_LATEST_RELEASE);
|
||||
|
||||
[Test]
|
||||
public void TheLatestAliasBecomesWhateverItsFamilyPointsAt()
|
||||
{
|
||||
var release = MistralReleases.Of(new ModelId("mistral-large-latest"), SOME_LATEST_RELEASE);
|
||||
|
||||
Assert.That(release, Is.EqualTo(SOME_LATEST_RELEASE));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AMarketingVersionBecomesTheReleaseItStandsFor()
|
||||
{
|
||||
//
|
||||
// And the more specific one has to win: read as plain text rather than as patterns, a rule
|
||||
// for "mistral-medium-3" would otherwise answer for "mistral-medium-3.5" as well and place
|
||||
// it eleven months too early, before the release which gave it reasoning.
|
||||
//
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(MistralReleases.Of(new ModelId("mistral-medium-3"), SOME_LATEST_RELEASE), Is.EqualTo(2505));
|
||||
Assert.That(MistralReleases.Of(new ModelId("mistral-medium-3.5"), SOME_LATEST_RELEASE), Is.EqualTo(2604));
|
||||
Assert.That(MistralReleases.Of(new ModelId("mistral-medium-3-5"), SOME_LATEST_RELEASE), Is.EqualTo(2604), "Mistral writes the version separator both ways for the same model.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void OneFamilyAnswersDifferentlyForTwoOfItsOwnReleases()
|
||||
{
|
||||
//
|
||||
// The point of the whole calculation, in one assertion: both names select the same family
|
||||
// and the same rule, and the model differs.
|
||||
//
|
||||
var beforeItCouldSee = ModelRegistry.Shared.Profile(LLMProviders.MISTRAL, "mistral-large-2411");
|
||||
var afterwards = ModelRegistry.Shared.Profile(LLMProviders.MISTRAL, "mistral-large-2512");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(beforeItCouldSee.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.False);
|
||||
Assert.That(beforeItCouldSee.Reasoning, Is.EqualTo(ReasoningSupport.NONE));
|
||||
|
||||
Assert.That(afterwards.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.True);
|
||||
Assert.That(afterwards.Reasoning, Is.EqualTo(ReasoningSupport.OPTIONAL));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AReleaseWhichCannotBeReadGrantsNothing()
|
||||
{
|
||||
//
|
||||
// The safe direction: offering an ability the model does not have makes the request fail,
|
||||
// while a missing one can be handed back by a person through the expert settings.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.MISTRAL, "mistral-large-whenever");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Has(Capability.FUNCTION_CALLING), Is.True, "What the family could always do is still stated.");
|
||||
Assert.That(profile.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.False);
|
||||
Assert.That(profile.Reasoning, Is.EqualTo(ReasoningSupport.NONE));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AFamilyWhichNeverReasonedDoesNotStartWithItsNewestRelease()
|
||||
{
|
||||
var newest = ModelRegistry.Shared.Profile(LLMProviders.MISTRAL, "ministral-3b-latest");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(newest.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.True, "Ministral 3 reads images.");
|
||||
Assert.That(newest.Reasoning, Is.EqualTo(ReasoningSupport.NONE), "No Ministral reasons, whatever its release.");
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
using AIStudio.Models;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the three types which have to be able to say "nobody knows".
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// They are tested together because they are tested for the same thing. Each of them is a value
|
||||
/// type sitting inside a model profile, so each of them has a default value somebody will read
|
||||
/// before anything was written into it, and that default has to mean unknown rather than zero. The
|
||||
/// day one of them answers "a context window of zero tokens" instead, a feature built on top of it
|
||||
/// will quietly do the wrong thing.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelFactsTests
|
||||
{
|
||||
[Test]
|
||||
public void AContextWindowNobodyWroteDownIsUnknown()
|
||||
{
|
||||
ContextWindow untouched = default;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(untouched.IsKnown, Is.False);
|
||||
Assert.That(untouched, Is.EqualTo(ContextWindow.UNKNOWN));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AContextWindowStatesWhatItShipsWithAndWhatItCanBeRaisedTo()
|
||||
{
|
||||
var window = ContextWindow.Of(128_000, 1_000_000);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(window.IsKnown, Is.True);
|
||||
Assert.That(window.DefaultTokens, Is.EqualTo(128_000));
|
||||
Assert.That(window.RaisableToTokens, Is.EqualTo(1_000_000));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AContextWindowWhichCannotBeRaisedSaysSoWithNothingRatherThanWithItsOwnSize()
|
||||
{
|
||||
var window = ContextWindow.Of(32_768);
|
||||
|
||||
Assert.That(window.RaisableToTokens, Is.Null);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AContextWindowOfNoTokensCannotBeStated() => Assert.Throws<ArgumentOutOfRangeException>(() => ContextWindow.Of(0));
|
||||
|
||||
[Test]
|
||||
public void AContextWindowCannotBeRaisedToLessThanItAlreadyIs() => Assert.Throws<ArgumentOutOfRangeException>(() => ContextWindow.Of(128_000, 32_768));
|
||||
|
||||
[Test]
|
||||
public void ATokenizerNobodyWroteDownIsTheBuiltInOne()
|
||||
{
|
||||
TokenizerRef untouched = default;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(untouched.IsKnown, Is.False);
|
||||
Assert.That(untouched.Kind, Is.EqualTo(TokenizerKind.UNKNOWN));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ATokenizerWithoutANameIsNotKnownEvenWhenItsKindIs()
|
||||
{
|
||||
var nameless = new TokenizerRef(TokenizerKind.HUGGING_FACE, string.Empty);
|
||||
|
||||
Assert.That(nameless.IsKnown, Is.False);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ImageLimitsNobodyWroteDownAreUnknown()
|
||||
{
|
||||
ImageLimits untouched = default;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(untouched.IsKnown, Is.False);
|
||||
Assert.That(untouched.MaxPerMessage, Is.Null);
|
||||
Assert.That(untouched.MaxPerRequest, Is.Null);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ImageLimitsTellNoImagesApartFromNobodyHavingSaid()
|
||||
{
|
||||
var noImages = new ImageLimits(MaxPerMessage: 0, MaxPerRequest: null);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(noImages.IsKnown, Is.True, "Zero images is a statement an operator can make.");
|
||||
Assert.That(noImages.MaxPerMessage, Is.EqualTo(0));
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,285 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks how a family states its rules, and what a variant inherits from the family it belongs to.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Inheritance here happens while the rules are being built, not while a name is being answered. A
|
||||
/// variant takes what its family stated and goes on from there, and what comes out is one complete
|
||||
/// rule -- so at runtime there is still exactly one selector winning, and the specificity remains
|
||||
/// the only thing deciding which.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelFamilyTests
|
||||
{
|
||||
[Test]
|
||||
public void AFamilyNamesItselfAsTheOriginOfItsRules()
|
||||
{
|
||||
var family = new SampleFamily();
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(family.Name, Is.EqualTo(nameof(SampleFamily)));
|
||||
Assert.That(family.Rules.Select(rule => rule.Origin), Is.All.EqualTo(nameof(SampleFamily)));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AFamilyStatesItsRulesOnlyOnce()
|
||||
{
|
||||
var family = new SampleFamily();
|
||||
var whenFirstAsked = family.Rules;
|
||||
var whenAskedAgain = family.Rules;
|
||||
|
||||
Assert.That(whenAskedAgain, Is.SameAs(whenFirstAsked));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ARuleIsAboutWholeNamePartsUnlessItSaysOtherwise()
|
||||
{
|
||||
var family = new PlainFamily();
|
||||
|
||||
Assert.That(family.Rules.Single().Pattern.Kind, Is.EqualTo(MatchKind.SEGMENT));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AVariantKeepsEverythingItsFamilyStatedAndOnlyChangesWhatItSays()
|
||||
{
|
||||
var index = ModelFamilyIndex.Build(new SampleFamily().Rules);
|
||||
var codex = index.Resolve(new ModelId("gpt-5.1-codex-max"), LLMProviders.OPEN_AI, ModelVendor.OPEN_AI);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(codex.Has(Capability.WEB_SEARCH), Is.False, "This is the one thing the variant takes away.");
|
||||
Assert.That(codex.Has(Capability.TEXT_INPUT | Capability.MULTIPLE_IMAGE_INPUT | Capability.FUNCTION_CALLING), Is.True);
|
||||
Assert.That(codex.Reasoning, Is.EqualTo(ReasoningSupport.OPTIONAL));
|
||||
Assert.That(codex.Context.DefaultTokens, Is.EqualTo(400_000));
|
||||
Assert.That(codex.Tokenizer.Id, Is.EqualTo("o200k_base"));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void WhatAVariantTakesAwayIsNotTakenAwayFromTheFamily()
|
||||
{
|
||||
var index = ModelFamilyIndex.Build(new SampleFamily().Rules);
|
||||
var plain = index.Resolve(new ModelId("gpt-5.1-mini"), LLMProviders.OPEN_AI, ModelVendor.OPEN_AI);
|
||||
|
||||
Assert.That(plain.Has(Capability.WEB_SEARCH), Is.True);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AVariantCanHandBackWhatItsFamilyTookAway()
|
||||
{
|
||||
var index = ModelFamilyIndex.Build(new FamilyWhichTakesSomethingBack().Rules);
|
||||
var withTools = index.Resolve(new ModelId("thing-with-tools"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN);
|
||||
|
||||
Assert.That(withTools.Has(Capability.FUNCTION_CALLING), Is.True);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AVariantMayNameTheRuleItInheritsFromInsteadOfTakingTheOneBefore()
|
||||
{
|
||||
var index = ModelFamilyIndex.Build(new FamilyWithTwoGenerations().Rules);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(index.Resolve(new ModelId("thing3-mini"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN).Reasoning, Is.EqualTo(ReasoningSupport.ALWAYS));
|
||||
Assert.That(index.Resolve(new ModelId("thing4"), LLMProviders.SELF_HOSTED, ModelVendor.UNKNOWN).Reasoning, Is.EqualTo(ReasoningSupport.NONE));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AFirstRuleHasNothingToInheritFromAndSaysSo()
|
||||
{
|
||||
var family = new FamilyInheritingFromNothing();
|
||||
var refused = Assert.Throws<InvalidOperationException>(() => _ = family.Rules);
|
||||
|
||||
Assert.That(refused?.Message, Does.Contain("first rule"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void InheritingFromARuleWhichWasNeverStatedSaysSo()
|
||||
{
|
||||
var family = new FamilyInheritingFromSomethingMissing();
|
||||
var refused = Assert.Throws<InvalidOperationException>(() => _ = family.Rules);
|
||||
|
||||
Assert.That(refused?.Message, Does.Contain("does not state"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void InheritingFromARuleTextWhichNamesTwoRulesSaysSo()
|
||||
{
|
||||
//
|
||||
// Stating one text twice is ordinary: a variant of a generation is written as the same
|
||||
// pattern with a condition on top. What cannot be done afterwards is naming that text to
|
||||
// inherit from, because it no longer names one rule -- and taking whichever came last
|
||||
// would be a coin toss nobody sees.
|
||||
//
|
||||
var family = new FamilyStatingOneTextTwice();
|
||||
var refused = Assert.Throws<InvalidOperationException>(() => _ = family.Rules);
|
||||
|
||||
Assert.That(refused?.Message, Does.Contain("more than once"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ARankNobodyAccountedForIsRefused()
|
||||
{
|
||||
//
|
||||
// The rank is the way past everything the specificity computes, and the sentence next to it
|
||||
// is the only thing keeping it accountable. The compiler asks for that sentence; this is
|
||||
// what keeps an empty one from passing for it, because a number without an explanation
|
||||
// reads as noise to whoever comes next -- and noise is what the computation replaced.
|
||||
//
|
||||
var family = new FamilyRankingWithoutSayingWhy();
|
||||
var refused = Assert.Throws<ArgumentException>(() => _ = family.Rules);
|
||||
|
||||
Assert.That(refused?.Message, Does.Contain("specificity gets wrong"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AFamilyWhichAdjustsRatherThanChoosesStatesAModifier()
|
||||
{
|
||||
var family = new FamilyWithAModifier();
|
||||
|
||||
Assert.That(family.Rules.Single().Kind, Is.EqualTo(ModelRuleKind.MODIFIER));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AFamilyLeavesTheProfileAloneUnlessItSaysItRefinesIt()
|
||||
{
|
||||
var family = new PlainFamily();
|
||||
var profile = new ModelProfile { Capabilities = Capability.TEXT_INPUT };
|
||||
|
||||
Assert.That(family.Refine(new ModelId("thing"), profile), Is.EqualTo(profile));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ASourceWithoutAPageOrADayIsNotAStatement()
|
||||
{
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(new SampleFamily().Source.IsStated, Is.True);
|
||||
Assert.That(new ModelSource(string.Empty, new DateOnly(2026, 9, 11), "a note").IsStated, Is.False);
|
||||
Assert.That(new ModelSource("https://example.invalid", default, "a note").IsStated, Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// The family from the plan, written the way a real one will be.
|
||||
/// </summary>
|
||||
private sealed class SampleFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.OPEN_AI;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/gpt-5.1", new DateOnly(2026, 9, 11), "Made up for this test, so that no real page is claimed to have been read.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder)
|
||||
{
|
||||
builder.Rule("gpt-5.1").AsPrefix()
|
||||
.Capabilities(Capability.TEXT_INPUT | Capability.MULTIPLE_IMAGE_INPUT | Capability.TEXT_OUTPUT | Capability.FUNCTION_CALLING | Capability.WEB_SEARCH)
|
||||
.Apis(Capability.RESPONSES_API | Capability.CHAT_COMPLETION_API)
|
||||
.Reasoning(ReasoningSupport.OPTIONAL)
|
||||
.ContextWindow(400_000)
|
||||
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
|
||||
|
||||
builder.Rule("gpt-5.1-codex").AsPrefix().Inherits().Removes(Capability.WEB_SEARCH);
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class PlainFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/plain", new DateOnly(2026, 9, 11), "A family stating one rule and nothing else.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("thing").Capabilities(Capability.TEXT_INPUT);
|
||||
}
|
||||
|
||||
private sealed class FamilyWhichTakesSomethingBack : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/back", new DateOnly(2026, 9, 11), "A family whose variant regains what the family lacks.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder)
|
||||
{
|
||||
builder.Rule("thing").Capabilities(Capability.TEXT_INPUT).Removes(Capability.FUNCTION_CALLING);
|
||||
builder.Rule("thing-with-tools").Inherits().Capabilities(Capability.FUNCTION_CALLING);
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class FamilyWithTwoGenerations : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/generations", new DateOnly(2026, 9, 11), "A family with two generations which reason differently.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder)
|
||||
{
|
||||
builder.Rule("thing3").AsPrefix().Capabilities(Capability.TEXT_INPUT).Reasoning(ReasoningSupport.ALWAYS);
|
||||
builder.Rule("thing4").AsPrefix().Capabilities(Capability.TEXT_INPUT).Reasoning(ReasoningSupport.NONE);
|
||||
|
||||
// Naming the generation rather than taking whatever stands above, which here is the
|
||||
// other one:
|
||||
builder.Rule("thing3-mini").AsPrefix().InheritsFrom("thing3");
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class FamilyInheritingFromNothing : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/nothing", new DateOnly(2026, 9, 11), "A family whose first rule inherits.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("thing").Inherits();
|
||||
}
|
||||
|
||||
private sealed class FamilyInheritingFromSomethingMissing : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/missing", new DateOnly(2026, 9, 11), "A family inheriting from a rule it never states.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder)
|
||||
{
|
||||
builder.Rule("thing").Capabilities(Capability.TEXT_INPUT);
|
||||
builder.Rule("thing-mini").InheritsFrom("something-else");
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class FamilyStatingOneTextTwice : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/twice", new DateOnly(2026, 9, 11), "A family stating one pattern text twice and then naming it to inherit from.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder)
|
||||
{
|
||||
builder.Rule("thing").Capabilities(Capability.TEXT_INPUT);
|
||||
builder.Rule("thing").AlsoContains("special").Capabilities(Capability.FUNCTION_CALLING);
|
||||
builder.Rule("thing-mini").InheritsFrom("thing");
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class FamilyRankingWithoutSayingWhy : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/rank", new DateOnly(2026, 9, 13), "A family moving one of its rules by hand without saying what it moves it past.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("thing").Rank(1, " ").Capabilities(Capability.TEXT_INPUT);
|
||||
}
|
||||
|
||||
private sealed class FamilyWithAModifier : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/modifier", new DateOnly(2026, 9, 11), "A family stating a modifier.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Modifier("base").Removes(Capability.FUNCTION_CALLING);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Tests.Models.Corpus;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Holds the rules to what a model is made for.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// What a model can do and what it is for are two questions, and they used to be answered by two
|
||||
/// pieces of code, each walking the same name with rules of its own. While both existed, the tests
|
||||
/// here held one against the other. The marker list is gone now, and with it the comparison: the
|
||||
/// corpus-wide check moved into the snapshot, which carries the kind of every model in a column of
|
||||
/// its own.
|
||||
///
|
||||
/// What is left says what a name is, in words a person can check against a model card -- and holds
|
||||
/// the handful of decisions where the rules deliberately answer something else than the markers did.
|
||||
/// Those stand in the corpus next to the name, with the reason.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelKindTests
|
||||
{
|
||||
[Test]
|
||||
public void EveryExampleIsRecognizedAsWhatItIsMadeFor()
|
||||
{
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
foreach (var example in ModelKindCorpus.ENTRIES)
|
||||
{
|
||||
var profile = ModelRegistry.Shared.Profile(example.Provider, example.ModelId);
|
||||
|
||||
Assert.That(profile.Kind, Is.EqualTo(example.Kind), $"{example.Provider} \"{example.ModelId}\"");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModelWhichIsNoKindOfItsOwnIsAChatModel()
|
||||
{
|
||||
//
|
||||
// The fallback, and the direction it points in. A model we fail to recognize stays visible
|
||||
// to the user rather than disappearing, because a provider adding a family we have never
|
||||
// seen is the normal case and a user paying for it is the one who would notice.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.SELF_HOSTED, "a-model-nobody-has-heard-of");
|
||||
|
||||
Assert.That(profile.Kind, Is.EqualTo(ModelKind.CHAT));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModelKeepsWhatItsFamilySaysWhenAnotherWordSaysWhatItIsFor()
|
||||
{
|
||||
//
|
||||
// The reason these are modifiers. Llama-Guard is a Llama, and everything the Llama rules
|
||||
// state about it stays true; it is simply not something to chat with. Written as a selector,
|
||||
// "guard" would have to beat "llama" -- two substrings of the same length, which is a tie,
|
||||
// which is an error rather than an answer.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.SELF_HOSTED, "llama-guard-3-8b");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Kind, Is.EqualTo(ModelKind.MODERATION));
|
||||
Assert.That(profile.Has(Capability.TEXT_INPUT), Is.True, "The family which chose the model still speaks for it.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ARerankerIsARerankerAndNotTheEmbeddingModelItIsNamedAfter()
|
||||
{
|
||||
var resolution = ModelRegistry.Shared.Explain(LLMProviders.SELF_HOSTED, "bge-reranker-v2-m3");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(resolution.Profile.Kind, Is.EqualTo(ModelKind.RERANKING));
|
||||
Assert.That(resolution.Modifiers.Select(modifier => modifier.Pattern.Text), Does.Contain("bge"), "Both words match; the ranked one has to be the one which gets the last word.");
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the answer object itself: what it says, and what it refuses to say.
|
||||
/// </summary>
|
||||
[TestFixture]
|
||||
public sealed class ModelProfileTests
|
||||
{
|
||||
[Test]
|
||||
public void AProfileNobodyWroteAnythingIntoKnowsNothingAndStillCountsAsAChatModel()
|
||||
{
|
||||
var untouched = ModelProfile.UNKNOWN;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(untouched.Capabilities, Is.EqualTo(Capability.NONE));
|
||||
Assert.That(untouched.Reasoning, Is.EqualTo(ReasoningSupport.NONE));
|
||||
Assert.That(untouched.Context.IsKnown, Is.False);
|
||||
|
||||
//
|
||||
// A model we fail to recognize has to stay visible to the user rather than disappear
|
||||
// from their list, which is why the unrecognized kind is chat rather than something
|
||||
// meaning "no idea".
|
||||
//
|
||||
Assert.That(untouched.Kind, Is.EqualTo(ModelKind.CHAT));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AskingWhetherAModelHasSeveralCapabilitiesAsksForAllOfThem()
|
||||
{
|
||||
var profile = new ModelProfile { Capabilities = Capability.TEXT_INPUT | Capability.TEXT_OUTPUT };
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Has(Capability.TEXT_INPUT | Capability.TEXT_OUTPUT), Is.True);
|
||||
Assert.That(profile.Has(Capability.TEXT_INPUT | Capability.WEB_SEARCH), Is.False);
|
||||
Assert.That(profile.HasAny(Capability.TEXT_INPUT | Capability.WEB_SEARCH), Is.True);
|
||||
Assert.That(profile.HasAny(Capability.WEB_SEARCH | Capability.EMBEDDING), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AskingForNoCapabilityAtAllIsAnsweredWithNo()
|
||||
{
|
||||
//
|
||||
// Without this, a variable which happens to hold NONE would report every model as able to
|
||||
// do it, because every set contains the empty set.
|
||||
//
|
||||
var profile = new ModelProfile { Capabilities = Capability.TEXT_INPUT };
|
||||
|
||||
Assert.That(profile.Has(Capability.NONE), Is.False);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AChangeOnlyTouchesWhatItStates()
|
||||
{
|
||||
var before = new ModelProfile
|
||||
{
|
||||
Capabilities = Capability.TEXT_INPUT | Capability.WEB_SEARCH,
|
||||
Reasoning = ReasoningSupport.OPTIONAL,
|
||||
Context = ContextWindow.Of(128_000),
|
||||
};
|
||||
|
||||
var after = new ModelProfileChange { Removes = Capability.WEB_SEARCH }.ApplyTo(before);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(after.Capabilities, Is.EqualTo(Capability.TEXT_INPUT));
|
||||
Assert.That(after.Reasoning, Is.EqualTo(ReasoningSupport.OPTIONAL), "A change saying nothing about reasoning must not reset it.");
|
||||
Assert.That(after.Context, Is.EqualTo(before.Context), "A change saying nothing about the context window must not reset it.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void WhatAChangeTakesAwayWinsOverWhatItAdds()
|
||||
{
|
||||
var change = new ModelProfileChange
|
||||
{
|
||||
Adds = Capability.TEXT_INPUT | Capability.WEB_SEARCH,
|
||||
Removes = Capability.WEB_SEARCH,
|
||||
};
|
||||
|
||||
Assert.That(change.ApplyTo(ModelProfile.UNKNOWN).Capabilities, Is.EqualTo(Capability.TEXT_INPUT));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AProfileNeverCarriesTheReasoningVocabulary()
|
||||
{
|
||||
//
|
||||
// The three reasoning members can be combined into answers no model can give, which is why
|
||||
// a profile states reasoning in one field instead. A rule declaring one of them has made a
|
||||
// mistake; that it cannot reach the answer is the second line of defence, not the first.
|
||||
//
|
||||
var change = new ModelProfileChange
|
||||
{
|
||||
Adds = Capability.TEXT_INPUT | Capability.ALWAYS_REASONING,
|
||||
Reasoning = ReasoningSupport.ALWAYS,
|
||||
};
|
||||
|
||||
var profile = change.ApplyTo(ModelProfile.UNKNOWN);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Capabilities, Is.EqualTo(Capability.TEXT_INPUT));
|
||||
Assert.That(profile.Has(Capability.ALWAYS_REASONING), Is.False);
|
||||
Assert.That(profile.Reasoning, Is.EqualTo(ReasoningSupport.ALWAYS));
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,180 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Models.Plugins;
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tests.Models.Plugins;
|
||||
|
||||
/// <summary>
|
||||
/// Checks where what an organization declares stands against what AI Studio works out itself.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Each test builds a registry of its own rather than asking the one the app uses. What is being
|
||||
/// checked is the order of the chain, and a test which had to name a real model to check it would
|
||||
/// start failing the day somebody corrects that model's rule.
|
||||
///
|
||||
/// The one exception borrows the registry the app uses, because only that one knows the hosts. It
|
||||
/// hands it back empty, and the fixture is kept out of any parallel run so that the borrowing
|
||||
/// cannot reach a test asking the same registry about a real model.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
[NonParallelizable]
|
||||
public sealed class DeclaredModelsTests
|
||||
{
|
||||
private static readonly Guid PLUGIN_ID = new("11111111-1111-1111-1111-111111111111");
|
||||
|
||||
[Test]
|
||||
public void WhatAnOrganizationDeclaresComesBeforeWhatTheRulesWorkOut()
|
||||
{
|
||||
var registry = ModelRegistry.Build([new AcmeFamily()], []);
|
||||
registry.Declare([Declaring("acme-assistant", MatchKind.PREFIX, Capability.TEXT_INPUT | Capability.TEXT_OUTPUT | Capability.MULTIPLE_IMAGE_INPUT)]);
|
||||
|
||||
var profile = registry.Profile(LLMProviders.SELF_HOSTED, "acme-assistant-7b");
|
||||
|
||||
Assert.That(profile.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.True, "Only the organization says this model reads images, and they are the ones running it.");
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ADeclarationIsTheWholeStatementAndNotAnAdditionToOne()
|
||||
{
|
||||
//
|
||||
// The part an administrator has to be able to rely on. Their entry says what the model can
|
||||
// do, so what AI Studio would have said instead is gone -- including the capabilities their
|
||||
// entry does not mention. Adding to the built-in answer would make it impossible to take
|
||||
// anything away, which is exactly what somebody correcting us is trying to do.
|
||||
//
|
||||
var registry = ModelRegistry.Build([new AcmeFamily()], []);
|
||||
registry.Declare([Declaring("acme-assistant", MatchKind.PREFIX, Capability.TEXT_INPUT | Capability.TEXT_OUTPUT)]);
|
||||
|
||||
var profile = registry.Profile(LLMProviders.SELF_HOSTED, "acme-assistant-7b");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(profile.Has(Capability.FUNCTION_CALLING), Is.False, "The built-in rule grants this one, and the declaration does not.");
|
||||
Assert.That(profile.Reasoning, Is.EqualTo(ReasoningSupport.NONE), "Nor does it reason, whatever the built-in rule says.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AModelNoDeclarationMentionsIsAnsweredByTheRulesAsBefore()
|
||||
{
|
||||
var registry = ModelRegistry.Build([new AcmeFamily()], []);
|
||||
registry.Declare([Declaring("something-else", MatchKind.SEGMENT, Capability.TEXT_INPUT)]);
|
||||
|
||||
var profile = registry.Profile(LLMProviders.SELF_HOSTED, "acme-assistant-7b");
|
||||
|
||||
Assert.That(profile.Has(Capability.FUNCTION_CALLING), Is.True);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TakingADeclarationAwayBringsTheBuiltInAnswerBack()
|
||||
{
|
||||
//
|
||||
// This is what happens when an organization withdraws a configuration, or when somebody
|
||||
// corrects their plugin and the plugins are reloaded. It is also the test that the kept
|
||||
// answers are dropped along with the declarations they were worked out under: a cache which
|
||||
// outlived them would go on answering with what a plugin said which is no longer there.
|
||||
//
|
||||
var registry = ModelRegistry.Build([new AcmeFamily()], []);
|
||||
registry.Declare([Declaring("acme-assistant", MatchKind.PREFIX, Capability.TEXT_INPUT | Capability.TEXT_OUTPUT)]);
|
||||
|
||||
var whileDeclared = registry.Profile(LLMProviders.SELF_HOSTED, "acme-assistant-7b");
|
||||
registry.Declare([]);
|
||||
var afterwards = registry.Profile(LLMProviders.SELF_HOSTED, "acme-assistant-7b");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(whileDeclared.Has(Capability.FUNCTION_CALLING), Is.False);
|
||||
Assert.That(afterwards.Has(Capability.FUNCTION_CALLING), Is.True);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AmongTheDeclarationsTheOneSayingMoreAboutTheNameWins()
|
||||
{
|
||||
//
|
||||
// Two plugins, or one plugin describing a family and then one of its variants. Nothing new
|
||||
// is needed for this: the declarations go through the same engine as the built-in rules, so
|
||||
// the specificity is computed here too and nobody writes an order.
|
||||
//
|
||||
var registry = ModelRegistry.Build([new AcmeFamily()], []);
|
||||
registry.Declare(
|
||||
[
|
||||
Declaring("acme-assistant", MatchKind.PREFIX, Capability.TEXT_INPUT | Capability.TEXT_OUTPUT),
|
||||
Declaring("acme-assistant-7b", MatchKind.EXACT, Capability.TEXT_INPUT | Capability.TEXT_OUTPUT | Capability.MULTIPLE_IMAGE_INPUT),
|
||||
]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(registry.Profile(LLMProviders.SELF_HOSTED, "acme-assistant-7b").Has(Capability.MULTIPLE_IMAGE_INPUT), Is.True);
|
||||
Assert.That(registry.Profile(LLMProviders.SELF_HOSTED, "acme-assistant-3b").Has(Capability.MULTIPLE_IMAGE_INPUT), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ADeclarationIsMeasuredAgainstTheNameWithoutTheProvidersWrapping()
|
||||
{
|
||||
//
|
||||
// An administrator writes the model's name, not the name plus whatever the gateway they
|
||||
// reach it through puts in front of it. Unwrapping happens before anything is asked, so the
|
||||
// same entry answers whichever way the model is reached.
|
||||
//
|
||||
var registry = ModelRegistry.Shared;
|
||||
var declaration = Declaring("gpt-5.1", MatchKind.PREFIX, Capability.TEXT_INPUT | Capability.TEXT_OUTPUT | Capability.VIDEO_INPUT);
|
||||
|
||||
try
|
||||
{
|
||||
registry.Declare([declaration]);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(registry.Profile(LLMProviders.OPEN_AI, "gpt-5.1").Has(Capability.VIDEO_INPUT), Is.True);
|
||||
Assert.That(registry.Profile(LLMProviders.OPEN_ROUTER, "openai/gpt-5.1").Has(Capability.VIDEO_INPUT), Is.True, "The same model, reached through a gateway which wraps the name.");
|
||||
});
|
||||
}
|
||||
finally
|
||||
{
|
||||
//
|
||||
// The registry the app uses is the only one which knows the hosts, so this test has to
|
||||
// borrow it. Handing it back empty is what keeps the borrowing from reaching the tests
|
||||
// which ask it about real models.
|
||||
//
|
||||
registry.Declare([]);
|
||||
}
|
||||
}
|
||||
|
||||
private static ModelDeclaration Declaring(string pattern, MatchKind matchKind, Capability capabilities) => new()
|
||||
{
|
||||
Pattern = new()
|
||||
{
|
||||
Kind = matchKind,
|
||||
Text = pattern,
|
||||
},
|
||||
|
||||
Change = new()
|
||||
{
|
||||
Adds = capabilities,
|
||||
},
|
||||
|
||||
Source = new("https://intranet.invalid/ai", new DateOnly(2026, 9, 12), "What a company says about its own models."),
|
||||
Origin = "Models of a company",
|
||||
EnterpriseConfigurationPluginId = PLUGIN_ID,
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// A family which says more about these models than the declarations of this test do.
|
||||
/// </summary>
|
||||
private sealed class AcmeFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/acme", new DateOnly(2026, 9, 12), "A family standing in for whatever AI Studio knows by itself.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) =>
|
||||
builder.Rule("acme-assistant").AsPrefix()
|
||||
.Capabilities(Capability.TEXT_INPUT | Capability.TEXT_OUTPUT | Capability.FUNCTION_CALLING)
|
||||
.Apis(Capability.CHAT_COMPLETION_API)
|
||||
.Reasoning(ReasoningSupport.ALWAYS);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,226 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Models.Plugins;
|
||||
using AIStudio.Provider;
|
||||
|
||||
using Lua;
|
||||
using Lua.Standard;
|
||||
|
||||
using Microsoft.Extensions.Logging.Abstractions;
|
||||
|
||||
namespace AIStudio.Tests.Models.Plugins;
|
||||
|
||||
/// <summary>
|
||||
/// Checks what AI Studio makes of a model an organization describes in a plugin of its own.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The entries below are written the way they are written in a plugin.lua, and they are read
|
||||
/// through a real Lua state rather than through a table put together in C#. What is being checked
|
||||
/// is the wire format an administrator types, so anything between their file and the declaration
|
||||
/// has to be part of the test.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelDeclarationTests
|
||||
{
|
||||
private static readonly Guid PLUGIN_ID = new("11111111-1111-1111-1111-111111111111");
|
||||
|
||||
private const string ORIGIN = "Models of a company";
|
||||
|
||||
private const string A_COMPLETE_DECLARATION = """
|
||||
["PATTERN"] = "acme-assistant",
|
||||
["MATCH"] = "PREFIX",
|
||||
["CAPABILITIES"] = { "TEXT_INPUT", "MULTIPLE_IMAGE_INPUT", "TEXT_OUTPUT", "FUNCTION_CALLING", "CHAT_COMPLETION_API" },
|
||||
["REASONING"] = "ON_BY_DEFAULT",
|
||||
["KIND"] = "CHAT",
|
||||
["CONTEXT_WINDOW"] = 131072,
|
||||
["CONTEXT_WINDOW_RAISABLE_TO"] = 262144,
|
||||
["TOKENIZER_KIND"] = "HUGGING_FACE",
|
||||
["TOKENIZER_ID"] = "acme/assistant",
|
||||
["MAX_IMAGES_PER_MESSAGE"] = 1,
|
||||
["MAX_IMAGES_PER_REQUEST"] = 8,
|
||||
["SOURCE_URL"] = "https://intranet.invalid/ai/acme-assistant",
|
||||
["SOURCE_CHECKED_ON"] = "2026-09-12",
|
||||
["SOURCE_NOTE"] = "Internal model card: tools, images, 128k context",
|
||||
""";
|
||||
|
||||
private const string THE_LEAST_A_DECLARATION_CAN_SAY = """
|
||||
["PATTERN"] = "acme-assistant",
|
||||
["CAPABILITIES"] = { "TEXT_INPUT", "TEXT_OUTPUT", "CHAT_COMPLETION_API" },
|
||||
["SOURCE_URL"] = "https://intranet.invalid/ai/acme-assistant",
|
||||
["SOURCE_CHECKED_ON"] = "2026-09-12",
|
||||
""";
|
||||
|
||||
[Test]
|
||||
public async Task ADeclarationIsReadTheWayItWasWritten()
|
||||
{
|
||||
var declaration = await ReadAsync(A_COMPLETE_DECLARATION);
|
||||
|
||||
Assert.That(declaration, Is.Not.Null);
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(declaration!.Pattern.Text, Is.EqualTo("acme-assistant"));
|
||||
Assert.That(declaration.Pattern.Kind, Is.EqualTo(MatchKind.PREFIX));
|
||||
Assert.That(declaration.Change.Adds, Is.EqualTo(Capability.TEXT_INPUT | Capability.MULTIPLE_IMAGE_INPUT | Capability.TEXT_OUTPUT | Capability.FUNCTION_CALLING | Capability.CHAT_COMPLETION_API));
|
||||
Assert.That(declaration.Change.Reasoning, Is.EqualTo(ReasoningSupport.ON_BY_DEFAULT));
|
||||
Assert.That(declaration.Change.Kind, Is.EqualTo(ModelKind.CHAT));
|
||||
Assert.That(declaration.Change.Context, Is.EqualTo(ContextWindow.Of(131_072, 262_144)));
|
||||
Assert.That(declaration.Change.Tokenizer, Is.EqualTo(new TokenizerRef(TokenizerKind.HUGGING_FACE, "acme/assistant")));
|
||||
Assert.That(declaration.Change.Images, Is.EqualTo(new ImageLimits(1, 8)));
|
||||
Assert.That(declaration.Source.CheckedOn, Is.EqualTo(new DateOnly(2026, 9, 12)));
|
||||
Assert.That(declaration.EnterpriseConfigurationPluginId, Is.EqualTo(PLUGIN_ID));
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task WhatADeclarationLeavesOutIsTheSameAsWhatAFamilyLeavesOut()
|
||||
{
|
||||
var declaration = await ReadAsync(THE_LEAST_A_DECLARATION_CAN_SAY);
|
||||
|
||||
Assert.That(declaration, Is.Not.Null);
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(declaration!.Pattern.Kind, Is.EqualTo(MatchKind.SEGMENT), "The kind to reach for by default, here as everywhere else.");
|
||||
Assert.That(declaration.Change.Reasoning, Is.EqualTo(ReasoningSupport.NONE));
|
||||
Assert.That(declaration.Change.Kind, Is.EqualTo(ModelKind.CHAT), "A model nobody said anything else about stays visible in the chat lists.");
|
||||
Assert.That(declaration.Change.Context, Is.Null);
|
||||
Assert.That(declaration.Change.Tokenizer, Is.Null);
|
||||
Assert.That(declaration.Change.Images, Is.Null);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task ADeclarationWithoutCapabilitiesIsRefused()
|
||||
{
|
||||
//
|
||||
// The one thing a declaration cannot leave out. It replaces what AI Studio would otherwise
|
||||
// say about these models, so an entry naming only a context window would take away every
|
||||
// capability the built-in rules knew -- and it would do so silently, because an entry which
|
||||
// matches is an answer.
|
||||
//
|
||||
var declaration = await ReadAsync("""
|
||||
["PATTERN"] = "acme-assistant",
|
||||
["CONTEXT_WINDOW"] = 131072,
|
||||
["SOURCE_URL"] = "https://intranet.invalid/ai/acme-assistant",
|
||||
["SOURCE_CHECKED_ON"] = "2026-09-12",
|
||||
""");
|
||||
|
||||
Assert.That(declaration, Is.Null);
|
||||
}
|
||||
|
||||
[TestCase("""["PATTERN"] = "Acme-Assistant",""", TestName = "A pattern in capitals", Description = "Names arrive in lower case, so this could never match.")]
|
||||
[TestCase("""["PATTERN"] = "acme_assistant",""", TestName = "A pattern with an underscore")]
|
||||
[TestCase("""["PATTERN"] = "acme assistant",""", TestName = "A pattern with a space")]
|
||||
[TestCase("""["PATTERN"] = "",""", TestName = "No pattern at all")]
|
||||
public async Task APatternWhichCouldNeverMatchAnythingIsRefused(string pattern)
|
||||
{
|
||||
var declaration = await ReadAsync($$"""
|
||||
{{pattern}}
|
||||
["CAPABILITIES"] = { "TEXT_INPUT", "TEXT_OUTPUT" },
|
||||
["SOURCE_URL"] = "https://intranet.invalid/ai/acme-assistant",
|
||||
["SOURCE_CHECKED_ON"] = "2026-09-12",
|
||||
""");
|
||||
|
||||
Assert.That(declaration, Is.Null, "A pattern which is not written the way a model name is written is a mistake, not a rule which happens to stay quiet.");
|
||||
}
|
||||
|
||||
[TestCase("ALWAYS_REASONING")]
|
||||
[TestCase("OPTIONAL_REASONING")]
|
||||
[TestCase("REASONING_BY_DEFAULT")]
|
||||
public async Task ReasoningStatedAsACapabilityIsRefusedRatherThanDropped(string reasoningWord)
|
||||
{
|
||||
//
|
||||
// The three words are the vocabulary of the expert settings, where a person answers three
|
||||
// questions with yes and no. Here one key says how a model reasons, and the three of them
|
||||
// together can state answers no model can give. A profile drops them anyway, so accepting
|
||||
// them would mean an administrator wrote something that never took effect.
|
||||
//
|
||||
var declaration = await ReadAsync($$"""
|
||||
["PATTERN"] = "acme-assistant",
|
||||
["CAPABILITIES"] = { "TEXT_INPUT", "TEXT_OUTPUT", "{{reasoningWord}}" },
|
||||
["SOURCE_URL"] = "https://intranet.invalid/ai/acme-assistant",
|
||||
["SOURCE_CHECKED_ON"] = "2026-09-12",
|
||||
""");
|
||||
|
||||
Assert.That(declaration, Is.Null);
|
||||
}
|
||||
|
||||
[TestCase("""["SOURCE_CHECKED_ON"] = "2026-09-12",""", TestName = "A page nobody named")]
|
||||
[TestCase("""["SOURCE_URL"] = "https://intranet.invalid/ai",""", TestName = "A day nobody named")]
|
||||
[TestCase("""["SOURCE_URL"] = "https://intranet.invalid/ai", ["SOURCE_CHECKED_ON"] = "12.09.2026",""", TestName = "A day written another way")]
|
||||
public async Task ADeclarationHasToSayWhereItWasReadAndWhen(string source)
|
||||
{
|
||||
//
|
||||
// The compiler asks a family in the source for this, and an organization's declaration
|
||||
// outlives whoever wrote it just the same. Naming the page and the day is what lets the next
|
||||
// administrator find out in a minute whether it still holds.
|
||||
//
|
||||
var declaration = await ReadAsync($$"""
|
||||
["PATTERN"] = "acme-assistant",
|
||||
["CAPABILITIES"] = { "TEXT_INPUT", "TEXT_OUTPUT" },
|
||||
{{source}}
|
||||
""");
|
||||
|
||||
Assert.That(declaration, Is.Null);
|
||||
}
|
||||
|
||||
[TestCase("""["TOKENIZER_KIND"] = "HUGGING_FACE",""", TestName = "A tokenizer kind without an ID")]
|
||||
[TestCase("""["TOKENIZER_ID"] = "acme/assistant",""", TestName = "A tokenizer ID without a kind")]
|
||||
[TestCase("""["CONTEXT_WINDOW_RAISABLE_TO"] = 262144,""", TestName = "A ceiling without a window")]
|
||||
[TestCase("""["CONTEXT_WINDOW"] = 262144, ["CONTEXT_WINDOW_RAISABLE_TO"] = 131072,""", TestName = "A ceiling below the window")]
|
||||
[TestCase("""["CONTEXT_WINDOW"] = 0,""", TestName = "A window of no tokens")]
|
||||
[TestCase("""["MAX_IMAGES_PER_REQUEST"] = -1,""", TestName = "Fewer than no images")]
|
||||
[TestCase("""["KIND"] = "SOMETHING_ELSE",""", TestName = "A kind of model nobody knows")]
|
||||
[TestCase("""["MATCH"] = "REGEX",""", TestName = "A way of matching which does not exist")]
|
||||
[TestCase("""["ONLY_ON"] = "ACME_CLOUD",""", TestName = "A provider which does not exist")]
|
||||
public async Task AnEntryWhichSaysSomethingUnreadableIsRefusedAsAWhole(string addition)
|
||||
{
|
||||
//
|
||||
// Never read in part: a declaration is one statement, and half of one would answer for the
|
||||
// models it matches just as firmly as a complete one, with the unreadable half missing and
|
||||
// nothing on screen saying so.
|
||||
//
|
||||
var declaration = await ReadAsync($"""
|
||||
{THE_LEAST_A_DECLARATION_CAN_SAY}
|
||||
{addition}
|
||||
""");
|
||||
|
||||
Assert.That(declaration, Is.Null);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task TwoDeclarationsCollideExactlyWhenTheyClaimTheSameNames()
|
||||
{
|
||||
//
|
||||
// What identifies a declaration is its pattern, because that is what a collision is here.
|
||||
// Two of them claiming the same names would both enter the index and tie there, and a tie
|
||||
// is something only a person can settle. Two about different names never meet.
|
||||
//
|
||||
var declaration = await ReadAsync(A_COMPLETE_DECLARATION);
|
||||
var theSameNames = await ReadAsync(A_COMPLETE_DECLARATION.Replace("""["CONTEXT_WINDOW"] = 131072,""", """["CONTEXT_WINDOW"] = 65536,""", StringComparison.Ordinal));
|
||||
var otherNames = await ReadAsync(A_COMPLETE_DECLARATION.Replace("""["MATCH"] = "PREFIX",""", """["MATCH"] = "SEGMENT",""", StringComparison.Ordinal));
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(theSameNames?.Id, Is.EqualTo(declaration?.Id), "The same pattern, so one of the two has to win.");
|
||||
Assert.That(otherNames?.Id, Is.Not.EqualTo(declaration?.Id), "Bound to the name differently, so they claim different sets of names.");
|
||||
});
|
||||
}
|
||||
|
||||
private static async Task<ModelDeclaration?> ReadAsync(string entry)
|
||||
{
|
||||
var state = LuaState.Create();
|
||||
state.OpenBasicLibrary();
|
||||
state.OpenTableLibrary();
|
||||
|
||||
await state.DoStringAsync($$"""
|
||||
MODEL = {
|
||||
{{entry}}
|
||||
}
|
||||
""");
|
||||
|
||||
if (!state.Environment["MODEL"].TryRead<LuaTable>(out var table))
|
||||
throw new InvalidOperationException("The entry of this test is not a Lua table.");
|
||||
|
||||
return ModelDeclaration.TryParse(1, table, PLUGIN_ID, ORIGIN, NullLogger.Instance, out var declaration) ? declaration : null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,115 @@
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Tests.Models.Corpus;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Holds the rules to the answers somebody decided on, over the whole corpus.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// While the old rules still stood, this was the test the rebuild was carried by: every model was
|
||||
/// asked of both and the two had to agree, except where the audit had found the old answer wrong.
|
||||
/// That comparison is over -- the old rules are gone, and the snapshot took over the job of noticing
|
||||
/// when an answer changes.
|
||||
///
|
||||
/// What remains is the part no snapshot can do, because it is about intent rather than about
|
||||
/// answers. The models the audit found wrong have to end up where the audit said. Every model
|
||||
/// reaching the global assumption has to be one somebody let reach it, and every model somebody
|
||||
/// listed there has to still be reaching it. And no name may be claimed by two rules with the same
|
||||
/// right.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class PortingDifferenceTests
|
||||
{
|
||||
[Test]
|
||||
public void EveryModelTheAuditFoundWrongIsNowAnsweredTheWayItShouldBe()
|
||||
{
|
||||
var corrected = ExpectedChanges.ENTRIES.Where(change => IsAnswered(change.Provider, change.ModelId)).ToList();
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(corrected, Is.Not.Empty, "Nothing the audit found wrong is answered by a rule at all, which would make this test green for the wrong reason.");
|
||||
|
||||
foreach (var change in corrected)
|
||||
{
|
||||
var entry = new CorpusEntry(change.Provider, change.ModelId, CorpusOrigin.NAMED_BY_NO_RULE);
|
||||
var rebuilt = CapabilitySnapshot.Describe(RebuiltRules.Ask(entry));
|
||||
|
||||
Assert.That(rebuilt, Is.EqualTo(CapabilitySnapshot.Describe(change.AnswerWanted)), $"{change.Provider} \"{change.ModelId}\": {change.Reason}");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryModelOfTheCorpusIsEitherAnsweredByARuleOrLeftToTheDefaultOnPurpose()
|
||||
{
|
||||
//
|
||||
// Comparing answers alone cannot catch a model falling through. It gets an empty profile,
|
||||
// the global default answers for it, and nothing about that looks wrong from the outside --
|
||||
// a family nobody got round to and a family nobody wanted are both simply missing. This is
|
||||
// the test which makes the difference visible, by asking for the reason.
|
||||
//
|
||||
var fallenThrough = ModelCorpus.ENTRIES
|
||||
.Where(entry => !IsAnswered(entry))
|
||||
.Where(entry => !IsLeftToTheDefault(entry))
|
||||
.Select(entry => $"{entry.Provider} \"{entry.ModelId}\"");
|
||||
|
||||
Assert.That(fallenThrough, Is.Empty, "No rule answers for these, and nothing says that is on purpose. Write a family for them, or put them into LeftToTheDefault with the reason.");
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void NothingLeftToTheDefaultIsAnsweredByARuleAfterAll()
|
||||
{
|
||||
//
|
||||
// The other direction, so the list cannot rot: once a family is written, the models it
|
||||
// answers for have no business standing among the ones nobody wrote a rule for.
|
||||
//
|
||||
var answeredAfterAll = LeftToTheDefault.ENTRIES
|
||||
.Where(left => IsAnswered(left.Provider, left.ModelId))
|
||||
.Select(left => $"{left.Provider} \"{left.ModelId}\"");
|
||||
|
||||
Assert.That(answeredAfterAll, Is.Empty, "A rule answers for these now, so they can be taken off the list of models left to the default.");
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void NoModelOfTheCorpusIsClaimedByTwoRulesWithTheSameRight()
|
||||
{
|
||||
//
|
||||
// Two rules of the same specificity which can both match one name are a mistake, not a coin
|
||||
// toss. Reading the rules alone cannot find it -- the two patterns are written differently
|
||||
// and only meet on a real name, which is what the corpus is full of.
|
||||
//
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
foreach (var entry in ModelCorpus.ENTRIES)
|
||||
{
|
||||
var resolution = ModelRegistry.Shared.Explain(entry.Provider, entry.ModelId);
|
||||
|
||||
Assert.That(resolution.IsAmbiguous, Is.False, $"{entry.Provider} \"{entry.ModelId}\" is claimed by {resolution.Selector} and, just as strongly, by {string.Join(", ", resolution.TiedSelectors)}.");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Whether any rule knows this model.
|
||||
/// </summary>
|
||||
/// <param name="entry">The corpus entry to ask about.</param>
|
||||
/// <returns>True, when a rule answers for it.</returns>
|
||||
private static bool IsAnswered(CorpusEntry entry) => IsAnswered(entry.Provider, entry.ModelId);
|
||||
|
||||
/// <summary>
|
||||
/// Whether any rule knows this model.
|
||||
/// </summary>
|
||||
/// <param name="provider">Who serves the model.</param>
|
||||
/// <param name="modelId">The model ID as that provider reports it.</param>
|
||||
/// <returns>True, when a rule answers for it.</returns>
|
||||
private static bool IsAnswered(LLMProviders provider, string modelId) => ModelRegistry.Shared.Explain(provider, modelId).IsKnown;
|
||||
|
||||
/// <summary>
|
||||
/// Whether this model reaches the global default because somebody decided it may.
|
||||
/// </summary>
|
||||
/// <param name="entry">The corpus entry to look up.</param>
|
||||
/// <returns>True, when it stands in the list of models left to the default.</returns>
|
||||
private static bool IsLeftToTheDefault(CorpusEntry entry) => LeftToTheDefault.ENTRIES.Any(left => left.Provider == entry.Provider && string.Equals(left.ModelId, entry.ModelId, StringComparison.Ordinal));
|
||||
}
|
||||
@@ -0,0 +1,176 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Matching;
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Tests.Models.Corpus;
|
||||
|
||||
namespace AIStudio.Tests.Models.Registry;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the registry itself, and the properties every rule in the app has to have.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The property tests below are the ones which cannot be written per family, because what they ask
|
||||
/// about only exists once all the families are together: whether two of them claim the same name,
|
||||
/// whether every rule can be traced back to somebody. They are cheap and they grow with the rules
|
||||
/// on their own, which is the point -- nobody has to remember to extend them when adding a family.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ModelRegistryTests
|
||||
{
|
||||
[Test]
|
||||
public void NoTwoRulesOfTheAppClaimTheSameNamesWithTheSameRight()
|
||||
{
|
||||
var ambiguities = ModelRegistry.Shared.Rules.Ambiguities.Select(ambiguity => $"{ambiguity.First} / {ambiguity.Second}: {ambiguity.Reason}");
|
||||
|
||||
Assert.That(ambiguities, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryRuleIsWrittenInTheFormNamesArriveIn()
|
||||
{
|
||||
//
|
||||
// The compile time rule says the same thing about every literal in the source. This says it
|
||||
// about the rules as they were actually built, which also covers a pattern that was put
|
||||
// together rather than written down.
|
||||
//
|
||||
var malformed = ModelRegistry.Shared.Rules.Rules.Where(rule => !rule.Pattern.IsWellFormed).Select(rule => rule.Description);
|
||||
|
||||
Assert.That(malformed, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryFamilySaysWhereItsStatementsCanBeCheckedAndWhen()
|
||||
{
|
||||
//
|
||||
// The further sources are asked the same question as the first one. A family which reads
|
||||
// its windows from one page and its image limits from another has two pages to name, and a
|
||||
// second page named without a day is exactly as uncheckable as no page at all.
|
||||
//
|
||||
var unstated = ModelRegistry.Shared.Families
|
||||
.Where(family => !family.Source.IsStated || family.FurtherSources.Any(source => !source.IsStated))
|
||||
.Select(family => family.Name);
|
||||
|
||||
Assert.That(unstated, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void NoFamilyStatesOneOfTheThreeReasoningWords()
|
||||
{
|
||||
//
|
||||
// They are override vocabulary: a person writes ALWAYS_REASONING to correct us, and a
|
||||
// profile answers the same question through its reasoning field, where the contradictory
|
||||
// combinations cannot be written down. A family reaching for the flag would be stating
|
||||
// something the profile then silently drops.
|
||||
//
|
||||
var confused = ModelRegistry.Shared.Rules.Rules
|
||||
.Where(rule => (rule.Change.Adds & ModelProfile.REASONING_VOCABULARY) is not Capability.NONE)
|
||||
.Select(rule => rule.Description);
|
||||
|
||||
Assert.That(confused, Is.Empty, "State how a model reasons with Reasoning(...) instead.");
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void EveryRuleNamesAFamilyTheRegistryCanFindAgain()
|
||||
{
|
||||
//
|
||||
// The origin is how a rule finds its way back to the family which wrote it, and that is what
|
||||
// decides whose Refine is asked. A name which leads nowhere would simply skip the refining.
|
||||
//
|
||||
var families = ModelRegistry.Shared.Families.Select(family => family.Name).ToHashSet(StringComparer.Ordinal);
|
||||
var orphans = ModelRegistry.Shared.Rules.Rules.Where(rule => !families.Contains(rule.Origin)).Select(rule => rule.Description);
|
||||
|
||||
Assert.That(orphans, Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void WithoutAProviderThereIsNothingToSayAboutAModel()
|
||||
{
|
||||
//
|
||||
// A model is reached through a provider, and without one there is no way to reach it. The
|
||||
// rules this replaces answered the same, by having no branch for it at all.
|
||||
//
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.NONE, "gpt-5.6");
|
||||
|
||||
Assert.That(RebuiltRules.AsCapabilities(profile), Is.Empty);
|
||||
}
|
||||
|
||||
[TestCase("")]
|
||||
[TestCase(" ")]
|
||||
public void AProviderWhichNamedNoModelIsAnsweredWithNothing(string modelId)
|
||||
{
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.OPEN_AI, modelId);
|
||||
|
||||
Assert.That(RebuiltRules.AsCapabilities(profile), Is.Empty);
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheSameModelReachedTwoWaysGetsTwoAnswers()
|
||||
{
|
||||
//
|
||||
// Also the test that the remembered answers are kept per provider: one key for both would
|
||||
// hand whichever was asked first to the other.
|
||||
//
|
||||
var atOpenAI = ModelRegistry.Shared.Profile(LLMProviders.OPEN_AI, "gpt-5.1");
|
||||
var throughAGateway = ModelRegistry.Shared.Profile(LLMProviders.OPEN_ROUTER, "openai/gpt-5.1");
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(atOpenAI.Has(Capability.RESPONSES_API), Is.True);
|
||||
Assert.That(throughAGateway.Has(Capability.RESPONSES_API), Is.False);
|
||||
Assert.That(throughAGateway.Has(Capability.FUNCTION_CALLING), Is.True, "Everything but the API survives the trip through a gateway.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheFamilyWhichChoseTheModelGetsToWorkSomethingOutOfTheName()
|
||||
{
|
||||
var registry = ModelRegistry.Build([new RefiningFamily()], []);
|
||||
var profile = registry.Profile(LLMProviders.SELF_HOSTED, "refined-thing");
|
||||
|
||||
Assert.That(profile.Has(Capability.WEB_SEARCH), Is.True, "The family adds this in Refine, which no rule can express.");
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TwoFamiliesOfTheSameNameAreRefused()
|
||||
{
|
||||
var refused = Assert.Throws<InvalidOperationException>(() => ModelRegistry.Build([new FirstPlace.TwiceNamedFamily(), new SecondPlace.TwiceNamedFamily()], []));
|
||||
|
||||
Assert.That(refused?.Message, Does.Contain(nameof(FirstPlace.TwiceNamedFamily)));
|
||||
}
|
||||
|
||||
private sealed class RefiningFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/refining", new DateOnly(2026, 9, 11), "A family which works something out of the name after a rule chose it.");
|
||||
|
||||
public override ModelProfile Refine(in ModelId id, in ModelProfile selected) => selected with { Capabilities = selected.Capabilities | Capability.WEB_SEARCH };
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("refined").Capabilities(Capability.TEXT_INPUT).Apis(Capability.CHAT_COMPLETION_API);
|
||||
}
|
||||
|
||||
private static class FirstPlace
|
||||
{
|
||||
internal sealed class TwiceNamedFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/first", new DateOnly(2026, 9, 11), "One of two families sharing a name.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("first");
|
||||
}
|
||||
}
|
||||
|
||||
private static class SecondPlace
|
||||
{
|
||||
internal sealed class TwiceNamedFamily : ModelFamily
|
||||
{
|
||||
public override ModelVendor Vendor => ModelVendor.UNKNOWN;
|
||||
|
||||
public override ModelSource Source => new("https://example.invalid/second", new DateOnly(2026, 9, 11), "The other of two families sharing a name.");
|
||||
|
||||
protected override void Declare(ModelFamilyBuilder builder) => builder.Rule("second");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
using AIStudio.Tests.Models.Corpus;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Writes the capability snapshot anew.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Marked explicit, so it never runs as part of the suite: it would make the characterization test
|
||||
/// pass by rewriting what that test compares against. Run it by hand, from the IDE or with
|
||||
/// "dotnet test --filter TakeTheSnapshotAnew", once a diff has been read and accepted, and commit
|
||||
/// the new file together with the change which caused it.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
[Explicit("Rewrites the file the characterization test compares against. Run it only after reading the diff.")]
|
||||
public sealed class SnapshotWriterTests
|
||||
{
|
||||
[Test]
|
||||
public void TakeTheSnapshotAnew()
|
||||
{
|
||||
File.WriteAllText(CapabilitySnapshot.FILE_PATH, CapabilitySnapshot.Render(ModelCorpus.ENTRIES));
|
||||
File.Delete(CapabilitySnapshot.ACTUAL_FILE_PATH);
|
||||
|
||||
TestContext.Out.WriteLine($"Wrote {CapabilitySnapshot.FILE_PATH}. Read the diff before committing it.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Settings;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks the test harness itself, before any test states something about the app.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Three things have to hold before a capability test can mean anything: the app assembly is
|
||||
/// referenced, this project counts as a friend assembly, and the assembly-wide setup has run. When
|
||||
/// one of them is missing, the failure looks like a broken rule rather than a broken harness, which
|
||||
/// is an expensive detour. These two tests make the difference visible right away.
|
||||
///
|
||||
/// Note the fully written type name below. AIStudio.Provider is a namespace and AIStudio.Settings
|
||||
/// .Provider is a type; inside a namespace under AIStudio, the namespace wins the lookup. That is a
|
||||
/// property of the app's own naming, not of the tests.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class TestHarnessTests
|
||||
{
|
||||
[Test]
|
||||
public void TheStaticApplicationStateIsAvailable()
|
||||
{
|
||||
//
|
||||
// Settings.Provider initializes a static logger from Program.LOGGER_FACTORY. Touching it
|
||||
// without the assembly-wide setup throws a TypeInitializationException.
|
||||
//
|
||||
Assert.That(AIStudio.Settings.Provider.NONE.UsedLLMProvider, Is.EqualTo(LLMProviders.NONE));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheCapabilityApiOfTheAppIsReachable()
|
||||
{
|
||||
var profile = LLMProviders.OPEN_AI.GetModelProfile(new Model("gpt-5.1", null));
|
||||
|
||||
Assert.That(profile.Has(Capability.FUNCTION_CALLING), Is.True);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Settings;
|
||||
|
||||
namespace AIStudio.Tests.Models;
|
||||
|
||||
/// <summary>
|
||||
/// Checks which tokenizer the rules name for a model.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Naming one changes nothing about the counting today: AI Studio counts with the tokenizer it
|
||||
/// ships unless somebody points it at a tokenizer.json file, and none of the references below is
|
||||
/// such a file. What they are for is the sentence in the provider dialog, which until now let a
|
||||
/// person guess -- including the ones who go looking for a file which was never published.
|
||||
///
|
||||
/// The kind matters as much as the name, and that is what the cases here pin. "o200k_base" is an
|
||||
/// encoding nobody can download, "/v1/messages/count_tokens" is an endpoint nobody can select in a
|
||||
/// file dialog, and telling them apart is the whole point of recording the kind alongside the name.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class TokenizerRuleTests
|
||||
{
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5.1", "o200k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5.6", "o200k_base", Description = "Every model of the 5 line inherits the encoding of its prefix.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-5-chat-latest", "o200k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-4o", "o200k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-4o-mini-search-preview", "o200k_base", Description = "Stated in full rather than inherited, so it has to say this itself.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "o1", "o200k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "o1-mini", "o200k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "o3-mini", "o200k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "o4-mini", "o200k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-4", "cl100k_base", Description = "The older encoding, which is where the 4 line stayed.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-4-turbo", "cl100k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-3.5-turbo", "cl100k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "text-embedding-3-small", "cl100k_base", Description = "The embedding models stayed on the older encoding as well, and their dialog asks the same question.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "text-embedding-3-large", "cl100k_base")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "text-embedding-ada-002", "cl100k_base")]
|
||||
public void OpenAINamesAnEncodingRatherThanAFile(LLMProviders provider, string modelId, string encoding)
|
||||
{
|
||||
var tokenizer = provider.GetModelProfile(new Model(modelId, null)).Tokenizer;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(tokenizer.Kind, Is.EqualTo(TokenizerKind.TIKTOKEN));
|
||||
Assert.That(tokenizer.Id, Is.EqualTo(encoding));
|
||||
});
|
||||
}
|
||||
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-opus-5", "/v1/messages/count_tokens")]
|
||||
[TestCase(LLMProviders.ANTHROPIC, "claude-3-5-haiku-latest", "/v1/messages/count_tokens")]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-3-pro", "countTokens")]
|
||||
[TestCase(LLMProviders.GOOGLE, "gemini-2.5-flash-lite", "countTokens")]
|
||||
public void AnthropicAndGoogleNameAnEndpointBecauseTheyPublishNoFile(LLMProviders provider, string modelId, string endpoint)
|
||||
{
|
||||
var tokenizer = provider.GetModelProfile(new Model(modelId, null)).Tokenizer;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(tokenizer.Kind, Is.EqualTo(TokenizerKind.PROVIDER_API));
|
||||
Assert.That(tokenizer.Id, Is.EqualTo(endpoint));
|
||||
});
|
||||
}
|
||||
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-6-astra", Description = "Newer than the mapping OpenAI publishes, so nothing is claimed for it.")]
|
||||
[TestCase(LLMProviders.OPEN_AI, "gpt-oss-120b", Description = "Open weights, and not in OpenAI's encoding table either.")]
|
||||
[TestCase(LLMProviders.SELF_HOSTED, "some-model-nobody-wrote-a-rule-for")]
|
||||
[TestCase(LLMProviders.MISTRAL, "mistral-large-2512")]
|
||||
public void AModelNobodyNamedATokenizerForSaysSo(LLMProviders provider, string modelId)
|
||||
{
|
||||
var tokenizer = provider.GetModelProfile(new Model(modelId, null)).Tokenizer;
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(tokenizer.IsKnown, Is.False);
|
||||
Assert.That(tokenizer.Kind, Is.EqualTo(TokenizerKind.UNKNOWN), "Which means the built-in tokenizer, the same as today.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void TheSameModelThroughAGatewayKeepsItsTokenizer()
|
||||
{
|
||||
//
|
||||
// A gateway cuts what its transport cannot carry, which is about APIs. Which tokenizer a
|
||||
// model was trained with is a property of the model and survives the trip.
|
||||
//
|
||||
var directly = ModelRegistry.Shared.Profile(LLMProviders.OPEN_AI, "gpt-5.1");
|
||||
var throughAGateway = ModelRegistry.Shared.Profile(LLMProviders.OPEN_ROUTER, "openai/gpt-5.1");
|
||||
|
||||
Assert.That(throughAGateway.Tokenizer, Is.EqualTo(directly.Tokenizer));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ANameWithoutAKindIsNotAReference()
|
||||
{
|
||||
//
|
||||
// Both halves have to be there. A kind without a name says nothing to act on, and a name
|
||||
// without a kind cannot be told apart from any other string -- whether it is a repository,
|
||||
// an encoding or an endpoint decides what a person can do with it.
|
||||
//
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(new TokenizerRef(TokenizerKind.HUGGING_FACE, string.Empty).IsKnown, Is.False);
|
||||
Assert.That(new TokenizerRef(TokenizerKind.UNKNOWN, "o200k_base").IsKnown, Is.False);
|
||||
Assert.That(TokenizerRef.UNKNOWN.IsKnown, Is.False);
|
||||
Assert.That(new TokenizerRef(TokenizerKind.TIKTOKEN, "o200k_base").IsKnown, Is.True);
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
using AIStudio.Models.Registry;
|
||||
using AIStudio.Provider;
|
||||
// ReSharper disable InconsistentNaming
|
||||
|
||||
namespace AIStudio.Tests.Models.ZAI;
|
||||
|
||||
/// <summary>
|
||||
/// Checks how a GLM name says that the model looks at pictures.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Z AI marks its vision models by gluing a "v" to the version number: glm-4v, glm-4.1v, glm-4.5v.
|
||||
/// That is not a name part, so no pattern can ask about it, and the family works it out of the name
|
||||
/// instead -- the second of the two places in the rebuilt rules where a capability is calculated.
|
||||
///
|
||||
/// These need tests of their own because the corpus cannot tell the calculation apart from a
|
||||
/// careless one. Looking for a bare "v" anywhere answers every corpus name the same way, and is
|
||||
/// still wrong: a quantized build carries one in "nvfp4", and so do the names of several inference
|
||||
/// providers. The corpus happens to hold that name only for a generation which reads images anyway.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class GlmFamilyTests
|
||||
{
|
||||
[TestCase("glm-4.5v", TestName = "The vision marker sits behind the version")]
|
||||
[TestCase("glm-4v", TestName = "A version without a dot carries the marker just the same")]
|
||||
[TestCase("glm-4.1v-9b", TestName = "A size may follow the marker")]
|
||||
public void AGlmWhoseVersionCarriesTheMarkerLooksAtPictures(string modelId)
|
||||
{
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.SELF_HOSTED, modelId);
|
||||
|
||||
Assert.That(profile.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.True);
|
||||
}
|
||||
|
||||
[TestCase("glm-4-9b-chat-nvfp4", TestName = "A quantized build is not a vision model")]
|
||||
[TestCase("glm-4-9b-chat", TestName = "The plain 4 line reads text only")]
|
||||
[TestCase("glm-4.6-latest", TestName = "A rolling tag says nothing about pictures")]
|
||||
public void AGlmCarryingAVSomewhereElseDoesNot(string modelId)
|
||||
{
|
||||
var profile = ModelRegistry.Shared.Profile(LLMProviders.SELF_HOSTED, modelId);
|
||||
|
||||
Assert.That(profile.Has(Capability.MULTIPLE_IMAGE_INPUT), Is.False);
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user