Rebuilt how AI Studio knows what a model can do (#960)
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions

This commit is contained in:
Thorsten Sommer authored and GitHub committed 2026-09-13 14:17:25 +02:00
1 parent d21e09dd1e
commit d85b4e71b6
287 files changed
+18341 -3677

No files matched your search

@@ -0,0 +1,91 @@
using static AIStudio.Provider.Capability;
namespace AIStudio.Models.Google;
/// <summary>
/// The Gemini chat models.
/// </summary>
/// <remarks>
/// A Gemini reads everything -- text, images, audio, speech, video -- writes text, and calls tools.
/// That is the first rule, and it is the family's own fallback for a Gemini nobody has written a
/// rule for yet. What the generations add to it is how they think, and the older exceptions take
/// something away instead.
///
/// Every generation gets a line of its own, including the dotted ones. The dot is a version
/// boundary rather than a name part boundary, deliberately -- it is what keeps llama3 and llama3.1
/// apart -- so a rule for "gemini-3" does not answer for "gemini-3.1", and each has to say so
/// itself. The previous rules searched for "gemini-3" anywhere in the name and covered unreleased
/// versions by accident; the price of not doing that is a line per generation, and the verification
/// run names any model of the corpus which finds no rule.
/// </remarks>
public sealed class GeminiFamily : ModelFamily
{
/// <inheritdoc />
public override ModelVendor Vendor => ModelVendor.GOOGLE;
/// <inheritdoc />
public override ModelSource Source => new("https://ai.google.dev/gemini-api/docs/gemini-3", new DateOnly(2026, 9, 12), "Capabilities ported unchanged from ProviderExtensions.Google.cs: one shape for all of Gemini, one sentence per generation about thinking. The Gemini 3 guide states a one million token input window; the 2.5 model pages state their input limit as 1,048,576, and both numbers are written here as their page gives them.");
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://ai.google.dev/gemini-api/docs/image-understanding", new DateOnly(2026, 9, 12), "States one number for the whole family: \"Gemini models support a maximum of 3,600 image files per request.\" The 20 MB it also names is a limit on the request body rather than on the number of images."),
new("https://ai.google.dev/gemini-api/docs/tokens", new DateOnly(2026, 9, 12), "Google publishes no tokenizer file for Gemini. Counting happens through the countTokens method of the API, which returns the number of tokens of the input alone.")
];
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
//
// Google states the image limit once, for all of Gemini, so it sits on the fallback and
// every generation inherits it. The two rules below which do not inherit from here say
// nothing about it: the live model looks at no still images at all, and for the 1.0 vision
// model Google's current pages state no number any more.
//
builder.Rule("gemini").AsSegment()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | AUDIO_INPUT | SPEECH_INPUT | VIDEO_INPUT | TEXT_OUTPUT | FUNCTION_CALLING)
.Apis(CHAT_COMPLETION_API)
.Images(maxPerRequest: 3_600)
.Tokenizer(TokenizerKind.PROVIDER_API, "countTokens");
// The one Gemini which only ever read text and images:
builder.Rule("gemini-1.0-pro-vision").AsPrefix()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT)
.Apis(CHAT_COMPLETION_API);
//
// The live model, which belongs to a different API: it speaks back, and it is the one
// Gemini that does not look at still images.
//
builder.Rule("gemini-2.0-flash-live").AsPrefix()
.Capabilities(TEXT_INPUT | AUDIO_INPUT | SPEECH_INPUT | VIDEO_INPUT | TEXT_OUTPUT | SPEECH_OUTPUT | FUNCTION_CALLING)
.Apis(CHAT_COMPLETION_API);
//
// Google states an input limit and an output limit rather than one window. The input limit
// is the one a conversation is measured against, because that is where the conversation
// accumulates, so that is the number written here.
//
builder.Rule("gemini-2.5").AsPrefix().InheritsFrom("gemini")
.Reasoning(ReasoningSupport.ALWAYS)
.ContextWindow(1_048_576);
//
// The one exception of the 2.5 line: it can think, but only when asked. From the 3.x line
// on, even the Flash Lite models think at their lowest level.
//
builder.Rule("gemini-2.5-flash-lite").AsPrefix().InheritsFrom("gemini-2.5")
.Reasoning(ReasoningSupport.OPTIONAL);
builder.Rule("gemini-3").AsPrefix().InheritsFrom("gemini")
.Reasoning(ReasoningSupport.ALWAYS)
.ContextWindow(1_000_000);
builder.Rule("gemini-3.1").AsPrefix().InheritsFrom("gemini-3");
builder.Rule("gemini-3.7").AsPrefix().InheritsFrom("gemini-3");
// The two rolling aliases, which carry no version number and point at the current line:
builder.Rule("gemini-flash-latest").AsExact().InheritsFrom("gemini-3");
builder.Rule("gemini-pro-latest").AsExact().InheritsFrom("gemini-3");
}
}
@@ -0,0 +1,42 @@
using static AIStudio.Provider.Capability;
namespace AIStudio.Models.Google;
/// <summary>
/// The Gemini models which draw as well as write.
/// </summary>
/// <remarks>
/// They are named like every other Gemini, with a version and a size, and the only thing setting
/// them apart is the name part "image". So the rules here are the generation rules of the chat
/// family with that one part required on top, and requiring it is exactly what makes them win: two
/// rules reaching equally far into a name are separated by how many conditions they carry.
///
/// What they can do is nearly the opposite of what their generation can. They write images, which
/// no chat Gemini does, and they call no tools, which every chat Gemini does. Reading them as chat
/// models of their line -- which is what happens when nobody asks about the image part first --
/// promises tool calling that is not there.
/// </remarks>
public sealed class GeminiImageFamily : ModelFamily
{
/// <inheritdoc />
public override ModelVendor Vendor => ModelVendor.GOOGLE;
/// <inheritdoc />
public override ModelSource Source => new("https://ai.google.dev/gemini-api/docs/image-generation", new DateOnly(2026, 9, 11), "Ported unchanged from the rules in ProviderExtensions.Google.cs: images out, no tool calling, and thinking from the 3 line on.");
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
builder.Rule("gemini-2.5").AsPrefix().AlsoContains("image")
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT | IMAGE_OUTPUT)
.Apis(CHAT_COMPLETION_API);
// From the 3 line on they think about a complicated prompt, and it cannot be switched off:
builder.Rule("gemini-3").AsPrefix().AlsoContains("image").Inherits()
.Reasoning(ReasoningSupport.ALWAYS);
// Only the 3.1 Flash image models watch video:
builder.Rule("gemini-3.1").AsPrefix().AlsoContains("image").Inherits()
.Capabilities(VIDEO_INPUT);
}
}
@@ -0,0 +1,78 @@
using static AIStudio.Provider.Capability;
namespace AIStudio.Models.Google;
/// <summary>
/// Gemma, the open weights Google publishes next to Gemini.
/// </summary>
/// <remarks>
/// Two generations and one spelling problem. Ollama writes "gemma3:27b" and the hub writes
/// "gemma-3-27b-it", and no normalization turns one into the other, so each statement stands twice.
/// What it buys is that the rules never have to ask who served the model.
///
/// Tool calling is the line between the generations. What Google documents for Gemma 3 is writing
/// the tool descriptions into the prompt by hand, which is a different thing from what the tools
/// field of an OpenAI-compatible request does: the chat template has neither a tool role nor tool
/// tokens, and Ollama refuses a request carrying tools for these models. Gemma 4 is the first with
/// tokens of its own, and the first that thinks -- when the request opens the thinking channel.
/// </remarks>
public sealed class GemmaFamily : ModelFamily
{
/// <inheritdoc />
public override ModelVendor Vendor => ModelVendor.GOOGLE;
/// <inheritdoc />
public override ModelSource Source => new("https://ai.google.dev/gemma/docs/core", new DateOnly(2026, 9, 11), "Ported unchanged from the Gemma block of ProviderExtensions.OpenSource.cs.");
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
// The early generations take text only and were not built for tools:
builder.Rule("gemma").AsSubstring()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT)
.Apis(CHAT_COMPLETION_API);
// Gemma 3 reads pictures from the 4B checkpoint upwards:
builder.Rule("gemma3").AsSubstring()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT)
.Apis(CHAT_COMPLETION_API);
builder.Rule("gemma-3").AsSubstring().Inherits();
// The 1B checkpoint is the one that does not:
builder.Rule("gemma3").AsSubstring().AlsoContains("1b")
.Capabilities(TEXT_INPUT | TEXT_OUTPUT)
.Apis(CHAT_COMPLETION_API);
builder.Rule("gemma-3").AsSubstring().AlsoContains("1b").Inherits();
//
// The 3n checkpoints listen as well. Video is not a modality of any Gemma: the model cards
// list text, image, and audio, and mention video only as frames somebody else cut it into.
//
builder.Rule("gemma3n").AsSubstring()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | AUDIO_INPUT | TEXT_OUTPUT)
.Apis(CHAT_COMPLETION_API);
builder.Rule("gemma-3n").AsSubstring().Inherits();
// Every checkpoint of Gemma 4 is multimodal; there is no text-only variant of it:
builder.Rule("gemma4").AsSubstring()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT | FUNCTION_CALLING)
.Apis(CHAT_COMPLETION_API)
.Reasoning(ReasoningSupport.OPTIONAL);
builder.Rule("gemma-4").AsSubstring().Inherits();
// Three of its checkpoints hear, and they are named one by one because that is all they
// have in common:
builder.Rule("gemma4").AsSubstring().AlsoContains("e2b").Inherits().Capabilities(AUDIO_INPUT);
builder.Rule("gemma-4").AsSubstring().AlsoContains("e2b").Inherits();
builder.Rule("gemma4").AsSubstring().AlsoContains("e4b").Inherits();
builder.Rule("gemma-4").AsSubstring().AlsoContains("e4b").Inherits();
builder.Rule("gemma4").AsSubstring().AlsoContains("12b").Inherits();
builder.Rule("gemma-4").AsSubstring().AlsoContains("12b").Inherits();
}
}
@@ -0,0 +1,35 @@
using AIStudio.Provider;
using static AIStudio.Provider.Capability;
namespace AIStudio.Models.Google;
/// <summary>
/// Google's embedding models, which turn text into a vector and answer nothing.
/// </summary>
/// <remarks>
/// The previous rules answered for these with the Google default: images in, text out, tool
/// calling. None of it is true, and the app already knows better -- it asks every provider for its
/// embedding models through a method of its own.
///
/// The Gemini one needs a rule of its own for another reason: its name begins with "gemini", so
/// without one it would be read as a chat model of the family.
/// </remarks>
public sealed class GoogleEmbeddingFamily : ModelFamily
{
/// <inheritdoc />
public override ModelVendor Vendor => ModelVendor.GOOGLE;
/// <inheritdoc />
public override ModelSource Source => new("https://ai.google.dev/gemini-api/docs/embeddings", new DateOnly(2026, 9, 11), "The app lists these under IProvider.GetEmbeddingModels, which is where the statement that they embed comes from.");
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
builder.Rule("text-embedding-004").AsExact()
.Capabilities(TEXT_INPUT | EMBEDDING)
.Kind(ModelKind.EMBEDDING);
builder.Rule("gemini-embedding").AsPrefix().Inherits();
}
}
@@ -0,0 +1,32 @@
using AIStudio.Provider;
using static AIStudio.Provider.Capability;
namespace AIStudio.Models.Google;
/// <summary>
/// Imagen, which draws a picture from a description and does nothing else.
/// </summary>
/// <remarks>
/// The previous rules had no branch for it. Its name does not contain "gemini", so it fell to the
/// last line of the Google function and was answered as a chat model: reads images, writes text,
/// calls functions. Not one of the three is true, and the one thing it does -- writing an image --
/// was not said at all.
///
/// Whole name parts, not a substring: "imagen" also sits inside "imagenet" and "reimagined", and a
/// chat model carrying such a word would be turned into an image generator by a careless match.
/// </remarks>
public sealed class ImagenFamily : ModelFamily
{
/// <inheritdoc />
public override ModelVendor Vendor => ModelVendor.GOOGLE;
/// <inheritdoc />
public override ModelSource Source => new("https://ai.google.dev/gemini-api/docs/imagen", new DateOnly(2026, 9, 11), "A description goes in and an image comes out; there is no conversation and no tool calling.");
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder) =>
builder.Rule("imagen").AsSegment()
.Capabilities(TEXT_INPUT | IMAGE_OUTPUT)
.Kind(ModelKind.IMAGE_GENERATION);
}