Rebuilt how AI Studio knows what a model can do (#960)
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions

This commit is contained in:
Thorsten Sommer authored and GitHub committed 2026-09-13 14:17:25 +02:00
1 parent d21e09dd1e
commit d85b4e71b6
287 files changed
+18341 -3677

No files matched your search

@@ -32,7 +32,7 @@ public sealed class ProviderAlibabaCloud() : BaseProvider(LLMProviders.ALIBABA_C
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -41,8 +41,8 @@ public sealed class ProviderAnthropic() : BaseProvider(LLMProviders.ANTHROPIC, n
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesAsync(
this.Provider, chatModel,
this.CreateSettingsProvider(chatModel),
// Anthropic-specific role mapping:
role => role switch
{
@@ -6,6 +6,8 @@ using System.Text.Json;
using System.Text.Json.Serialization;
using AIStudio.Chat;
using AIStudio.Models;
using AIStudio.Models.Live;
using AIStudio.Provider.Anthropic;
using AIStudio.Provider.OpenAI;
using AIStudio.Provider.SelfHosted;
@@ -191,6 +193,7 @@ public abstract class BaseProvider : IProvider, ISecretId
Action<HttpRequestMessage, string>? requestConfigurator = null,
JsonSerializerOptions? jsonSerializerOptions = null,
bool isTryingSecret = false,
Func<TResponse, IEnumerable<ModelListing>>? listingFactory = null,
CancellationToken token = default)
{
var secretKey = await this.GetModelLoadingSecretKey(storeType, apiKeyProvisional, isTryingSecret);
@@ -220,6 +223,16 @@ public abstract class BaseProvider : IProvider, ISecretId
if (parsedResponse is null)
return FailedModelLoadResult(ModelLoadFailureReason.INVALID_RESPONSE, "Model list response could not be deserialized.");
//
// What the list stated about the models, read before anything is filtered out of
// it: a model left out below as an embedding model is still a model somebody may
// have configured this instance with, and a list like this one is the only place
// its window is ever stated. Only pass a whole list in here -- reporting a part of
// one would tell the app that everything left out has stopped existing.
//
if (listingFactory is not null)
ListedModels.Shared.Report(this.ConfiguredProviderId, listingFactory(parsedResponse));
return SuccessfulModelLoadResult(modelFactory(parsedResponse));
}
catch (Exception e)
@@ -235,13 +248,34 @@ public abstract class BaseProvider : IProvider, ISecretId
}
}
protected virtual string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason) => failureReason switch
/// <summary>
/// Says what a failed request means for the user.
/// </summary>
/// <remarks>
/// The window is only ever known where the caller knows which model the request was for, which
/// is why it is optional rather than a second required argument: most failures say nothing
/// about a length and need no number to explain themselves.
/// </remarks>
/// <param name="failureReason">Why the request failed.</param>
/// <param name="contextWindow">What the model reads, where that is known.</param>
/// <returns>The message to show, or an empty string when we have nothing to say.</returns>
protected virtual string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason, ContextWindow contextWindow = default) => failureReason switch
{
ProviderRequestFailureReason.TOO_MANY_REQUESTS => TB("The provider rejected the request because too many requests were sent. Please wait a moment and try again."),
ProviderRequestFailureReason.INVALID_OR_MISSING_API_KEY => string.Format(TB("The API key for the provider '{0}' is missing or was rejected. Please check the key in the settings."), this.InstanceName),
ProviderRequestFailureReason.AUTHENTICATION_OR_PERMISSION_ERROR => string.Format(TB("The provider '{0}' refused the request. Your account might not be allowed to use the selected model, or the provider might not serve your region."), this.InstanceName),
ProviderRequestFailureReason.PROVIDER_UNAVAILABLE => string.Format(TB("The provider '{0}' could not be reached. Please check whether it is running and reachable, then try again."), this.InstanceName),
ProviderRequestFailureReason.MODEL_NOT_FOUND => string.Format(TB("The provider '{0}' does not know the selected model. Please select another model."), this.InstanceName),
//
// Naming the number is the whole point of knowing it: "too long" leaves the user guessing
// by how much, while the window turns the next step into arithmetic. Where nobody knows the
// window, no number is invented -- the sentence below says the same thing without one.
//
// Written out in full rather than shortened the way the chat shortens it. The sentence ends
// by asking the user to set a chunk size, and 32.77k is not a number anybody types into a
// field.
//
ProviderRequestFailureReason.CONTEXT_LENGTH_EXCEEDED when contextWindow.IsKnown => string.Format(TB("The text was longer than the selected model accepts, which is {0} tokens. Please select a model which takes longer texts, or reduce the chunk size of the data source."), contextWindow.DefaultTokens.ToString("N0", I18N.I.Culture)),
ProviderRequestFailureReason.CONTEXT_LENGTH_EXCEEDED => TB("The text was longer than the selected model accepts. Please select a model which takes longer texts, or reduce the chunk size of the data source."),
ProviderRequestFailureReason.TOOLS_NOT_SUPPORTED => string.Format(TB("The selected model is not able to use tools. Please select a model which can, or open the settings of the provider '{0}', show its expert settings, and switch the function calling capability off there."), this.InstanceName),
ProviderRequestFailureReason.EMBEDDINGS_NOT_SUPPORTED => string.Format(TB("The provider '{0}' cannot create embeddings. Please select a provider which offers an embedding model."), this.InstanceName),
@@ -267,10 +301,18 @@ public abstract class BaseProvider : IProvider, ISecretId
/// Shared with the providers which talk to an embedding endpoint of their own: what the user
/// needs to know does not depend on which route the request took.
/// </remarks>
protected ProviderRequestException CreateEmbeddingRequestException(HttpStatusCode statusCode, string reasonPhrase, string responseBody)
protected ProviderRequestException CreateEmbeddingRequestException(HttpStatusCode statusCode, string reasonPhrase, string responseBody, Model embeddingModel)
{
//
// What the rules know about this model, corrected by whatever this installation reported
// about it. That is the same walk a configured chat provider takes, minus the expert
// settings: an embedding provider has none, so there is nothing above the two to ask.
//
var stated = this.Provider.GetModelProfile(embeddingModel);
var contextWindow = ListedModels.Shared.Of(this.ConfiguredProviderId, embeddingModel.Id).ApplyTo(stated).Context;
var failureReason = this.ClassifyEmbeddingRequestFailure(statusCode, responseBody);
var userMessage = this.GetProviderRequestFailureUserMessage(failureReason);
var userMessage = this.GetProviderRequestFailureUserMessage(failureReason, contextWindow);
// We know nothing about this failure, so we pass on what the provider said about it:
if (string.IsNullOrWhiteSpace(userMessage))
@@ -1539,7 +1581,7 @@ public abstract class BaseProvider : IProvider, ISecretId
// thousands being indexed in the background or the one thing the user just asked
// for, and only it can decide how often the user should hear about it.
//
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody);
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody, embeddingModel);
}
var embeddingResponse = JsonSerializer.Deserialize<EmbeddingResponse>(responseBody, JSON_SERIALIZER_OPTIONS);
+72 -42
View File
@@ -3,115 +3,145 @@ namespace AIStudio.Provider;
/// <summary>
/// Represents the capabilities of an AI model.
/// </summary>
public enum Capability
/// <remarks>
/// A set of capabilities is one value, not a collection: a model profile carries this enum as a
/// single field, and asking whether a capability is present is one bit test instead of a walk
/// through a list. That is why the members are powers of two.
///
/// The numeric values are an implementation detail and are never written anywhere. Overrides,
/// plugins, and the settings file all address a capability by its name, so the names are the part
/// which must not change. Removing a member would silently drop the override an organization wrote
/// for it, which is why the members we no longer hand out ourselves are still here.
///
/// Adding a member means adding the next free bit. Sixty-four of them fit; should they ever run
/// out, the answer is a second enum next to this one rather than a wider underlying type, because
/// widening changes the meaning of every value already written down.
/// </remarks>
[Flags]
public enum Capability : ulong
{
/// <summary>
/// No capabilities specified.
/// </summary>
NONE,
NONE = 0,
/// <summary>
/// We don't know what the AI model can do.
/// </summary>
UNKNOWN,
UNKNOWN = 1UL << 0,
/// <summary>
/// The AI model can perform text input.
/// </summary>
TEXT_INPUT,
TEXT_INPUT = 1UL << 1,
/// <summary>
/// The AI model can perform audio input, such as music or sound.
/// </summary>
AUDIO_INPUT,
AUDIO_INPUT = 1UL << 2,
/// <summary>
/// The AI model can perform one image input, such as one photo or drawing.
/// </summary>
SINGLE_IMAGE_INPUT,
SINGLE_IMAGE_INPUT = 1UL << 3,
/// <summary>
/// The AI model can perform multiple images as input, such as multiple photos or drawings.
/// </summary>
MULTIPLE_IMAGE_INPUT,
MULTIPLE_IMAGE_INPUT = 1UL << 4,
/// <summary>
/// The AI model can perform speech input.
/// </summary>
SPEECH_INPUT,
SPEECH_INPUT = 1UL << 5,
/// <summary>
/// The AI model can perform video input, such as video files or streams.
/// </summary>
VIDEO_INPUT,
VIDEO_INPUT = 1UL << 6,
/// <summary>
/// The AI model can generate text output.
/// </summary>
TEXT_OUTPUT,
TEXT_OUTPUT = 1UL << 7,
/// <summary>
/// The AI model can generate audio output, such as music or sound.
/// </summary>
AUDIO_OUTPUT,
AUDIO_OUTPUT = 1UL << 8,
/// <summary>
/// The AI model can generate image output, such as photos or drawings.
/// </summary>
IMAGE_OUTPUT,
IMAGE_OUTPUT = 1UL << 9,
/// <summary>
/// The AI model can generate speech output.
/// </summary>
SPEECH_OUTPUT,
SPEECH_OUTPUT = 1UL << 10,
/// <summary>
/// The AI model can generate video output.
/// </summary>
VIDEO_OUTPUT,
VIDEO_OUTPUT = 1UL << 11,
/// <summary>
/// The AI model can perform reasoning tasks. You can enable reasoning optionally, but it is disabled by default.
/// </summary>
OPTIONAL_REASONING,
/// <remarks>
/// Override vocabulary. A model profile states how a model reasons through its ReasoningSupport
/// field and never sets this flag, because the three reasoning flags can be combined into
/// answers no model can give. Asking a profile whether it has this capability always says no.
/// </remarks>
OPTIONAL_REASONING = 1UL << 12,
/// <summary>
/// The AI model always performs reasoning. There is no option to disable reasoning.
/// </summary>
ALWAYS_REASONING,
/// <remarks>
/// Override vocabulary. A model profile states how a model reasons through its ReasoningSupport
/// field and never sets this flag, because the three reasoning flags can be combined into
/// answers no model can give. Asking a profile whether it has this capability always says no.
/// </remarks>
ALWAYS_REASONING = 1UL << 13,
/// <summary>
/// The AI model performs optional reasoning, but it is enabled by default.
/// </summary>
REASONING_BY_DEFAULT,
/// <remarks>
/// Override vocabulary. A model profile states how a model reasons through its ReasoningSupport
/// field and never sets this flag, because the three reasoning flags can be combined into
/// answers no model can give. Asking a profile whether it has this capability always says no.
/// </remarks>
REASONING_BY_DEFAULT = 1UL << 14,
/// <summary>
/// The AI model can embed information or data.
/// </summary>
EMBEDDING,
EMBEDDING = 1UL << 15,
/// <summary>
/// The AI model can perform in real-time.
/// </summary>
REALTIME,
REALTIME = 1UL << 16,
/// <summary>
/// The AI model can perform function calling, such as invoking APIs or executing functions.
/// </summary>
FUNCTION_CALLING,
FUNCTION_CALLING = 1UL << 17,
/// <summary>
/// The AI model can perform web search to retrieve information from the internet.
/// </summary>
WEB_SEARCH,
WEB_SEARCH = 1UL << 18,
/// <summary>
/// The AI model is used via the Chat Completion API.
/// </summary>
CHAT_COMPLETION_API,
CHAT_COMPLETION_API = 1UL << 19,
/// <summary>
/// The AI model is used via the Responses API.
/// </summary>
RESPONSES_API,
RESPONSES_API = 1UL << 20,
}
@@ -32,7 +32,7 @@ public sealed class ProviderDeepSeek() : BaseProvider(LLMProviders.DEEP_SEEK, ne
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -103,7 +103,7 @@ public sealed class ProviderDeepSeek() : BaseProvider(LLMProviders.DEEP_SEEK, ne
return this.LoadModelsResponse<ModelsResponse>(
storeType,
"models",
modelResponse => modelResponse.Data.Where(model => model.IsChatModel()),
modelResponse => modelResponse.Data.Where(model => model.IsChatModel(this.Provider)),
apiKeyProvisional, token: token);
}
}
@@ -32,7 +32,7 @@ public class ProviderFireworks() : BaseProvider(LLMProviders.FIREWORKS, new Uri(
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -40,7 +40,7 @@ public sealed class ProviderGWDG() : BaseProvider(LLMProviders.GWDG, new Uri("ht
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -88,7 +88,7 @@ public sealed class ProviderGWDG() : BaseProvider(LLMProviders.GWDG, new Uri("ht
var result = await this.LoadModels(SecretStoreType.LLM_PROVIDER, apiKeyProvisional, token);
return result with
{
Models = [..result.Models.Where(model => model.IsChatModel())]
Models = [..result.Models.Where(model => model.IsChatModel(this.Provider))]
};
}
@@ -112,7 +112,7 @@ public sealed class ProviderGWDG() : BaseProvider(LLMProviders.GWDG, new Uri("ht
if (!result.Success)
return result;
var embeddingModels = result.Models.Where(model => model.IsEmbeddingModel()).ToList();
var embeddingModels = result.Models.Where(model => model.IsEmbeddingModel(this.Provider)).ToList();
if (embeddingModels.Count is 0)
return ModelLoadResult.FromModels(KNOWN_EMBEDDING_MODELS);
@@ -34,7 +34,7 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -116,7 +116,7 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
if (!response.IsSuccessStatusCode)
{
LOGGER.LogError("Embedding request failed with status code {ResponseStatusCode} and body: '{ResponseBody}'.", response.StatusCode, responseBody);
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody);
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody, embeddingModel);
}
var embeddingResponse = JsonSerializer.Deserialize<GoogleEmbeddingResponse>(responseBody, JSON_SERIALIZER_OPTIONS);
@@ -168,9 +168,15 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
{
Models =
[
//
// Asking what a model is made for, rather than only ruling out the embedding ones.
// Google names everything after the chat model it grew out of, so the catalog is
// full of names which look like something to talk to and are not: the image models,
// and the computer use model whose API refuses a request without its tool.
//
..result.Models.Where(model =>
model.Id.StartsWith("gemini-", StringComparison.OrdinalIgnoreCase) &&
!this.IsEmbeddingModel(model.Id))
model.IsChatModel(this.Provider))
.Select(this.WithDisplayNameFallback)
]
};
@@ -189,7 +195,7 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
{
Models =
[
..result.Models.Where(model => this.IsEmbeddingModel(model.Id))
..result.Models.Where(model => model.IsEmbeddingModel(this.Provider))
.Select(this.WithDisplayNameFallback)
]
};
@@ -222,12 +228,6 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
token: token);
}
private bool IsEmbeddingModel(string modelId)
{
return modelId.Contains("embedding", StringComparison.OrdinalIgnoreCase) ||
modelId.Contains("embed", StringComparison.OrdinalIgnoreCase);
}
private Model WithDisplayNameFallback(Model model)
{
return string.IsNullOrWhiteSpace(model.DisplayName)
@@ -0,0 +1,15 @@
using System.Text.Json.Serialization;
namespace AIStudio.Provider.Groq;
/// <summary>
/// One model as Groq lists it.
/// </summary>
/// <remarks>
/// Groq says more about a model than the shared OpenAI-compatible list does, which is why this
/// provider brings a data model of its own instead of using that one: the shared record is read by
/// a dozen providers, and a field only one of them sends has no business in it.
/// </remarks>
/// <param name="Id">The model's ID.</param>
/// <param name="ContextWindowTokens">How much the model reads and writes in one conversation, in tokens.</param>
public readonly record struct GroqModel(string Id, [property: JsonPropertyName("context_window")] int? ContextWindowTokens);
@@ -0,0 +1,7 @@
namespace AIStudio.Provider.Groq;
/// <summary>
/// A data model for the response from the Groq models endpoint.
/// </summary>
/// <param name="Data">The models Groq serves.</param>
public readonly record struct GroqModelsResponse(IList<GroqModel> Data);
@@ -1,6 +1,7 @@
using System.Runtime.CompilerServices;
using AIStudio.Chat;
using AIStudio.Models.Live;
using AIStudio.Provider.OpenAI;
using AIStudio.Settings;
@@ -35,7 +36,7 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
apiParameters["seed"] = parsedSeed;
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -83,7 +84,7 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
var result = await this.LoadModels(SecretStoreType.LLM_PROVIDER, apiKeyProvisional, token);
return result with
{
Models = [..result.Models.Where(model => model.IsChatModel())]
Models = [..result.Models.Where(model => model.IsChatModel(this.Provider))]
};
}
@@ -105,7 +106,7 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
var result = await this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, apiKeyProvisional, token);
return result with
{
Models = [..result.Models.Where(model => model.IsTranscriptionModel())]
Models = [..result.Models.Where(model => model.IsTranscriptionModel(this.Provider))]
};
}
@@ -113,10 +114,12 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
private Task<ModelLoadResult> LoadModels(SecretStoreType storeType, string? apiKeyProvisional, CancellationToken token)
{
return this.LoadModelsResponse<ModelsResponse>(
return this.LoadModelsResponse<GroqModelsResponse>(
storeType,
"models",
modelResponse => modelResponse.Data,
apiKeyProvisional, token: token);
modelResponse => modelResponse.Data.Select(n => new Model(n.Id, null)),
apiKeyProvisional,
listingFactory: modelResponse => modelResponse.Data.Select(n => ModelListing.For(n.Id, n.ContextWindowTokens)),
token: token);
}
}
@@ -34,7 +34,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -84,7 +84,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
{
Models =
[
..result.Models.Where(model => model.IsChatModel())
..result.Models.Where(model => model.IsChatModel(this.Provider))
]
};
}
@@ -103,7 +103,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
{
Models =
[
..result.Models.Where(model => model.IsEmbeddingModel())
..result.Models.Where(model => model.IsEmbeddingModel(this.Provider))
]
};
}
@@ -116,7 +116,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
{
Models =
[
..result.Models.Where(model => model.IsTranscriptionModel())
..result.Models.Where(model => model.IsTranscriptionModel(this.Provider))
]
};
}
@@ -31,7 +31,7 @@ public sealed class ProviderHetzner() : BaseProvider(LLMProviders.HETZNER, new U
settingsManager,
async (systemPrompt, apiParameters, tools) =>
{
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -69,7 +69,7 @@ public sealed class ProviderHetzner() : BaseProvider(LLMProviders.HETZNER, new U
/// <inheritdoc />
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", modelResponse => modelResponse.Data.Where(model => model.IsChatModel()), apiKeyProvisional, token: token);
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", modelResponse => modelResponse.Data.Where(model => model.IsChatModel(this.Provider)), apiKeyProvisional, token: token);
}
/// <inheritdoc />
@@ -5,4 +5,30 @@ namespace AIStudio.Provider.HuggingFace;
/// </summary>
/// <param name="Id">The ID of the model, written as "org/model".</param>
/// <param name="Providers">The inference providers serving this model.</param>
public readonly record struct HFModel(string Id, IList<HFModelProvider>? Providers);
public readonly record struct HFModel(string Id, IList<HFModelProvider>? Providers)
{
/// <summary>
/// The window this model has when it is reached the way this user set things up.
/// </summary>
/// <remarks>
/// A window belongs to an inference provider here, not to the model: the same weights run
/// behind several of them, each configured by somebody else. Where the user named one, its
/// number is the answer. Where they let the router choose, the smallest window among the
/// providers currently serving the model is -- nobody knows which one the router will take, and
/// a number promising more than the chosen provider delivers would walk a conversation into an
/// error the user could not see coming.
/// </remarks>
/// <param name="providerSlug">The inference provider the user chose, or empty when the router chooses.</param>
/// <returns>The window in tokens, or null where nobody stated one.</returns>
public int? ContextWindowTokens(string providerSlug)
{
if (this.Providers is null)
return null;
var serving = this.Providers.Where(provider => provider.IsLive);
if (!string.IsNullOrEmpty(providerSlug))
serving = serving.Where(provider => string.Equals(provider.Provider, providerSlug, StringComparison.OrdinalIgnoreCase));
return serving.Where(provider => provider.ContextWindowTokens is > 0).Min(provider => provider.ContextWindowTokens);
}
}
@@ -1,3 +1,5 @@
using System.Text.Json.Serialization;
namespace AIStudio.Provider.HuggingFace;
/// <summary>
@@ -5,4 +7,13 @@ namespace AIStudio.Provider.HuggingFace;
/// </summary>
/// <param name="Provider">The slug of the inference provider, e.g. "novita".</param>
/// <param name="Status">Whether the provider currently serves the model. Known value: "live".</param>
public readonly record struct HFModelProvider(string Provider, string Status);
/// <param name="ContextWindowTokens">How much this provider reads and writes in one conversation, in tokens.</param>
public readonly record struct HFModelProvider(string Provider, string Status, [property: JsonPropertyName("context_length")] int? ContextWindowTokens)
{
private const string LIVE = "live";
/// <summary>
/// Whether this provider serves the model right now.
/// </summary>
public bool IsLive => string.Equals(this.Status, LIVE, StringComparison.OrdinalIgnoreCase);
}
@@ -2,6 +2,8 @@
using System.Runtime.CompilerServices;
using AIStudio.Chat;
using AIStudio.Models;
using AIStudio.Models.Live;
using AIStudio.Provider.OpenAI;
using AIStudio.Settings;
using AIStudio.Tools.PluginSystem;
@@ -127,10 +129,10 @@ public sealed class ProviderHuggingFace : BaseProvider
}
/// <inheritdoc />
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason)
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason, ContextWindow contextWindow = default)
{
if (failureReason is not ProviderRequestFailureReason.MODEL_NOT_SUPPORTED_BY_PROVIDER)
return base.GetProviderRequestFailureUserMessage(failureReason);
return base.GetProviderRequestFailureUserMessage(failureReason, contextWindow);
//
// When Hugging Face chose the provider itself, naming it back to the user would help
@@ -166,7 +168,7 @@ public sealed class ProviderHuggingFace : BaseProvider
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -221,7 +223,23 @@ public sealed class ProviderHuggingFace : BaseProvider
/// <inheritdoc />
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", this.SelectChatModels, apiKeyProvisional, token: token);
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", this.SelectChatModels, apiKeyProvisional, listingFactory: this.ListingsOf, token: token);
}
/// <summary>
/// What the router stated about the models it knows.
/// </summary>
/// <remarks>
/// Every model the router reports, not only the ones offered for chatting below: which models
/// are offered depends on the chosen inference provider, while a window belongs to whoever is
/// configured here, and both questions are asked of the same list.
/// </remarks>
/// <param name="response">The response of the model endpoint.</param>
/// <returns>One listing per model, which says nothing for the models nobody stated a window for.</returns>
private IEnumerable<ModelListing> ListingsOf(ModelsResponse response)
{
var providerSlug = this.hfProvider.EndpointsId();
return response.Data.Select(hfModel => ModelListing.For(hfModel.Id, hfModel.ContextWindowTokens(providerSlug)));
}
/// <summary>
@@ -237,7 +255,7 @@ public sealed class ProviderHuggingFace : BaseProvider
/// <returns>The models to offer.</returns>
private IEnumerable<Model> SelectChatModels(ModelsResponse response)
{
var chatModels = response.Data.Where(hfModel => new Model(hfModel.Id, null).IsChatModel());
var chatModels = response.Data.Where(hfModel => new Model(hfModel.Id, null).IsChatModel(this.Provider));
var providerSlug = this.hfProvider.EndpointsId();
if (string.IsNullOrEmpty(providerSlug))
return ToModels(chatModels);
@@ -253,8 +271,8 @@ public sealed class ProviderHuggingFace : BaseProvider
return false;
return hfModel.Providers.Any(provider =>
string.Equals(provider.Provider, providerSlug, StringComparison.OrdinalIgnoreCase) &&
string.Equals(provider.Status, "live", StringComparison.OrdinalIgnoreCase));
provider.IsLive &&
string.Equals(provider.Provider, providerSlug, StringComparison.OrdinalIgnoreCase));
}
/// <inheritdoc />
@@ -39,7 +39,7 @@ public sealed class ProviderIONOS() : BaseProvider(LLMProviders.IONOS, new Uri("
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -84,7 +84,7 @@ public sealed class ProviderIONOS() : BaseProvider(LLMProviders.IONOS, new Uri("
/// <inheritdoc />
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(this.Provider), apiKeyProvisional, token);
}
/// <inheritdoc />
@@ -96,7 +96,7 @@ public sealed class ProviderIONOS() : BaseProvider(LLMProviders.IONOS, new Uri("
/// <inheritdoc />
public override Task<ModelLoadResult> GetEmbeddingModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(this.Provider), apiKeyProvisional, token);
}
/// <inheritdoc />
@@ -32,7 +32,7 @@ public sealed class ProviderLiteLLM(string hostname) : BaseProvider(LLMProviders
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -77,7 +77,7 @@ public sealed class ProviderLiteLLM(string hostname) : BaseProvider(LLMProviders
/// <inheritdoc />
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.LLM_PROVIDER, static model => model.IsChatModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(this.Provider), apiKeyProvisional, token);
}
/// <inheritdoc />
@@ -89,13 +89,13 @@ public sealed class ProviderLiteLLM(string hostname) : BaseProvider(LLMProviders
/// <inheritdoc />
public override Task<ModelLoadResult> GetEmbeddingModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, static model => model.IsEmbeddingModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(this.Provider), apiKeyProvisional, token);
}
/// <inheritdoc />
public override Task<ModelLoadResult> GetTranscriptionModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, static model => model.IsTranscriptionModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, model => model.IsTranscriptionModel(this.Provider), apiKeyProvisional, token);
}
#endregion
@@ -1,3 +1,13 @@
using System.Text.Json.Serialization;
namespace AIStudio.Provider.Mistral;
public readonly record struct Model(string Id, string Object, int Created, string OwnedBy);
/// <summary>
/// One model as Mistral lists it.
/// </summary>
/// <param name="Id">The model's ID.</param>
/// <param name="Object">What kind of thing the entry is. Known value: "model".</param>
/// <param name="Created">When the model was published, as seconds since the epoch.</param>
/// <param name="OwnedBy">Who Mistral names as the owner of the model.</param>
/// <param name="ContextWindowTokens">How much the model reads and writes in one conversation, in tokens.</param>
public readonly record struct Model(string Id, string Object, int Created, string OwnedBy, [property: JsonPropertyName("max_context_length")] int? ContextWindowTokens);
@@ -1,6 +1,7 @@
using System.Runtime.CompilerServices;
using AIStudio.Chat;
using AIStudio.Models.Live;
using AIStudio.Provider.OpenAI;
using AIStudio.Settings;
@@ -38,7 +39,7 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
apiParameters["random_seed"] = parsedRandomSeed;
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -97,7 +98,7 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
// kind detection:
..modelResponse.Models.Where(n =>
!n.Id.StartsWith("code", StringComparison.OrdinalIgnoreCase) &&
n.IsChatModel())
n.IsChatModel(this.Provider))
]
};
}
@@ -111,7 +112,7 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
return modelResponse with
{
Models = [..modelResponse.Models.Where(n => n.IsEmbeddingModel())]
Models = [..modelResponse.Models.Where(n => n.IsEmbeddingModel(this.Provider))]
};
}
@@ -139,6 +140,8 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
storeType,
"models",
modelResponse => modelResponse.Data.Select(n => new Provider.Model(n.Id, null)),
apiKeyProvisional, token: token);
apiKeyProvisional,
listingFactory: modelResponse => modelResponse.Data.Select(n => ModelListing.For(n.Id, n.ContextWindowTokens)),
token: token);
}
}
@@ -76,6 +76,16 @@ public enum ModelKind
/// </remarks>
REALTIME,
/// <summary>
/// The model drives a computer: it looks at a screen and says what to click next.
/// </summary>
/// <remarks>
/// These refuse a plain conversation outright. Google's answer to a request without the computer
/// use tool is "This model requires the use of the Computer Use tool", so the model belongs in no
/// chat list, however much its name looks like the chat model it grew out of.
/// </remarks>
COMPUTER_USE,
/// <summary>
/// The model extracts text from images or scanned documents.
/// </summary>
@@ -1,228 +0,0 @@
namespace AIStudio.Provider;
/// <summary>
/// Determines what kind of model we are dealing with, based on its name.
/// </summary>
/// <remarks>
/// Many providers serve every kind of model through one models endpoint, without telling us what
/// kind each model is. Before this class existed, every provider carried its own list of name
/// fragments to sort those models apart. Those lists disagreed with each other: a model like
/// nomic-embed-text was recognized as an embedding model by some providers, while others offered it
/// as a chat model. The knowledge about model families is the same for all providers, so it lives
/// here now.
///
/// This class recognizes what a model is NOT made for. Everything we do not recognize is reported as
/// a chat model. That direction matters: when a provider adds a model family we have never seen, the
/// user still gets to use it. Getting it wrong the other way around would hide a model the user is
/// paying for.
///
/// What this class must not become is a place for provider-specific knowledge. That a model called
/// "codestral" is a fill-in-the-middle model at Mistral, or that Alibaba's chat models all start
/// with a "q", is true for that one provider only. Such rules stay in the provider.
/// </remarks>
public static class ModelKindExtensions
{
//
// Checked first, because these entries are no models at all: whatever else their name might
// suggest, none of the other kinds applies to them.
//
private static readonly string[] OTHER_MARKERS = ["container"];
//
// Reranking is checked before embedding: rerankers are commonly named after the embedding model
// they belong to, e.g. Qwen3-VL-Reranker-8B next to Qwen3-VL-Embedding-8B.
//
private static readonly string[] RERANKING_MARKERS = ["rerank"];
private static readonly string[] EMBEDDING_MARKERS = ["embed", "bge", "mpnet", "paraphrase", "sentence-transformers", "gte-", "e5-", "gritlm"];
//
// The models from before chat completions existed. Providers keep offering some of them, and
// Helmholtz Blablador still reports 'text-davinci-003', but asking any of them for a chat
// completion fails. We deliberately do not look for 'ada' here: three letters appear in far too
// many unrelated model names, and losing a chat model weighs heavier than keeping a dead one.
//
private static readonly string[] TEXT_COMPLETION_MARKERS = ["davinci", "babbage", "curie", "gpt-3.5-turbo-instruct"];
private static readonly string[] IMAGE_GENERATION_MARKERS = ["flux", "stable-diffusion", "sdxl", "dall-e", "midjourney", "gpt-image"];
//
// Google names its image models after the chat model they grew out of and appends "image":
// gemini-3-pro-image, gemini-3.1-flash-image, gemini-2.5-flash-image. Read as a plain substring,
// that word is too greedy -- it also sits inside "imagenet" and "reimagined", and a chat model
// carrying such a word would disappear from the user's list. It therefore counts only where a
// name segment begins and ends with it.
//
private static readonly string[] IMAGE_GENERATION_WORD_MARKERS = ["image"];
private static readonly string[] VIDEO_GENERATION_MARKERS = ["sora", "veo-", "runway", "hailuo"];
//
// Markers which have to stand as a word of their own. "kling" is such a case: taken as a plain
// substring, it also matches the organization "Klingspor", the model "Inkling", and the
// fine-tune "Llama-2-7b-chat-klingon" -- all of them models to chat with, which would vanish
// from the user's list. The video models themselves are named "kling-v1" or "kling-video",
// where the name ends at a separator.
//
private static readonly string[] VIDEO_GENERATION_WORD_MARKERS = ["kling"];
//
// Voxtral is marketed as an audio model which understands speech, so one could expect it to work
// in a chat as well. It does not: asking Mistral for a chat completion with 'voxtral-mini-latest'
// is answered with 'Invalid model'. Voxtral therefore belongs here, next to the models which do
// nothing but transcribe.
//
private static readonly string[] TRANSCRIPTION_MARKERS = ["whisper", "-transcribe", "wav2vec", "parakeet", "voxtral"];
//
// Besides the pure text-to-speech models, this covers the models which answer in audio, such as
// 'gpt-audio' and 'gpt-4o-audio-preview'. Those do accept a text-only request, but they are made
// for spoken conversations, and the providers offering them directly keep them out of their chat
// model lists as well.
//
private static readonly string[] SPEECH_SYNTHESIS_MARKERS = ["-tts", "tts-", "-speech", "speech-", "-audio", "audio-"];
//
// The models for spoken conversations over a live connection. They speak their own protocol,
// usually a WebSocket, and answer a chat completion request with an error. Checked before
// transcription, because some of them carry the name of a transcription model, such as
// OpenAI's 'gpt-realtime-whisper'. Those still need the live connection.
//
private static readonly string[] REALTIME_MARKERS = ["realtime"];
private static readonly string[] OCR_MARKERS = ["ocr"];
private static readonly string[] MODERATION_MARKERS = ["moderation", "guard"];
/// <summary>
/// Determines what kind of model this is, based on its name.
/// </summary>
/// <param name="model">The model to inspect.</param>
/// <returns>The recognized kind, or ModelKind.CHAT when we recognize no other kind.</returns>
public static ModelKind DetermineKind(this Model model)
{
if (string.IsNullOrWhiteSpace(model.Id) || model.IsSystemModel)
return ModelKind.CHAT;
if (HasAnyMarker(model.Id, OTHER_MARKERS))
return ModelKind.OTHER;
if (HasAnyMarker(model.Id, RERANKING_MARKERS))
return ModelKind.RERANKING;
if (HasAnyMarker(model.Id, EMBEDDING_MARKERS))
return ModelKind.EMBEDDING;
if (HasAnyMarker(model.Id, TEXT_COMPLETION_MARKERS))
return ModelKind.TEXT_COMPLETION;
if (HasAnyMarker(model.Id, IMAGE_GENERATION_MARKERS) || HasAnyWordMarker(model.Id, IMAGE_GENERATION_WORD_MARKERS))
return ModelKind.IMAGE_GENERATION;
if (HasAnyMarker(model.Id, VIDEO_GENERATION_MARKERS) || HasAnyWordMarker(model.Id, VIDEO_GENERATION_WORD_MARKERS))
return ModelKind.VIDEO_GENERATION;
if (HasAnyMarker(model.Id, REALTIME_MARKERS))
return ModelKind.REALTIME;
if (HasAnyMarker(model.Id, TRANSCRIPTION_MARKERS))
return ModelKind.TRANSCRIPTION;
if (HasAnyMarker(model.Id, SPEECH_SYNTHESIS_MARKERS))
return ModelKind.SPEECH_SYNTHESIS;
if (HasAnyMarker(model.Id, OCR_MARKERS))
return ModelKind.OCR;
if (HasAnyMarker(model.Id, MODERATION_MARKERS))
return ModelKind.MODERATION;
return ModelKind.CHAT;
}
/// <summary>
/// Checks whether this model can be used for chatting.
/// </summary>
/// <param name="model">The model to check.</param>
/// <returns>True, when the model is a chat model or when we recognize no other kind.</returns>
public static bool IsChatModel(this Model model) => model.DetermineKind() is ModelKind.CHAT;
/// <summary>
/// Checks whether this model creates embeddings.
/// </summary>
/// <param name="model">The model to check.</param>
/// <returns>True, when the model is an embedding model.</returns>
public static bool IsEmbeddingModel(this Model model) => model.DetermineKind() is ModelKind.EMBEDDING;
/// <summary>
/// Checks whether this model transcribes audio.
/// </summary>
/// <param name="model">The model to check.</param>
/// <returns>True, when the model is a transcription model.</returns>
public static bool IsTranscriptionModel(this Model model) => model.DetermineKind() is ModelKind.TRANSCRIPTION;
/// <summary>
/// Checks whether this model generates images.
/// </summary>
/// <param name="model">The model to check.</param>
/// <returns>True, when the model is an image generation model.</returns>
public static bool IsImageModel(this Model model) => model.DetermineKind() is ModelKind.IMAGE_GENERATION;
private static bool HasAnyMarker(string modelId, string[] markers)
{
foreach (var marker in markers)
if (modelId.Contains(marker, StringComparison.OrdinalIgnoreCase))
return true;
return false;
}
/// <summary>
/// Checks whether the model name contains one of the markers as a word of its own.
/// </summary>
/// <remarks>
/// A short marker which is also a common syllable cannot be looked for as a plain substring:
/// it would match names which have nothing to do with it, and the model would be sorted into
/// the wrong kind. Such a marker counts only where a name segment begins and ends with it.
/// </remarks>
/// <param name="modelId">The ID of the model.</param>
/// <param name="markers">The markers to look for.</param>
/// <returns>True, when one of the markers stands as a word of its own.</returns>
private static bool HasAnyWordMarker(string modelId, string[] markers)
{
foreach (var marker in markers)
{
var searchIndex = 0;
while (searchIndex <= modelId.Length - marker.Length)
{
var markerIndex = modelId.IndexOf(marker, searchIndex, StringComparison.OrdinalIgnoreCase);
if (markerIndex is -1)
break;
if (IsWholeWord(modelId, marker, markerIndex))
return true;
// The same marker may appear again later in the name, so we keep looking:
searchIndex = markerIndex + 1;
}
}
return false;
}
private static bool IsWholeWord(string modelId, string marker, int markerIndex)
{
if (markerIndex > 0 && !IsSeparator(modelId[markerIndex - 1]))
return false;
var endIndex = markerIndex + marker.Length;
return endIndex >= modelId.Length || IsSeparator(modelId[endIndex]);
}
/// <summary>
/// The characters which separate the parts of a model name, such as in "fal-ai/kling-video".
/// </summary>
/// <param name="character">The character to check.</param>
/// <returns>True, when the character separates two parts of a name.</returns>
private static bool IsSeparator(char character) => character is '/' or '-' or '_' or '.' or ' ' or ':';
}
@@ -49,7 +49,5 @@ public class NoProvider : IProvider
public Task<IReadOnlyList<IReadOnlyList<float>>> EmbedTextAsync(Model embeddingModel, SettingsManager settingsManager, CancellationToken token = default, params List<string> texts) => Task.FromResult<IReadOnlyList<IReadOnlyList<float>>>([]);
public IReadOnlyCollection<Capability> GetModelCapabilities(Model model) => [ Capability.NONE ];
#endregion
}
@@ -5,6 +5,7 @@ using System.Text;
using System.Text.Json;
using AIStudio.Chat;
using AIStudio.Models;
using AIStudio.Settings;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Rust;
@@ -48,10 +49,10 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
return base.ClassifyProviderRequestFailure(errorCode, errorType, errorMessage, responseBody);
}
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason) => failureReason switch
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason, ContextWindow contextWindow = default) => failureReason switch
{
ProviderRequestFailureReason.INSUFFICIENT_QUOTA => TB("It looks like you do not have any API credits left with OpenAI. Please add credits to your account and try again."),
_ => base.GetProviderRequestFailureUserMessage(failureReason),
_ => base.GetProviderRequestFailureUserMessage(failureReason, contextWindow),
};
/// <inheritdoc />
@@ -92,10 +93,10 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
// Read the model capabilities. Through the settings provider, so that the user's expert
// capability overrides apply:
var providerSettings = this.CreateSettingsProvider(chatModel);
var modelCapabilities = providerSettings.GetModelCapabilities();
var modelProfile = providerSettings.GetModelProfile();
// Check if we are using the Responses API or the Chat Completion API:
var usingResponsesAPI = modelCapabilities.Contains(Capability.RESPONSES_API);
var usingResponsesAPI = modelProfile.Has(Capability.RESPONSES_API);
// Prepare the request path based on the API we are using:
var requestPath = usingResponsesAPI ? "responses" : "chat/completions";
@@ -115,7 +116,7 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
var minimumWebSearchConfidence = toolRegistry?.GetMinimumProviderConfidence(ToolSelectionRules.WEB_SEARCH_TOOL_ID) ?? ConfidenceLevel.NONE;
var isWebSearchAllowed = settingsManager.IsToolActive(ToolSelectionRules.WEB_SEARCH_TOOL_ID) &&
ToolSelectionRules.IsProviderConfidenceAllowed(providerConfidence, minimumWebSearchConfidence);
IList<object> providerTools = modelCapabilities.Contains(Capability.WEB_SEARCH) && isWebSearchAllowed
IList<object> providerTools = modelProfile.Has(Capability.WEB_SEARCH) && isWebSearchAllowed
? [ ProviderTools.WEB_SEARCH ]
: [];
@@ -133,8 +134,7 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
async (systemPrompt, apiParameters, tools) =>
{
var messages = await chatThread.Blocks.BuildMessagesAsync(
this.Provider,
chatModel,
providerSettings,
role => role switch
{
ChatRole.USER => "user",
@@ -198,7 +198,7 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesAsync(
this.Provider, chatModel,
providerSettings,
role => role switch
{
ChatRole.USER => "user",
@@ -368,25 +368,25 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
/// <inheritdoc />
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.LLM_PROVIDER, static model => model.IsChatModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(this.Provider), apiKeyProvisional, token);
}
/// <inheritdoc />
public override Task<ModelLoadResult> GetImageModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.IMAGE_PROVIDER, static model => model.IsImageModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.IMAGE_PROVIDER, model => model.IsImageModel(this.Provider), apiKeyProvisional, token);
}
/// <inheritdoc />
public override Task<ModelLoadResult> GetEmbeddingModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, static model => model.IsEmbeddingModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(this.Provider), apiKeyProvisional, token);
}
/// <inheritdoc />
public override Task<ModelLoadResult> GetTranscriptionModels(string? apiKeyProvisional = null, CancellationToken token = default)
{
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, static model => model.IsTranscriptionModel(), apiKeyProvisional, token);
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, model => model.IsTranscriptionModel(this.Provider), apiKeyProvisional, token);
}
#endregion
@@ -1,8 +1,16 @@
using System.Text.Json.Serialization;
namespace AIStudio.Provider.OpenRouter;
/// <summary>
/// A data model for an OpenRouter model from the API.
/// </summary>
/// <remarks>
/// The window is the model's, not that of any one provider behind it. OpenRouter also states a
/// window per provider it currently prefers, but it picks one per request, so a number taken from
/// there would describe a choice nobody has made yet.
/// </remarks>
/// <param name="Id">The model's ID.</param>
/// <param name="Name">The model's human-readable display name.</param>
public readonly record struct OpenRouterModel(string Id, string? Name);
/// <param name="ContextWindowTokens">How much the model reads and writes in one conversation, in tokens.</param>
public readonly record struct OpenRouterModel(string Id, string? Name, [property: JsonPropertyName("context_length")] int? ContextWindowTokens);
@@ -2,6 +2,7 @@ using System.Net.Http.Headers;
using System.Runtime.CompilerServices;
using AIStudio.Chat;
using AIStudio.Models.Live;
using AIStudio.Provider.OpenAI;
using AIStudio.Settings;
@@ -36,7 +37,7 @@ public sealed class ProviderOpenRouter() : BaseProvider(LLMProviders.OPEN_ROUTER
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -117,7 +118,7 @@ public sealed class ProviderOpenRouter() : BaseProvider(LLMProviders.OPEN_ROUTER
"models",
modelResponse => modelResponse.Data
.Select(n => new Model(n.Id, n.Name))
.Where(model => model.IsChatModel()),
.Where(model => model.IsChatModel(this.Provider)),
apiKeyProvisional,
requestConfigurator: (request, secretKey) =>
{
@@ -125,9 +126,21 @@ public sealed class ProviderOpenRouter() : BaseProvider(LLMProviders.OPEN_ROUTER
request.Headers.Add("HTTP-Referer", PROJECT_WEBSITE);
request.Headers.Add("X-Title", PROJECT_NAME);
},
listingFactory: modelResponse => modelResponse.Data.Select(n => ModelListing.For(n.Id, n.ContextWindowTokens)),
token: token);
}
/// <summary>
/// Loads the models OpenRouter offers for embedding, which live on a route of their own.
/// </summary>
/// <remarks>
/// Nothing is reported from here: this route answers with the embedding models alone, and what
/// is reported replaces everything an instance said before. The windows of the chat models
/// would go missing the moment somebody opens the embedding settings.
/// </remarks>
/// <param name="apiKeyProvisional">An API key which is not stored yet.</param>
/// <param name="token">The cancellation token to use.</param>
/// <returns>The embedding models.</returns>
private Task<ModelLoadResult> LoadEmbeddingModels(string? apiKeyProvisional, CancellationToken token)
{
return this.LoadModelsResponse<OpenRouterModelsResponse>(
@@ -41,7 +41,7 @@ public sealed class ProviderPerplexity() : BaseProvider(LLMProviders.PERPLEXITY,
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -0,0 +1,45 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// Anthropic's extended thinking, written as a "thinking" object.
/// </summary>
/// <remarks>
/// The object carries a type, and the two types which switch thinking on are named outright:
/// "enabled" and "adaptive". Everything else falls through to the ordinary reading of a value, so
/// that a person writing "thinking": false is understood as well.
/// </remarks>
public sealed class AnthropicThinkingDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.ANTHROPIC_THINKING;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
{
if (!ReasoningParameters.TryGet(parameters, "thinking", out var thinking))
return ReasoningConfigurationState.NOT_CONFIGURED;
return thinking switch
{
IDictionary<string, object> thinkingObject when ReasoningParameters.TryGet(thinkingObject, "type", out var type) => TypeOf(type),
_ => ReasoningParameters.LevelOf(thinking),
};
}
/// <summary>
/// Reads the "type" of an Anthropic thinking object.
/// </summary>
/// <param name="value">The configured thinking type.</param>
/// <returns>What it says.</returns>
private static ReasoningConfigurationState TypeOf(object? value) => value switch
{
string text when text.Equals("enabled", StringComparison.OrdinalIgnoreCase) ||
text.Equals("adaptive", StringComparison.OrdinalIgnoreCase)
=> ReasoningConfigurationState.EXPLICITLY_ENABLED,
string text when ReasoningParameters.IsDisabledText(text) => ReasoningConfigurationState.EXPLICITLY_DISABLED,
_ => ReasoningParameters.LevelOf(value),
};
}
@@ -0,0 +1,87 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// Google's thinking config, thinking level, and thought summaries.
/// </summary>
/// <remarks>
/// Google offers the same settings in several places at once: directly, under "generation_config",
/// and in both spellings of each key, because their own libraries write snake case while the REST
/// API answers in camel case. All of them are read, and the answers put together.
///
/// Summaries are the one setting which only ever says yes. Asking for thought summaries proves that
/// thinking is on; switching them off proves nothing, because a model can think without showing it.
/// </remarks>
public sealed class GoogleThinkingDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.GOOGLE_THINKING;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
{
var states = new List<ReasoningConfigurationState>();
if (ReasoningParameters.TryGet(parameters, "thinking_config", out var thinkingConfig) &&
thinkingConfig is IDictionary<string, object> thinkingConfigObject)
states.Add(ConfigOf(thinkingConfigObject));
if (ReasoningParameters.TryGet(parameters, "generation_config", out var generationConfig) &&
generationConfig is IDictionary<string, object> generationConfigObject)
{
if (ReasoningParameters.TryGet(generationConfigObject, "thinking_config", out var nestedThinkingConfig) &&
nestedThinkingConfig is IDictionary<string, object> nestedThinkingConfigObject)
states.Add(ConfigOf(nestedThinkingConfigObject));
if (ReasoningParameters.TryGet(generationConfigObject, "thinking_summaries", out var thinkingSummaries))
states.Add(SummariesOf(thinkingSummaries));
if (ReasoningParameters.TryGet(generationConfigObject, "thinking_level", out var thinkingLevel))
states.Add(ReasoningParameters.LevelOf(thinkingLevel));
}
if (ReasoningParameters.TryGet(parameters, "thinking_summaries", out var topLevelThinkingSummaries))
states.Add(SummariesOf(topLevelThinkingSummaries));
if (ReasoningParameters.TryGet(parameters, "thinking_level", out var topLevelThinkingLevel))
states.Add(ReasoningParameters.LevelOf(topLevelThinkingLevel));
return ReasoningParameters.Merge(states);
}
/// <summary>
/// Reads a thinking config, in either spelling of its keys.
/// </summary>
/// <param name="thinkingConfig">The parsed thinking config object.</param>
/// <returns>What it says.</returns>
private static ReasoningConfigurationState ConfigOf(IDictionary<string, object> thinkingConfig)
{
var states = new List<ReasoningConfigurationState>();
if (ReasoningParameters.TryGet(thinkingConfig, "thinking_budget", out var thinkingBudget) ||
ReasoningParameters.TryGet(thinkingConfig, "thinkingBudget", out thinkingBudget))
states.Add(ReasoningParameters.BudgetOf(thinkingBudget));
if (ReasoningParameters.TryGet(thinkingConfig, "include_thoughts", out var includeThoughts) ||
ReasoningParameters.TryGet(thinkingConfig, "includeThoughts", out includeThoughts))
states.Add(ReasoningParameters.LevelOf(includeThoughts));
return ReasoningParameters.Merge(states);
}
/// <summary>
/// Reads a thought summary setting, which can only ever say yes.
/// </summary>
/// <param name="value">The configured summary setting.</param>
/// <returns>Yes, when it asks for summaries; nothing otherwise.</returns>
private static ReasoningConfigurationState SummariesOf(object? value) => value switch
{
string text when text.Equals("auto", StringComparison.OrdinalIgnoreCase) ||
text.Equals("on", StringComparison.OrdinalIgnoreCase) ||
text.Equals("summarized", StringComparison.OrdinalIgnoreCase)
=> ReasoningConfigurationState.EXPLICITLY_ENABLED,
true => ReasoningConfigurationState.EXPLICITLY_ENABLED,
_ => ReasoningConfigurationState.NOT_CONFIGURED,
};
}
@@ -0,0 +1,47 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// The reasoning mode and budget of the llama.cpp server.
/// </summary>
/// <remarks>
/// Its "reasoning" key is a mode rather than an object, and one of its three values means neither
/// yes nor no: "auto" hands the decision to the model's own template, which is exactly the case
/// where nobody has decided anything.
/// </remarks>
public sealed class LlamaCppReasoningDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.LLAMA_CPP;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
{
var states = new List<ReasoningConfigurationState>();
if (ReasoningParameters.TryGet(parameters, "reasoning", out var reasoning))
states.Add(ModeOf(reasoning));
if (ReasoningParameters.TryGet(parameters, "reasoning_budget", out var reasoningBudget))
states.Add(ReasoningParameters.BudgetOf(reasoningBudget));
if (ReasoningParameters.TryGet(parameters, "chat_template_kwargs", out var chatTemplateKwargs) &&
chatTemplateKwargs is IDictionary<string, object> chatTemplateKwargsObject)
states.Add(QwenThinkingDialect.In(chatTemplateKwargsObject));
return ReasoningParameters.Merge(states);
}
/// <summary>
/// Reads the reasoning mode.
/// </summary>
/// <param name="value">The configured mode.</param>
/// <returns>What it says, which for "auto" is nothing.</returns>
private static ReasoningConfigurationState ModeOf(object? value) => value switch
{
string text when text.Equals("on", StringComparison.OrdinalIgnoreCase) => ReasoningConfigurationState.EXPLICITLY_ENABLED,
string text when text.Equals("off", StringComparison.OrdinalIgnoreCase) => ReasoningConfigurationState.EXPLICITLY_DISABLED,
string text when text.Equals("auto", StringComparison.OrdinalIgnoreCase) => ReasoningConfigurationState.NOT_CONFIGURED,
_ => ReasoningParameters.LevelOf(value),
};
}
@@ -0,0 +1,20 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// Ollama's "think" parameter.
/// </summary>
/// <remarks>
/// One key, and it takes a boolean as readily as a level, which is why it needs no reading of its
/// own beyond the ordinary one.
/// </remarks>
public sealed class OllamaThinkDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.OLLAMA_THINK;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters) =>
ReasoningParameters.TryGet(parameters, "think", out var think)
? ReasoningParameters.LevelOf(think)
: ReasoningConfigurationState.NOT_CONFIGURED;
}
@@ -0,0 +1,31 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// The nested "reasoning" object almost every OpenAI-compatible server accepts.
/// </summary>
/// <remarks>
/// The object may carry an effort or a summary setting, and it may be written as a plain value
/// instead. An object carrying neither says nothing: somebody who wrote "reasoning": {} has not
/// asked for anything yet.
/// </remarks>
public sealed class OpenAICompatibleDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.OPEN_AI_COMPATIBLE;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
{
if (!ReasoningParameters.TryGet(parameters, "reasoning", out var reasoning))
return ReasoningConfigurationState.NOT_CONFIGURED;
return reasoning switch
{
IDictionary<string, object> reasoningObject when ReasoningParameters.TryGet(reasoningObject, "effort", out var effort) => ReasoningParameters.LevelOf(effort),
IDictionary<string, object> reasoningObject when ReasoningParameters.TryGet(reasoningObject, "summary", out var summary) => ReasoningParameters.LevelOf(summary),
IDictionary<string, object> => ReasoningConfigurationState.NOT_CONFIGURED,
_ => ReasoningParameters.LevelOf(reasoning),
};
}
}
@@ -0,0 +1,38 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// The "enable_thinking" switch Qwen introduced and other servers took over.
/// </summary>
/// <remarks>
/// It is accepted at the top level and inside "chat_template_kwargs", because it is really an
/// argument to the chat template rather than to the API -- which is also why two other dialects ask
/// this one about their own kwargs object instead of repeating the two keys.
/// </remarks>
public sealed class QwenThinkingDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.QWEN_THINKING;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters) => In(parameters);
/// <summary>
/// Reads the switch out of any parameter object, which need not be the top-level one.
/// </summary>
/// <param name="parameters">The object to look in.</param>
/// <returns>What it says.</returns>
public static ReasoningConfigurationState In(IDictionary<string, object> parameters)
{
var states = new List<ReasoningConfigurationState>();
if (ReasoningParameters.TryGet(parameters, "enable_thinking", out var enableThinking))
states.Add(ReasoningParameters.LevelOf(enableThinking));
if (ReasoningParameters.TryGet(parameters, "chat_template_kwargs", out var chatTemplateKwargs) &&
chatTemplateKwargs is IDictionary<string, object> chatTemplateKwargsObject &&
ReasoningParameters.TryGet(chatTemplateKwargsObject, "enable_thinking", out var nestedEnableThinking))
states.Add(ReasoningParameters.LevelOf(nestedEnableThinking));
return ReasoningParameters.Merge(states);
}
}
@@ -0,0 +1,21 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// The top-level "reasoning_effort" parameter.
/// </summary>
/// <remarks>
/// A dialect of its own although it is one key, because it travels on its own: providers accept it
/// without the nested object next to it, and the code this replaces had to remember to check for it
/// separately at every one of them. Here it is one line in the table instead.
/// </remarks>
public sealed class ReasoningEffortDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.REASONING_EFFORT;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters) =>
ReasoningParameters.TryGet(parameters, "reasoning_effort", out var effort)
? ReasoningParameters.LevelOf(effort)
: ReasoningConfigurationState.NOT_CONFIGURED;
}
@@ -0,0 +1,34 @@
namespace AIStudio.Provider.Reasoning.Dialects;
/// <summary>
/// The thinking token budget and chat template kwargs of vLLM.
/// </summary>
/// <remarks>
/// What vLLM accepts depends on the model family it was pointed at and on which reasoning parser
/// the operator started it with, so both the budget and the template arguments are read.
/// </remarks>
public sealed class VllmReasoningDialect : IReasoningDialect
{
/// <inheritdoc />
public ReasoningDialect Dialect => ReasoningDialect.VLLM;
/// <inheritdoc />
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
{
var states = new List<ReasoningConfigurationState>();
if (ReasoningParameters.TryGet(parameters, "thinking_token_budget", out var thinkingTokenBudget))
states.Add(ReasoningParameters.BudgetOf(thinkingTokenBudget));
if (ReasoningParameters.TryGet(parameters, "chat_template_kwargs", out var chatTemplateKwargs) &&
chatTemplateKwargs is IDictionary<string, object> chatTemplateKwargsObject)
{
states.Add(QwenThinkingDialect.In(chatTemplateKwargsObject));
if (ReasoningParameters.TryGet(chatTemplateKwargsObject, "thinking", out var thinking))
states.Add(ReasoningParameters.LevelOf(thinking));
}
return ReasoningParameters.Merge(states);
}
}
@@ -0,0 +1,24 @@
namespace AIStudio.Provider.Reasoning;
/// <summary>
/// One way of asking a request to think, and how to recognize it.
/// </summary>
/// <remarks>
/// A dialect reads parameters and says nothing else. It does not know which provider it is being
/// asked for, it keeps no state, and it never looks at the model -- what a model is able to do comes
/// from the rules, and mixing the two is what made the code this replaces hard to follow.
/// </remarks>
public interface IReasoningDialect
{
/// <summary>
/// Which dialect this is, which is also where it stands in the order.
/// </summary>
ReasoningDialect Dialect { get; }
/// <summary>
/// Reads what these parameters say about reasoning.
/// </summary>
/// <param name="parameters">The parsed additional API parameters.</param>
/// <returns>What they say, which is usually nothing.</returns>
ReasoningConfigurationState Detect(IDictionary<string, object> parameters);
}
@@ -0,0 +1,28 @@
namespace AIStudio.Provider.Reasoning;
/// <summary>
/// What the additional API parameters of a provider say about reasoning.
/// </summary>
/// <remarks>
/// This answers a different question than ReasoningSupport does. That one says what a model is able
/// to do, and it comes from the rules. This one says what the person asked their provider for, in
/// the free-text parameters they wrote themselves -- and most of the time it says nothing at all,
/// which is a statement of its own rather than a missing answer.
/// </remarks>
public enum ReasoningConfigurationState
{
/// <summary>
/// No recognized reasoning parameter was found.
/// </summary>
NOT_CONFIGURED,
/// <summary>
/// A recognized reasoning parameter explicitly enables reasoning.
/// </summary>
EXPLICITLY_ENABLED,
/// <summary>
/// A recognized reasoning parameter explicitly disables reasoning.
/// </summary>
EXPLICITLY_DISABLED,
}
@@ -0,0 +1,53 @@
namespace AIStudio.Provider.Reasoning;
/// <summary>
/// The ways a request can be asked to think, one per way of writing it down.
/// </summary>
/// <remarks>
/// Every provider speaks one or more of these, and which ones is stated in the dispatcher rather
/// than worked out from anything. The order here is the order they are asked in: the answer does not
/// depend on it -- a "no" wins wherever it stands -- but a report which named them in whatever order
/// a container handed them over would read differently on another machine.
/// </remarks>
public enum ReasoningDialect
{
/// <summary>
/// The nested "reasoning" object most OpenAI-compatible servers accept.
/// </summary>
OPEN_AI_COMPATIBLE,
/// <summary>
/// The top-level "reasoning_effort" parameter.
/// </summary>
REASONING_EFFORT,
/// <summary>
/// Anthropic's extended thinking, written as a "thinking" object.
/// </summary>
ANTHROPIC_THINKING,
/// <summary>
/// Google's thinking config, thinking level, and thought summaries.
/// </summary>
GOOGLE_THINKING,
/// <summary>
/// The "enable_thinking" switch Qwen introduced and other servers took over.
/// </summary>
QWEN_THINKING,
/// <summary>
/// Ollama's "think" parameter.
/// </summary>
OLLAMA_THINK,
/// <summary>
/// The reasoning mode and budget of the llama.cpp server.
/// </summary>
LLAMA_CPP,
/// <summary>
/// The thinking token budget and chat template kwargs of vLLM.
/// </summary>
VLLM,
}
@@ -0,0 +1,147 @@
using System.Collections.Concurrent;
using System.Collections.Frozen;
using AIStudio.Provider.Reasoning.Dialects;
using Host = AIStudio.Provider.SelfHosted.Host;
namespace AIStudio.Provider.Reasoning;
/// <summary>
/// Decides which dialects a provider speaks, and reads its parameters in all of them.
/// </summary>
/// <remarks>
/// Which dialect answers for which provider is a table here rather than a chain of checks spread
/// through the reading itself. That is the whole point of the split: adding a provider means adding
/// a line, and reading what one accepts means reading one line.
///
/// The answer is worked out once per provider setting. The question is asked from the provider list,
/// which re-renders whenever anything on the page changes, and the old code parsed the JSON a person
/// typed into their expert settings on every one of those renders. Nothing here reaches for
/// application state, so a test can ask it without the app having started.
/// </remarks>
public static class ReasoningDispatcher
{
/// <summary>
/// Every dialect there is, in the order the enum names them.
/// </summary>
private static readonly FrozenDictionary<ReasoningDialect, IReasoningDialect> DIALECTS = new IReasoningDialect[]
{
new OpenAICompatibleDialect(),
new ReasoningEffortDialect(),
new AnthropicThinkingDialect(),
new GoogleThinkingDialect(),
new QwenThinkingDialect(),
new OllamaThinkDialect(),
new LlamaCppReasoningDialect(),
new VllmReasoningDialect(),
}.ToFrozenDictionary(dialect => dialect.Dialect);
/// <summary>
/// Every dialect there is, in the order the enum names them.
/// </summary>
public static IReadOnlyList<IReasoningDialect> Dialects { get; } = DIALECTS.Values.OrderBy(dialect => dialect.Dialect).ToList();
/// <summary>
/// What an OpenAI-compatible server understands when nothing more is known about it.
/// </summary>
/// <remarks>
/// The gateways and resellers serve everybody's models, so they are asked in every dialect a
/// model of any vendor might answer to. Reading one dialect too many costs a dictionary lookup;
/// reading one too few hides a switch the person has set.
/// </remarks>
private static readonly ReasoningDialect[] EVERYTHING_A_GATEWAY_MIGHT_SERVE =
[
ReasoningDialect.OPEN_AI_COMPATIBLE,
ReasoningDialect.REASONING_EFFORT,
ReasoningDialect.QWEN_THINKING,
ReasoningDialect.GOOGLE_THINKING,
];
private static readonly ReasoningDialect[] NOTHING = [];
/// <summary>
/// The answers already worked out, so that the same settings are read once.
/// </summary>
private static readonly ConcurrentDictionary<(LLMProviders Provider, Host Host, string Parameters), ReasoningConfigurationState> ANSWERED = new();
/// <summary>
/// Reads what a provider's additional API parameters say about reasoning.
/// </summary>
/// <param name="provider">The LLM provider.</param>
/// <param name="host">The engine behind it, which only matters for self-hosted providers.</param>
/// <param name="additionalParameters">The additional API parameters, as the person wrote them.</param>
/// <returns>What they say, which is usually nothing.</returns>
public static ReasoningConfigurationState WhatTheParametersSay(LLMProviders provider, Host host, string? additionalParameters)
{
if (string.IsNullOrWhiteSpace(additionalParameters))
return ReasoningConfigurationState.NOT_CONFIGURED;
return ANSWERED.GetOrAdd((provider, host, additionalParameters), static key => Read(key.Provider, key.Host, key.Parameters));
}
/// <summary>
/// Which dialects this provider speaks.
/// </summary>
/// <remarks>
/// The commercial providers are asked only in their own dialect plus whatever their API
/// documents, because a parameter they do not accept says nothing about what they will do. The
/// self-hosted engines are the other case: the operator picked the engine, so what it accepts is
/// known, and it is the engine rather than the model which decides.
/// </remarks>
/// <param name="provider">The LLM provider.</param>
/// <param name="host">The engine behind it.</param>
/// <returns>The dialects to read the parameters in.</returns>
public static IReadOnlyList<ReasoningDialect> DialectsOf(LLMProviders provider, Host host) => provider switch
{
LLMProviders.OPEN_AI => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT],
LLMProviders.ANTHROPIC => [ReasoningDialect.ANTHROPIC_THINKING],
LLMProviders.MISTRAL or LLMProviders.PERPLEXITY => [ReasoningDialect.REASONING_EFFORT],
LLMProviders.GOOGLE => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.GOOGLE_THINKING],
LLMProviders.ALIBABA_CLOUD => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING],
LLMProviders.OPEN_ROUTER or
LLMProviders.HETZNER or
LLMProviders.IONOS or
LLMProviders.LITE_LLM or
LLMProviders.X or
LLMProviders.DEEP_SEEK or
LLMProviders.GROQ or
LLMProviders.FIREWORKS or
LLMProviders.HUGGINGFACE or
LLMProviders.HELMHOLTZ or
LLMProviders.GWDG => EVERYTHING_A_GATEWAY_MIGHT_SERVE,
LLMProviders.SELF_HOSTED => host switch
{
Host.OLLAMA => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING, ReasoningDialect.OLLAMA_THINK],
Host.LLAMA_CPP => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING, ReasoningDialect.LLAMA_CPP],
Host.VLLM => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING, ReasoningDialect.GOOGLE_THINKING, ReasoningDialect.VLLM],
_ => EVERYTHING_A_GATEWAY_MIGHT_SERVE,
},
_ => NOTHING,
};
/// <summary>
/// Parses the parameters and asks every dialect this provider speaks.
/// </summary>
/// <param name="provider">The LLM provider.</param>
/// <param name="host">The engine behind it.</param>
/// <param name="additionalParameters">The additional API parameters.</param>
/// <returns>What they say.</returns>
private static ReasoningConfigurationState Read(LLMProviders provider, Host host, string additionalParameters)
{
if (!AdditionalApiParametersParser.TryParse(additionalParameters, out var parameters, out _))
return ReasoningConfigurationState.NOT_CONFIGURED;
return ReasoningParameters.Merge(DialectsOf(provider, host).Select(key => DIALECTS[key].Detect(parameters)));
}
}
@@ -0,0 +1,131 @@
namespace AIStudio.Provider.Reasoning;
/// <summary>
/// Reading the values a person wrote into their additional API parameters.
/// </summary>
/// <remarks>
/// Every dialect ends up asking the same two questions: is this key there, and does this value mean
/// yes or no. The answers are the same whoever asks them -- "off" is off at every provider -- so
/// they live here rather than once per dialect.
/// </remarks>
public static class ReasoningParameters
{
/// <summary>
/// Try to read a parameter, matching the key regardless of how it was capitalized.
/// </summary>
/// <param name="parameters">The parsed parameter dictionary.</param>
/// <param name="key">The parameter name to find.</param>
/// <param name="value">The matched parameter value, if found.</param>
/// <returns>True, when a matching key was found.</returns>
public static bool TryGet(IDictionary<string, object> parameters, string key, out object? value)
{
value = null;
if (parameters.Count is 0)
return false;
var foundKey = parameters.Keys.FirstOrDefault(candidate => string.Equals(candidate, key, StringComparison.OrdinalIgnoreCase));
if (foundKey is null)
return false;
value = parameters[foundKey];
return true;
}
/// <summary>
/// Reads a value which is written as a boolean, a number, or a level.
/// </summary>
/// <param name="value">The raw parsed parameter value.</param>
/// <returns>What the value says.</returns>
public static ReasoningConfigurationState LevelOf(object? value) => value switch
{
bool booleanValue => booleanValue ? ReasoningConfigurationState.EXPLICITLY_ENABLED : ReasoningConfigurationState.EXPLICITLY_DISABLED,
int i => i is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
long l => l is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
double d => Math.Abs(d) < double.Epsilon ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
decimal m => m is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
string text when IsDisabledText(text) => ReasoningConfigurationState.EXPLICITLY_DISABLED,
string text when IsEnabledText(text) => ReasoningConfigurationState.EXPLICITLY_ENABLED,
_ => ReasoningConfigurationState.NOT_CONFIGURED,
};
/// <summary>
/// Reads a token budget, which several providers use to say the same thing with a number.
/// </summary>
/// <remarks>
/// A budget of zero switches thinking off. Everything else, negative budgets included, leaves it
/// available -- a negative one usually means "as much as it takes".
/// </remarks>
/// <param name="value">The configured budget value.</param>
/// <returns>What the budget says.</returns>
public static ReasoningConfigurationState BudgetOf(object? value) => value switch
{
int i => i is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
long l => l is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
double d => Math.Abs(d) < double.Epsilon ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
decimal m => m is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
_ => LevelOf(value),
};
/// <summary>
/// Puts several answers together into one.
/// </summary>
/// <remarks>
/// A "no" wins over a "yes", wherever the two stand. Somebody who switched thinking off in one
/// place meant to switch it off, and an indicator lighting up anyway because another parameter
/// could be read as a yes would be the app arguing with them.
/// </remarks>
/// <param name="states">What the dialects found.</param>
/// <returns>The one answer.</returns>
public static ReasoningConfigurationState Merge(IEnumerable<ReasoningConfigurationState> states)
{
var result = ReasoningConfigurationState.NOT_CONFIGURED;
foreach (var state in states)
{
if (state is ReasoningConfigurationState.EXPLICITLY_DISABLED)
return ReasoningConfigurationState.EXPLICITLY_DISABLED;
if (state is ReasoningConfigurationState.EXPLICITLY_ENABLED)
result = ReasoningConfigurationState.EXPLICITLY_ENABLED;
}
return result;
}
/// <summary>
/// Puts several answers together into one.
/// </summary>
/// <param name="states">What the dialects found.</param>
/// <returns>The one answer.</returns>
public static ReasoningConfigurationState Merge(params ReasoningConfigurationState[] states) => Merge(states.AsEnumerable());
/// <summary>
/// Whether a text means yes.
/// </summary>
/// <param name="text">The string value to inspect.</param>
/// <returns>True, when the value switches reasoning on.</returns>
public static bool IsEnabledText(string text) =>
text.Equals("true", StringComparison.OrdinalIgnoreCase) ||
text.Equals("yes", StringComparison.OrdinalIgnoreCase) ||
text.Equals("on", StringComparison.OrdinalIgnoreCase) ||
text.Equals("enabled", StringComparison.OrdinalIgnoreCase) ||
text.Equals("low", StringComparison.OrdinalIgnoreCase) ||
text.Equals("minimal", StringComparison.OrdinalIgnoreCase) ||
text.Equals("medium", StringComparison.OrdinalIgnoreCase) ||
text.Equals("high", StringComparison.OrdinalIgnoreCase) ||
text.Equals("max", StringComparison.OrdinalIgnoreCase);
/// <summary>
/// Whether a text means no.
/// </summary>
/// <param name="text">The string value to inspect.</param>
/// <returns>True, when the value switches reasoning off.</returns>
public static bool IsDisabledText(string text) =>
string.IsNullOrWhiteSpace(text) ||
text.Equals("false", StringComparison.OrdinalIgnoreCase) ||
text.Equals("no", StringComparison.OrdinalIgnoreCase) ||
text.Equals("off", StringComparison.OrdinalIgnoreCase) ||
text.Equals("none", StringComparison.OrdinalIgnoreCase) ||
text.Equals("disabled", StringComparison.OrdinalIgnoreCase);
}
@@ -1,3 +1,23 @@
using System.Text.Json.Serialization;
namespace AIStudio.Provider.SelfHosted;
public readonly record struct Model(string Id, string? Object, string? OwnedBy, ModelArchitecture? Architecture);
/// <summary>
/// One model as an OpenAI-compatible engine lists it.
/// </summary>
/// <remarks>
/// The context window is vLLM's addition to that route: it reports the window the operator started
/// the engine with, which is the one number no rule about the weights could ever know. Ollama,
/// LM Studio, and llama.cpp answer the same route without it, so it stays unknown there instead of
/// being guessed.
///
/// vLLM calls that field max_model_len, which reads like a limit on the model rather than on a
/// conversation. The wire keeps their spelling, and this record says what the number means, so that
/// nobody has to remember the translation while reading the code that uses it.
/// </remarks>
/// <param name="Id">The model's ID.</param>
/// <param name="Object">What kind of thing the entry is. Known value: "model".</param>
/// <param name="OwnedBy">Who the engine names as the owner of the model.</param>
/// <param name="Architecture">Which kinds of input and output the model takes, where the engine says.</param>
/// <param name="ContextWindowTokens">The context window the engine was started with, in tokens, where it says.</param>
public readonly record struct Model(string Id, string? Object, string? OwnedBy, ModelArchitecture? Architecture, [property: JsonPropertyName("max_model_len")] int? ContextWindowTokens);
@@ -3,6 +3,7 @@ using System.Runtime.CompilerServices;
using System.Text.Json;
using AIStudio.Chat;
using AIStudio.Models.Live;
using AIStudio.Provider.OpenAI;
using AIStudio.Settings;
using AIStudio.Tools.PluginSystem;
@@ -42,8 +43,8 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide
// - LM Studio, vLLM, and llama.cpp use the nested image URL format: { "type": "image_url", "image_url": { "url": "data:..." } }
var messages = host switch
{
Host.OLLAMA => await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, effectiveChatModel),
_ => await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, effectiveChatModel),
Host.OLLAMA => await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(effectiveChatModel)),
_ => await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(effectiveChatModel)),
};
return new ChatCompletionAPIRequest
@@ -188,8 +189,23 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide
return FailedModelLoadResult(this.GetModelLoadFailureReason(lmStudioResponse, responseBody), $"Status={(int)lmStudioResponse.StatusCode} {lmStudioResponse.ReasonPhrase}; Body='{responseBody}'");
}
var lmStudioModelResponse = await lmStudioResponse.Content.ReadFromJsonAsync<ModelsResponse>(token);
//
// Read with the shared options, the way every other model list of this app is read.
// This one route did without them, which quietly cost it every field an engine spells
// in snake case: owned_by has been arriving as nothing all along, and the next field
// somebody adds here would have gone the same way without anything failing.
//
var lmStudioModelResponse = await lmStudioResponse.Content.ReadFromJsonAsync<ModelsResponse>(JSON_SERIALIZER_OPTIONS, token);
var models = lmStudioModelResponse.Data ?? [];
//
// What the engine said about its own models, taken from the whole list rather than
// from what is offered below: a model filtered out here as an embedding model is still
// a model somebody may have configured this instance with, and this list is the only
// place its window is ever stated.
//
ListedModels.Shared.Report(this.ConfiguredProviderId, ListingsOf(models));
return SuccessfulModelLoadResult(models.
Where(model => !string.IsNullOrWhiteSpace(model.Id) &&
!ignorePhrases.Any(ignorePhrase => model.Id.Contains(ignorePhrase, StringComparison.InvariantCulture)) &&
@@ -302,6 +318,13 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide
}
}
/// <summary>
/// What an engine stated about the models it serves.
/// </summary>
/// <param name="models">The models exactly as the engine listed them.</param>
/// <returns>One listing per model, which says nothing for the models the engine was silent about.</returns>
private static IEnumerable<ModelListing> ListingsOf(IEnumerable<Model> models) => models.Select(model => ModelListing.For(model.Id, model.ContextWindowTokens));
private static bool IsMatchingLlamaCppTextModel(Model model, string[] ignorePhrases, string[] filterPhrases)
{
if (string.IsNullOrWhiteSpace(model.Id))
@@ -32,7 +32,7 @@ public sealed class ProviderX() : BaseProvider(LLMProviders.X, new Uri("https://
async (systemPrompt, apiParameters, tools) =>
{
// Build the list of messages:
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
return new ChatCompletionAPIRequest
{
@@ -79,7 +79,12 @@ public sealed class ProviderX() : BaseProvider(LLMProviders.X, new Uri("https://
var result = await this.LoadModels(SecretStoreType.LLM_PROVIDER, ["grok-"], apiKeyProvisional, token);
return result with
{
Models = [..result.Models.Where(n => !n.Id.Contains("-image", StringComparison.OrdinalIgnoreCase))]
//
// Asking what a model is made for rather than testing its name for a word. The word was
// "-image", which said nothing about grok-imagine-video: that one made films and stood
// in the list of things to chat with.
//
Models = [..result.Models.Where(model => model.IsChatModel(this.Provider))]
};
}