mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-08 19:09:40 +00:00
Rebuilt how AI Studio knows what a model can do (#960)
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
This commit is contained in:
1 parent
d21e09dd1e
commit
d85b4e71b6
287 files changed
+18341
-3677
No files matched your search
@@ -32,7 +32,7 @@ public sealed class ProviderAlibabaCloud() : BaseProvider(LLMProviders.ALIBABA_C
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
|
||||
@@ -41,8 +41,8 @@ public sealed class ProviderAnthropic() : BaseProvider(LLMProviders.ANTHROPIC, n
|
||||
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesAsync(
|
||||
this.Provider, chatModel,
|
||||
|
||||
this.CreateSettingsProvider(chatModel),
|
||||
|
||||
// Anthropic-specific role mapping:
|
||||
role => role switch
|
||||
{
|
||||
|
||||
@@ -6,6 +6,8 @@ using System.Text.Json;
|
||||
using System.Text.Json.Serialization;
|
||||
|
||||
using AIStudio.Chat;
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Live;
|
||||
using AIStudio.Provider.Anthropic;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
using AIStudio.Provider.SelfHosted;
|
||||
@@ -191,6 +193,7 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
Action<HttpRequestMessage, string>? requestConfigurator = null,
|
||||
JsonSerializerOptions? jsonSerializerOptions = null,
|
||||
bool isTryingSecret = false,
|
||||
Func<TResponse, IEnumerable<ModelListing>>? listingFactory = null,
|
||||
CancellationToken token = default)
|
||||
{
|
||||
var secretKey = await this.GetModelLoadingSecretKey(storeType, apiKeyProvisional, isTryingSecret);
|
||||
@@ -220,6 +223,16 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
if (parsedResponse is null)
|
||||
return FailedModelLoadResult(ModelLoadFailureReason.INVALID_RESPONSE, "Model list response could not be deserialized.");
|
||||
|
||||
//
|
||||
// What the list stated about the models, read before anything is filtered out of
|
||||
// it: a model left out below as an embedding model is still a model somebody may
|
||||
// have configured this instance with, and a list like this one is the only place
|
||||
// its window is ever stated. Only pass a whole list in here -- reporting a part of
|
||||
// one would tell the app that everything left out has stopped existing.
|
||||
//
|
||||
if (listingFactory is not null)
|
||||
ListedModels.Shared.Report(this.ConfiguredProviderId, listingFactory(parsedResponse));
|
||||
|
||||
return SuccessfulModelLoadResult(modelFactory(parsedResponse));
|
||||
}
|
||||
catch (Exception e)
|
||||
@@ -235,13 +248,34 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
}
|
||||
}
|
||||
|
||||
protected virtual string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason) => failureReason switch
|
||||
/// <summary>
|
||||
/// Says what a failed request means for the user.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The window is only ever known where the caller knows which model the request was for, which
|
||||
/// is why it is optional rather than a second required argument: most failures say nothing
|
||||
/// about a length and need no number to explain themselves.
|
||||
/// </remarks>
|
||||
/// <param name="failureReason">Why the request failed.</param>
|
||||
/// <param name="contextWindow">What the model reads, where that is known.</param>
|
||||
/// <returns>The message to show, or an empty string when we have nothing to say.</returns>
|
||||
protected virtual string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason, ContextWindow contextWindow = default) => failureReason switch
|
||||
{
|
||||
ProviderRequestFailureReason.TOO_MANY_REQUESTS => TB("The provider rejected the request because too many requests were sent. Please wait a moment and try again."),
|
||||
ProviderRequestFailureReason.INVALID_OR_MISSING_API_KEY => string.Format(TB("The API key for the provider '{0}' is missing or was rejected. Please check the key in the settings."), this.InstanceName),
|
||||
ProviderRequestFailureReason.AUTHENTICATION_OR_PERMISSION_ERROR => string.Format(TB("The provider '{0}' refused the request. Your account might not be allowed to use the selected model, or the provider might not serve your region."), this.InstanceName),
|
||||
ProviderRequestFailureReason.PROVIDER_UNAVAILABLE => string.Format(TB("The provider '{0}' could not be reached. Please check whether it is running and reachable, then try again."), this.InstanceName),
|
||||
ProviderRequestFailureReason.MODEL_NOT_FOUND => string.Format(TB("The provider '{0}' does not know the selected model. Please select another model."), this.InstanceName),
|
||||
//
|
||||
// Naming the number is the whole point of knowing it: "too long" leaves the user guessing
|
||||
// by how much, while the window turns the next step into arithmetic. Where nobody knows the
|
||||
// window, no number is invented -- the sentence below says the same thing without one.
|
||||
//
|
||||
// Written out in full rather than shortened the way the chat shortens it. The sentence ends
|
||||
// by asking the user to set a chunk size, and 32.77k is not a number anybody types into a
|
||||
// field.
|
||||
//
|
||||
ProviderRequestFailureReason.CONTEXT_LENGTH_EXCEEDED when contextWindow.IsKnown => string.Format(TB("The text was longer than the selected model accepts, which is {0} tokens. Please select a model which takes longer texts, or reduce the chunk size of the data source."), contextWindow.DefaultTokens.ToString("N0", I18N.I.Culture)),
|
||||
ProviderRequestFailureReason.CONTEXT_LENGTH_EXCEEDED => TB("The text was longer than the selected model accepts. Please select a model which takes longer texts, or reduce the chunk size of the data source."),
|
||||
ProviderRequestFailureReason.TOOLS_NOT_SUPPORTED => string.Format(TB("The selected model is not able to use tools. Please select a model which can, or open the settings of the provider '{0}', show its expert settings, and switch the function calling capability off there."), this.InstanceName),
|
||||
ProviderRequestFailureReason.EMBEDDINGS_NOT_SUPPORTED => string.Format(TB("The provider '{0}' cannot create embeddings. Please select a provider which offers an embedding model."), this.InstanceName),
|
||||
@@ -267,10 +301,18 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
/// Shared with the providers which talk to an embedding endpoint of their own: what the user
|
||||
/// needs to know does not depend on which route the request took.
|
||||
/// </remarks>
|
||||
protected ProviderRequestException CreateEmbeddingRequestException(HttpStatusCode statusCode, string reasonPhrase, string responseBody)
|
||||
protected ProviderRequestException CreateEmbeddingRequestException(HttpStatusCode statusCode, string reasonPhrase, string responseBody, Model embeddingModel)
|
||||
{
|
||||
//
|
||||
// What the rules know about this model, corrected by whatever this installation reported
|
||||
// about it. That is the same walk a configured chat provider takes, minus the expert
|
||||
// settings: an embedding provider has none, so there is nothing above the two to ask.
|
||||
//
|
||||
var stated = this.Provider.GetModelProfile(embeddingModel);
|
||||
var contextWindow = ListedModels.Shared.Of(this.ConfiguredProviderId, embeddingModel.Id).ApplyTo(stated).Context;
|
||||
|
||||
var failureReason = this.ClassifyEmbeddingRequestFailure(statusCode, responseBody);
|
||||
var userMessage = this.GetProviderRequestFailureUserMessage(failureReason);
|
||||
var userMessage = this.GetProviderRequestFailureUserMessage(failureReason, contextWindow);
|
||||
|
||||
// We know nothing about this failure, so we pass on what the provider said about it:
|
||||
if (string.IsNullOrWhiteSpace(userMessage))
|
||||
@@ -1539,7 +1581,7 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
// thousands being indexed in the background or the one thing the user just asked
|
||||
// for, and only it can decide how often the user should hear about it.
|
||||
//
|
||||
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody);
|
||||
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody, embeddingModel);
|
||||
}
|
||||
|
||||
var embeddingResponse = JsonSerializer.Deserialize<EmbeddingResponse>(responseBody, JSON_SERIALIZER_OPTIONS);
|
||||
|
||||
@@ -3,115 +3,145 @@ namespace AIStudio.Provider;
|
||||
/// <summary>
|
||||
/// Represents the capabilities of an AI model.
|
||||
/// </summary>
|
||||
public enum Capability
|
||||
/// <remarks>
|
||||
/// A set of capabilities is one value, not a collection: a model profile carries this enum as a
|
||||
/// single field, and asking whether a capability is present is one bit test instead of a walk
|
||||
/// through a list. That is why the members are powers of two.
|
||||
///
|
||||
/// The numeric values are an implementation detail and are never written anywhere. Overrides,
|
||||
/// plugins, and the settings file all address a capability by its name, so the names are the part
|
||||
/// which must not change. Removing a member would silently drop the override an organization wrote
|
||||
/// for it, which is why the members we no longer hand out ourselves are still here.
|
||||
///
|
||||
/// Adding a member means adding the next free bit. Sixty-four of them fit; should they ever run
|
||||
/// out, the answer is a second enum next to this one rather than a wider underlying type, because
|
||||
/// widening changes the meaning of every value already written down.
|
||||
/// </remarks>
|
||||
[Flags]
|
||||
public enum Capability : ulong
|
||||
{
|
||||
/// <summary>
|
||||
/// No capabilities specified.
|
||||
/// </summary>
|
||||
NONE,
|
||||
|
||||
NONE = 0,
|
||||
|
||||
/// <summary>
|
||||
/// We don't know what the AI model can do.
|
||||
/// </summary>
|
||||
UNKNOWN,
|
||||
|
||||
UNKNOWN = 1UL << 0,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform text input.
|
||||
/// </summary>
|
||||
TEXT_INPUT,
|
||||
|
||||
TEXT_INPUT = 1UL << 1,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform audio input, such as music or sound.
|
||||
/// </summary>
|
||||
AUDIO_INPUT,
|
||||
|
||||
AUDIO_INPUT = 1UL << 2,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform one image input, such as one photo or drawing.
|
||||
/// </summary>
|
||||
SINGLE_IMAGE_INPUT,
|
||||
|
||||
SINGLE_IMAGE_INPUT = 1UL << 3,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform multiple images as input, such as multiple photos or drawings.
|
||||
/// </summary>
|
||||
MULTIPLE_IMAGE_INPUT,
|
||||
|
||||
MULTIPLE_IMAGE_INPUT = 1UL << 4,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform speech input.
|
||||
/// </summary>
|
||||
SPEECH_INPUT,
|
||||
|
||||
SPEECH_INPUT = 1UL << 5,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform video input, such as video files or streams.
|
||||
/// </summary>
|
||||
VIDEO_INPUT,
|
||||
|
||||
VIDEO_INPUT = 1UL << 6,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can generate text output.
|
||||
/// </summary>
|
||||
TEXT_OUTPUT,
|
||||
|
||||
TEXT_OUTPUT = 1UL << 7,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can generate audio output, such as music or sound.
|
||||
/// </summary>
|
||||
AUDIO_OUTPUT,
|
||||
|
||||
AUDIO_OUTPUT = 1UL << 8,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can generate image output, such as photos or drawings.
|
||||
/// </summary>
|
||||
IMAGE_OUTPUT,
|
||||
|
||||
IMAGE_OUTPUT = 1UL << 9,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can generate speech output.
|
||||
/// </summary>
|
||||
SPEECH_OUTPUT,
|
||||
|
||||
SPEECH_OUTPUT = 1UL << 10,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can generate video output.
|
||||
/// </summary>
|
||||
VIDEO_OUTPUT,
|
||||
|
||||
VIDEO_OUTPUT = 1UL << 11,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform reasoning tasks. You can enable reasoning optionally, but it is disabled by default.
|
||||
/// </summary>
|
||||
OPTIONAL_REASONING,
|
||||
|
||||
/// <remarks>
|
||||
/// Override vocabulary. A model profile states how a model reasons through its ReasoningSupport
|
||||
/// field and never sets this flag, because the three reasoning flags can be combined into
|
||||
/// answers no model can give. Asking a profile whether it has this capability always says no.
|
||||
/// </remarks>
|
||||
OPTIONAL_REASONING = 1UL << 12,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model always performs reasoning. There is no option to disable reasoning.
|
||||
/// </summary>
|
||||
ALWAYS_REASONING,
|
||||
/// <remarks>
|
||||
/// Override vocabulary. A model profile states how a model reasons through its ReasoningSupport
|
||||
/// field and never sets this flag, because the three reasoning flags can be combined into
|
||||
/// answers no model can give. Asking a profile whether it has this capability always says no.
|
||||
/// </remarks>
|
||||
ALWAYS_REASONING = 1UL << 13,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model performs optional reasoning, but it is enabled by default.
|
||||
/// </summary>
|
||||
REASONING_BY_DEFAULT,
|
||||
/// <remarks>
|
||||
/// Override vocabulary. A model profile states how a model reasons through its ReasoningSupport
|
||||
/// field and never sets this flag, because the three reasoning flags can be combined into
|
||||
/// answers no model can give. Asking a profile whether it has this capability always says no.
|
||||
/// </remarks>
|
||||
REASONING_BY_DEFAULT = 1UL << 14,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can embed information or data.
|
||||
/// </summary>
|
||||
EMBEDDING,
|
||||
|
||||
EMBEDDING = 1UL << 15,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform in real-time.
|
||||
/// </summary>
|
||||
REALTIME,
|
||||
|
||||
REALTIME = 1UL << 16,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform function calling, such as invoking APIs or executing functions.
|
||||
/// </summary>
|
||||
FUNCTION_CALLING,
|
||||
|
||||
FUNCTION_CALLING = 1UL << 17,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model can perform web search to retrieve information from the internet.
|
||||
/// </summary>
|
||||
WEB_SEARCH,
|
||||
|
||||
WEB_SEARCH = 1UL << 18,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model is used via the Chat Completion API.
|
||||
/// </summary>
|
||||
CHAT_COMPLETION_API,
|
||||
|
||||
CHAT_COMPLETION_API = 1UL << 19,
|
||||
|
||||
/// <summary>
|
||||
/// The AI model is used via the Responses API.
|
||||
/// </summary>
|
||||
RESPONSES_API,
|
||||
RESPONSES_API = 1UL << 20,
|
||||
}
|
||||
@@ -32,7 +32,7 @@ public sealed class ProviderDeepSeek() : BaseProvider(LLMProviders.DEEP_SEEK, ne
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -103,7 +103,7 @@ public sealed class ProviderDeepSeek() : BaseProvider(LLMProviders.DEEP_SEEK, ne
|
||||
return this.LoadModelsResponse<ModelsResponse>(
|
||||
storeType,
|
||||
"models",
|
||||
modelResponse => modelResponse.Data.Where(model => model.IsChatModel()),
|
||||
modelResponse => modelResponse.Data.Where(model => model.IsChatModel(this.Provider)),
|
||||
apiKeyProvisional, token: token);
|
||||
}
|
||||
}
|
||||
@@ -32,7 +32,7 @@ public class ProviderFireworks() : BaseProvider(LLMProviders.FIREWORKS, new Uri(
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
|
||||
@@ -40,7 +40,7 @@ public sealed class ProviderGWDG() : BaseProvider(LLMProviders.GWDG, new Uri("ht
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -88,7 +88,7 @@ public sealed class ProviderGWDG() : BaseProvider(LLMProviders.GWDG, new Uri("ht
|
||||
var result = await this.LoadModels(SecretStoreType.LLM_PROVIDER, apiKeyProvisional, token);
|
||||
return result with
|
||||
{
|
||||
Models = [..result.Models.Where(model => model.IsChatModel())]
|
||||
Models = [..result.Models.Where(model => model.IsChatModel(this.Provider))]
|
||||
};
|
||||
}
|
||||
|
||||
@@ -112,7 +112,7 @@ public sealed class ProviderGWDG() : BaseProvider(LLMProviders.GWDG, new Uri("ht
|
||||
if (!result.Success)
|
||||
return result;
|
||||
|
||||
var embeddingModels = result.Models.Where(model => model.IsEmbeddingModel()).ToList();
|
||||
var embeddingModels = result.Models.Where(model => model.IsEmbeddingModel(this.Provider)).ToList();
|
||||
if (embeddingModels.Count is 0)
|
||||
return ModelLoadResult.FromModels(KNOWN_EMBEDDING_MODELS);
|
||||
|
||||
|
||||
@@ -34,7 +34,7 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -116,7 +116,7 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
|
||||
if (!response.IsSuccessStatusCode)
|
||||
{
|
||||
LOGGER.LogError("Embedding request failed with status code {ResponseStatusCode} and body: '{ResponseBody}'.", response.StatusCode, responseBody);
|
||||
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody);
|
||||
throw this.CreateEmbeddingRequestException(response.StatusCode, response.ReasonPhrase ?? string.Empty, responseBody, embeddingModel);
|
||||
}
|
||||
|
||||
var embeddingResponse = JsonSerializer.Deserialize<GoogleEmbeddingResponse>(responseBody, JSON_SERIALIZER_OPTIONS);
|
||||
@@ -168,9 +168,15 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
|
||||
{
|
||||
Models =
|
||||
[
|
||||
//
|
||||
// Asking what a model is made for, rather than only ruling out the embedding ones.
|
||||
// Google names everything after the chat model it grew out of, so the catalog is
|
||||
// full of names which look like something to talk to and are not: the image models,
|
||||
// and the computer use model whose API refuses a request without its tool.
|
||||
//
|
||||
..result.Models.Where(model =>
|
||||
model.Id.StartsWith("gemini-", StringComparison.OrdinalIgnoreCase) &&
|
||||
!this.IsEmbeddingModel(model.Id))
|
||||
model.IsChatModel(this.Provider))
|
||||
.Select(this.WithDisplayNameFallback)
|
||||
]
|
||||
};
|
||||
@@ -189,7 +195,7 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
|
||||
{
|
||||
Models =
|
||||
[
|
||||
..result.Models.Where(model => this.IsEmbeddingModel(model.Id))
|
||||
..result.Models.Where(model => model.IsEmbeddingModel(this.Provider))
|
||||
.Select(this.WithDisplayNameFallback)
|
||||
]
|
||||
};
|
||||
@@ -222,12 +228,6 @@ public class ProviderGoogle() : BaseProvider(LLMProviders.GOOGLE, new Uri("https
|
||||
token: token);
|
||||
}
|
||||
|
||||
private bool IsEmbeddingModel(string modelId)
|
||||
{
|
||||
return modelId.Contains("embedding", StringComparison.OrdinalIgnoreCase) ||
|
||||
modelId.Contains("embed", StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
|
||||
private Model WithDisplayNameFallback(Model model)
|
||||
{
|
||||
return string.IsNullOrWhiteSpace(model.DisplayName)
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
using System.Text.Json.Serialization;
|
||||
|
||||
namespace AIStudio.Provider.Groq;
|
||||
|
||||
/// <summary>
|
||||
/// One model as Groq lists it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Groq says more about a model than the shared OpenAI-compatible list does, which is why this
|
||||
/// provider brings a data model of its own instead of using that one: the shared record is read by
|
||||
/// a dozen providers, and a field only one of them sends has no business in it.
|
||||
/// </remarks>
|
||||
/// <param name="Id">The model's ID.</param>
|
||||
/// <param name="ContextWindowTokens">How much the model reads and writes in one conversation, in tokens.</param>
|
||||
public readonly record struct GroqModel(string Id, [property: JsonPropertyName("context_window")] int? ContextWindowTokens);
|
||||
@@ -0,0 +1,7 @@
|
||||
namespace AIStudio.Provider.Groq;
|
||||
|
||||
/// <summary>
|
||||
/// A data model for the response from the Groq models endpoint.
|
||||
/// </summary>
|
||||
/// <param name="Data">The models Groq serves.</param>
|
||||
public readonly record struct GroqModelsResponse(IList<GroqModel> Data);
|
||||
@@ -1,6 +1,7 @@
|
||||
using System.Runtime.CompilerServices;
|
||||
|
||||
using AIStudio.Chat;
|
||||
using AIStudio.Models.Live;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
using AIStudio.Settings;
|
||||
|
||||
@@ -35,7 +36,7 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
|
||||
apiParameters["seed"] = parsedSeed;
|
||||
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -83,7 +84,7 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
|
||||
var result = await this.LoadModels(SecretStoreType.LLM_PROVIDER, apiKeyProvisional, token);
|
||||
return result with
|
||||
{
|
||||
Models = [..result.Models.Where(model => model.IsChatModel())]
|
||||
Models = [..result.Models.Where(model => model.IsChatModel(this.Provider))]
|
||||
};
|
||||
}
|
||||
|
||||
@@ -105,7 +106,7 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
|
||||
var result = await this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, apiKeyProvisional, token);
|
||||
return result with
|
||||
{
|
||||
Models = [..result.Models.Where(model => model.IsTranscriptionModel())]
|
||||
Models = [..result.Models.Where(model => model.IsTranscriptionModel(this.Provider))]
|
||||
};
|
||||
}
|
||||
|
||||
@@ -113,10 +114,12 @@ public class ProviderGroq() : BaseProvider(LLMProviders.GROQ, new Uri("https://a
|
||||
|
||||
private Task<ModelLoadResult> LoadModels(SecretStoreType storeType, string? apiKeyProvisional, CancellationToken token)
|
||||
{
|
||||
return this.LoadModelsResponse<ModelsResponse>(
|
||||
return this.LoadModelsResponse<GroqModelsResponse>(
|
||||
storeType,
|
||||
"models",
|
||||
modelResponse => modelResponse.Data,
|
||||
apiKeyProvisional, token: token);
|
||||
modelResponse => modelResponse.Data.Select(n => new Model(n.Id, null)),
|
||||
apiKeyProvisional,
|
||||
listingFactory: modelResponse => modelResponse.Data.Select(n => ModelListing.For(n.Id, n.ContextWindowTokens)),
|
||||
token: token);
|
||||
}
|
||||
}
|
||||
@@ -34,7 +34,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -84,7 +84,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
|
||||
{
|
||||
Models =
|
||||
[
|
||||
..result.Models.Where(model => model.IsChatModel())
|
||||
..result.Models.Where(model => model.IsChatModel(this.Provider))
|
||||
]
|
||||
};
|
||||
}
|
||||
@@ -103,7 +103,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
|
||||
{
|
||||
Models =
|
||||
[
|
||||
..result.Models.Where(model => model.IsEmbeddingModel())
|
||||
..result.Models.Where(model => model.IsEmbeddingModel(this.Provider))
|
||||
]
|
||||
};
|
||||
}
|
||||
@@ -116,7 +116,7 @@ public sealed class ProviderHelmholtz() : BaseProvider(LLMProviders.HELMHOLTZ, n
|
||||
{
|
||||
Models =
|
||||
[
|
||||
..result.Models.Where(model => model.IsTranscriptionModel())
|
||||
..result.Models.Where(model => model.IsTranscriptionModel(this.Provider))
|
||||
]
|
||||
};
|
||||
}
|
||||
|
||||
@@ -31,7 +31,7 @@ public sealed class ProviderHetzner() : BaseProvider(LLMProviders.HETZNER, new U
|
||||
settingsManager,
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -69,7 +69,7 @@ public sealed class ProviderHetzner() : BaseProvider(LLMProviders.HETZNER, new U
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", modelResponse => modelResponse.Data.Where(model => model.IsChatModel()), apiKeyProvisional, token: token);
|
||||
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", modelResponse => modelResponse.Data.Where(model => model.IsChatModel(this.Provider)), apiKeyProvisional, token: token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
|
||||
@@ -5,4 +5,30 @@ namespace AIStudio.Provider.HuggingFace;
|
||||
/// </summary>
|
||||
/// <param name="Id">The ID of the model, written as "org/model".</param>
|
||||
/// <param name="Providers">The inference providers serving this model.</param>
|
||||
public readonly record struct HFModel(string Id, IList<HFModelProvider>? Providers);
|
||||
public readonly record struct HFModel(string Id, IList<HFModelProvider>? Providers)
|
||||
{
|
||||
/// <summary>
|
||||
/// The window this model has when it is reached the way this user set things up.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A window belongs to an inference provider here, not to the model: the same weights run
|
||||
/// behind several of them, each configured by somebody else. Where the user named one, its
|
||||
/// number is the answer. Where they let the router choose, the smallest window among the
|
||||
/// providers currently serving the model is -- nobody knows which one the router will take, and
|
||||
/// a number promising more than the chosen provider delivers would walk a conversation into an
|
||||
/// error the user could not see coming.
|
||||
/// </remarks>
|
||||
/// <param name="providerSlug">The inference provider the user chose, or empty when the router chooses.</param>
|
||||
/// <returns>The window in tokens, or null where nobody stated one.</returns>
|
||||
public int? ContextWindowTokens(string providerSlug)
|
||||
{
|
||||
if (this.Providers is null)
|
||||
return null;
|
||||
|
||||
var serving = this.Providers.Where(provider => provider.IsLive);
|
||||
if (!string.IsNullOrEmpty(providerSlug))
|
||||
serving = serving.Where(provider => string.Equals(provider.Provider, providerSlug, StringComparison.OrdinalIgnoreCase));
|
||||
|
||||
return serving.Where(provider => provider.ContextWindowTokens is > 0).Min(provider => provider.ContextWindowTokens);
|
||||
}
|
||||
}
|
||||
@@ -1,3 +1,5 @@
|
||||
using System.Text.Json.Serialization;
|
||||
|
||||
namespace AIStudio.Provider.HuggingFace;
|
||||
|
||||
/// <summary>
|
||||
@@ -5,4 +7,13 @@ namespace AIStudio.Provider.HuggingFace;
|
||||
/// </summary>
|
||||
/// <param name="Provider">The slug of the inference provider, e.g. "novita".</param>
|
||||
/// <param name="Status">Whether the provider currently serves the model. Known value: "live".</param>
|
||||
public readonly record struct HFModelProvider(string Provider, string Status);
|
||||
/// <param name="ContextWindowTokens">How much this provider reads and writes in one conversation, in tokens.</param>
|
||||
public readonly record struct HFModelProvider(string Provider, string Status, [property: JsonPropertyName("context_length")] int? ContextWindowTokens)
|
||||
{
|
||||
private const string LIVE = "live";
|
||||
|
||||
/// <summary>
|
||||
/// Whether this provider serves the model right now.
|
||||
/// </summary>
|
||||
public bool IsLive => string.Equals(this.Status, LIVE, StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
@@ -2,6 +2,8 @@
|
||||
using System.Runtime.CompilerServices;
|
||||
|
||||
using AIStudio.Chat;
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Models.Live;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
using AIStudio.Settings;
|
||||
using AIStudio.Tools.PluginSystem;
|
||||
@@ -127,10 +129,10 @@ public sealed class ProviderHuggingFace : BaseProvider
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason)
|
||||
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason, ContextWindow contextWindow = default)
|
||||
{
|
||||
if (failureReason is not ProviderRequestFailureReason.MODEL_NOT_SUPPORTED_BY_PROVIDER)
|
||||
return base.GetProviderRequestFailureUserMessage(failureReason);
|
||||
return base.GetProviderRequestFailureUserMessage(failureReason, contextWindow);
|
||||
|
||||
//
|
||||
// When Hugging Face chose the provider itself, naming it back to the user would help
|
||||
@@ -166,7 +168,7 @@ public sealed class ProviderHuggingFace : BaseProvider
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -221,7 +223,23 @@ public sealed class ProviderHuggingFace : BaseProvider
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", this.SelectChatModels, apiKeyProvisional, token: token);
|
||||
return this.LoadModelsResponse<ModelsResponse>(SecretStoreType.LLM_PROVIDER, "models", this.SelectChatModels, apiKeyProvisional, listingFactory: this.ListingsOf, token: token);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// What the router stated about the models it knows.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every model the router reports, not only the ones offered for chatting below: which models
|
||||
/// are offered depends on the chosen inference provider, while a window belongs to whoever is
|
||||
/// configured here, and both questions are asked of the same list.
|
||||
/// </remarks>
|
||||
/// <param name="response">The response of the model endpoint.</param>
|
||||
/// <returns>One listing per model, which says nothing for the models nobody stated a window for.</returns>
|
||||
private IEnumerable<ModelListing> ListingsOf(ModelsResponse response)
|
||||
{
|
||||
var providerSlug = this.hfProvider.EndpointsId();
|
||||
return response.Data.Select(hfModel => ModelListing.For(hfModel.Id, hfModel.ContextWindowTokens(providerSlug)));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
@@ -237,7 +255,7 @@ public sealed class ProviderHuggingFace : BaseProvider
|
||||
/// <returns>The models to offer.</returns>
|
||||
private IEnumerable<Model> SelectChatModels(ModelsResponse response)
|
||||
{
|
||||
var chatModels = response.Data.Where(hfModel => new Model(hfModel.Id, null).IsChatModel());
|
||||
var chatModels = response.Data.Where(hfModel => new Model(hfModel.Id, null).IsChatModel(this.Provider));
|
||||
var providerSlug = this.hfProvider.EndpointsId();
|
||||
if (string.IsNullOrEmpty(providerSlug))
|
||||
return ToModels(chatModels);
|
||||
@@ -253,8 +271,8 @@ public sealed class ProviderHuggingFace : BaseProvider
|
||||
return false;
|
||||
|
||||
return hfModel.Providers.Any(provider =>
|
||||
string.Equals(provider.Provider, providerSlug, StringComparison.OrdinalIgnoreCase) &&
|
||||
string.Equals(provider.Status, "live", StringComparison.OrdinalIgnoreCase));
|
||||
provider.IsLive &&
|
||||
string.Equals(provider.Provider, providerSlug, StringComparison.OrdinalIgnoreCase));
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
|
||||
@@ -39,7 +39,7 @@ public sealed class ProviderIONOS() : BaseProvider(LLMProviders.IONOS, new Uri("
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -84,7 +84,7 @@ public sealed class ProviderIONOS() : BaseProvider(LLMProviders.IONOS, new Uri("
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
@@ -96,7 +96,7 @@ public sealed class ProviderIONOS() : BaseProvider(LLMProviders.IONOS, new Uri("
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetEmbeddingModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
|
||||
@@ -32,7 +32,7 @@ public sealed class ProviderLiteLLM(string hostname) : BaseProvider(LLMProviders
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -77,7 +77,7 @@ public sealed class ProviderLiteLLM(string hostname) : BaseProvider(LLMProviders
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.LLM_PROVIDER, static model => model.IsChatModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
@@ -89,13 +89,13 @@ public sealed class ProviderLiteLLM(string hostname) : BaseProvider(LLMProviders
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetEmbeddingModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, static model => model.IsEmbeddingModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetTranscriptionModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, static model => model.IsTranscriptionModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, model => model.IsTranscriptionModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
#endregion
|
||||
|
||||
@@ -1,3 +1,13 @@
|
||||
using System.Text.Json.Serialization;
|
||||
|
||||
namespace AIStudio.Provider.Mistral;
|
||||
|
||||
public readonly record struct Model(string Id, string Object, int Created, string OwnedBy);
|
||||
/// <summary>
|
||||
/// One model as Mistral lists it.
|
||||
/// </summary>
|
||||
/// <param name="Id">The model's ID.</param>
|
||||
/// <param name="Object">What kind of thing the entry is. Known value: "model".</param>
|
||||
/// <param name="Created">When the model was published, as seconds since the epoch.</param>
|
||||
/// <param name="OwnedBy">Who Mistral names as the owner of the model.</param>
|
||||
/// <param name="ContextWindowTokens">How much the model reads and writes in one conversation, in tokens.</param>
|
||||
public readonly record struct Model(string Id, string Object, int Created, string OwnedBy, [property: JsonPropertyName("max_context_length")] int? ContextWindowTokens);
|
||||
@@ -1,6 +1,7 @@
|
||||
using System.Runtime.CompilerServices;
|
||||
|
||||
using AIStudio.Chat;
|
||||
using AIStudio.Models.Live;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
using AIStudio.Settings;
|
||||
|
||||
@@ -38,7 +39,7 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
|
||||
apiParameters["random_seed"] = parsedRandomSeed;
|
||||
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -97,7 +98,7 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
|
||||
// kind detection:
|
||||
..modelResponse.Models.Where(n =>
|
||||
!n.Id.StartsWith("code", StringComparison.OrdinalIgnoreCase) &&
|
||||
n.IsChatModel())
|
||||
n.IsChatModel(this.Provider))
|
||||
]
|
||||
};
|
||||
}
|
||||
@@ -111,7 +112,7 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
|
||||
|
||||
return modelResponse with
|
||||
{
|
||||
Models = [..modelResponse.Models.Where(n => n.IsEmbeddingModel())]
|
||||
Models = [..modelResponse.Models.Where(n => n.IsEmbeddingModel(this.Provider))]
|
||||
};
|
||||
}
|
||||
|
||||
@@ -139,6 +140,8 @@ public sealed class ProviderMistral() : BaseProvider(LLMProviders.MISTRAL, new U
|
||||
storeType,
|
||||
"models",
|
||||
modelResponse => modelResponse.Data.Select(n => new Provider.Model(n.Id, null)),
|
||||
apiKeyProvisional, token: token);
|
||||
apiKeyProvisional,
|
||||
listingFactory: modelResponse => modelResponse.Data.Select(n => ModelListing.For(n.Id, n.ContextWindowTokens)),
|
||||
token: token);
|
||||
}
|
||||
}
|
||||
@@ -76,6 +76,16 @@ public enum ModelKind
|
||||
/// </remarks>
|
||||
REALTIME,
|
||||
|
||||
/// <summary>
|
||||
/// The model drives a computer: it looks at a screen and says what to click next.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// These refuse a plain conversation outright. Google's answer to a request without the computer
|
||||
/// use tool is "This model requires the use of the Computer Use tool", so the model belongs in no
|
||||
/// chat list, however much its name looks like the chat model it grew out of.
|
||||
/// </remarks>
|
||||
COMPUTER_USE,
|
||||
|
||||
/// <summary>
|
||||
/// The model extracts text from images or scanned documents.
|
||||
/// </summary>
|
||||
|
||||
@@ -1,228 +0,0 @@
|
||||
namespace AIStudio.Provider;
|
||||
|
||||
/// <summary>
|
||||
/// Determines what kind of model we are dealing with, based on its name.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Many providers serve every kind of model through one models endpoint, without telling us what
|
||||
/// kind each model is. Before this class existed, every provider carried its own list of name
|
||||
/// fragments to sort those models apart. Those lists disagreed with each other: a model like
|
||||
/// nomic-embed-text was recognized as an embedding model by some providers, while others offered it
|
||||
/// as a chat model. The knowledge about model families is the same for all providers, so it lives
|
||||
/// here now.
|
||||
///
|
||||
/// This class recognizes what a model is NOT made for. Everything we do not recognize is reported as
|
||||
/// a chat model. That direction matters: when a provider adds a model family we have never seen, the
|
||||
/// user still gets to use it. Getting it wrong the other way around would hide a model the user is
|
||||
/// paying for.
|
||||
///
|
||||
/// What this class must not become is a place for provider-specific knowledge. That a model called
|
||||
/// "codestral" is a fill-in-the-middle model at Mistral, or that Alibaba's chat models all start
|
||||
/// with a "q", is true for that one provider only. Such rules stay in the provider.
|
||||
/// </remarks>
|
||||
public static class ModelKindExtensions
|
||||
{
|
||||
//
|
||||
// Checked first, because these entries are no models at all: whatever else their name might
|
||||
// suggest, none of the other kinds applies to them.
|
||||
//
|
||||
private static readonly string[] OTHER_MARKERS = ["container"];
|
||||
|
||||
//
|
||||
// Reranking is checked before embedding: rerankers are commonly named after the embedding model
|
||||
// they belong to, e.g. Qwen3-VL-Reranker-8B next to Qwen3-VL-Embedding-8B.
|
||||
//
|
||||
private static readonly string[] RERANKING_MARKERS = ["rerank"];
|
||||
|
||||
private static readonly string[] EMBEDDING_MARKERS = ["embed", "bge", "mpnet", "paraphrase", "sentence-transformers", "gte-", "e5-", "gritlm"];
|
||||
|
||||
//
|
||||
// The models from before chat completions existed. Providers keep offering some of them, and
|
||||
// Helmholtz Blablador still reports 'text-davinci-003', but asking any of them for a chat
|
||||
// completion fails. We deliberately do not look for 'ada' here: three letters appear in far too
|
||||
// many unrelated model names, and losing a chat model weighs heavier than keeping a dead one.
|
||||
//
|
||||
private static readonly string[] TEXT_COMPLETION_MARKERS = ["davinci", "babbage", "curie", "gpt-3.5-turbo-instruct"];
|
||||
|
||||
private static readonly string[] IMAGE_GENERATION_MARKERS = ["flux", "stable-diffusion", "sdxl", "dall-e", "midjourney", "gpt-image"];
|
||||
|
||||
//
|
||||
// Google names its image models after the chat model they grew out of and appends "image":
|
||||
// gemini-3-pro-image, gemini-3.1-flash-image, gemini-2.5-flash-image. Read as a plain substring,
|
||||
// that word is too greedy -- it also sits inside "imagenet" and "reimagined", and a chat model
|
||||
// carrying such a word would disappear from the user's list. It therefore counts only where a
|
||||
// name segment begins and ends with it.
|
||||
//
|
||||
private static readonly string[] IMAGE_GENERATION_WORD_MARKERS = ["image"];
|
||||
|
||||
private static readonly string[] VIDEO_GENERATION_MARKERS = ["sora", "veo-", "runway", "hailuo"];
|
||||
|
||||
//
|
||||
// Markers which have to stand as a word of their own. "kling" is such a case: taken as a plain
|
||||
// substring, it also matches the organization "Klingspor", the model "Inkling", and the
|
||||
// fine-tune "Llama-2-7b-chat-klingon" -- all of them models to chat with, which would vanish
|
||||
// from the user's list. The video models themselves are named "kling-v1" or "kling-video",
|
||||
// where the name ends at a separator.
|
||||
//
|
||||
private static readonly string[] VIDEO_GENERATION_WORD_MARKERS = ["kling"];
|
||||
|
||||
//
|
||||
// Voxtral is marketed as an audio model which understands speech, so one could expect it to work
|
||||
// in a chat as well. It does not: asking Mistral for a chat completion with 'voxtral-mini-latest'
|
||||
// is answered with 'Invalid model'. Voxtral therefore belongs here, next to the models which do
|
||||
// nothing but transcribe.
|
||||
//
|
||||
private static readonly string[] TRANSCRIPTION_MARKERS = ["whisper", "-transcribe", "wav2vec", "parakeet", "voxtral"];
|
||||
|
||||
//
|
||||
// Besides the pure text-to-speech models, this covers the models which answer in audio, such as
|
||||
// 'gpt-audio' and 'gpt-4o-audio-preview'. Those do accept a text-only request, but they are made
|
||||
// for spoken conversations, and the providers offering them directly keep them out of their chat
|
||||
// model lists as well.
|
||||
//
|
||||
private static readonly string[] SPEECH_SYNTHESIS_MARKERS = ["-tts", "tts-", "-speech", "speech-", "-audio", "audio-"];
|
||||
|
||||
//
|
||||
// The models for spoken conversations over a live connection. They speak their own protocol,
|
||||
// usually a WebSocket, and answer a chat completion request with an error. Checked before
|
||||
// transcription, because some of them carry the name of a transcription model, such as
|
||||
// OpenAI's 'gpt-realtime-whisper'. Those still need the live connection.
|
||||
//
|
||||
private static readonly string[] REALTIME_MARKERS = ["realtime"];
|
||||
|
||||
private static readonly string[] OCR_MARKERS = ["ocr"];
|
||||
|
||||
private static readonly string[] MODERATION_MARKERS = ["moderation", "guard"];
|
||||
|
||||
/// <summary>
|
||||
/// Determines what kind of model this is, based on its name.
|
||||
/// </summary>
|
||||
/// <param name="model">The model to inspect.</param>
|
||||
/// <returns>The recognized kind, or ModelKind.CHAT when we recognize no other kind.</returns>
|
||||
public static ModelKind DetermineKind(this Model model)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(model.Id) || model.IsSystemModel)
|
||||
return ModelKind.CHAT;
|
||||
|
||||
if (HasAnyMarker(model.Id, OTHER_MARKERS))
|
||||
return ModelKind.OTHER;
|
||||
|
||||
if (HasAnyMarker(model.Id, RERANKING_MARKERS))
|
||||
return ModelKind.RERANKING;
|
||||
|
||||
if (HasAnyMarker(model.Id, EMBEDDING_MARKERS))
|
||||
return ModelKind.EMBEDDING;
|
||||
|
||||
if (HasAnyMarker(model.Id, TEXT_COMPLETION_MARKERS))
|
||||
return ModelKind.TEXT_COMPLETION;
|
||||
|
||||
if (HasAnyMarker(model.Id, IMAGE_GENERATION_MARKERS) || HasAnyWordMarker(model.Id, IMAGE_GENERATION_WORD_MARKERS))
|
||||
return ModelKind.IMAGE_GENERATION;
|
||||
|
||||
if (HasAnyMarker(model.Id, VIDEO_GENERATION_MARKERS) || HasAnyWordMarker(model.Id, VIDEO_GENERATION_WORD_MARKERS))
|
||||
return ModelKind.VIDEO_GENERATION;
|
||||
|
||||
if (HasAnyMarker(model.Id, REALTIME_MARKERS))
|
||||
return ModelKind.REALTIME;
|
||||
|
||||
if (HasAnyMarker(model.Id, TRANSCRIPTION_MARKERS))
|
||||
return ModelKind.TRANSCRIPTION;
|
||||
|
||||
if (HasAnyMarker(model.Id, SPEECH_SYNTHESIS_MARKERS))
|
||||
return ModelKind.SPEECH_SYNTHESIS;
|
||||
|
||||
if (HasAnyMarker(model.Id, OCR_MARKERS))
|
||||
return ModelKind.OCR;
|
||||
|
||||
if (HasAnyMarker(model.Id, MODERATION_MARKERS))
|
||||
return ModelKind.MODERATION;
|
||||
|
||||
return ModelKind.CHAT;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Checks whether this model can be used for chatting.
|
||||
/// </summary>
|
||||
/// <param name="model">The model to check.</param>
|
||||
/// <returns>True, when the model is a chat model or when we recognize no other kind.</returns>
|
||||
public static bool IsChatModel(this Model model) => model.DetermineKind() is ModelKind.CHAT;
|
||||
|
||||
/// <summary>
|
||||
/// Checks whether this model creates embeddings.
|
||||
/// </summary>
|
||||
/// <param name="model">The model to check.</param>
|
||||
/// <returns>True, when the model is an embedding model.</returns>
|
||||
public static bool IsEmbeddingModel(this Model model) => model.DetermineKind() is ModelKind.EMBEDDING;
|
||||
|
||||
/// <summary>
|
||||
/// Checks whether this model transcribes audio.
|
||||
/// </summary>
|
||||
/// <param name="model">The model to check.</param>
|
||||
/// <returns>True, when the model is a transcription model.</returns>
|
||||
public static bool IsTranscriptionModel(this Model model) => model.DetermineKind() is ModelKind.TRANSCRIPTION;
|
||||
|
||||
/// <summary>
|
||||
/// Checks whether this model generates images.
|
||||
/// </summary>
|
||||
/// <param name="model">The model to check.</param>
|
||||
/// <returns>True, when the model is an image generation model.</returns>
|
||||
public static bool IsImageModel(this Model model) => model.DetermineKind() is ModelKind.IMAGE_GENERATION;
|
||||
|
||||
private static bool HasAnyMarker(string modelId, string[] markers)
|
||||
{
|
||||
foreach (var marker in markers)
|
||||
if (modelId.Contains(marker, StringComparison.OrdinalIgnoreCase))
|
||||
return true;
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Checks whether the model name contains one of the markers as a word of its own.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A short marker which is also a common syllable cannot be looked for as a plain substring:
|
||||
/// it would match names which have nothing to do with it, and the model would be sorted into
|
||||
/// the wrong kind. Such a marker counts only where a name segment begins and ends with it.
|
||||
/// </remarks>
|
||||
/// <param name="modelId">The ID of the model.</param>
|
||||
/// <param name="markers">The markers to look for.</param>
|
||||
/// <returns>True, when one of the markers stands as a word of its own.</returns>
|
||||
private static bool HasAnyWordMarker(string modelId, string[] markers)
|
||||
{
|
||||
foreach (var marker in markers)
|
||||
{
|
||||
var searchIndex = 0;
|
||||
while (searchIndex <= modelId.Length - marker.Length)
|
||||
{
|
||||
var markerIndex = modelId.IndexOf(marker, searchIndex, StringComparison.OrdinalIgnoreCase);
|
||||
if (markerIndex is -1)
|
||||
break;
|
||||
|
||||
if (IsWholeWord(modelId, marker, markerIndex))
|
||||
return true;
|
||||
|
||||
// The same marker may appear again later in the name, so we keep looking:
|
||||
searchIndex = markerIndex + 1;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
private static bool IsWholeWord(string modelId, string marker, int markerIndex)
|
||||
{
|
||||
if (markerIndex > 0 && !IsSeparator(modelId[markerIndex - 1]))
|
||||
return false;
|
||||
|
||||
var endIndex = markerIndex + marker.Length;
|
||||
return endIndex >= modelId.Length || IsSeparator(modelId[endIndex]);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// The characters which separate the parts of a model name, such as in "fal-ai/kling-video".
|
||||
/// </summary>
|
||||
/// <param name="character">The character to check.</param>
|
||||
/// <returns>True, when the character separates two parts of a name.</returns>
|
||||
private static bool IsSeparator(char character) => character is '/' or '-' or '_' or '.' or ' ' or ':';
|
||||
}
|
||||
@@ -49,7 +49,5 @@ public class NoProvider : IProvider
|
||||
|
||||
public Task<IReadOnlyList<IReadOnlyList<float>>> EmbedTextAsync(Model embeddingModel, SettingsManager settingsManager, CancellationToken token = default, params List<string> texts) => Task.FromResult<IReadOnlyList<IReadOnlyList<float>>>([]);
|
||||
|
||||
public IReadOnlyCollection<Capability> GetModelCapabilities(Model model) => [ Capability.NONE ];
|
||||
|
||||
#endregion
|
||||
}
|
||||
@@ -5,6 +5,7 @@ using System.Text;
|
||||
using System.Text.Json;
|
||||
|
||||
using AIStudio.Chat;
|
||||
using AIStudio.Models;
|
||||
using AIStudio.Settings;
|
||||
using AIStudio.Tools.PluginSystem;
|
||||
using AIStudio.Tools.Rust;
|
||||
@@ -48,10 +49,10 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
|
||||
return base.ClassifyProviderRequestFailure(errorCode, errorType, errorMessage, responseBody);
|
||||
}
|
||||
|
||||
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason) => failureReason switch
|
||||
protected override string GetProviderRequestFailureUserMessage(ProviderRequestFailureReason failureReason, ContextWindow contextWindow = default) => failureReason switch
|
||||
{
|
||||
ProviderRequestFailureReason.INSUFFICIENT_QUOTA => TB("It looks like you do not have any API credits left with OpenAI. Please add credits to your account and try again."),
|
||||
_ => base.GetProviderRequestFailureUserMessage(failureReason),
|
||||
_ => base.GetProviderRequestFailureUserMessage(failureReason, contextWindow),
|
||||
};
|
||||
|
||||
/// <inheritdoc />
|
||||
@@ -92,10 +93,10 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
|
||||
// Read the model capabilities. Through the settings provider, so that the user's expert
|
||||
// capability overrides apply:
|
||||
var providerSettings = this.CreateSettingsProvider(chatModel);
|
||||
var modelCapabilities = providerSettings.GetModelCapabilities();
|
||||
var modelProfile = providerSettings.GetModelProfile();
|
||||
|
||||
// Check if we are using the Responses API or the Chat Completion API:
|
||||
var usingResponsesAPI = modelCapabilities.Contains(Capability.RESPONSES_API);
|
||||
var usingResponsesAPI = modelProfile.Has(Capability.RESPONSES_API);
|
||||
|
||||
// Prepare the request path based on the API we are using:
|
||||
var requestPath = usingResponsesAPI ? "responses" : "chat/completions";
|
||||
@@ -115,7 +116,7 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
|
||||
var minimumWebSearchConfidence = toolRegistry?.GetMinimumProviderConfidence(ToolSelectionRules.WEB_SEARCH_TOOL_ID) ?? ConfidenceLevel.NONE;
|
||||
var isWebSearchAllowed = settingsManager.IsToolActive(ToolSelectionRules.WEB_SEARCH_TOOL_ID) &&
|
||||
ToolSelectionRules.IsProviderConfidenceAllowed(providerConfidence, minimumWebSearchConfidence);
|
||||
IList<object> providerTools = modelCapabilities.Contains(Capability.WEB_SEARCH) && isWebSearchAllowed
|
||||
IList<object> providerTools = modelProfile.Has(Capability.WEB_SEARCH) && isWebSearchAllowed
|
||||
? [ ProviderTools.WEB_SEARCH ]
|
||||
: [];
|
||||
|
||||
@@ -133,8 +134,7 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
var messages = await chatThread.Blocks.BuildMessagesAsync(
|
||||
this.Provider,
|
||||
chatModel,
|
||||
providerSettings,
|
||||
role => role switch
|
||||
{
|
||||
ChatRole.USER => "user",
|
||||
@@ -198,7 +198,7 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
|
||||
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesAsync(
|
||||
this.Provider, chatModel,
|
||||
providerSettings,
|
||||
role => role switch
|
||||
{
|
||||
ChatRole.USER => "user",
|
||||
@@ -368,25 +368,25 @@ public sealed class ProviderOpenAI() : BaseProvider(LLMProviders.OPEN_AI, new Ur
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetTextModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.LLM_PROVIDER, static model => model.IsChatModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.LLM_PROVIDER, model => model.IsChatModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetImageModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.IMAGE_PROVIDER, static model => model.IsImageModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.IMAGE_PROVIDER, model => model.IsImageModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetEmbeddingModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, static model => model.IsEmbeddingModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, model => model.IsEmbeddingModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public override Task<ModelLoadResult> GetTranscriptionModels(string? apiKeyProvisional = null, CancellationToken token = default)
|
||||
{
|
||||
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, static model => model.IsTranscriptionModel(), apiKeyProvisional, token);
|
||||
return this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, model => model.IsTranscriptionModel(this.Provider), apiKeyProvisional, token);
|
||||
}
|
||||
|
||||
#endregion
|
||||
|
||||
@@ -1,8 +1,16 @@
|
||||
using System.Text.Json.Serialization;
|
||||
|
||||
namespace AIStudio.Provider.OpenRouter;
|
||||
|
||||
/// <summary>
|
||||
/// A data model for an OpenRouter model from the API.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The window is the model's, not that of any one provider behind it. OpenRouter also states a
|
||||
/// window per provider it currently prefers, but it picks one per request, so a number taken from
|
||||
/// there would describe a choice nobody has made yet.
|
||||
/// </remarks>
|
||||
/// <param name="Id">The model's ID.</param>
|
||||
/// <param name="Name">The model's human-readable display name.</param>
|
||||
public readonly record struct OpenRouterModel(string Id, string? Name);
|
||||
/// <param name="ContextWindowTokens">How much the model reads and writes in one conversation, in tokens.</param>
|
||||
public readonly record struct OpenRouterModel(string Id, string? Name, [property: JsonPropertyName("context_length")] int? ContextWindowTokens);
|
||||
@@ -2,6 +2,7 @@ using System.Net.Http.Headers;
|
||||
using System.Runtime.CompilerServices;
|
||||
|
||||
using AIStudio.Chat;
|
||||
using AIStudio.Models.Live;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
using AIStudio.Settings;
|
||||
|
||||
@@ -36,7 +37,7 @@ public sealed class ProviderOpenRouter() : BaseProvider(LLMProviders.OPEN_ROUTER
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -117,7 +118,7 @@ public sealed class ProviderOpenRouter() : BaseProvider(LLMProviders.OPEN_ROUTER
|
||||
"models",
|
||||
modelResponse => modelResponse.Data
|
||||
.Select(n => new Model(n.Id, n.Name))
|
||||
.Where(model => model.IsChatModel()),
|
||||
.Where(model => model.IsChatModel(this.Provider)),
|
||||
apiKeyProvisional,
|
||||
requestConfigurator: (request, secretKey) =>
|
||||
{
|
||||
@@ -125,9 +126,21 @@ public sealed class ProviderOpenRouter() : BaseProvider(LLMProviders.OPEN_ROUTER
|
||||
request.Headers.Add("HTTP-Referer", PROJECT_WEBSITE);
|
||||
request.Headers.Add("X-Title", PROJECT_NAME);
|
||||
},
|
||||
listingFactory: modelResponse => modelResponse.Data.Select(n => ModelListing.For(n.Id, n.ContextWindowTokens)),
|
||||
token: token);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Loads the models OpenRouter offers for embedding, which live on a route of their own.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Nothing is reported from here: this route answers with the embedding models alone, and what
|
||||
/// is reported replaces everything an instance said before. The windows of the chat models
|
||||
/// would go missing the moment somebody opens the embedding settings.
|
||||
/// </remarks>
|
||||
/// <param name="apiKeyProvisional">An API key which is not stored yet.</param>
|
||||
/// <param name="token">The cancellation token to use.</param>
|
||||
/// <returns>The embedding models.</returns>
|
||||
private Task<ModelLoadResult> LoadEmbeddingModels(string? apiKeyProvisional, CancellationToken token)
|
||||
{
|
||||
return this.LoadModelsResponse<OpenRouterModelsResponse>(
|
||||
|
||||
@@ -41,7 +41,7 @@ public sealed class ProviderPerplexity() : BaseProvider(LLMProviders.PERPLEXITY,
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// Anthropic's extended thinking, written as a "thinking" object.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The object carries a type, and the two types which switch thinking on are named outright:
|
||||
/// "enabled" and "adaptive". Everything else falls through to the ordinary reading of a value, so
|
||||
/// that a person writing "thinking": false is understood as well.
|
||||
/// </remarks>
|
||||
public sealed class AnthropicThinkingDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.ANTHROPIC_THINKING;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
|
||||
{
|
||||
if (!ReasoningParameters.TryGet(parameters, "thinking", out var thinking))
|
||||
return ReasoningConfigurationState.NOT_CONFIGURED;
|
||||
|
||||
return thinking switch
|
||||
{
|
||||
IDictionary<string, object> thinkingObject when ReasoningParameters.TryGet(thinkingObject, "type", out var type) => TypeOf(type),
|
||||
|
||||
_ => ReasoningParameters.LevelOf(thinking),
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reads the "type" of an Anthropic thinking object.
|
||||
/// </summary>
|
||||
/// <param name="value">The configured thinking type.</param>
|
||||
/// <returns>What it says.</returns>
|
||||
private static ReasoningConfigurationState TypeOf(object? value) => value switch
|
||||
{
|
||||
string text when text.Equals("enabled", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("adaptive", StringComparison.OrdinalIgnoreCase)
|
||||
=> ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
|
||||
string text when ReasoningParameters.IsDisabledText(text) => ReasoningConfigurationState.EXPLICITLY_DISABLED,
|
||||
|
||||
_ => ReasoningParameters.LevelOf(value),
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// Google's thinking config, thinking level, and thought summaries.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Google offers the same settings in several places at once: directly, under "generation_config",
|
||||
/// and in both spellings of each key, because their own libraries write snake case while the REST
|
||||
/// API answers in camel case. All of them are read, and the answers put together.
|
||||
///
|
||||
/// Summaries are the one setting which only ever says yes. Asking for thought summaries proves that
|
||||
/// thinking is on; switching them off proves nothing, because a model can think without showing it.
|
||||
/// </remarks>
|
||||
public sealed class GoogleThinkingDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.GOOGLE_THINKING;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
|
||||
{
|
||||
var states = new List<ReasoningConfigurationState>();
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "thinking_config", out var thinkingConfig) &&
|
||||
thinkingConfig is IDictionary<string, object> thinkingConfigObject)
|
||||
states.Add(ConfigOf(thinkingConfigObject));
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "generation_config", out var generationConfig) &&
|
||||
generationConfig is IDictionary<string, object> generationConfigObject)
|
||||
{
|
||||
if (ReasoningParameters.TryGet(generationConfigObject, "thinking_config", out var nestedThinkingConfig) &&
|
||||
nestedThinkingConfig is IDictionary<string, object> nestedThinkingConfigObject)
|
||||
states.Add(ConfigOf(nestedThinkingConfigObject));
|
||||
|
||||
if (ReasoningParameters.TryGet(generationConfigObject, "thinking_summaries", out var thinkingSummaries))
|
||||
states.Add(SummariesOf(thinkingSummaries));
|
||||
|
||||
if (ReasoningParameters.TryGet(generationConfigObject, "thinking_level", out var thinkingLevel))
|
||||
states.Add(ReasoningParameters.LevelOf(thinkingLevel));
|
||||
}
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "thinking_summaries", out var topLevelThinkingSummaries))
|
||||
states.Add(SummariesOf(topLevelThinkingSummaries));
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "thinking_level", out var topLevelThinkingLevel))
|
||||
states.Add(ReasoningParameters.LevelOf(topLevelThinkingLevel));
|
||||
|
||||
return ReasoningParameters.Merge(states);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reads a thinking config, in either spelling of its keys.
|
||||
/// </summary>
|
||||
/// <param name="thinkingConfig">The parsed thinking config object.</param>
|
||||
/// <returns>What it says.</returns>
|
||||
private static ReasoningConfigurationState ConfigOf(IDictionary<string, object> thinkingConfig)
|
||||
{
|
||||
var states = new List<ReasoningConfigurationState>();
|
||||
|
||||
if (ReasoningParameters.TryGet(thinkingConfig, "thinking_budget", out var thinkingBudget) ||
|
||||
ReasoningParameters.TryGet(thinkingConfig, "thinkingBudget", out thinkingBudget))
|
||||
states.Add(ReasoningParameters.BudgetOf(thinkingBudget));
|
||||
|
||||
if (ReasoningParameters.TryGet(thinkingConfig, "include_thoughts", out var includeThoughts) ||
|
||||
ReasoningParameters.TryGet(thinkingConfig, "includeThoughts", out includeThoughts))
|
||||
states.Add(ReasoningParameters.LevelOf(includeThoughts));
|
||||
|
||||
return ReasoningParameters.Merge(states);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reads a thought summary setting, which can only ever say yes.
|
||||
/// </summary>
|
||||
/// <param name="value">The configured summary setting.</param>
|
||||
/// <returns>Yes, when it asks for summaries; nothing otherwise.</returns>
|
||||
private static ReasoningConfigurationState SummariesOf(object? value) => value switch
|
||||
{
|
||||
string text when text.Equals("auto", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("on", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("summarized", StringComparison.OrdinalIgnoreCase)
|
||||
=> ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
|
||||
true => ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
|
||||
_ => ReasoningConfigurationState.NOT_CONFIGURED,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// The reasoning mode and budget of the llama.cpp server.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Its "reasoning" key is a mode rather than an object, and one of its three values means neither
|
||||
/// yes nor no: "auto" hands the decision to the model's own template, which is exactly the case
|
||||
/// where nobody has decided anything.
|
||||
/// </remarks>
|
||||
public sealed class LlamaCppReasoningDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.LLAMA_CPP;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
|
||||
{
|
||||
var states = new List<ReasoningConfigurationState>();
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "reasoning", out var reasoning))
|
||||
states.Add(ModeOf(reasoning));
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "reasoning_budget", out var reasoningBudget))
|
||||
states.Add(ReasoningParameters.BudgetOf(reasoningBudget));
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "chat_template_kwargs", out var chatTemplateKwargs) &&
|
||||
chatTemplateKwargs is IDictionary<string, object> chatTemplateKwargsObject)
|
||||
states.Add(QwenThinkingDialect.In(chatTemplateKwargsObject));
|
||||
|
||||
return ReasoningParameters.Merge(states);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reads the reasoning mode.
|
||||
/// </summary>
|
||||
/// <param name="value">The configured mode.</param>
|
||||
/// <returns>What it says, which for "auto" is nothing.</returns>
|
||||
private static ReasoningConfigurationState ModeOf(object? value) => value switch
|
||||
{
|
||||
string text when text.Equals("on", StringComparison.OrdinalIgnoreCase) => ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
string text when text.Equals("off", StringComparison.OrdinalIgnoreCase) => ReasoningConfigurationState.EXPLICITLY_DISABLED,
|
||||
string text when text.Equals("auto", StringComparison.OrdinalIgnoreCase) => ReasoningConfigurationState.NOT_CONFIGURED,
|
||||
|
||||
_ => ReasoningParameters.LevelOf(value),
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// Ollama's "think" parameter.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// One key, and it takes a boolean as readily as a level, which is why it needs no reading of its
|
||||
/// own beyond the ordinary one.
|
||||
/// </remarks>
|
||||
public sealed class OllamaThinkDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.OLLAMA_THINK;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters) =>
|
||||
ReasoningParameters.TryGet(parameters, "think", out var think)
|
||||
? ReasoningParameters.LevelOf(think)
|
||||
: ReasoningConfigurationState.NOT_CONFIGURED;
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// The nested "reasoning" object almost every OpenAI-compatible server accepts.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The object may carry an effort or a summary setting, and it may be written as a plain value
|
||||
/// instead. An object carrying neither says nothing: somebody who wrote "reasoning": {} has not
|
||||
/// asked for anything yet.
|
||||
/// </remarks>
|
||||
public sealed class OpenAICompatibleDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.OPEN_AI_COMPATIBLE;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
|
||||
{
|
||||
if (!ReasoningParameters.TryGet(parameters, "reasoning", out var reasoning))
|
||||
return ReasoningConfigurationState.NOT_CONFIGURED;
|
||||
|
||||
return reasoning switch
|
||||
{
|
||||
IDictionary<string, object> reasoningObject when ReasoningParameters.TryGet(reasoningObject, "effort", out var effort) => ReasoningParameters.LevelOf(effort),
|
||||
IDictionary<string, object> reasoningObject when ReasoningParameters.TryGet(reasoningObject, "summary", out var summary) => ReasoningParameters.LevelOf(summary),
|
||||
IDictionary<string, object> => ReasoningConfigurationState.NOT_CONFIGURED,
|
||||
|
||||
_ => ReasoningParameters.LevelOf(reasoning),
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// The "enable_thinking" switch Qwen introduced and other servers took over.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// It is accepted at the top level and inside "chat_template_kwargs", because it is really an
|
||||
/// argument to the chat template rather than to the API -- which is also why two other dialects ask
|
||||
/// this one about their own kwargs object instead of repeating the two keys.
|
||||
/// </remarks>
|
||||
public sealed class QwenThinkingDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.QWEN_THINKING;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters) => In(parameters);
|
||||
|
||||
/// <summary>
|
||||
/// Reads the switch out of any parameter object, which need not be the top-level one.
|
||||
/// </summary>
|
||||
/// <param name="parameters">The object to look in.</param>
|
||||
/// <returns>What it says.</returns>
|
||||
public static ReasoningConfigurationState In(IDictionary<string, object> parameters)
|
||||
{
|
||||
var states = new List<ReasoningConfigurationState>();
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "enable_thinking", out var enableThinking))
|
||||
states.Add(ReasoningParameters.LevelOf(enableThinking));
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "chat_template_kwargs", out var chatTemplateKwargs) &&
|
||||
chatTemplateKwargs is IDictionary<string, object> chatTemplateKwargsObject &&
|
||||
ReasoningParameters.TryGet(chatTemplateKwargsObject, "enable_thinking", out var nestedEnableThinking))
|
||||
states.Add(ReasoningParameters.LevelOf(nestedEnableThinking));
|
||||
|
||||
return ReasoningParameters.Merge(states);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// The top-level "reasoning_effort" parameter.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A dialect of its own although it is one key, because it travels on its own: providers accept it
|
||||
/// without the nested object next to it, and the code this replaces had to remember to check for it
|
||||
/// separately at every one of them. Here it is one line in the table instead.
|
||||
/// </remarks>
|
||||
public sealed class ReasoningEffortDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.REASONING_EFFORT;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters) =>
|
||||
ReasoningParameters.TryGet(parameters, "reasoning_effort", out var effort)
|
||||
? ReasoningParameters.LevelOf(effort)
|
||||
: ReasoningConfigurationState.NOT_CONFIGURED;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
namespace AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
/// <summary>
|
||||
/// The thinking token budget and chat template kwargs of vLLM.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// What vLLM accepts depends on the model family it was pointed at and on which reasoning parser
|
||||
/// the operator started it with, so both the budget and the template arguments are read.
|
||||
/// </remarks>
|
||||
public sealed class VllmReasoningDialect : IReasoningDialect
|
||||
{
|
||||
/// <inheritdoc />
|
||||
public ReasoningDialect Dialect => ReasoningDialect.VLLM;
|
||||
|
||||
/// <inheritdoc />
|
||||
public ReasoningConfigurationState Detect(IDictionary<string, object> parameters)
|
||||
{
|
||||
var states = new List<ReasoningConfigurationState>();
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "thinking_token_budget", out var thinkingTokenBudget))
|
||||
states.Add(ReasoningParameters.BudgetOf(thinkingTokenBudget));
|
||||
|
||||
if (ReasoningParameters.TryGet(parameters, "chat_template_kwargs", out var chatTemplateKwargs) &&
|
||||
chatTemplateKwargs is IDictionary<string, object> chatTemplateKwargsObject)
|
||||
{
|
||||
states.Add(QwenThinkingDialect.In(chatTemplateKwargsObject));
|
||||
|
||||
if (ReasoningParameters.TryGet(chatTemplateKwargsObject, "thinking", out var thinking))
|
||||
states.Add(ReasoningParameters.LevelOf(thinking));
|
||||
}
|
||||
|
||||
return ReasoningParameters.Merge(states);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
namespace AIStudio.Provider.Reasoning;
|
||||
|
||||
/// <summary>
|
||||
/// One way of asking a request to think, and how to recognize it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A dialect reads parameters and says nothing else. It does not know which provider it is being
|
||||
/// asked for, it keeps no state, and it never looks at the model -- what a model is able to do comes
|
||||
/// from the rules, and mixing the two is what made the code this replaces hard to follow.
|
||||
/// </remarks>
|
||||
public interface IReasoningDialect
|
||||
{
|
||||
/// <summary>
|
||||
/// Which dialect this is, which is also where it stands in the order.
|
||||
/// </summary>
|
||||
ReasoningDialect Dialect { get; }
|
||||
|
||||
/// <summary>
|
||||
/// Reads what these parameters say about reasoning.
|
||||
/// </summary>
|
||||
/// <param name="parameters">The parsed additional API parameters.</param>
|
||||
/// <returns>What they say, which is usually nothing.</returns>
|
||||
ReasoningConfigurationState Detect(IDictionary<string, object> parameters);
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
namespace AIStudio.Provider.Reasoning;
|
||||
|
||||
/// <summary>
|
||||
/// What the additional API parameters of a provider say about reasoning.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// This answers a different question than ReasoningSupport does. That one says what a model is able
|
||||
/// to do, and it comes from the rules. This one says what the person asked their provider for, in
|
||||
/// the free-text parameters they wrote themselves -- and most of the time it says nothing at all,
|
||||
/// which is a statement of its own rather than a missing answer.
|
||||
/// </remarks>
|
||||
public enum ReasoningConfigurationState
|
||||
{
|
||||
/// <summary>
|
||||
/// No recognized reasoning parameter was found.
|
||||
/// </summary>
|
||||
NOT_CONFIGURED,
|
||||
|
||||
/// <summary>
|
||||
/// A recognized reasoning parameter explicitly enables reasoning.
|
||||
/// </summary>
|
||||
EXPLICITLY_ENABLED,
|
||||
|
||||
/// <summary>
|
||||
/// A recognized reasoning parameter explicitly disables reasoning.
|
||||
/// </summary>
|
||||
EXPLICITLY_DISABLED,
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
namespace AIStudio.Provider.Reasoning;
|
||||
|
||||
/// <summary>
|
||||
/// The ways a request can be asked to think, one per way of writing it down.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every provider speaks one or more of these, and which ones is stated in the dispatcher rather
|
||||
/// than worked out from anything. The order here is the order they are asked in: the answer does not
|
||||
/// depend on it -- a "no" wins wherever it stands -- but a report which named them in whatever order
|
||||
/// a container handed them over would read differently on another machine.
|
||||
/// </remarks>
|
||||
public enum ReasoningDialect
|
||||
{
|
||||
/// <summary>
|
||||
/// The nested "reasoning" object most OpenAI-compatible servers accept.
|
||||
/// </summary>
|
||||
OPEN_AI_COMPATIBLE,
|
||||
|
||||
/// <summary>
|
||||
/// The top-level "reasoning_effort" parameter.
|
||||
/// </summary>
|
||||
REASONING_EFFORT,
|
||||
|
||||
/// <summary>
|
||||
/// Anthropic's extended thinking, written as a "thinking" object.
|
||||
/// </summary>
|
||||
ANTHROPIC_THINKING,
|
||||
|
||||
/// <summary>
|
||||
/// Google's thinking config, thinking level, and thought summaries.
|
||||
/// </summary>
|
||||
GOOGLE_THINKING,
|
||||
|
||||
/// <summary>
|
||||
/// The "enable_thinking" switch Qwen introduced and other servers took over.
|
||||
/// </summary>
|
||||
QWEN_THINKING,
|
||||
|
||||
/// <summary>
|
||||
/// Ollama's "think" parameter.
|
||||
/// </summary>
|
||||
OLLAMA_THINK,
|
||||
|
||||
/// <summary>
|
||||
/// The reasoning mode and budget of the llama.cpp server.
|
||||
/// </summary>
|
||||
LLAMA_CPP,
|
||||
|
||||
/// <summary>
|
||||
/// The thinking token budget and chat template kwargs of vLLM.
|
||||
/// </summary>
|
||||
VLLM,
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
using System.Collections.Concurrent;
|
||||
using System.Collections.Frozen;
|
||||
|
||||
using AIStudio.Provider.Reasoning.Dialects;
|
||||
|
||||
using Host = AIStudio.Provider.SelfHosted.Host;
|
||||
|
||||
namespace AIStudio.Provider.Reasoning;
|
||||
|
||||
/// <summary>
|
||||
/// Decides which dialects a provider speaks, and reads its parameters in all of them.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Which dialect answers for which provider is a table here rather than a chain of checks spread
|
||||
/// through the reading itself. That is the whole point of the split: adding a provider means adding
|
||||
/// a line, and reading what one accepts means reading one line.
|
||||
///
|
||||
/// The answer is worked out once per provider setting. The question is asked from the provider list,
|
||||
/// which re-renders whenever anything on the page changes, and the old code parsed the JSON a person
|
||||
/// typed into their expert settings on every one of those renders. Nothing here reaches for
|
||||
/// application state, so a test can ask it without the app having started.
|
||||
/// </remarks>
|
||||
public static class ReasoningDispatcher
|
||||
{
|
||||
/// <summary>
|
||||
/// Every dialect there is, in the order the enum names them.
|
||||
/// </summary>
|
||||
private static readonly FrozenDictionary<ReasoningDialect, IReasoningDialect> DIALECTS = new IReasoningDialect[]
|
||||
{
|
||||
new OpenAICompatibleDialect(),
|
||||
new ReasoningEffortDialect(),
|
||||
new AnthropicThinkingDialect(),
|
||||
new GoogleThinkingDialect(),
|
||||
new QwenThinkingDialect(),
|
||||
new OllamaThinkDialect(),
|
||||
new LlamaCppReasoningDialect(),
|
||||
new VllmReasoningDialect(),
|
||||
}.ToFrozenDictionary(dialect => dialect.Dialect);
|
||||
|
||||
/// <summary>
|
||||
/// Every dialect there is, in the order the enum names them.
|
||||
/// </summary>
|
||||
public static IReadOnlyList<IReasoningDialect> Dialects { get; } = DIALECTS.Values.OrderBy(dialect => dialect.Dialect).ToList();
|
||||
|
||||
/// <summary>
|
||||
/// What an OpenAI-compatible server understands when nothing more is known about it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The gateways and resellers serve everybody's models, so they are asked in every dialect a
|
||||
/// model of any vendor might answer to. Reading one dialect too many costs a dictionary lookup;
|
||||
/// reading one too few hides a switch the person has set.
|
||||
/// </remarks>
|
||||
private static readonly ReasoningDialect[] EVERYTHING_A_GATEWAY_MIGHT_SERVE =
|
||||
[
|
||||
ReasoningDialect.OPEN_AI_COMPATIBLE,
|
||||
ReasoningDialect.REASONING_EFFORT,
|
||||
ReasoningDialect.QWEN_THINKING,
|
||||
ReasoningDialect.GOOGLE_THINKING,
|
||||
];
|
||||
|
||||
private static readonly ReasoningDialect[] NOTHING = [];
|
||||
|
||||
/// <summary>
|
||||
/// The answers already worked out, so that the same settings are read once.
|
||||
/// </summary>
|
||||
private static readonly ConcurrentDictionary<(LLMProviders Provider, Host Host, string Parameters), ReasoningConfigurationState> ANSWERED = new();
|
||||
|
||||
/// <summary>
|
||||
/// Reads what a provider's additional API parameters say about reasoning.
|
||||
/// </summary>
|
||||
/// <param name="provider">The LLM provider.</param>
|
||||
/// <param name="host">The engine behind it, which only matters for self-hosted providers.</param>
|
||||
/// <param name="additionalParameters">The additional API parameters, as the person wrote them.</param>
|
||||
/// <returns>What they say, which is usually nothing.</returns>
|
||||
public static ReasoningConfigurationState WhatTheParametersSay(LLMProviders provider, Host host, string? additionalParameters)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(additionalParameters))
|
||||
return ReasoningConfigurationState.NOT_CONFIGURED;
|
||||
|
||||
return ANSWERED.GetOrAdd((provider, host, additionalParameters), static key => Read(key.Provider, key.Host, key.Parameters));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Which dialects this provider speaks.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The commercial providers are asked only in their own dialect plus whatever their API
|
||||
/// documents, because a parameter they do not accept says nothing about what they will do. The
|
||||
/// self-hosted engines are the other case: the operator picked the engine, so what it accepts is
|
||||
/// known, and it is the engine rather than the model which decides.
|
||||
/// </remarks>
|
||||
/// <param name="provider">The LLM provider.</param>
|
||||
/// <param name="host">The engine behind it.</param>
|
||||
/// <returns>The dialects to read the parameters in.</returns>
|
||||
public static IReadOnlyList<ReasoningDialect> DialectsOf(LLMProviders provider, Host host) => provider switch
|
||||
{
|
||||
LLMProviders.OPEN_AI => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT],
|
||||
|
||||
LLMProviders.ANTHROPIC => [ReasoningDialect.ANTHROPIC_THINKING],
|
||||
|
||||
LLMProviders.MISTRAL or LLMProviders.PERPLEXITY => [ReasoningDialect.REASONING_EFFORT],
|
||||
|
||||
LLMProviders.GOOGLE => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.GOOGLE_THINKING],
|
||||
|
||||
LLMProviders.ALIBABA_CLOUD => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING],
|
||||
|
||||
LLMProviders.OPEN_ROUTER or
|
||||
LLMProviders.HETZNER or
|
||||
LLMProviders.IONOS or
|
||||
LLMProviders.LITE_LLM or
|
||||
LLMProviders.X or
|
||||
LLMProviders.DEEP_SEEK or
|
||||
LLMProviders.GROQ or
|
||||
LLMProviders.FIREWORKS or
|
||||
LLMProviders.HUGGINGFACE or
|
||||
LLMProviders.HELMHOLTZ or
|
||||
LLMProviders.GWDG => EVERYTHING_A_GATEWAY_MIGHT_SERVE,
|
||||
|
||||
LLMProviders.SELF_HOSTED => host switch
|
||||
{
|
||||
Host.OLLAMA => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING, ReasoningDialect.OLLAMA_THINK],
|
||||
|
||||
Host.LLAMA_CPP => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING, ReasoningDialect.LLAMA_CPP],
|
||||
|
||||
Host.VLLM => [ReasoningDialect.OPEN_AI_COMPATIBLE, ReasoningDialect.REASONING_EFFORT, ReasoningDialect.QWEN_THINKING, ReasoningDialect.GOOGLE_THINKING, ReasoningDialect.VLLM],
|
||||
|
||||
_ => EVERYTHING_A_GATEWAY_MIGHT_SERVE,
|
||||
},
|
||||
|
||||
_ => NOTHING,
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// Parses the parameters and asks every dialect this provider speaks.
|
||||
/// </summary>
|
||||
/// <param name="provider">The LLM provider.</param>
|
||||
/// <param name="host">The engine behind it.</param>
|
||||
/// <param name="additionalParameters">The additional API parameters.</param>
|
||||
/// <returns>What they say.</returns>
|
||||
private static ReasoningConfigurationState Read(LLMProviders provider, Host host, string additionalParameters)
|
||||
{
|
||||
if (!AdditionalApiParametersParser.TryParse(additionalParameters, out var parameters, out _))
|
||||
return ReasoningConfigurationState.NOT_CONFIGURED;
|
||||
|
||||
return ReasoningParameters.Merge(DialectsOf(provider, host).Select(key => DIALECTS[key].Detect(parameters)));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
namespace AIStudio.Provider.Reasoning;
|
||||
|
||||
/// <summary>
|
||||
/// Reading the values a person wrote into their additional API parameters.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every dialect ends up asking the same two questions: is this key there, and does this value mean
|
||||
/// yes or no. The answers are the same whoever asks them -- "off" is off at every provider -- so
|
||||
/// they live here rather than once per dialect.
|
||||
/// </remarks>
|
||||
public static class ReasoningParameters
|
||||
{
|
||||
/// <summary>
|
||||
/// Try to read a parameter, matching the key regardless of how it was capitalized.
|
||||
/// </summary>
|
||||
/// <param name="parameters">The parsed parameter dictionary.</param>
|
||||
/// <param name="key">The parameter name to find.</param>
|
||||
/// <param name="value">The matched parameter value, if found.</param>
|
||||
/// <returns>True, when a matching key was found.</returns>
|
||||
public static bool TryGet(IDictionary<string, object> parameters, string key, out object? value)
|
||||
{
|
||||
value = null;
|
||||
if (parameters.Count is 0)
|
||||
return false;
|
||||
|
||||
var foundKey = parameters.Keys.FirstOrDefault(candidate => string.Equals(candidate, key, StringComparison.OrdinalIgnoreCase));
|
||||
if (foundKey is null)
|
||||
return false;
|
||||
|
||||
value = parameters[foundKey];
|
||||
return true;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Reads a value which is written as a boolean, a number, or a level.
|
||||
/// </summary>
|
||||
/// <param name="value">The raw parsed parameter value.</param>
|
||||
/// <returns>What the value says.</returns>
|
||||
public static ReasoningConfigurationState LevelOf(object? value) => value switch
|
||||
{
|
||||
bool booleanValue => booleanValue ? ReasoningConfigurationState.EXPLICITLY_ENABLED : ReasoningConfigurationState.EXPLICITLY_DISABLED,
|
||||
int i => i is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
long l => l is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
double d => Math.Abs(d) < double.Epsilon ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
decimal m => m is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
string text when IsDisabledText(text) => ReasoningConfigurationState.EXPLICITLY_DISABLED,
|
||||
string text when IsEnabledText(text) => ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
|
||||
_ => ReasoningConfigurationState.NOT_CONFIGURED,
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// Reads a token budget, which several providers use to say the same thing with a number.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A budget of zero switches thinking off. Everything else, negative budgets included, leaves it
|
||||
/// available -- a negative one usually means "as much as it takes".
|
||||
/// </remarks>
|
||||
/// <param name="value">The configured budget value.</param>
|
||||
/// <returns>What the budget says.</returns>
|
||||
public static ReasoningConfigurationState BudgetOf(object? value) => value switch
|
||||
{
|
||||
int i => i is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
long l => l is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
double d => Math.Abs(d) < double.Epsilon ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
decimal m => m is 0 ? ReasoningConfigurationState.EXPLICITLY_DISABLED : ReasoningConfigurationState.EXPLICITLY_ENABLED,
|
||||
|
||||
_ => LevelOf(value),
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// Puts several answers together into one.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A "no" wins over a "yes", wherever the two stand. Somebody who switched thinking off in one
|
||||
/// place meant to switch it off, and an indicator lighting up anyway because another parameter
|
||||
/// could be read as a yes would be the app arguing with them.
|
||||
/// </remarks>
|
||||
/// <param name="states">What the dialects found.</param>
|
||||
/// <returns>The one answer.</returns>
|
||||
public static ReasoningConfigurationState Merge(IEnumerable<ReasoningConfigurationState> states)
|
||||
{
|
||||
var result = ReasoningConfigurationState.NOT_CONFIGURED;
|
||||
foreach (var state in states)
|
||||
{
|
||||
if (state is ReasoningConfigurationState.EXPLICITLY_DISABLED)
|
||||
return ReasoningConfigurationState.EXPLICITLY_DISABLED;
|
||||
|
||||
if (state is ReasoningConfigurationState.EXPLICITLY_ENABLED)
|
||||
result = ReasoningConfigurationState.EXPLICITLY_ENABLED;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Puts several answers together into one.
|
||||
/// </summary>
|
||||
/// <param name="states">What the dialects found.</param>
|
||||
/// <returns>The one answer.</returns>
|
||||
public static ReasoningConfigurationState Merge(params ReasoningConfigurationState[] states) => Merge(states.AsEnumerable());
|
||||
|
||||
/// <summary>
|
||||
/// Whether a text means yes.
|
||||
/// </summary>
|
||||
/// <param name="text">The string value to inspect.</param>
|
||||
/// <returns>True, when the value switches reasoning on.</returns>
|
||||
public static bool IsEnabledText(string text) =>
|
||||
text.Equals("true", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("yes", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("on", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("enabled", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("low", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("minimal", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("medium", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("high", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("max", StringComparison.OrdinalIgnoreCase);
|
||||
|
||||
/// <summary>
|
||||
/// Whether a text means no.
|
||||
/// </summary>
|
||||
/// <param name="text">The string value to inspect.</param>
|
||||
/// <returns>True, when the value switches reasoning off.</returns>
|
||||
public static bool IsDisabledText(string text) =>
|
||||
string.IsNullOrWhiteSpace(text) ||
|
||||
text.Equals("false", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("no", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("off", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("none", StringComparison.OrdinalIgnoreCase) ||
|
||||
text.Equals("disabled", StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
@@ -1,3 +1,23 @@
|
||||
using System.Text.Json.Serialization;
|
||||
|
||||
namespace AIStudio.Provider.SelfHosted;
|
||||
|
||||
public readonly record struct Model(string Id, string? Object, string? OwnedBy, ModelArchitecture? Architecture);
|
||||
/// <summary>
|
||||
/// One model as an OpenAI-compatible engine lists it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The context window is vLLM's addition to that route: it reports the window the operator started
|
||||
/// the engine with, which is the one number no rule about the weights could ever know. Ollama,
|
||||
/// LM Studio, and llama.cpp answer the same route without it, so it stays unknown there instead of
|
||||
/// being guessed.
|
||||
///
|
||||
/// vLLM calls that field max_model_len, which reads like a limit on the model rather than on a
|
||||
/// conversation. The wire keeps their spelling, and this record says what the number means, so that
|
||||
/// nobody has to remember the translation while reading the code that uses it.
|
||||
/// </remarks>
|
||||
/// <param name="Id">The model's ID.</param>
|
||||
/// <param name="Object">What kind of thing the entry is. Known value: "model".</param>
|
||||
/// <param name="OwnedBy">Who the engine names as the owner of the model.</param>
|
||||
/// <param name="Architecture">Which kinds of input and output the model takes, where the engine says.</param>
|
||||
/// <param name="ContextWindowTokens">The context window the engine was started with, in tokens, where it says.</param>
|
||||
public readonly record struct Model(string Id, string? Object, string? OwnedBy, ModelArchitecture? Architecture, [property: JsonPropertyName("max_model_len")] int? ContextWindowTokens);
|
||||
@@ -3,6 +3,7 @@ using System.Runtime.CompilerServices;
|
||||
using System.Text.Json;
|
||||
|
||||
using AIStudio.Chat;
|
||||
using AIStudio.Models.Live;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
using AIStudio.Settings;
|
||||
using AIStudio.Tools.PluginSystem;
|
||||
@@ -42,8 +43,8 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide
|
||||
// - LM Studio, vLLM, and llama.cpp use the nested image URL format: { "type": "image_url", "image_url": { "url": "data:..." } }
|
||||
var messages = host switch
|
||||
{
|
||||
Host.OLLAMA => await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.Provider, effectiveChatModel),
|
||||
_ => await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, effectiveChatModel),
|
||||
Host.OLLAMA => await chatThread.Blocks.BuildMessagesUsingDirectImageUrlAsync(this.CreateSettingsProvider(effectiveChatModel)),
|
||||
_ => await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(effectiveChatModel)),
|
||||
};
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
@@ -188,8 +189,23 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide
|
||||
return FailedModelLoadResult(this.GetModelLoadFailureReason(lmStudioResponse, responseBody), $"Status={(int)lmStudioResponse.StatusCode} {lmStudioResponse.ReasonPhrase}; Body='{responseBody}'");
|
||||
}
|
||||
|
||||
var lmStudioModelResponse = await lmStudioResponse.Content.ReadFromJsonAsync<ModelsResponse>(token);
|
||||
//
|
||||
// Read with the shared options, the way every other model list of this app is read.
|
||||
// This one route did without them, which quietly cost it every field an engine spells
|
||||
// in snake case: owned_by has been arriving as nothing all along, and the next field
|
||||
// somebody adds here would have gone the same way without anything failing.
|
||||
//
|
||||
var lmStudioModelResponse = await lmStudioResponse.Content.ReadFromJsonAsync<ModelsResponse>(JSON_SERIALIZER_OPTIONS, token);
|
||||
var models = lmStudioModelResponse.Data ?? [];
|
||||
|
||||
//
|
||||
// What the engine said about its own models, taken from the whole list rather than
|
||||
// from what is offered below: a model filtered out here as an embedding model is still
|
||||
// a model somebody may have configured this instance with, and this list is the only
|
||||
// place its window is ever stated.
|
||||
//
|
||||
ListedModels.Shared.Report(this.ConfiguredProviderId, ListingsOf(models));
|
||||
|
||||
return SuccessfulModelLoadResult(models.
|
||||
Where(model => !string.IsNullOrWhiteSpace(model.Id) &&
|
||||
!ignorePhrases.Any(ignorePhrase => model.Id.Contains(ignorePhrase, StringComparison.InvariantCulture)) &&
|
||||
@@ -302,6 +318,13 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// What an engine stated about the models it serves.
|
||||
/// </summary>
|
||||
/// <param name="models">The models exactly as the engine listed them.</param>
|
||||
/// <returns>One listing per model, which says nothing for the models the engine was silent about.</returns>
|
||||
private static IEnumerable<ModelListing> ListingsOf(IEnumerable<Model> models) => models.Select(model => ModelListing.For(model.Id, model.ContextWindowTokens));
|
||||
|
||||
private static bool IsMatchingLlamaCppTextModel(Model model, string[] ignorePhrases, string[] filterPhrases)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(model.Id))
|
||||
|
||||
@@ -32,7 +32,7 @@ public sealed class ProviderX() : BaseProvider(LLMProviders.X, new Uri("https://
|
||||
async (systemPrompt, apiParameters, tools) =>
|
||||
{
|
||||
// Build the list of messages:
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.Provider, chatModel);
|
||||
var messages = await chatThread.Blocks.BuildMessagesUsingNestedImageUrlAsync(this.CreateSettingsProvider(chatModel));
|
||||
|
||||
return new ChatCompletionAPIRequest
|
||||
{
|
||||
@@ -79,7 +79,12 @@ public sealed class ProviderX() : BaseProvider(LLMProviders.X, new Uri("https://
|
||||
var result = await this.LoadModels(SecretStoreType.LLM_PROVIDER, ["grok-"], apiKeyProvisional, token);
|
||||
return result with
|
||||
{
|
||||
Models = [..result.Models.Where(n => !n.Id.Contains("-image", StringComparison.OrdinalIgnoreCase))]
|
||||
//
|
||||
// Asking what a model is made for rather than testing its name for a word. The word was
|
||||
// "-image", which said nothing about grok-imagine-video: that one made films and stood
|
||||
// in the list of things to chat with.
|
||||
//
|
||||
Models = [..result.Models.Where(model => model.IsChatModel(this.Provider))]
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user