From c6e36a11d37379d9863b7a0178e7def9858b8f2c Mon Sep 17 00:00:00 2001 From: Thorsten Sommer Date: Sat, 19 Sep 2026 12:10:31 +0200 Subject: [PATCH] Sort the self-hosted model lists by what a model is made for --- .../Provider/SelfHosted/ProviderSelfHosted.cs | 80 ++++++++--- .../Provider/SelfHostedModelListTests.cs | 135 ++++++++++++++++++ 2 files changed, 195 insertions(+), 20 deletions(-) create mode 100644 app/Tests/Provider/SelfHostedModelListTests.cs diff --git a/app/MindWork AI Studio/Provider/SelfHosted/ProviderSelfHosted.cs b/app/MindWork AI Studio/Provider/SelfHosted/ProviderSelfHosted.cs index c24e7238..999e0b44 100644 --- a/app/MindWork AI Studio/Provider/SelfHosted/ProviderSelfHosted.cs +++ b/app/MindWork AI Studio/Provider/SelfHosted/ProviderSelfHosted.cs @@ -97,12 +97,16 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide switch (host) { case Host.LLAMA_CPP: - return await this.LoadLlamaCppTextModels(["embed"], [], apiKeyProvisional, token); - + return await this.LoadLlamaCppTextModels(apiKeyProvisional, token); + case Host.LM_STUDIO: case Host.OLLAMA: case Host.VLLM: - return await this.LoadModels( SecretStoreType.LLM_PROVIDER, ["embed"], [], apiKeyProvisional, token); + var result = await this.LoadModels(SecretStoreType.LLM_PROVIDER, apiKeyProvisional, token); + return result with + { + Models = [..result.Models.Where(model => model.IsChatModel(this.Provider))] + }; } return ModelLoadResult.FromModels([]); @@ -129,14 +133,18 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide case Host.LM_STUDIO: case Host.OLLAMA: case Host.VLLM: - return await this.LoadModels( SecretStoreType.EMBEDDING_PROVIDER, [], ["embed"], apiKeyProvisional, token); + var result = await this.LoadModels(SecretStoreType.EMBEDDING_PROVIDER, apiKeyProvisional, token); + return result with + { + Models = [..result.Models.Where(model => model.IsEmbeddingModel(this.Provider))] + }; } return ModelLoadResult.FromModels([]); } catch(Exception e) { - LOGGER.LogError($"Failed to load text models from self-hosted provider: {e.Message}"); + LOGGER.LogError($"Failed to load embedding models from self-hosted provider: {e.Message}"); return ModelLoadResult.Failure(ModelLoadFailureReason.UNKNOWN, e.Message); } } @@ -154,10 +162,22 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide new Provider.Model("loaded-model", TB("Model as configured by whisper.cpp")), ]); + // + // These two answer the models endpoint with everything they serve, and nothing in + // that answer says which of them listens. Asking what each model is made for is the + // only thing standing between this list and every chat and embedding model of the + // installation, which is what it used to hold. An engine running no speech model at + // all therefore offers nothing here, and says so, rather than offering models which + // would fail the moment audio reaches them. + // case Host.OLLAMA: case Host.VLLM: - return await this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, [], [], apiKeyProvisional, token); - + var result = await this.LoadModels(SecretStoreType.TRANSCRIPTION_PROVIDER, apiKeyProvisional, token); + return result with + { + Models = [..result.Models.Where(model => model.IsTranscriptionModel(this.Provider))] + }; + default: return ModelLoadResult.FromModels([]); } @@ -171,7 +191,21 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide #endregion - private async Task LoadModels(SecretStoreType storeType, string[] ignorePhrases, string[] filterPhrases, string? apiKeyProvisional, CancellationToken token) + /// + /// Everything the engine lists, in the order it listed it. + /// + /// + /// What kind of model each of these is stays unanswered here. It used to be answered right in + /// this method, by looking for the word "embed" in the name: the text models were the ones + /// without it, the embedding models the ones with it. That reading lost bge-m3 and all-minilm, + /// which say what they are through another word, and handed them to the chat list instead. The + /// callers ask the shared model kind detection now, the way every other provider does. + /// + /// Which key to send along. + /// A key from a dialog which has not stored it yet. + /// The cancellation token. + /// The models the engine named, unsorted and unfiltered. + private async Task LoadModels(SecretStoreType storeType, string? apiKeyProvisional, CancellationToken token) { var secretKey = await this.GetModelLoadingSecretKey(storeType, apiKeyProvisional, isTryingSecret: true); @@ -211,10 +245,8 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide // ListedModels.Shared.Report(this.ConfiguredProviderId, ListingsOf(models)); - return SuccessfulModelLoadResult(models. - Where(model => !string.IsNullOrWhiteSpace(model.Id) && - !ignorePhrases.Any(ignorePhrase => model.Id.Contains(ignorePhrase, StringComparison.InvariantCulture)) && - filterPhrases.All( filter => model.Id.Contains(filter, StringComparison.InvariantCulture))) + return SuccessfulModelLoadResult(models + .Where(model => !string.IsNullOrWhiteSpace(model.Id)) .Select(n => new Provider.Model(n.Id, null))); } catch (Exception e) when (this.IsTimeoutException(e, token)) @@ -229,7 +261,7 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide if (host is not Host.LLAMA_CPP || !chatModel.IsSystemModel) return chatModel; - var modelLoadResult = await this.LoadLlamaCppTextModels(["embed"], [], null, token); + var modelLoadResult = await this.LoadLlamaCppTextModels(null, token); if (!modelLoadResult.Success) return chatModel; @@ -266,7 +298,7 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide return chatModel; } - private async Task LoadLlamaCppTextModels(string[] ignorePhrases, string[] filterPhrases, string? apiKeyProvisional, CancellationToken token) + private async Task LoadLlamaCppTextModels(string? apiKeyProvisional, CancellationToken token) { var secretKey = await this.GetModelLoadingSecretKey(SecretStoreType.LLM_PROVIDER, apiKeyProvisional, true); @@ -298,7 +330,7 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide return LlamaCppLegacyModelResult(); var models = responseModels - .Where(model => IsMatchingLlamaCppTextModel(model, ignorePhrases, filterPhrases)) + .Where(this.IsMatchingLlamaCppTextModel) .Select(model => new Provider.Model(model.Id, null)) .ToList(); @@ -330,15 +362,23 @@ public sealed class ProviderSelfHosted(Host host, string hostname) : BaseProvide /// One listing per model, which says nothing for the models the engine was silent about. private static IEnumerable ListingsOf(IEnumerable models) => models.Select(model => ModelListing.For(model.Id, model.ContextWindowTokens)); - private static bool IsMatchingLlamaCppTextModel(Model model, string[] ignorePhrases, string[] filterPhrases) + /// + /// Whether this is a model somebody can chat with, as far as llama.cpp and the rules say. + /// + /// + /// Two sources, and both have to agree. What a model is made for comes from the shared rules, + /// the same answer the other engines get. What the running build of it puts out comes from + /// llama.cpp itself, which states the modalities on this route: an engine serving a model that + /// answers in something other than text knows that before any rule about the name could. + /// + /// The model as llama.cpp listed it. + /// True when both agree that it answers a chat in text. + private bool IsMatchingLlamaCppTextModel(Model model) { if (string.IsNullOrWhiteSpace(model.Id)) return false; - if (ignorePhrases.Any(ignorePhrase => model.Id.Contains(ignorePhrase, StringComparison.InvariantCultureIgnoreCase))) - return false; - - if (!filterPhrases.All(filter => model.Id.Contains(filter, StringComparison.InvariantCultureIgnoreCase))) + if (!new Provider.Model(model.Id, null).IsChatModel(this.Provider)) return false; var outputModalities = model.Architecture?.OutputModalities; diff --git a/app/Tests/Provider/SelfHostedModelListTests.cs b/app/Tests/Provider/SelfHostedModelListTests.cs new file mode 100644 index 00000000..21cd4989 --- /dev/null +++ b/app/Tests/Provider/SelfHostedModelListTests.cs @@ -0,0 +1,135 @@ +using System.Text.Json; + +using AIStudio.Provider; +using AIStudio.Settings; + +using SelfHostedModelsResponse = AIStudio.Provider.SelfHosted.ModelsResponse; + +namespace AIStudio.Tests.Provider; + +/// +/// Checks how the models of somebody's own server are sorted into the three lists they are offered in. +/// +/// +/// The engines answer one route with everything they serve, and that answer says nothing about what +/// any of it is made for: an ID, the word "model", and at Ollama a timestamp. Which list a model +/// ends up in is therefore decided afterwards, and for a long time it was decided by looking for +/// the word "embed" in the name -- the chat list was everything without it, the embedding list +/// everything with it, and the transcription list was not filtered at all. +/// +/// The body below is the real answer of a local Ollama, copied off the route rather than written +/// from memory, and it holds the two names which that reading got wrong. What it costs is visible +/// in the assertions: an embedding model in the chat list is one somebody picks and then waits for +/// an answer which never comes. +/// +[TestFixture] +public sealed class SelfHostedModelListTests +{ + private static readonly JsonSerializerOptions AS_THE_PROVIDERS_READ_IT = new() + { + PropertyNamingPolicy = JsonNamingPolicy.SnakeCaseLower, + }; + + /// + /// What "GET /v1/models" answers on a local Ollama, shortened to the fields it sends. + /// + private const string WHAT_A_LOCAL_OLLAMA_ANSWERS = + """ + { + "object": "list", + "data": [ + { "id": "all-minilm:latest", "object": "model", "created": 1789809739, "owned_by": "library" }, + { "id": "bge-m3:latest", "object": "model", "created": 1789809702, "owned_by": "library" }, + { "id": "qwen3-embedding:0.6b", "object": "model", "created": 1788699716, "owned_by": "library" }, + { "id": "qwen3-embedding:4b", "object": "model", "created": 1788699586, "owned_by": "library" }, + { "id": "qwen3-embedding:latest", "object": "model", "created": 1788699253, "owned_by": "library" }, + { "id": "qwen3.8:latest", "object": "model", "created": 1788698492, "owned_by": "library" }, + { "id": "gpt-oss:latest", "object": "model", "created": 1756645805, "owned_by": "library" } + ] + } + """; + + private static IReadOnlyList TheModelsTheEngineListed() + { + var response = JsonSerializer.Deserialize(WHAT_A_LOCAL_OLLAMA_ANSWERS, AS_THE_PROVIDERS_READ_IT); + Assert.That(response.Data, Is.Not.Null, "The answer has to be readable before anything can be sorted out of it."); + + return response.Data! + .Where(model => !string.IsNullOrWhiteSpace(model.Id)) + .Select(model => new Model(model.Id, null)) + .ToList(); + } + + [Test] + public void TheChatListHoldsWhatSomebodyCanTalkTo() + { + var chatModels = TheModelsTheEngineListed() + .Where(model => model.IsChatModel(LLMProviders.SELF_HOSTED)) + .Select(model => model.Id) + .ToList(); + + Assert.That(chatModels, Is.EquivalentTo(new[] { "qwen3.8:latest", "gpt-oss:latest" })); + } + + [Test] + public void TheEmbeddingListHoldsTheModelsWhichSayNothingAboutEmbedding() + { + var embeddingModels = TheModelsTheEngineListed() + .Where(model => model.IsEmbeddingModel(LLMProviders.SELF_HOSTED)) + .Select(model => model.Id) + .ToList(); + + Assert.Multiple(() => + { + Assert.That(embeddingModels, Does.Contain("bge-m3:latest"), "Named after the family which built it, with no word about what it does."); + Assert.That(embeddingModels, Does.Contain("all-minilm:latest"), "The same, and without the organization which used to be the only marker."); + Assert.That(embeddingModels, Has.Count.EqualTo(5), "The three Qwen embedding tags belong here as well, and nothing else does."); + }); + } + + [Test] + public void NothingIsInTwoListsAtOnce() + { + // + // The two lists were cut from one name with one word, so a model could only ever be in one + // of them. They are cut by two questions now, and two questions can both say yes. + // + var models = TheModelsTheEngineListed(); + var inBothLists = models + .Where(model => model.IsChatModel(LLMProviders.SELF_HOSTED) && model.IsEmbeddingModel(LLMProviders.SELF_HOSTED)) + .Select(model => model.Id) + .ToList(); + + Assert.That(inBothLists, Is.Empty); + } + + [Test] + public void AnEngineWithoutASpeechModelOffersNoneForTranscription() + { + // + // Ollama serves no speech-to-text model of its own, and the list said otherwise: it was + // handed through unfiltered, so all seven of these stood there to be picked. + // + var transcriptionModels = TheModelsTheEngineListed() + .Where(model => model.IsTranscriptionModel(LLMProviders.SELF_HOSTED)) + .ToList(); + + Assert.That(transcriptionModels, Is.Empty); + } + + [Test] + public void ASpeechModelOnSuchAServerIsOfferedForTranscription() + { + // + // The other half of the one above: the empty list has to come from there being no speech + // model, not from the question never saying yes on this provider. + // + var models = new[] { "whisper-large-v3", "faster-whisper-large-v3", "canary-1b-flash" } + .Select(id => new Model(id, null)) + .Where(model => model.IsTranscriptionModel(LLMProviders.SELF_HOSTED)) + .Select(model => model.Id) + .ToList(); + + Assert.That(models, Has.Count.EqualTo(3)); + } +} \ No newline at end of file