diff --git a/app/MindWork AI Studio/Settings/ProviderExtensions.Alibaba.cs b/app/MindWork AI Studio/Settings/ProviderExtensions.Alibaba.cs index 86944157..6a26b0c8 100644 --- a/app/MindWork AI Studio/Settings/ProviderExtensions.Alibaba.cs +++ b/app/MindWork AI Studio/Settings/ProviderExtensions.Alibaba.cs @@ -11,8 +11,24 @@ public static partial class ProviderExtensions // Qwen models: if (modelName.StartsWith("qwen")) { - // Check for omni models: + // Check for omni models. Alibaba lists the Qwen3 and Qwen3.5 Omni series among the + // models which call functions; the older qwen-omni ones are not on that list, which + // is what the version check separates here: if (modelName.IndexOf("omni") is not -1) + { + if (modelName.StartsWith("qwen3")) + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, + Capability.AUDIO_INPUT, Capability.SPEECH_INPUT, + Capability.VIDEO_INPUT, + + Capability.TEXT_OUTPUT, Capability.SPEECH_OUTPUT, + + Capability.FUNCTION_CALLING, + Capability.CHAT_COMPLETION_API, + ]; + return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, @@ -20,10 +36,11 @@ public static partial class ProviderExtensions Capability.VIDEO_INPUT, Capability.TEXT_OUTPUT, Capability.SPEECH_OUTPUT, - + Capability.CHAT_COMPLETION_API, ]; - + } + // Check for Qwen 3.5: if(modelName.StartsWith("qwen3.5")) return @@ -47,6 +64,44 @@ public static partial class ProviderExtensions Capability.CHAT_COMPLETION_API, ]; + // Check for the Qwen 3.7 family. Thinking is optional here and switched on by + // default, except for the two preview snapshots, which do nothing else: + if(modelName.StartsWith("qwen3.7")) + { + if(modelName.IndexOf("-preview") is not -1 || + modelName.IndexOf("-2026-05-17") is not -1) + return + [ + Capability.TEXT_INPUT, + Capability.TEXT_OUTPUT, + + Capability.ALWAYS_REASONING, Capability.FUNCTION_CALLING, + Capability.CHAT_COMPLETION_API, + ]; + + // Vision arrived in the middle of the series. The rolling qwen3.7-max alias + // still answers as the text-only May snapshot, so only the June one may be + // told that it reads images and video: + if(modelName.IndexOf("-2026-06-08") is not -1) + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.VIDEO_INPUT, + Capability.TEXT_OUTPUT, + + Capability.REASONING_BY_DEFAULT, Capability.FUNCTION_CALLING, + Capability.CHAT_COMPLETION_API, + ]; + + return + [ + Capability.TEXT_INPUT, + Capability.TEXT_OUTPUT, + + Capability.REASONING_BY_DEFAULT, Capability.FUNCTION_CALLING, + Capability.CHAT_COMPLETION_API, + ]; + } + // Check for the Qwen 3.8 family: if(modelName.StartsWith("qwen3.8")) { @@ -84,15 +139,28 @@ public static partial class ProviderExtensions ]; } - // Check for the 3.0 VL models: + // Check for the VL models. Alibaba names the Qwen3-VL Plus and Flash series as + // function callers; the older qwen-vl models are absent from that list: if(modelName.IndexOf("-vl-") is not -1) + { + if(modelName.StartsWith("qwen3")) + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, + Capability.TEXT_OUTPUT, + + Capability.FUNCTION_CALLING, + Capability.CHAT_COMPLETION_API, + ]; + return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.TEXT_OUTPUT, - + Capability.CHAT_COMPLETION_API, ]; + } // Check for Qwen 3: if(modelName.StartsWith("qwen3")) diff --git a/app/MindWork AI Studio/Settings/ProviderExtensions.Google.cs b/app/MindWork AI Studio/Settings/ProviderExtensions.Google.cs index 451605b0..1073b7ae 100644 --- a/app/MindWork AI Studio/Settings/ProviderExtensions.Google.cs +++ b/app/MindWork AI Studio/Settings/ProviderExtensions.Google.cs @@ -10,10 +10,57 @@ public static partial class ProviderExtensions if (modelName.IndexOf("gemini-") is not -1) { + // + // Image generation models. They carry a version number like every other model and + // have to be asked about first, or gemini-3-pro-image would be read as a chat model + // of the 3.x line and be promised function calling. No image model of the family + // offers that; what they do offer, and the chat models do not, is writing images. + // + if (modelName.IndexOf("-image") is not -1) + { + // Of the image models, only the 3.1 Flash ones read video. They think about + // complex prompts, and, as with the 3.x chat models, thinking cannot be + // switched off: + if (modelName.IndexOf("gemini-3.1-flash") is not -1) + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.VIDEO_INPUT, + + Capability.TEXT_OUTPUT, Capability.IMAGE_OUTPUT, + + Capability.ALWAYS_REASONING, + Capability.CHAT_COMPLETION_API, + ]; + + // Every other Gemini 3 image model thinks as well, it just does not read video: + if (modelName.IndexOf("gemini-3") is not -1) + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, + + Capability.TEXT_OUTPUT, Capability.IMAGE_OUTPUT, + + Capability.ALWAYS_REASONING, + Capability.CHAT_COMPLETION_API, + ]; + + // The older image models, such as the 2.5 Flash one, do not think: + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, + + Capability.TEXT_OUTPUT, Capability.IMAGE_OUTPUT, + Capability.CHAT_COMPLETION_API, + ]; + } + // Chat-compatible Gemini 3.x reasoning models. We match the entire 3.x line // so that new releases are covered as well: they all reason, and the - // thinking level can only be lowered, never turned off. The two rolling - // aliases carry no version number and are listed separately: + // thinking level can only be lowered, never turned off. That holds for the + // Flash Lite models of this line too, which is what sets them apart from + // Gemini 2.5 Flash Lite below: there, thinking is off until it is asked for, + // while here the lowest level still thinks. The two rolling aliases carry no + // version number and are listed separately: if (modelName.IndexOf("gemini-3") is not -1 || modelName is "gemini-flash-latest" || modelName is "gemini-pro-latest") @@ -54,17 +101,6 @@ public static partial class ProviderExtensions Capability.CHAT_COMPLETION_API, ]; - // Image generation: - if(modelName.IndexOf("-2.0-flash-preview-image-") is not -1) - return - [ - Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.AUDIO_INPUT, - Capability.SPEECH_INPUT, Capability.VIDEO_INPUT, - - Capability.TEXT_OUTPUT, Capability.IMAGE_OUTPUT, - Capability.CHAT_COMPLETION_API, - ]; - // Realtime model: if(modelName.IndexOf("-2.0-flash-live-") is not -1) return @@ -78,17 +114,17 @@ public static partial class ProviderExtensions Capability.CHAT_COMPLETION_API, ]; - // The 2.0 flash models cannot call functions: - if(modelName.IndexOf("-2.0-flash-") is not -1) - return - [ - Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.AUDIO_INPUT, - Capability.SPEECH_INPUT, Capability.VIDEO_INPUT, - - Capability.TEXT_OUTPUT, - Capability.CHAT_COMPLETION_API, - ]; - + // + // There used to be a branch here which withheld function calling from the 2.0 Flash + // models. It said the wrong thing about them, and it only ever caught the dated IDs + // because it asked for a trailing hyphen: the plain gemini-2.0-flash alias walked + // past it and got a different answer than gemini-2.0-flash-001, which is the same + // model. Both questions are moot now, because Google shut the 2.0 Flash chat models + // down on 1 June 2026. Anything still asking for one of those names gets the default + // below. The live model above keeps its branch: it belongs to a different API whose + // retirement Google announces separately. + // + // The old 1.0 pro vision model: if(modelName.IndexOf("pro-vision") is not -1) return diff --git a/app/MindWork AI Studio/Settings/ProviderExtensions.OpenAI.cs b/app/MindWork AI Studio/Settings/ProviderExtensions.OpenAI.cs index eff57469..5f4b10e6 100644 --- a/app/MindWork AI Studio/Settings/ProviderExtensions.OpenAI.cs +++ b/app/MindWork AI Studio/Settings/ProviderExtensions.OpenAI.cs @@ -54,14 +54,6 @@ public static partial class ProviderExtensions Capability.CHAT_COMPLETION_API, ]; - if (modelName.StartsWith("chatgpt-4o-")) - return - [ - Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, - Capability.TEXT_OUTPUT, - Capability.RESPONSES_API, - ]; - if (modelName.StartsWith("o3-mini")) return [ @@ -116,8 +108,9 @@ public static partial class ProviderExtensions [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.TEXT_OUTPUT, - + Capability.FUNCTION_CALLING, Capability.ALWAYS_REASONING, + Capability.WEB_SEARCH, Capability.RESPONSES_API, ]; @@ -132,12 +125,19 @@ public static partial class ProviderExtensions Capability.RESPONSES_API, ]; + // + // None of the GPT-5 models writes images itself. They can ask for one through the + // image generation tool, which is a tool call like any other and produces a picture + // from a separate model. That is a different thing from an output modality, and we + // must not report it as one: the chat would then offer to receive images which never + // arrive. + // if(modelName is "gpt-5.1" || modelName.StartsWith("gpt-5.1-")) return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, - Capability.TEXT_OUTPUT, Capability.IMAGE_OUTPUT, - + Capability.TEXT_OUTPUT, + Capability.FUNCTION_CALLING, Capability.OPTIONAL_REASONING, Capability.WEB_SEARCH, Capability.RESPONSES_API, Capability.CHAT_COMPLETION_API, @@ -147,8 +147,8 @@ public static partial class ProviderExtensions return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, - Capability.TEXT_OUTPUT, Capability.IMAGE_OUTPUT, - + Capability.TEXT_OUTPUT, + Capability.FUNCTION_CALLING, Capability.OPTIONAL_REASONING, Capability.WEB_SEARCH, Capability.RESPONSES_API, Capability.CHAT_COMPLETION_API, @@ -180,7 +180,7 @@ public static partial class ProviderExtensions return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, - Capability.TEXT_OUTPUT, Capability.IMAGE_OUTPUT, + Capability.TEXT_OUTPUT, Capability.FUNCTION_CALLING, Capability.REASONING_BY_DEFAULT, Capability.WEB_SEARCH, @@ -197,7 +197,22 @@ public static partial class ProviderExtensions Capability.WEB_SEARCH, Capability.RESPONSES_API, Capability.CHAT_COMPLETION_API, ]; - + + // + // GPT-6 Astra. Unlike the 5.5 and 5.6 models, it reasons on every request: the effort + // reaches from low to max, and there is no setting which switches thinking off. + // + if(modelName is "gpt-6-astra" || modelName.StartsWith("gpt-6-astra-")) + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, + Capability.TEXT_OUTPUT, + + Capability.FUNCTION_CALLING, Capability.ALWAYS_REASONING, + Capability.WEB_SEARCH, + Capability.RESPONSES_API, Capability.CHAT_COMPLETION_API, + ]; + return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, diff --git a/app/MindWork AI Studio/Settings/ProviderExtensions.OpenSource.cs b/app/MindWork AI Studio/Settings/ProviderExtensions.OpenSource.cs index b0c2d472..441202b5 100644 --- a/app/MindWork AI Studio/Settings/ProviderExtensions.OpenSource.cs +++ b/app/MindWork AI Studio/Settings/ProviderExtensions.OpenSource.cs @@ -547,12 +547,27 @@ public static partial class ProviderExtensions Capability.CHAT_COMPLETION_API, ]; - // Grok 4 models take text, images, and video natively. Reasoning is always - // on, only the reasoning effort can be configured: + // One member of the 4.20 line answers without thinking, and it says so in its + // name. It has to be asked about before the general Grok 4 rule, which would + // otherwise claim the opposite of what the name states: + if(modelName.IndexOf("-non-reasoning") is not -1) + return + [ + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, + Capability.TEXT_OUTPUT, + + Capability.FUNCTION_CALLING, + Capability.CHAT_COMPLETION_API, + ]; + + // Grok 4 models take text and images. Reasoning is always on, only the + // reasoning effort can be configured. Video is not among their modalities: + // xAI serves audio, image, and video through models and APIs of their own, + // and the model pages of the 4.x line say "text, image" and nothing else: if(modelName.IndexOf("grok-4") is not -1) return [ - Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.VIDEO_INPUT, + Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, Capability.TEXT_OUTPUT, Capability.ALWAYS_REASONING, Capability.FUNCTION_CALLING, diff --git a/app/MindWork AI Studio/Settings/ProviderExtensions.Perplexity.cs b/app/MindWork AI Studio/Settings/ProviderExtensions.Perplexity.cs index 50510323..4665d240 100644 --- a/app/MindWork AI Studio/Settings/ProviderExtensions.Perplexity.cs +++ b/app/MindWork AI Studio/Settings/ProviderExtensions.Perplexity.cs @@ -8,29 +8,32 @@ public static partial class ProviderExtensions { var modelName = NormalizeModelId(model.Id).AsSpan(); + // + // No Sonar model writes images. What looked like it does is the option to have the + // answer come with images: those are pictures the search found on the pages it read, + // handed back as links, not something the model drew. + // if(modelName.IndexOf("reasoning") is not -1 || modelName.IndexOf("deep-research") is not -1) return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, - + Capability.TEXT_OUTPUT, - Capability.IMAGE_OUTPUT, - + Capability.ALWAYS_REASONING, Capability.WEB_SEARCH, Capability.CHAT_COMPLETION_API, ]; - + return [ Capability.TEXT_INPUT, Capability.MULTIPLE_IMAGE_INPUT, - + Capability.TEXT_OUTPUT, - Capability.IMAGE_OUTPUT, - + Capability.WEB_SEARCH, Capability.CHAT_COMPLETION_API, ];