mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-09-28 21:43:38 +00:00
Resolved 29 conflicting files. The notable decisions: Confidence: main's tool-calling gate (RequiredProviderConfidence) and this branch's local-RAG gate (DataConfidenceLevel) turned out to be the same rule on the same axis, so they are now one field. Both tool results and data sources raise it through RequireProviderConfidence(). The gate checks the level strictly and no longer exempts providers trusted by configuration: TrustedProviderIds is documented as applying to data-source security checks only, and organizations set confidence through DataConfidence .CustomConfidenceScheme instead. The security axis (DataSecurity, ERI, IsTrustedForDataSourceSecurityChecks) is unchanged. Provider creation: main's CreateProvider signature won (hfEndpointKind, capabilityOverrides, no model parameter); tokenizerPath was added to it and is set for every provider, including the new Hetzner, IONOS and LiteLLM. Provider and EmbeddingProvider combine the record parameters, Lua parsing and Lua serialization of both sides. File types: main's hierarchy (ODT leaf, WORD parent, PowerPoint without the legacy .ppt, TABULAR instead of DELIMITED_TABLE) plus this branch's SPREADSHEET parent with ODS and the xlsm/xlsb/xla/xlam extensions, which the runtime already reads. Both sides had added a conflicting HTML filter; the reading family keeps the name, and the export path uses a narrow HTML_DOCUMENT, following the existing LATEX/TEX split. Runtime: main's file_data.rs is the base, including the prompt-injection sanitizer and the extraction routes. Token counting and chunk segmentation moved into take_released, so they act on the text the filter has released rather than on text it is still holding. A failed count is logged and left out instead of ending the extraction, because the app counts such a segment itself. Data sources: the participating-provider checks of this branch are kept, and main's GetAllowedDataSources overload now builds on them. DirectChatService resolves the launched chat's data source options before the check, so filter and chat see the same options. .NET and Rust both build clean; I18N regenerated to 4060 keys.
193 lines
7.0 KiB
C#
193 lines
7.0 KiB
C#
using AIStudio.Provider;
|
|
using AIStudio.Provider.HuggingFace;
|
|
using AIStudio.Tools.PluginSystem;
|
|
|
|
using Host = AIStudio.Provider.SelfHosted.Host;
|
|
|
|
namespace AIStudio.Tools.Validation;
|
|
|
|
public sealed class ProviderValidation
|
|
{
|
|
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(ProviderValidation).Namespace, nameof(ProviderValidation));
|
|
|
|
public Func<LLMProviders> GetProvider { get; init; } = () => LLMProviders.NONE;
|
|
|
|
public Func<string> GetAPIKeyStorageIssue { get; init; } = () => string.Empty;
|
|
|
|
public Func<string> GetPreviousInstanceName { get; init; } = () => string.Empty;
|
|
|
|
public Func<IEnumerable<string>> GetUsedInstanceNames { get; init; } = () => [];
|
|
|
|
public Func<Host> GetHost { get; init; } = () => Host.NONE;
|
|
|
|
public Func<bool> IsModelProvidedManually { get; init; } = () => false;
|
|
|
|
public Func<string> GetCustomTokenizerValidationIssue { get; init; } = () => string.Empty;
|
|
public Func<bool> IsModelSelectionHidden { get; init; } = () => false;
|
|
|
|
public string? ValidatingHostname(string hostname)
|
|
{
|
|
//
|
|
// Every provider for which IsHostnameNeeded is true must be validated here. Otherwise,
|
|
// the dialog shows a hostname field which nobody checks, and the provider silently ends
|
|
// up as a NoProvider later on, because its base URI cannot be built:
|
|
//
|
|
if(this.GetProvider() is not (LLMProviders.SELF_HOSTED or LLMProviders.LITE_LLM))
|
|
return null;
|
|
|
|
if(string.IsNullOrWhiteSpace(hostname))
|
|
return TB("Please enter a hostname, e.g., http://localhost:1234");
|
|
|
|
if(!hostname.StartsWith("http://", StringComparison.InvariantCultureIgnoreCase) && !hostname.StartsWith("https://", StringComparison.InvariantCultureIgnoreCase))
|
|
return TB("The hostname must start with either http:// or https://");
|
|
|
|
if(!Uri.TryCreate(hostname, UriKind.Absolute, out _))
|
|
return TB("The hostname is not a valid HTTP(S) URL.");
|
|
|
|
return null;
|
|
}
|
|
|
|
public string? ValidatingAPIKey(string apiKey)
|
|
{
|
|
if(this.GetProvider() is LLMProviders.SELF_HOSTED)
|
|
return null;
|
|
|
|
var apiKeyStorageIssue = this.GetAPIKeyStorageIssue();
|
|
if(!string.IsNullOrWhiteSpace(apiKeyStorageIssue))
|
|
return apiKeyStorageIssue;
|
|
|
|
if(string.IsNullOrWhiteSpace(apiKey))
|
|
return TB("Please enter an API key.");
|
|
|
|
return null;
|
|
}
|
|
|
|
public string? ValidatingInstanceName(string instanceName)
|
|
{
|
|
if (string.IsNullOrWhiteSpace(instanceName))
|
|
return TB("Please enter an instance name.");
|
|
|
|
if (instanceName.Length > 40)
|
|
return TB("The instance name must not exceed 40 characters.");
|
|
|
|
// The instance name must be unique:
|
|
var lowerInstanceName = instanceName.ToLowerInvariant();
|
|
if (lowerInstanceName != this.GetPreviousInstanceName() && this.GetUsedInstanceNames().Contains(lowerInstanceName))
|
|
return TB("The instance name must be unique; the chosen name is already in use.");
|
|
|
|
return null;
|
|
}
|
|
|
|
public string? ValidatingModel(Model model)
|
|
{
|
|
// For NONE providers, no validation is needed:
|
|
if (this.GetProvider() is LLMProviders.NONE)
|
|
return null;
|
|
|
|
// For self-hosted whisper.cpp, no model selection needed
|
|
// (model is loaded at startup):
|
|
if (this.GetProvider() is LLMProviders.SELF_HOSTED && this.GetHost() is Host.WHISPER_CPP)
|
|
return null;
|
|
|
|
// For legacy hosts without model selection, no selection validation is needed:
|
|
if (this.IsModelSelectionHidden())
|
|
return null;
|
|
|
|
// For manually entered models, this validation doesn't apply:
|
|
if (this.IsModelProvidedManually())
|
|
return null;
|
|
|
|
if (model == default)
|
|
return TB("Please select a model.");
|
|
|
|
return null;
|
|
}
|
|
|
|
public string? ValidatingProvider(LLMProviders llmProvider)
|
|
{
|
|
if (llmProvider == LLMProviders.NONE)
|
|
return TB("Please select a provider.");
|
|
|
|
return null;
|
|
}
|
|
|
|
public string? ValidatingHost(Host host)
|
|
{
|
|
if(this.GetProvider() is not LLMProviders.SELF_HOSTED)
|
|
return null;
|
|
|
|
if (host == Host.NONE)
|
|
return TB("Please select a host.");
|
|
|
|
return null;
|
|
}
|
|
|
|
public string? ValidatingHFInstanceProvider(HFInferenceProvider inferenceProvider)
|
|
{
|
|
if(this.GetProvider() is not LLMProviders.HUGGINGFACE)
|
|
return null;
|
|
|
|
if (!inferenceProvider.SupportsChat())
|
|
return TB("Please select an Hugging Face inference provider.");
|
|
|
|
return null;
|
|
}
|
|
|
|
public string? ValidatingCustomTokenizer(string _)
|
|
{
|
|
var issue = this.GetCustomTokenizerValidationIssue();
|
|
if (string.IsNullOrWhiteSpace(issue))
|
|
return null;
|
|
|
|
return issue;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Validates the Hugging Face inference provider chosen for embeddings.
|
|
/// </summary>
|
|
/// <remarks>
|
|
/// Far fewer providers create embeddings for us than serve chat models, so a selection which is
|
|
/// fine for chatting may not be for embeddings. A provider configured before the choice narrowed
|
|
/// is no longer among the options, which would leave the user with an empty field and no reason
|
|
/// given.
|
|
/// </remarks>
|
|
/// <param name="inferenceProvider">The inference provider to validate.</param>
|
|
/// <returns>The message to show, or null when the selection is fine.</returns>
|
|
public string? ValidatingHFInstanceProviderForEmbeddings(HFInferenceProvider inferenceProvider)
|
|
{
|
|
if(this.GetProvider() is not LLMProviders.HUGGINGFACE)
|
|
return null;
|
|
|
|
if (inferenceProvider is HFInferenceProvider.NONE)
|
|
return TB("Please select an Hugging Face inference provider.");
|
|
|
|
if (!inferenceProvider.SupportsEmbeddings())
|
|
return TB("This Hugging Face inference provider does not create embeddings. Please select another one.");
|
|
|
|
return null;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Validates the Hugging Face inference provider chosen for transcription.
|
|
/// </summary>
|
|
/// <remarks>
|
|
/// As with embeddings, only some of the inference providers transcribe audio for us, so the
|
|
/// choice is narrower than it is for chatting.
|
|
/// </remarks>
|
|
/// <param name="inferenceProvider">The inference provider to validate.</param>
|
|
/// <returns>The message to show, or null when the selection is fine.</returns>
|
|
public string? ValidatingHFInstanceProviderForTranscription(HFInferenceProvider inferenceProvider)
|
|
{
|
|
if(this.GetProvider() is not LLMProviders.HUGGINGFACE)
|
|
return null;
|
|
|
|
if (inferenceProvider is HFInferenceProvider.NONE)
|
|
return TB("Please select an Hugging Face inference provider.");
|
|
|
|
if (!inferenceProvider.SupportsTranscription())
|
|
return TB("This Hugging Face inference provider does not transcribe audio. Please select another one.");
|
|
|
|
return null;
|
|
}
|
|
}
|