Rebuilt how AI Studio knows what a model can do (#960)
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions

This commit is contained in:
Thorsten Sommer authored and GitHub committed 2026-09-13 14:17:25 +02:00
1 parent d21e09dd1e
commit d85b4e71b6
287 files changed
+18341 -3677

No files matched your search

@@ -1,17 +1,38 @@
using System.Globalization;
namespace AIStudio.Tools.PluginSystem;
public class I18N : ILang
{
public static readonly I18N I = new();
private static readonly ILogger<I18N> LOG = Program.LOGGER_FACTORY.CreateLogger<I18N>();
private ILanguagePlugin? language;
private I18N()
{
}
public static void Init(ILanguagePlugin language) => I.language = language;
/// <summary>
/// How the language in use writes its numbers, or the invariant culture while none is loaded.
/// </summary>
/// <remarks>
/// A number standing inside a translated sentence has to be written the way that language
/// writes numbers. AI Studio's language is chosen in its own settings and never moves the
/// thread's culture along with it, so a number formatted from the thread comes out with English
/// separators inside a German sentence. It lives here because it is the same decision as the
/// texts: whoever picked the language picked how its numbers look.
///
/// Components which already hold the active plugin may keep deriving it themselves. This is for
/// the code which has no plugin to ask -- a provider building an error message, say.
/// </remarks>
public CultureInfo Culture { get; private set; } = CultureInfo.InvariantCulture;
public static void Init(ILanguagePlugin language)
{
I.language = language;
I.Culture = CommonTools.DeriveActiveCultureOrInvariant(language.IETFTag);
}
#region Implementation of ILang
@@ -0,0 +1,19 @@
namespace AIStudio.Tools.PluginSystem;
/// <summary>
/// A plugin which contributes live content, and therefore takes part in deciding a collision.
/// </summary>
/// <remarks>
/// Two plugins may well say something about the same thing. Which of them is heard is decided the
/// same way for every kind of content: a plugin acting on behalf of the organization wins, and
/// among plugins of the same origin the declared priority does. Where the plugin was stored is
/// known from its path; what it declared has to come from the plugin itself, which is all this
/// interface is for.
/// </remarks>
public interface ILivePluginContentSource
{
/// <summary>
/// The priority this plugin declares. Zero when it declares none.
/// </summary>
public int Priority { get; }
}
@@ -9,7 +9,7 @@ using Lua;
namespace AIStudio.Tools.PluginSystem;
public sealed class PluginConfiguration(bool isInternal, LuaState state, PluginType type) : PluginBase(isInternal, state, type)
public sealed class PluginConfiguration(bool isInternal, LuaState state, PluginType type) : PluginBase(isInternal, state, type), ILivePluginContentSource
{
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(PluginConfiguration).Namespace, nameof(PluginConfiguration));
private static SettingsManager SettingsManagerAccess => Program.SERVICE_PROVIDER.GetRequiredService<SettingsManager>();
@@ -427,7 +427,12 @@ public static partial class PluginFactory
var assistantPlugin = new PluginAssistants(isInternal, state, type);
assistantPlugin.TryLoad();
return assistantPlugin;
case PluginType.MODEL:
var modelPlugin = new PluginModels(isInternal, state, type);
modelPlugin.TryLoad();
return modelPlugin;
default:
return new NoPlugin("This plugin type is not supported yet. Please try again with a future version of AI Studio.");
}
@@ -1,3 +1,5 @@
using AIStudio.Models.Registry;
namespace AIStudio.Tools.PluginSystem;
public static partial class PluginFactory
@@ -64,6 +66,7 @@ public static partial class PluginFactory
// declare an ID which differs from its directory name, and a single directory may even hold
// several plugins:
//
var unloadedAModelPlugin = false;
foreach (var plugin in AVAILABLE_PLUGINS.Where(plugin => IsPathInside(configurationDirectory, plugin.LocalPath)).ToList())
{
AVAILABLE_PLUGINS.Remove(plugin);
@@ -71,6 +74,7 @@ public static partial class PluginFactory
if (RUNNING_PLUGINS.FirstOrDefault(runningPlugin => runningPlugin.Id == plugin.Id) is { } runningPluginToRemove)
{
RUNNING_PLUGINS.Remove(runningPluginToRemove);
unloadedAModelPlugin |= runningPluginToRemove is PluginModels;
// The plugin is unloaded, so its Lua runtime is of no use anymore:
runningPluginToRemove.Dispose();
@@ -79,6 +83,15 @@ public static partial class PluginFactory
LOG.LogInformation("Unloaded the plugin '{PluginName}' ({PluginId}). Reason: {Reason}.", plugin.Name, plugin.Id, reason);
}
//
// This clean-up runs after the plugins were started, and nothing starts them again
// afterwards. A model plugin whose configuration is gone would otherwise go on describing
// models until the next restart, which is the one thing withdrawing a configuration has to
// stop:
//
if (unloadedAModelPlugin)
ModelRegistry.Shared.Declare(GetModelDeclarations());
if (!Directory.Exists(configurationDirectory))
return;
@@ -1,4 +1,5 @@
using System.Text;
using AIStudio.Models.Registry;
using AIStudio.Settings;
using AIStudio.Settings.DataModel;
using AIStudio.Tools.PluginSystem.Assistants;
@@ -92,7 +93,13 @@ public static partial class PluginFactory
try
{
if (availablePlugin.IsInternal || SettingsManagerAccess.IsPluginEnabled(availablePlugin) || availablePlugin.Type == PluginType.CONFIGURATION || availablePlugin.Type == PluginType.ASSISTANT)
//
// A model plugin runs like a configuration plugin, without anybody switching it on:
// it describes models an organization deployed it to describe, and a description
// somebody has to enable first would leave half the installations answering
// differently from the other half for no reason anyone could see.
//
if (availablePlugin.IsInternal || SettingsManagerAccess.IsPluginEnabled(availablePlugin) || availablePlugin.Type is PluginType.CONFIGURATION or PluginType.ASSISTANT or PluginType.MODEL)
if(await Start(availablePlugin, cancellationToken) is { IsValid: true } plugin)
{
if (plugin is PluginConfiguration configPlugin)
@@ -108,7 +115,15 @@ public static partial class PluginFactory
}
LogAssistantPluginStartupState();
//
// Hand what the model plugins declare to the registry before anything is told that the
// plugins are up. Whoever reacts to that message may ask about a model right away, and the
// registry keeps the answers it gives: an answer handed out before the declarations arrived
// would be the answer everybody gets until the next reload.
//
ModelRegistry.Shared.Declare(GetModelDeclarations());
// Inform all components that the plugins have been reloaded or started:
await MessageBus.INSTANCE.SendMessage<bool>(null, Event.PLUGINS_RELOADED);
return configObjects;
@@ -1,3 +1,4 @@
using AIStudio.Models.Plugins;
using AIStudio.Settings;
using AIStudio.Settings.DataModel;
@@ -437,37 +438,53 @@ public static partial class PluginFactory
public static IReadOnlyList<DataMandatoryInfo> GetMandatoryInfos()
{
return ResolveLivePluginContent<DataMandatoryInfo>("mandatory info", plugin => plugin.MandatoryInfos).ToList();
return ResolveLivePluginContent<PluginConfiguration, DataMandatoryInfo>("mandatory info", plugin => plugin.MandatoryInfos).ToList();
}
public static IReadOnlyList<DataIntroduction> GetIntroductions()
{
return ResolveLivePluginContent<DataIntroduction>("introduction", plugin => plugin.Introductions)
return ResolveLivePluginContent<PluginConfiguration, DataIntroduction>("introduction", plugin => plugin.Introductions)
.OrderBy(introduction => introduction.Index)
.ThenBy(introduction => introduction.Id, StringComparer.Ordinal)
.ToList();
}
/// <summary>
/// Collects live content from all running configuration plugins, so that each content ID appears exactly once.
/// Collects what the running model plugins declare about models.
/// </summary>
/// <remarks>
/// The IDs of live content are chosen by whoever writes the configuration, so two configuration
/// plugins may use the same ID. We resolve such a collision the same way a collision on a setting
/// is resolved: a configuration which acts on behalf of the organization wins, so nobody can push
/// aside what an organization deployed. Among configurations of the same origin, the declared
/// priority decides, and when even that is equal, the plugin which started later wins.<br/><br/>
/// A declaration is identified by its pattern, so two plugins claiming exactly the same model
/// names are a collision like any other and are settled the same way. Two plugins describing
/// different models never meet, and both are heard.
/// </remarks>
/// <returns>The declarations of all model plugins, with every pattern resolved to one winner.</returns>
public static IReadOnlyList<ModelDeclaration> GetModelDeclarations()
{
return ResolveLivePluginContent<PluginModels, ModelDeclaration>("model declaration", plugin => plugin.Declarations).ToList();
}
/// <summary>
/// Collects live content from all running plugins of one kind, so that each content ID appears exactly once.
/// </summary>
/// <remarks>
/// The IDs of live content are chosen by whoever writes the plugin, so two plugins may use the
/// same ID. We resolve such a collision the same way a collision on a setting is resolved: a
/// plugin which acts on behalf of the organization wins, so nobody can push aside what an
/// organization deployed. Among plugins of the same origin, the declared priority decides, and
/// when even that is equal, the plugin which started later wins.<br/><br/>
/// Duplicates are not merely a cosmetic problem: the home page keys its panels by the introduction
/// ID, and the acceptance of a mandatory info is stored per ID as well.
/// ID, the acceptance of a mandatory info is stored per ID as well, and two model declarations
/// claiming the same names would tie in the matching engine, which only a person can settle.
/// </remarks>
/// <param name="contentKind">The kind of content, used to report a collision in the log.</param>
/// <param name="selector">Selects the content of one configuration plugin.</param>
/// <param name="selector">Selects the content of one plugin.</param>
/// <typeparam name="TPlugin">The kind of plugin providing the content.</typeparam>
/// <typeparam name="T">The type of the live plugin content.</typeparam>
/// <returns>The content of all configuration plugins, with every ID resolved to one winner.</returns>
private static IEnumerable<T> ResolveLivePluginContent<T>(string contentKind, Func<PluginConfiguration, IEnumerable<T>> selector) where T : ILivePluginContent
/// <returns>The content of all those plugins, with every ID resolved to one winner.</returns>
private static IEnumerable<T> ResolveLivePluginContent<TPlugin, T>(string contentKind, Func<TPlugin, IEnumerable<T>> selector) where TPlugin : PluginBase, ILivePluginContentSource where T : ILivePluginContent
{
var contentById = new Dictionary<string, (T Content, int Authority, int Priority)>(StringComparer.Ordinal);
foreach (var plugin in RUNNING_PLUGINS.OfType<PluginConfiguration>())
foreach (var plugin in RUNNING_PLUGINS.OfType<TPlugin>())
{
var authority = GetConfigurationAuthority(plugin.PluginPath);
foreach (var content in selector(plugin))
@@ -484,14 +501,14 @@ public static partial class PluginFactory
var ignoredPluginId = isTakingOver ? currentWinner.Content.EnterpriseConfigurationPluginId : content.EnterpriseConfigurationPluginId;
if (winnerPluginId == ignoredPluginId)
LOG.LogWarning($"The configuration plugin '{winnerPluginId}' defines the {contentKind} ID '{content.Id}' more than once. Using its last definition and ignoring the earlier one. Please use each ID only once.");
LOG.LogWarning($"The plugin '{winnerPluginId}' defines the {contentKind} ID '{content.Id}' more than once. Using its last definition and ignoring the earlier one. Please use each ID only once.");
else
{
var reason = isTakingOver
? DescribeConfigurationPrecedence(authority, plugin.Priority, currentWinner.Authority, currentWinner.Priority)
: DescribeConfigurationPrecedence(currentWinner.Authority, currentWinner.Priority, authority, plugin.Priority);
LOG.LogWarning($"Multiple configuration plugins define the {contentKind} ID '{content.Id}'. Using the one from the configuration plugin '{winnerPluginId}' and ignoring the one from the configuration plugin '{ignoredPluginId}', because {reason}.");
LOG.LogWarning($"Multiple plugins define the {contentKind} ID '{content.Id}'. Using the one from the plugin '{winnerPluginId}' and ignoring the one from the plugin '{ignoredPluginId}', because {reason}.");
}
if (!isTakingOver)
@@ -506,7 +523,7 @@ public static partial class PluginFactory
}
/// <summary>
/// Explains in one phrase why one configuration plugin won a collision against another.
/// Explains in one phrase why one plugin won a collision against another.
/// </summary>
/// <remarks>
/// Administrators read this in the log while they are testing their configuration. Naming the
@@ -515,11 +532,11 @@ public static partial class PluginFactory
private static string DescribeConfigurationPrecedence(int winnerAuthority, int winnerPriority, int ignoredAuthority, int ignoredPriority)
{
if (winnerAuthority != ignoredAuthority)
return "a configuration which acts on behalf of your organization takes precedence over a locally placed one";
return "a plugin which acts on behalf of your organization takes precedence over a locally placed one";
if (winnerPriority != ignoredPriority)
return $"it declares the higher priority ({winnerPriority} instead of {ignoredPriority})";
return $"both declare the same priority ({winnerPriority}), so the configuration plugin which started later wins";
return $"both declare the same priority ({winnerPriority}), so the plugin which started later wins";
}
}
@@ -0,0 +1,72 @@
using AIStudio.Models.Plugins;
using Lua;
namespace AIStudio.Tools.PluginSystem;
/// <summary>
/// A plugin which tells AI Studio about models it does not know, or knows wrongly.
/// </summary>
/// <remarks>
/// Organizations run models nobody outside them has ever heard of: their own fine-tunes, a model
/// behind an internal name, an engine an operator configured differently from the model card. Until
/// now the only way to tell AI Studio about those was the expert settings of each configured
/// provider, one person and one provider at a time.
///
/// A model plugin describes, and that is all it does. It names no endpoint, carries no key, runs no
/// code of its own and reaches nothing over the network, which is why it needs none of the checks an
/// assistant plugin goes through. Where it was deployed is what says how much it may claim, exactly
/// as for every other kind of plugin.
/// </remarks>
public sealed class PluginModels(bool isInternal, LuaState state, PluginType type) : PluginBase(isInternal, state, type), ILivePluginContentSource
{
private static readonly ILogger LOG = Program.LOGGER_FACTORY.CreateLogger(nameof(PluginModels));
private readonly List<ModelDeclaration> declarations = [];
/// <summary>
/// The models this plugin declares.
/// </summary>
public IReadOnlyList<ModelDeclaration> Declarations => this.declarations;
/// <inheritdoc />
public int Priority { get; } = ReadPriority(state);
/// <summary>
/// Reads the MODELS table of the plugin.
/// </summary>
/// <remarks>
/// An entry which cannot be read is reported and skipped, and the rest of the table still
/// counts. A single mistyped capability in the twentieth entry must not take the nineteen
/// working ones with it -- the plugin would then be silently doing nothing at all.
/// </remarks>
public void TryLoad()
{
if (!this.State.Environment["MODELS"].TryRead<LuaTable>(out var modelsTable))
{
this.PluginIssues.Add(TB("The table MODELS does not exist or is using an invalid syntax."));
return;
}
for (var i = 1; i <= modelsTable.ArrayLength; i++)
{
if (!modelsTable[i].TryRead<LuaTable>(out var modelTable))
{
LOG.LogWarning("The table 'MODELS' entry at index {Index} is not a valid table (model plugin id: {PluginId}).", i, this.Id);
continue;
}
if (ModelDeclaration.TryParse(i, modelTable, this.Id, this.Name, LOG, out var declaration))
this.declarations.Add(declaration);
else
LOG.LogWarning("The table 'MODELS' entry at index {Index} does not contain a valid model declaration and is ignored (model plugin id: {PluginId}).", i, this.Id);
}
if (this.declarations.Count is 0)
LOG.LogWarning("The model plugin '{PluginId}' declares no model AI Studio could read. It has no effect.", this.Id);
}
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(PluginModels).Namespace, nameof(PluginModels));
private static int ReadPriority(LuaState state) => state.Environment["PRIORITY"].TryRead<int>(out var priority) ? priority : 0;
}
@@ -8,4 +8,5 @@ public enum PluginType
ASSISTANT,
CONFIGURATION,
THEME,
MODEL,
}
@@ -10,7 +10,8 @@ public static class PluginTypeExtensions
PluginType.ASSISTANT => TB("Assistant plugin"),
PluginType.CONFIGURATION => TB("Configuration plugin"),
PluginType.THEME => TB("Theme plugin"),
PluginType.MODEL => TB("Model plugin"),
_ => TB("Unknown plugin type"),
};
@@ -20,7 +21,8 @@ public static class PluginTypeExtensions
PluginType.ASSISTANT => "assistants",
PluginType.CONFIGURATION => "configurations",
PluginType.THEME => "themes",
PluginType.MODEL => "models",
_ => "unknown",
};
}
@@ -0,0 +1,258 @@
using System.Collections.Concurrent;
using System.Security.Cryptography;
using System.Text;
using AIStudio.Chat;
using AIStudio.Provider;
using AIStudio.Settings;
namespace AIStudio.Tools.Services;
//
// Inside the namespace on purpose. A "Provider" written here would otherwise be the namespace
// AIStudio.Provider, which every namespace below AIStudio sees before it sees a file's aliases.
//
using Provider = AIStudio.Settings.Provider;
/// <summary>
/// Counts what a conversation takes out of a model's context window.
/// </summary>
/// <remarks>
/// Asked from the chat while somebody types, so what it must not do is as important as what it
/// does. Every document is read and measured once and then remembered, because extracting a
/// thousand-page PDF on each keystroke would be unusable. The conversation so far is remembered the
/// same way, so typing measures the sentence being typed rather than the whole chat again.
///
/// The numbers are estimates and are shown as such. Unless somebody configured the model's own
/// tokenizer for their provider, the built-in one does the counting, and two tokenizers disagree by
/// a few percent on prose and by more than that on code.
/// </remarks>
public sealed class ConversationTokenCounter(RustService rustService, ILogger<ConversationTokenCounter> logger)
{
/// <summary>
/// How much text goes into one counting request.
/// </summary>
/// <remarks>
/// The same bound the rest of the app uses when it hands text to the tokenizer. Longer
/// conversations are counted in several pieces and added up, which costs a handful of special
/// tokens per piece -- a rounding error against a window of hundreds of thousands.
/// </remarks>
private const int CHUNK_SIZE = RustService.MAX_TOKEN_COUNT_REQUEST_TEXT_LENGTH;
/// <summary>
/// What separates the parts of a key.
/// </summary>
/// <remarks>
/// A character which cannot occur in a path, a tokenizer name or a hash, so that no two
/// different keys can be spelled the same way by accident. Written as an escape rather than as
/// the character itself: a source file carrying a raw zero byte is a binary file as far as Git
/// is concerned, and stops being reviewable.
/// </remarks>
private const string KEY_SEPARATOR = "\0";
private readonly ConcurrentDictionary<string, int> counted = new(StringComparer.Ordinal);
/// <summary>
/// What the texts which were still being written cost during the previous count.
/// </summary>
/// <remarks>
/// One run's worth, replaced by the next -- so at most the draft and the answer being streamed
/// stand in here. It exists for the case where nothing about them changed: a draft somebody left
/// standing while they think would otherwise be measured again on every heartbeat, and that is a
/// call to the tokenizer for an answer we already have.
/// </remarks>
private IReadOnlyDictionary<string, int> stillGrowing = new Dictionary<string, int>(StringComparer.Ordinal);
/// <summary>
/// Counts what the next request would carry.
/// </summary>
/// <param name="provider">The configured provider, which decides both the tokenizer and the window.</param>
/// <param name="parts">What the conversation would send, collected beforehand.</param>
/// <param name="token">Ends the counting when nobody needs the answer anymore.</param>
/// <returns>What the conversation costs, or that nothing could be counted.</returns>
public async Task<ConversationTokens> CountAsync(Provider provider, ConversationParts parts, CancellationToken token = default)
{
if (provider.UsedLLMProvider is LLMProviders.NONE)
return ConversationTokens.UNAVAILABLE;
var profile = provider.GetModelProfile();
var previouslyGrowing = this.stillGrowing;
var growing = new Dictionary<string, int>(StringComparer.Ordinal);
var tokens = 0;
try
{
foreach (var text in parts.Texts)
tokens += await this.CountTextAsync(provider, text, token);
//
// A text which is still being written is measured whole every time rather than by its
// increment. Two counts meet at a token boundary, and adding up the pieces drifts
// further from the truth with every three seconds an answer goes on.
//
foreach (var text in parts.GrowingTexts)
{
var key = Key(provider, text);
if (!previouslyGrowing.TryGetValue(key, out var known))
known = await this.MeasureAsync(provider, text, token);
growing[key] = known;
tokens += known;
}
foreach (var document in parts.Documents)
tokens += await this.CountDocumentAsync(provider, document, token);
}
catch (OperationCanceledException)
{
return ConversationTokens.UNAVAILABLE;
}
catch (Exception e)
{
logger.LogWarning(e, "Could not count the tokens of this conversation.");
return ConversationTokens.UNAVAILABLE;
}
this.stillGrowing = growing;
return new()
{
IsKnown = true,
Tokens = tokens,
IsEstimate = string.IsNullOrWhiteSpace(provider.TokenizerPath),
Window = profile.Context,
UncountedImages = parts.Images,
ImageLimits = profile.Images,
};
}
/// <summary>
/// Forgets everything counted so far.
/// </summary>
/// <remarks>
/// Needed when a file changed behind our back in a way its size and time do not show, which is
/// rare enough that nothing calls this today. It exists so that the cache has a way out other
/// than restarting the app.
/// </remarks>
public void Forget()
{
this.counted.Clear();
this.stillGrowing = new Dictionary<string, int>(StringComparer.Ordinal);
}
/// <summary>
/// Counts one text, remembering the answer under a fingerprint of it.
/// </summary>
/// <remarks>
/// This is what makes typing affordable. A message which was sent an hour ago says exactly what
/// it said then, and its tokens are the same number every time -- so the whole conversation is
/// measured once and every keystroke afterwards measures the sentence being written.
///
/// Keyed by a hash rather than by the text, because the key of the cache would otherwise be a
/// second copy of the whole conversation in memory. Hashing is not free, but it is two orders of
/// magnitude cheaper than tokenizing the same bytes, so the trade pays for itself on the first
/// repeat.
/// </remarks>
private async Task<int> CountTextAsync(Provider provider, string text, CancellationToken token)
{
if (string.IsNullOrWhiteSpace(text))
return 0;
var key = Key(provider, text);
if (this.counted.TryGetValue(key, out var known))
return known;
var tokens = await this.MeasureAsync(provider, text, token);
this.counted[key] = tokens;
return tokens;
}
/// <summary>
/// Measures one text without remembering the answer.
/// </summary>
/// <remarks>
/// A text longer than one request is split. Cutting between characters rather than between
/// words costs a token or two where the cut falls, which is the cheapest way to stay inside the
/// bound without pretending to know the language.
/// </remarks>
private async Task<int> MeasureAsync(Provider provider, string text, CancellationToken token)
{
var tokens = 0;
for (var start = 0; start < text.Length; start += CHUNK_SIZE)
tokens += await this.AskTokenizerAsync(provider, text.Substring(start, Math.Min(CHUNK_SIZE, text.Length - start)), token);
return tokens;
}
/// <summary>
/// Under which name one text is remembered.
/// </summary>
/// <remarks>
/// The tokenizer travels in the key: the same text counted for two providers is two different
/// numbers, and handing one of them to the other would be wrong in exactly the case somebody
/// switches providers to see whether their chat fits.
/// </remarks>
private static string Key(Provider provider, string text) => $"{provider.TokenizerPath}{KEY_SEPARATOR}{Fingerprint(text)}";
/// <summary>
/// A short, stable name for a piece of text.
/// </summary>
/// <remarks>
/// The length travels along with the hash. Two texts colliding on the hash and agreeing on
/// their length as well is not something which happens by accident, and nothing here is a
/// security decision: the worst a collision could do is show a number which is a few tokens off.
/// </remarks>
private static string Fingerprint(string text) => $"{text.Length}:{Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text)))}";
/// <summary>
/// Counts one document, reading it the first time and remembering it afterwards.
/// </summary>
/// <remarks>
/// The key carries the tokenizer as well as the file: the same document counted for two
/// providers is two different numbers, and handing one of them to the other would be wrong in
/// exactly the case somebody switches providers to see whether their chat fits.
/// </remarks>
private async Task<int> CountDocumentAsync(Provider provider, FileAttachment document, CancellationToken token)
{
var file = new FileInfo(document.FilePath);
if (!file.Exists)
return 0;
var key = $"{provider.TokenizerPath}{KEY_SEPARATOR}{file.FullName}{KEY_SEPARATOR}{file.Length}{KEY_SEPARATOR}{file.LastWriteTimeUtc.Ticks}";
if (this.counted.TryGetValue(key, out var known))
return known;
//
// Read without telling the user about filtered passages. Nothing here is sent anywhere: the
// text is measured and dropped, and the warning belongs to the moment the document actually
// travels -- where it is still given.
//
var extraction = await rustService.ReadArbitraryFileData(document.FilePath, int.MaxValue, reportPromptInjections: false, token: token);
if (!extraction.HasUsableContent)
{
//
// A document which cannot be read is not sent either, so it costs nothing. Remembered
// as zero so that a broken file is not read again on every keystroke.
//
logger.LogInformation("The attachment '{FilePath}' could not be read and is therefore counted as nothing.", document.FilePath);
this.counted[key] = 0;
return 0;
}
var tokens = await this.CountTextAsync(provider, extraction.Content, token);
this.counted[key] = tokens;
return tokens;
}
private async Task<int> AskTokenizerAsync(Provider provider, string text, CancellationToken token)
{
if (string.IsNullOrWhiteSpace(text))
return 0;
var response = await rustService.GetTokenCount(provider, text, token);
if (response is null || !response.Value.Success)
throw new InvalidOperationException($"The tokenizer did not answer: {response?.Message}");
return response.Value.TokenCount;
}
}
@@ -31,7 +31,13 @@ public sealed partial class RustService
/// already gone.
/// </param>
/// <returns>The result of reading the file.</returns>
public async Task<FileExtractionResult> ReadArbitraryFileData(string path, int maxChunks, bool extractImages = false, CancellationToken token = default)
/// <param name="reportPromptInjections">
/// Whether to tell the user about passages which were filtered out of the file. Pass false only
/// where the content is measured and thrown away again, such as counting the tokens of an
/// attachment: nothing leaves the app on that path, so there is nothing to warn about, and
/// reporting it there would warn a second time when the file is actually sent.
/// </param>
public async Task<FileExtractionResult> ReadArbitraryFileData(string path, int maxChunks, bool extractImages = false, bool reportPromptInjections = true, CancellationToken token = default)
{
//
// The runtime filters prompt injections while it streams the file. Doing it there rather
@@ -238,9 +244,12 @@ public sealed partial class RustService
//
// Reported from here rather than from the callers: every way of reading a file passes
// through this method, so this is the one place where no caller can forget it.
// through this method, so this is the one place where no caller can forget it. The
// filtering itself has already happened either way -- only the telling is skipped, and only
// where the content never leaves the app.
//
await guardService.ReportAsync(new(PromptInjectionSource.FileContent(path), promptInjectionFindings, promptInjectionRedactedCount));
if (reportPromptInjections)
await guardService.ReportAsync(new(PromptInjectionSource.FileContent(path), promptInjectionFindings, promptInjectionRedactedCount));
//
// Filtering does not change the outcome: the passages were removed and the document
@@ -13,12 +13,10 @@ public static class ToolCallingAvailabilityExtensions
if (provider == AIStudio.Settings.Provider.NONE || provider.UsedLLMProvider is LLMProviders.NONE)
return new(false, TB("Please select an LLM provider."));
var modelCapabilities = provider.GetModelCapabilities();
var supportsRequiredApis =
modelCapabilities.Contains(Capability.CHAT_COMPLETION_API) ||
modelCapabilities.Contains(Capability.RESPONSES_API);
var modelProfile = provider.GetModelProfile();
var supportsRequiredApis = modelProfile.HasAny(Capability.CHAT_COMPLETION_API | Capability.RESPONSES_API);
if (!supportsRequiredApis || !modelCapabilities.Contains(Capability.FUNCTION_CALLING))
if (!supportsRequiredApis || !modelProfile.Has(Capability.FUNCTION_CALLING))
return new(false, TB("Tool calling support is not enabled by default for this model, but you can enable this capability in the expert settings of the provider if you are sure the model supports it."));
return ToolCallingAvailability.Available();