mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-10 11:13:47 +00:00
Rebuilt how AI Studio knows what a model can do (#960)
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
This commit is contained in:
1 parent
d21e09dd1e
commit
d85b4e71b6
287 files changed
+18341
-3677
No files matched your search
@@ -146,10 +146,8 @@ public sealed record ChatThread
|
||||
/// </summary>
|
||||
public bool MayRunTools(SettingsManager settingsManager) => this.RuntimeToolsAreAssistantManaged || settingsManager.IsToolSelectionVisible(this.RuntimeComponent);
|
||||
|
||||
private bool allowProfile = true;
|
||||
|
||||
/// <summary>
|
||||
/// Prepares the system prompt for the chat thread.
|
||||
/// Prepares the system prompt for the chat thread, and remembers what it was built from.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The actual system prompt depends on the selected profile. If no profile is selected,
|
||||
@@ -161,7 +159,35 @@ public sealed record ChatThread
|
||||
/// <returns>The prepared system prompt.</returns>
|
||||
public string PrepareSystemPrompt(SettingsManager settingsManager, IEnumerable<ToolDefinition>? runnableToolDefinitions = null)
|
||||
{
|
||||
this.allowProfile = true;
|
||||
var prepared = this.BuildSystemPrompt(settingsManager, runnableToolDefinitions);
|
||||
|
||||
// We need a way to save the changed system prompt in our chat thread.
|
||||
// Otherwise, the chat thread will always tell us that it is using the
|
||||
// default system prompt:
|
||||
this.SystemPrompt = prepared.BasePrompt;
|
||||
LOGGER.LogInformation(prepared.Explanation);
|
||||
|
||||
return prepared.Text;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Works out the system prompt without changing anything about the thread.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Split off from the preparation above so that somebody can ask how long the next request
|
||||
/// would be. Counting the tokens of a conversation has to ask the same question the request
|
||||
/// asks -- a count against the prompt a person typed, rather than against the one a chat
|
||||
/// template, a data source, a profile and the tool policy make of it, is a number about a
|
||||
/// request which is never sent.
|
||||
///
|
||||
/// Nothing here writes to the thread and nothing logs, because this runs while somebody types.
|
||||
/// </remarks>
|
||||
/// <param name="settingsManager">The settings manager instance to use.</param>
|
||||
/// <param name="runnableToolDefinitions">The tools which may run in this thread. Null when the thread runs without tools.</param>
|
||||
/// <returns>The system prompt and what building it decided.</returns>
|
||||
public PreparedSystemPrompt BuildSystemPrompt(SettingsManager settingsManager, IEnumerable<ToolDefinition>? runnableToolDefinitions = null)
|
||||
{
|
||||
var allowProfile = true;
|
||||
|
||||
//
|
||||
// Use the information from the chat template, if provided. Otherwise, use the default system prompt
|
||||
@@ -186,18 +212,12 @@ public sealed record ChatThread
|
||||
else
|
||||
{
|
||||
logMessage = $"Using chat template '{chatTemplate.Name}' for chat thread '{this.Name}'.";
|
||||
this.allowProfile = chatTemplate.AllowProfileUsage;
|
||||
allowProfile = chatTemplate.AllowProfileUsage;
|
||||
systemPromptTextWithChatTemplate = chatTemplate.ToSystemPrompt();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// We need a way to save the changed system prompt in our chat thread.
|
||||
// Otherwise, the chat thread will always tell us that it is using the
|
||||
// default system prompt:
|
||||
this.SystemPrompt = systemPromptTextWithChatTemplate;
|
||||
LOGGER.LogInformation(logMessage);
|
||||
|
||||
//
|
||||
// Add augmented data, if available:
|
||||
@@ -214,18 +234,16 @@ public sealed record ChatThread
|
||||
false => systemPromptTextWithChatTemplate,
|
||||
};
|
||||
|
||||
if(isAugmentedDataAvailable)
|
||||
LOGGER.LogInformation("Augmented data is available for the chat thread.");
|
||||
else
|
||||
LOGGER.LogInformation("No augmented data is available for the chat thread.");
|
||||
|
||||
|
||||
logMessage = isAugmentedDataAvailable
|
||||
? $"{logMessage} Augmented data is available for the chat thread."
|
||||
: $"{logMessage} No augmented data is available for the chat thread.";
|
||||
|
||||
//
|
||||
// Add information from the profile if available and allowed:
|
||||
//
|
||||
string systemPromptText;
|
||||
logMessage = $"Using no profile for chat thread '{this.Name}'.";
|
||||
if (string.IsNullOrWhiteSpace(this.SelectedProfile) || !this.allowProfile)
|
||||
var profileNote = $"Using no profile for chat thread '{this.Name}'.";
|
||||
if (string.IsNullOrWhiteSpace(this.SelectedProfile) || !allowProfile)
|
||||
systemPromptText = systemPromptWithAugmentedData;
|
||||
else
|
||||
{
|
||||
@@ -242,7 +260,7 @@ public sealed record ChatThread
|
||||
systemPromptText = systemPromptWithAugmentedData;
|
||||
else
|
||||
{
|
||||
logMessage = $"Using profile '{profile.Name}' for chat thread '{this.Name}'.";
|
||||
profileNote = $"Using profile '{profile.Name}' for chat thread '{this.Name}'.";
|
||||
systemPromptText = $"""
|
||||
{systemPromptWithAugmentedData}
|
||||
|
||||
@@ -252,8 +270,6 @@ public sealed record ChatThread
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
LOGGER.LogInformation(logMessage);
|
||||
|
||||
var toolPolicy = ToolSelectionRules.BuildToolPolicyPrompt(runnableToolDefinitions ?? []);
|
||||
if (!string.IsNullOrWhiteSpace(toolPolicy))
|
||||
@@ -265,9 +281,10 @@ public sealed record ChatThread
|
||||
""";
|
||||
}
|
||||
|
||||
var explanation = $"{logMessage} {profileNote}";
|
||||
if(!this.IncludeDateTime)
|
||||
return systemPromptText;
|
||||
|
||||
return new(systemPromptText, systemPromptTextWithChatTemplate, allowProfile, explanation);
|
||||
|
||||
//
|
||||
// Prepend the current date and time to the system prompt:
|
||||
//
|
||||
@@ -278,11 +295,13 @@ public sealed record ChatThread
|
||||
$"Today is {nowUtc:dddd, MMMM d, yyyy h:mm tt} (UTC) and {nowLocal:dddd, MMMM d, yyyy h:mm tt} (local time)."
|
||||
);
|
||||
|
||||
return $"""
|
||||
{currentDateTime}
|
||||
var withDateTime = $"""
|
||||
{currentDateTime}
|
||||
|
||||
{systemPromptText}
|
||||
""";
|
||||
{systemPromptText}
|
||||
""";
|
||||
|
||||
return new(withDateTime, systemPromptTextWithChatTemplate, allowProfile, explanation);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
|
||||
@@ -0,0 +1,136 @@
|
||||
namespace AIStudio.Chat;
|
||||
|
||||
/// <summary>
|
||||
/// Everything a conversation would put into the next request, sorted by how it can be counted.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Collected here rather than while counting, so that what counts towards a token budget is one
|
||||
/// question with one answer which a test can ask. It follows what the message builder actually
|
||||
/// sends: the system prompt, the text of every block, and the attachments hanging off those
|
||||
/// blocks -- plus whatever is standing in the composer but has not been sent yet, because that is
|
||||
/// the part a person is deciding about while they look at the number.
|
||||
/// </remarks>
|
||||
public sealed record ConversationParts
|
||||
{
|
||||
/// <summary>
|
||||
/// A conversation with nothing in it.
|
||||
/// </summary>
|
||||
public static readonly ConversationParts NOTHING = new();
|
||||
|
||||
/// <summary>
|
||||
/// The texts which go into the request as they are.
|
||||
/// </summary>
|
||||
public IReadOnlyList<string> Texts { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// The texts which are still being written.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// They cost exactly what the others cost; what sets them apart is that they will never be seen
|
||||
/// again in this shape. The sentence somebody is typing changes with the next pause, and an
|
||||
/// answer being streamed is a different text three seconds later -- so remembering what they
|
||||
/// cost fills memory with answers nobody will ask for again.
|
||||
/// </remarks>
|
||||
public IReadOnlyList<string> GrowingTexts { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// The documents whose content is put into the request.
|
||||
/// </summary>
|
||||
public IReadOnlyList<FileAttachment> Documents { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// How many images travel along.
|
||||
/// </summary>
|
||||
public int Images { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Collects what a conversation would send.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Blocks without text are skipped, because the message builder skips them too: a block whose
|
||||
/// text is empty never becomes a message, whatever else hangs off it.
|
||||
/// </remarks>
|
||||
/// <param name="thread">The conversation so far, or null when there is none yet.</param>
|
||||
/// <param name="systemPrompt">
|
||||
/// The system prompt as it would be sent, which is not the one a person typed: a chat template
|
||||
/// may replace it, the retrieved data of a data source is appended to it, a profile adds a
|
||||
/// paragraph, and the tool policy adds another.
|
||||
/// </param>
|
||||
/// <param name="draft">What stands in the composer.</param>
|
||||
/// <param name="draftAttachments">What is attached to the composer.</param>
|
||||
/// <param name="imagesAreSent">Whether the model takes images at all. When it does not, none are sent.</param>
|
||||
/// <returns>The parts of the conversation.</returns>
|
||||
public static ConversationParts Of(ChatThread? thread, string systemPrompt, string draft, IEnumerable<FileAttachment>? draftAttachments, bool imagesAreSent)
|
||||
{
|
||||
var texts = new List<string>();
|
||||
var growing = new List<string>();
|
||||
var documents = new List<FileAttachment>();
|
||||
var images = 0;
|
||||
|
||||
if (!string.IsNullOrWhiteSpace(systemPrompt))
|
||||
texts.Add(systemPrompt);
|
||||
|
||||
if (thread is not null)
|
||||
{
|
||||
//
|
||||
// Blocks hidden from the user are counted like any other. They are hidden on the screen,
|
||||
// not in the request: the message builder sends them, so they take their tokens whether
|
||||
// or not anybody can see them.
|
||||
//
|
||||
foreach (var block in thread.Blocks)
|
||||
{
|
||||
if (block.ContentType is not ContentType.TEXT || block.Content is not ContentText text || string.IsNullOrWhiteSpace(text.Text))
|
||||
continue;
|
||||
|
||||
if (text.IsStreaming)
|
||||
growing.Add(text.Text);
|
||||
else
|
||||
texts.Add(text.Text);
|
||||
|
||||
Sort(text.FileAttachments, documents, ref images);
|
||||
}
|
||||
}
|
||||
|
||||
if (!string.IsNullOrWhiteSpace(draft))
|
||||
growing.Add(draft);
|
||||
|
||||
if (draftAttachments is not null)
|
||||
Sort(draftAttachments, documents, ref images);
|
||||
|
||||
return new()
|
||||
{
|
||||
Texts = texts,
|
||||
GrowingTexts = growing,
|
||||
Documents = documents,
|
||||
Images = imagesAreSent ? images : 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Puts attachments into the two groups they are counted in.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// An attachment whose file is gone is left out of both. It is not sent either: the message
|
||||
/// builder drops it and tells the person about it, so counting it would promise a request which
|
||||
/// is never made.
|
||||
/// </remarks>
|
||||
private static void Sort(IEnumerable<FileAttachment> attachments, List<FileAttachment> documents, ref int images)
|
||||
{
|
||||
foreach (var attachment in attachments)
|
||||
{
|
||||
if (!attachment.Exists)
|
||||
continue;
|
||||
|
||||
switch (attachment.Type)
|
||||
{
|
||||
case FileAttachmentType.DOCUMENT:
|
||||
documents.Add(attachment);
|
||||
break;
|
||||
|
||||
case FileAttachmentType.IMAGE:
|
||||
images++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
namespace AIStudio.Chat;
|
||||
|
||||
/// <summary>
|
||||
/// Keeps a number up to date which nothing announces.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A conversation is a plain list of plain objects. Nothing raises an event when a block is added,
|
||||
/// when a document is attached, or when an answer grows by another sentence -- so a number derived
|
||||
/// from all of that cannot be wired to the places which change it. It was tried: fifteen call sites,
|
||||
/// and four review rounds each found another one which was missing.
|
||||
///
|
||||
/// So the number is recomputed instead of notified. Whoever thinks something may have changed nudges
|
||||
/// this tracker, and the tracker decides when to do the work: many nudges in a row become one run, a
|
||||
/// nudge arriving during a run becomes exactly one further run, and a minimum distance keeps a burst
|
||||
/// of them from turning into a burst of counting.
|
||||
///
|
||||
/// The heartbeat is not distrust of the nudges. Attachments are read from disk every time they are
|
||||
/// sent, so a file somebody edits in another program changes what the next message costs without
|
||||
/// anything happening in AI Studio which anyone could nudge from.
|
||||
/// </remarks>
|
||||
/// <param name="recount">Does the actual work. Gets a token which ends it when the tracker goes away.</param>
|
||||
/// <param name="quietTime">
|
||||
/// How long to stay quiet after a run before honouring the next nudge. Asked again each time,
|
||||
/// because what is reasonable depends on what is going on: a person who just switched a profile is
|
||||
/// waiting for the number, while an answer being written moves it with every word and wants a
|
||||
/// slower pace than the words arrive at.
|
||||
/// </param>
|
||||
/// <param name="heartbeat">How long to wait for a nudge before running anyway.</param>
|
||||
public sealed class ConversationTokenTracker(Func<CancellationToken, Task> recount, Func<TimeSpan> quietTime, TimeSpan heartbeat) : IAsyncDisposable
|
||||
{
|
||||
/// <summary>
|
||||
/// How long a tracker which is going away waits for its own loop.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The loop ends on cancellation, so this is only ever reached when something it called does
|
||||
/// not. Whoever is leaving the screen must not be the one who waits for that.
|
||||
/// </remarks>
|
||||
private static readonly TimeSpan SHUTDOWN_PATIENCE = TimeSpan.FromSeconds(2);
|
||||
|
||||
private readonly SemaphoreSlim wakeUp = new(0, 1);
|
||||
private readonly CancellationTokenSource stopping = new();
|
||||
|
||||
private Task? loop;
|
||||
|
||||
/// <summary>
|
||||
/// Starts the loop. Calling this twice does nothing the second time.
|
||||
/// </summary>
|
||||
public void Start() => this.loop ??= Task.Run(this.RunAsync);
|
||||
|
||||
/// <summary>
|
||||
/// Says that something may have changed.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Cheap on purpose, because it is called from the render path. It says "maybe", never "yes":
|
||||
/// asking for a run which turns out to change nothing costs a few lookups, while missing one is
|
||||
/// the bug this whole class exists to make impossible.
|
||||
/// </remarks>
|
||||
public void Nudge()
|
||||
{
|
||||
//
|
||||
// One pending wake-up is all a loop can act on. A second one would only make it run again
|
||||
// with the same answer.
|
||||
//
|
||||
if (this.wakeUp.CurrentCount > 0)
|
||||
return;
|
||||
|
||||
try
|
||||
{
|
||||
this.wakeUp.Release();
|
||||
}
|
||||
catch (SemaphoreFullException)
|
||||
{
|
||||
//
|
||||
// Two threads got past the check above at the same time. The one which won left the
|
||||
// wake-up we wanted, so there is nothing left to do here.
|
||||
//
|
||||
}
|
||||
catch (ObjectDisposedException)
|
||||
{
|
||||
// The tracker is going away, and a number nobody will look at needs no update.
|
||||
}
|
||||
}
|
||||
|
||||
private async Task RunAsync()
|
||||
{
|
||||
var token = this.stopping.Token;
|
||||
while (!token.IsCancellationRequested)
|
||||
{
|
||||
try
|
||||
{
|
||||
//
|
||||
// Sleeps until somebody nudges -- or until the heartbeat is due, which is what the
|
||||
// timeout returning false means. Both lead to the same run, so the result is not
|
||||
// even looked at.
|
||||
//
|
||||
await this.wakeUp.WaitAsync(heartbeat, token);
|
||||
if (token.IsCancellationRequested)
|
||||
return;
|
||||
|
||||
//
|
||||
// Deliberately without draining further wake-ups first. A nudge which arrives while
|
||||
// this run reads the conversation may well be about a change this run is already
|
||||
// seeing -- and then the extra run costs a few lookups. Draining would risk the
|
||||
// other case, where the change comes after the read and nobody asks again.
|
||||
//
|
||||
try
|
||||
{
|
||||
await recount(token);
|
||||
}
|
||||
catch (Exception) when (!token.IsCancellationRequested)
|
||||
{
|
||||
//
|
||||
// One failed run must not end the loop: a tracker which died on a single bad
|
||||
// answer would leave a stale number standing forever, which is the failure this
|
||||
// class was built to rule out. Saying what went wrong is the job of the work
|
||||
// itself, which is the only side that has a logger.
|
||||
//
|
||||
}
|
||||
|
||||
//
|
||||
// The quiet time is kept after the work, not before it: the first nudge of a burst
|
||||
// is answered at once, and the rest of the burst collapses into the single run which
|
||||
// follows this delay.
|
||||
//
|
||||
// It is also what paces a run which feeds itself. Showing a new number renders, and
|
||||
// a render nudges -- so while something changes continuously, this delay is the
|
||||
// whole cadence.
|
||||
//
|
||||
await Task.Delay(quietTime(), token);
|
||||
}
|
||||
catch (OperationCanceledException)
|
||||
{
|
||||
return;
|
||||
}
|
||||
catch (ObjectDisposedException)
|
||||
{
|
||||
// The tracker was disposed underneath this loop, which is another way of stopping.
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#region Implementation of IAsyncDisposable
|
||||
|
||||
public async ValueTask DisposeAsync()
|
||||
{
|
||||
await this.stopping.CancelAsync();
|
||||
|
||||
if (this.loop is not null)
|
||||
{
|
||||
try
|
||||
{
|
||||
//
|
||||
// Awaited rather than abandoned, so that nothing is still counting into a component
|
||||
// which is already gone. The counting itself takes the same token, so a run which
|
||||
// sits in an IPC call ends with it -- and the patience is there for the case where
|
||||
// it does not, because a chat being closed is not worth hanging on to.
|
||||
//
|
||||
await this.loop.WaitAsync(SHUTDOWN_PATIENCE);
|
||||
}
|
||||
catch (Exception)
|
||||
{
|
||||
// The loop ends on cancellation; whatever else it carries out is of no use here.
|
||||
}
|
||||
}
|
||||
|
||||
this.stopping.Dispose();
|
||||
this.wakeUp.Dispose();
|
||||
}
|
||||
|
||||
#endregion
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
using AIStudio.Models;
|
||||
|
||||
namespace AIStudio.Chat;
|
||||
|
||||
/// <summary>
|
||||
/// What a conversation costs, as far as the app can count it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Three separate statements, and keeping them apart is the point. How many tokens were counted is
|
||||
/// one; what the model's window is, if anybody has written it down, is the second; and how much of
|
||||
/// the conversation could not be counted at all is the third. Folding any of them into the others
|
||||
/// would turn a gap into a number somebody reads as a fact.
|
||||
/// </remarks>
|
||||
public readonly record struct ConversationTokens
|
||||
{
|
||||
/// <summary>
|
||||
/// The answer when nothing could be counted, which is what a broken tokenizer leaves behind.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Deliberately not a zero. A conversation of no tokens and a conversation nobody could measure
|
||||
/// look the same as a number and are not the same thing, so the display shows nothing at all
|
||||
/// rather than claiming an empty chat.
|
||||
/// </remarks>
|
||||
public static readonly ConversationTokens UNAVAILABLE = new();
|
||||
|
||||
/// <summary>
|
||||
/// Whether anything could be counted.
|
||||
/// </summary>
|
||||
public bool IsKnown { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many tokens the counted parts of the conversation take.
|
||||
/// </summary>
|
||||
public int Tokens { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Whether the number is an estimate rather than the model's own count.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// True whenever the built-in tokenizer did the counting, which is the normal case: a model's
|
||||
/// own tokenizer is only used where somebody configured one for their provider. Two tokenizers
|
||||
/// disagree by a few percent on ordinary prose and by a lot more on code or a language they were
|
||||
/// not trained on, so the number is shown as an approximation unless we counted with the
|
||||
/// tokenizer the model itself uses.
|
||||
/// </remarks>
|
||||
public bool IsEstimate { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How much the model reads, where anybody has stated it.
|
||||
/// </summary>
|
||||
public ContextWindow Window { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many images travel along which nobody can count.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every vendor charges images differently -- OpenAI by tiles of the scaled image, Anthropic by
|
||||
/// its area, Google by tiles of another size -- and none of those numbers can be had from the
|
||||
/// file without decoding it first. So they are reported as a number of images instead of being
|
||||
/// guessed at, or worse, counted as the base64 text they are sent as: that text is two to three
|
||||
/// orders of magnitude longer than what any vendor charges for the picture.
|
||||
/// </remarks>
|
||||
public int UncountedImages { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many images the model takes, where its vendor stated a number.
|
||||
/// </summary>
|
||||
public ImageLimits ImageLimits { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Whether more images travel than the model is documented to accept.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Counted over the whole conversation rather than over the message being written, because that
|
||||
/// is what a request carries: every picture anybody attached is sent again with every further
|
||||
/// message, so a chat crosses this line long after the message which added the picture -- and
|
||||
/// the person who crosses it has usually forgotten that the pictures are still there.
|
||||
///
|
||||
/// False whenever nobody stated a limit, which is most models. An invented ceiling would refuse
|
||||
/// something that works.
|
||||
/// </remarks>
|
||||
public bool TooManyImages => this.ImageLimits.MaxInOneMessage is { } allowed && this.UncountedImages > allowed;
|
||||
}
|
||||
@@ -11,23 +11,25 @@ public static class ListContentBlockExtensions
|
||||
/// </summary>
|
||||
/// <param name="blocks">The list of content blocks to process.</param>
|
||||
/// <param name="roleTransformer">A function that transforms each content block into a message result asynchronously.</param>
|
||||
/// <param name="selectedProvider">The selected LLM provider.</param>
|
||||
/// <param name="selectedModel">The selected model.</param>
|
||||
/// <param name="provider">The configured provider, whose model is being written to.</param>
|
||||
/// <param name="textSubContentFactory">A factory function to create text sub-content.</param>
|
||||
/// <param name="imageSubContentFactory">A factory function to create image sub-content.</param>
|
||||
/// <returns>An asynchronous task that resolves to a list of transformed results.</returns>
|
||||
public static async Task<IList<IMessageBase>> BuildMessagesAsync(
|
||||
this List<ContentBlock> blocks,
|
||||
LLMProviders selectedProvider,
|
||||
Model selectedModel,
|
||||
AIStudio.Settings.Provider provider,
|
||||
Func<ChatRole, string> roleTransformer,
|
||||
Func<string, ISubContent> textSubContentFactory,
|
||||
Func<FileAttachmentImage, Task<ISubContent>> imageSubContentFactory)
|
||||
{
|
||||
var capabilities = selectedProvider.GetModelCapabilities(selectedModel);
|
||||
var canProcessImages = capabilities.Contains(Capability.MULTIPLE_IMAGE_INPUT) ||
|
||||
capabilities.Contains(Capability.SINGLE_IMAGE_INPUT);
|
||||
|
||||
//
|
||||
// Asked through the configured provider, so that what a person set in their expert settings
|
||||
// counts here too. It did not: this path read the automatic answer alone, so somebody who
|
||||
// switched image input on saw it work while attaching the picture and saw it ignored while
|
||||
// the message was built -- every chat round and every tool round.
|
||||
//
|
||||
var canProcessImages = provider.SupportsImageInput();
|
||||
|
||||
var messageTaskList = new List<Task<IMessageBase>>(blocks.Count);
|
||||
foreach (var block in blocks)
|
||||
{
|
||||
@@ -102,8 +104,7 @@ public static class ListContentBlockExtensions
|
||||
/// Processes a list of content blocks using direct image URL format to create message results asynchronously.
|
||||
/// </summary>
|
||||
/// <param name="blocks">The list of content blocks to process.</param>
|
||||
/// <param name="selectedProvider">The selected LLM provider.</param>
|
||||
/// <param name="selectedModel">The selected model.</param>
|
||||
/// <param name="provider">The configured provider, whose model is being written to.</param>
|
||||
/// <returns>An asynchronous task that resolves to a list of transformed message results.</returns>
|
||||
/// <remarks>
|
||||
/// Uses direct image URL format where the image data is placed directly in the image_url field:
|
||||
@@ -114,10 +115,8 @@ public static class ListContentBlockExtensions
|
||||
/// </remarks>
|
||||
public static async Task<IList<IMessageBase>> BuildMessagesUsingDirectImageUrlAsync(
|
||||
this List<ContentBlock> blocks,
|
||||
LLMProviders selectedProvider,
|
||||
Model selectedModel) => await blocks.BuildMessagesAsync(
|
||||
selectedProvider,
|
||||
selectedModel,
|
||||
AIStudio.Settings.Provider provider) => await blocks.BuildMessagesAsync(
|
||||
provider,
|
||||
StandardRoleTransformer,
|
||||
StandardTextSubContentFactory,
|
||||
DirectImageSubContentFactory);
|
||||
@@ -126,8 +125,7 @@ public static class ListContentBlockExtensions
|
||||
/// Processes a list of content blocks using nested image URL format to create message results asynchronously.
|
||||
/// </summary>
|
||||
/// <param name="blocks">The list of content blocks to process.</param>
|
||||
/// <param name="selectedProvider">The selected LLM provider.</param>
|
||||
/// <param name="selectedModel">The selected model.</param>
|
||||
/// <param name="provider">The configured provider, whose model is being written to.</param>
|
||||
/// <returns>An asynchronous task that resolves to a list of transformed message results.</returns>
|
||||
/// <remarks>
|
||||
/// Uses nested image URL format where the image data is wrapped in an object:
|
||||
@@ -138,10 +136,8 @@ public static class ListContentBlockExtensions
|
||||
/// </remarks>
|
||||
public static async Task<IList<IMessageBase>> BuildMessagesUsingNestedImageUrlAsync(
|
||||
this List<ContentBlock> blocks,
|
||||
LLMProviders selectedProvider,
|
||||
Model selectedModel) => await blocks.BuildMessagesAsync(
|
||||
selectedProvider,
|
||||
selectedModel,
|
||||
AIStudio.Settings.Provider provider) => await blocks.BuildMessagesAsync(
|
||||
provider,
|
||||
StandardRoleTransformer,
|
||||
StandardTextSubContentFactory,
|
||||
NestedImageSubContentFactory);
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
namespace AIStudio.Chat;
|
||||
|
||||
/// <summary>
|
||||
/// The system prompt of a chat thread as it would be sent, together with what building it decided.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The system prompt is not the text a person typed into it. A chat template may replace it, the
|
||||
/// retrieved data of a data source is appended to it, a profile adds its own paragraph, the tool
|
||||
/// policy adds another, and the current date goes in front of everything. Whoever wants to know how
|
||||
/// long the next request is has to ask the same question the request does.
|
||||
/// </remarks>
|
||||
/// <param name="Text">The whole system prompt, as the provider receives it.</param>
|
||||
/// <param name="BasePrompt">
|
||||
/// The prompt without any of the parts added around it. The thread keeps this one, so that it can
|
||||
/// still say which prompt it was configured with rather than the assembled result.
|
||||
/// </param>
|
||||
/// <param name="ProfileIsAllowed">Whether the chat template let a profile take part.</param>
|
||||
/// <param name="Explanation">What was used, in one sentence, for the log.</param>
|
||||
public sealed record PreparedSystemPrompt(string Text, string BasePrompt, bool ProfileIsAllowed, string Explanation);
|
||||
@@ -0,0 +1,47 @@
|
||||
using System.Globalization;
|
||||
|
||||
namespace AIStudio.Chat;
|
||||
|
||||
/// <summary>
|
||||
/// Writes a number of tokens the way a person reads it next to their input field.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A context window of a million tokens written out in full is eight characters of noise under a
|
||||
/// text field, and nobody reads the last five of them. So everything from a thousand on is
|
||||
/// shortened, and two decimals keep the resolution a person acts on: the difference between 1.20k
|
||||
/// and 1.80k is one they can see, while the last three digits of 1,234 are not.
|
||||
///
|
||||
/// The culture is passed in rather than taken from the thread. AI Studio's language is chosen in
|
||||
/// its settings and does not move the thread's culture along with it, so a German who picked German
|
||||
/// would otherwise read English separators inside a German sentence.
|
||||
/// </remarks>
|
||||
public static class TokenAmount
|
||||
{
|
||||
/// <summary>
|
||||
/// Below this, the exact number is shown.
|
||||
/// </summary>
|
||||
private const int EXACT_BELOW = 1_000;
|
||||
|
||||
/// <summary>
|
||||
/// Writes a number of tokens.
|
||||
/// </summary>
|
||||
/// <param name="tokens">The number of tokens.</param>
|
||||
/// <param name="culture">The culture whose separators the number is written with.</param>
|
||||
/// <returns>The number, shortened from a thousand on.</returns>
|
||||
public static string Format(int tokens, CultureInfo culture)
|
||||
{
|
||||
if (tokens < EXACT_BELOW)
|
||||
return tokens.ToString("N0", culture);
|
||||
|
||||
//
|
||||
// Rounded before the unit is chosen, not after. Otherwise the few hundred tokens just below
|
||||
// a million round up inside their own unit and read as "1,000.00k", which is a number
|
||||
// nobody writes.
|
||||
//
|
||||
var thousands = tokens / 1_000d;
|
||||
if (Math.Round(thousands, 2) < 1_000d)
|
||||
return $"{thousands.ToString("N2", culture)}k";
|
||||
|
||||
return $"{(tokens / 1_000_000d).ToString("N2", culture)}M";
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user