mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-05 03:49:40 +00:00
Added the exact token count where the provider reports it (#989)
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
1 parent
82986afe62
commit
be1e6532fb
34 files changed
+1360
-104
No files matched your search
@@ -392,6 +392,50 @@ public sealed record ChatThread
|
||||
return true;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Finds what a provider reported for this conversation, as long as the report still describes it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A report describes one request: the conversation up to the answer which carries it. It is
|
||||
/// worth something only while the thread still is that conversation, so only the last block is
|
||||
/// asked, and no earlier answer ever stands in for it. Whatever came after an older report -- a
|
||||
/// message whose request was turned down, an answer which is still being written, an answer
|
||||
/// without a report of its own -- is missing from that report's number, and the estimate is
|
||||
/// closer to the truth than a figure which leaves it out.<br/><br/>
|
||||
///
|
||||
/// A report also stops counting when the thread holds a different number of blocks than the
|
||||
/// request did, which means an earlier message was deleted, and when the next request goes to
|
||||
/// another model, which counts the same conversation with another tokenizer. Editing the last
|
||||
/// message or rolling the chat back makes an earlier answer the last block again, with exactly
|
||||
/// the blocks it was reported for, so its report counts once more.<br/><br/>
|
||||
///
|
||||
/// What goes unnoticed is a change beside the messages: another system prompt, profile, or
|
||||
/// selection of tools. That shows only with the next answer. Noticing it would take a
|
||||
/// fingerprint of the system prompt as it was sent, after the data sources added to it, which
|
||||
/// is a lot of machinery for a number which corrects itself one answer later.
|
||||
/// </remarks>
|
||||
/// <param name="model">The model the next request would go to.</param>
|
||||
/// <returns>
|
||||
/// What the provider reported, together with the answer which followed it, or
|
||||
/// ReportedHistory.UNKNOWN when no report describes this conversation.
|
||||
/// </returns>
|
||||
public ReportedHistory ReportedHistoryFor(Model model)
|
||||
{
|
||||
if (this.Blocks.Count is 0)
|
||||
return ReportedHistory.UNKNOWN;
|
||||
|
||||
if (this.Blocks[^1].Content is not ContentText { IsStreaming: false, ReportedUsage: { } reported } answer)
|
||||
return ReportedHistory.UNKNOWN;
|
||||
|
||||
if (reported.BlockCount != this.Blocks.Count)
|
||||
return ReportedHistory.UNKNOWN;
|
||||
|
||||
if (!string.Equals(reported.ModelId, model.Id, StringComparison.Ordinal))
|
||||
return ReportedHistory.UNKNOWN;
|
||||
|
||||
return ReportedHistory.Of(reported.ToTokenUsage(), answer.Text);
|
||||
}
|
||||
|
||||
private static void DeleteManagedAttachments(ContentBlock block)
|
||||
{
|
||||
if (block.Content is not ContentText textContent)
|
||||
|
||||
@@ -52,6 +52,21 @@ public sealed class ContentText : IContent
|
||||
|
||||
public List<ToolInvocationTrace> ToolInvocations { get; set; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// What the provider said everything sent along with this answer cost, where it said anything.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Kept on the answer rather than beside the chat, so that it is stored, loaded, and exported
|
||||
/// with the message it belongs to -- and so that it goes away when the message does. An edited
|
||||
/// or regenerated answer removes its block, which removes these numbers with it. Null for every
|
||||
/// answer written before this was recorded, and at every provider which reports nothing.
|
||||
///
|
||||
/// Having a report is not the same as the report still being true. Whether it still describes
|
||||
/// the conversation is decided by ChatThread.ReportedHistoryFor, which looks at the thread around
|
||||
/// the answer, not just at the answer.
|
||||
/// </remarks>
|
||||
public ReportedTokenUsage? ReportedUsage { get; set; }
|
||||
|
||||
[JsonIgnore]
|
||||
public ToolRuntimeStatus ToolRuntimeStatus { get; set; } = new();
|
||||
|
||||
@@ -98,6 +113,30 @@ public sealed class ContentText : IContent
|
||||
/// </remarks>
|
||||
public void EndToolRun() => this.PendingToolConversation = [];
|
||||
|
||||
/// <summary>
|
||||
/// Keeps what the provider says the request behind this answer cost.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The one place where a stream's usage becomes part of the answer, for every path which writes
|
||||
/// one. It arrives on one line of the stream, usually the last one, and only where the provider
|
||||
/// reports it at all -- so a usage which states nothing leaves what was reported before alone.
|
||||
/// </remarks>
|
||||
/// <param name="usage">What the current chunk of the stream says, which is mostly nothing.</param>
|
||||
/// <param name="modelId">The model the request went to.</param>
|
||||
/// <param name="blockCount">How many blocks the conversation had when the request went out, this answer included.</param>
|
||||
public void RecordReportedUsage(TokenUsage usage, string modelId, int blockCount)
|
||||
{
|
||||
if (!usage.IsKnown)
|
||||
return;
|
||||
|
||||
this.ReportedUsage = new()
|
||||
{
|
||||
PromptTokens = usage.PromptTokens,
|
||||
ModelId = modelId,
|
||||
BlockCount = blockCount,
|
||||
};
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
public async Task<ChatThread> CreateFromProviderAsync(IProvider provider, Model chatModel, IContent? lastUserPrompt, ChatThread? chatThread, CancellationToken token = default)
|
||||
{
|
||||
@@ -158,7 +197,10 @@ public sealed class ContentText : IContent
|
||||
|
||||
// Get the settings manager:
|
||||
var settings = Program.SERVICE_PROVIDER.GetService<SettingsManager>()!;
|
||||
|
||||
|
||||
// What the request carries, for telling later whether a report still describes this chat:
|
||||
var blocksSent = chatThread.Blocks.Count;
|
||||
|
||||
// Start another thread by using a task to uncouple
|
||||
// the UI thread from the AI processing:
|
||||
try
|
||||
@@ -187,6 +229,9 @@ public sealed class ContentText : IContent
|
||||
// Merge the sources:
|
||||
this.Sources.MergeSources(contentStreamChunk.Sources);
|
||||
|
||||
// Keep what the provider says the request cost:
|
||||
this.RecordReportedUsage(contentStreamChunk.Usage, chatModel.Id, blocksSent);
|
||||
|
||||
// Notify the UI that the content has changed,
|
||||
// depending on the energy saving mode:
|
||||
var now = DateTimeOffset.Now;
|
||||
@@ -310,6 +355,12 @@ public sealed class ContentText : IContent
|
||||
}
|
||||
|
||||
/// <inheritdoc />
|
||||
/// <remarks>
|
||||
/// The reported usage stays behind on purpose. A clone continues somewhere else -- as the
|
||||
/// example conversation of a chat template, or as an assistant's conversation carried over into
|
||||
/// a chat -- with another system prompt around it, and what the provider stated was for the
|
||||
/// request this answer came out of.
|
||||
/// </remarks>
|
||||
public IContent DeepClone() => new ContentText
|
||||
{
|
||||
Text = this.Text,
|
||||
|
||||
@@ -19,6 +19,11 @@ namespace AIStudio.Chat;
|
||||
/// which is not about the next request but about the one in flight: it is what the model is
|
||||
/// reading at this moment, it is what fills the window while somebody watches, and it is gone
|
||||
/// again once the answer stands.
|
||||
///
|
||||
/// The three are kept apart, and nothing stands in two of them. The conversation so far is what a
|
||||
/// provider may already have counted exactly; the draft is what nobody has counted yet; and the
|
||||
/// tool conversation is what the number will lose again once the answer is there. Each of them is
|
||||
/// counted once and named on its own, so that a person can tell which part they are looking at.
|
||||
/// </remarks>
|
||||
public sealed record ConversationParts
|
||||
{
|
||||
@@ -28,35 +33,62 @@ public sealed record ConversationParts
|
||||
public static readonly ConversationParts NOTHING = new();
|
||||
|
||||
/// <summary>
|
||||
/// The texts which go into the request as they are.
|
||||
/// The texts of the conversation which go into the request as they are.
|
||||
/// </summary>
|
||||
public IReadOnlyList<string> Texts { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// The texts which belong to this moment alone.
|
||||
/// The texts of the conversation which are still being written.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// They cost exactly what the others cost; what sets them apart is that they will never be seen
|
||||
/// again in this shape. The sentence somebody is typing changes with the next pause, and an
|
||||
/// answer being streamed is a different text three seconds later -- so remembering what they
|
||||
/// cost fills memory with answers nobody will ask for again.
|
||||
///
|
||||
/// What a model's tools have returned so far belongs here for the same reason, although nobody
|
||||
/// is writing it: it travels with every further round of one request and with nothing after
|
||||
/// that, so it is measured while it matters and forgotten when the answer is there.
|
||||
/// again in this shape. An answer being streamed is a different text three seconds later -- so
|
||||
/// remembering what it cost fills memory with answers nobody will ask for again.
|
||||
/// </remarks>
|
||||
public IReadOnlyList<string> GrowingTexts { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// The documents whose content is put into the request.
|
||||
/// What the tools of the running request have returned so far, along with the calls to them.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Growing in the same way as the answer being streamed, and measured the same way. It travels
|
||||
/// with every further round of one request and with nothing after that, so it is measured while
|
||||
/// it matters and forgotten when the answer is there.
|
||||
/// </remarks>
|
||||
public IReadOnlyList<string> ToolConversation { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// The documents of the conversation whose content is put into the request.
|
||||
/// </summary>
|
||||
public IReadOnlyList<FileAttachment> Documents { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// How many images travel along.
|
||||
/// How many images of the conversation travel along.
|
||||
/// </summary>
|
||||
public int Images { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// What stands in the composer, or an empty string when nothing does.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Changes with the next pause, so it is measured like the texts which are still being written.
|
||||
/// </remarks>
|
||||
public string DraftText { get; init; } = string.Empty;
|
||||
|
||||
/// <summary>
|
||||
/// The documents attached to the composer.
|
||||
/// </summary>
|
||||
public IReadOnlyList<FileAttachment> DraftDocuments { get; init; } = [];
|
||||
|
||||
/// <summary>
|
||||
/// How many images attached to the composer travel along.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Apart from the images of the conversation, because only those can be part of what a
|
||||
/// provider has already counted.
|
||||
/// </remarks>
|
||||
public int DraftImages { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Collects what a conversation would send.
|
||||
/// </summary>
|
||||
@@ -84,6 +116,7 @@ public sealed record ConversationParts
|
||||
{
|
||||
var texts = new List<string>();
|
||||
var growing = new List<string>();
|
||||
var toolConversation = new List<string>();
|
||||
var documents = new List<FileAttachment>();
|
||||
var images = 0;
|
||||
|
||||
@@ -117,7 +150,7 @@ public sealed record ConversationParts
|
||||
// the tool conversation. A block skipped for having nothing to say is exactly the
|
||||
// block whose request is growing the fastest.
|
||||
//
|
||||
growing.AddRange(text.PendingToolConversation);
|
||||
toolConversation.AddRange(text.PendingToolConversation);
|
||||
|
||||
if (string.IsNullOrWhiteSpace(text.Text))
|
||||
continue;
|
||||
@@ -131,18 +164,24 @@ public sealed record ConversationParts
|
||||
}
|
||||
}
|
||||
|
||||
if (!string.IsNullOrWhiteSpace(draft))
|
||||
growing.Add(draft);
|
||||
|
||||
//
|
||||
// Sorted the same way as the attachments of the conversation, into lists of their own.
|
||||
//
|
||||
var draftDocuments = new List<FileAttachment>();
|
||||
var draftImages = 0;
|
||||
if (draftAttachments is not null)
|
||||
Sort(draftAttachments, documents, ref images);
|
||||
Sort(draftAttachments, draftDocuments, ref draftImages);
|
||||
|
||||
return new()
|
||||
{
|
||||
Texts = texts,
|
||||
GrowingTexts = growing,
|
||||
ToolConversation = toolConversation,
|
||||
Documents = documents,
|
||||
Images = imagesAreSent ? images : 0,
|
||||
DraftText = string.IsNullOrWhiteSpace(draft) ? string.Empty : draft,
|
||||
DraftDocuments = draftDocuments,
|
||||
DraftImages = imagesAreSent ? draftImages : 0,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -29,9 +29,26 @@ public readonly record struct ConversationTokens
|
||||
public bool IsKnown { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many tokens the counted parts of the conversation take.
|
||||
/// How many tokens the counted parts of the conversation take, the draft included.
|
||||
/// </summary>
|
||||
public int Tokens { get; init; }
|
||||
public int Tokens => this.HistoryTokens + this.DraftTokens;
|
||||
|
||||
/// <summary>
|
||||
/// How many tokens the conversation takes without the message being written right now.
|
||||
/// </summary>
|
||||
public int HistoryTokens { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How much of HistoryTokens the tools of the running request add: the calls to them and what
|
||||
/// they returned so far.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Named on its own because it is the one share which goes away again. The model reads it in
|
||||
/// every round of the request, so it fills the window while the tools work -- and none of it is
|
||||
/// sent with the next message. Without saying so, the number would climb by tens of thousands
|
||||
/// and then drop back once the answer stands, and nobody could tell why.
|
||||
/// </remarks>
|
||||
public int ToolTokens { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Whether the number is an estimate rather than the model's own count.
|
||||
@@ -45,13 +62,46 @@ public readonly record struct ConversationTokens
|
||||
/// </remarks>
|
||||
public bool IsEstimate { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many tokens the message being written right now takes.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Kept apart from the rest for the same reason the statements above are kept apart: what the
|
||||
/// conversation has already cost is something a provider can be asked about, while a sentence
|
||||
/// nobody has sent yet can only be estimated.
|
||||
/// </remarks>
|
||||
public int DraftTokens { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Whether the conversation so far was counted by the provider rather than by this app.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// True once a provider has stated what a request of this conversation cost, which makes
|
||||
/// everything up to the last answer an exact number. That answer is counted by this app, since
|
||||
/// the provider's number for it includes reasoning which no request carries; next to the rest
|
||||
/// of the conversation, its share of the error is small enough to still call the history exact.
|
||||
/// It says nothing about the draft, which stays an estimate either way -- nobody has charged
|
||||
/// for that one yet.
|
||||
/// </remarks>
|
||||
public bool HistoryIsReported { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How much the model reads, where anybody has stated it.
|
||||
/// </summary>
|
||||
public ContextWindow Window { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many images travel along which nobody can count.
|
||||
/// How many images travel along, those of the conversation and those of the draft.
|
||||
/// </summary>
|
||||
public int Images { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many of those images are attached to the message being written right now.
|
||||
/// </summary>
|
||||
public int DraftImages { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// How many images travel along which nobody has counted.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every vendor charges images differently -- OpenAI by tiles of the scaled image, Anthropic by
|
||||
@@ -59,8 +109,11 @@ public readonly record struct ConversationTokens
|
||||
/// file without decoding it first. So they are reported as a number of images instead of being
|
||||
/// guessed at, or worse, counted as the base64 text they are sent as: that text is two to three
|
||||
/// orders of magnitude longer than what any vendor charges for the picture.
|
||||
///
|
||||
/// Where the provider reported the conversation so far, it counted the pictures in it as well,
|
||||
/// however it charges them. Then only those of the draft are left uncounted.
|
||||
/// </remarks>
|
||||
public int UncountedImages { get; init; }
|
||||
public int UncountedImages => this.HistoryIsReported ? this.DraftImages : this.Images;
|
||||
|
||||
/// <summary>
|
||||
/// How many images the model takes, where its vendor stated a number.
|
||||
@@ -77,7 +130,8 @@ public readonly record struct ConversationTokens
|
||||
/// the person who crosses it has usually forgotten that the pictures are still there.
|
||||
///
|
||||
/// False whenever nobody stated a limit, which is most models. An invented ceiling would refuse
|
||||
/// something that works.
|
||||
/// something that works. Whether anybody counted the pictures plays no part: the limit is on
|
||||
/// how many travel, not on what they cost.
|
||||
/// </remarks>
|
||||
public bool TooManyImages => this.ImageLimits.MaxInOneMessage is { } allowed && this.UncountedImages > allowed;
|
||||
public bool TooManyImages => this.ImageLimits.MaxInOneMessage is { } allowed && this.Images > allowed;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Chat;
|
||||
|
||||
/// <summary>
|
||||
/// What a conversation carries into its next request, as far as a provider has stated it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Two parts, and only one of them is the provider's statement. PromptTokens is what the provider
|
||||
/// counted for the request behind the last answer: the system prompt, the tools, and every message
|
||||
/// up to the question. The answer travels in the next request as well, though not in the shape the
|
||||
/// provider charged for: its completion included the model's reasoning and whatever the model wrote
|
||||
/// between think tags. So the answer comes along as the text it will be sent as, and is counted the
|
||||
/// same way every other text is.
|
||||
///
|
||||
/// That text is the answer alone. Reasoning never belongs to it, whether it was thrown away or kept
|
||||
/// to be read next to the answer: it is there for a person, and no request carries it. An answer
|
||||
/// which consists of reasoning only has no text at all, is not sent, and adds nothing to the prompt.
|
||||
///
|
||||
/// What is estimated that way is one answer, next to a prompt which holds the whole conversation
|
||||
/// before it. The error is the tokenizer's error on that one answer, not on the chat.
|
||||
/// </remarks>
|
||||
public sealed record ReportedHistory
|
||||
{
|
||||
/// <summary>
|
||||
/// The history of a conversation no report describes.
|
||||
/// </summary>
|
||||
public static readonly ReportedHistory UNKNOWN = new();
|
||||
|
||||
/// <summary>
|
||||
/// Whether a report describes the conversation. When false, nothing else here means anything.
|
||||
/// </summary>
|
||||
public bool IsKnown { get; private init; }
|
||||
|
||||
/// <summary>
|
||||
/// What the provider counted for the request behind the last answer.
|
||||
/// </summary>
|
||||
public int PromptTokens { get; private init; }
|
||||
|
||||
/// <summary>
|
||||
/// The text of the last answer, as the next request will carry it, without any reasoning.
|
||||
/// </summary>
|
||||
public string LastAnswer { get; private init; } = string.Empty;
|
||||
|
||||
/// <summary>
|
||||
/// States what a report says about the conversation.
|
||||
/// </summary>
|
||||
/// <param name="usage">What the provider reported for the request behind the last answer.</param>
|
||||
/// <param name="lastAnswer">The text of that answer.</param>
|
||||
/// <returns>The history, or UNKNOWN when the usage states nothing.</returns>
|
||||
public static ReportedHistory Of(TokenUsage usage, string lastAnswer) => usage.IsKnown
|
||||
? new()
|
||||
{
|
||||
IsKnown = true,
|
||||
PromptTokens = usage.PromptTokens,
|
||||
LastAnswer = lastAnswer,
|
||||
}
|
||||
: UNKNOWN;
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Chat;
|
||||
|
||||
/// <summary>
|
||||
/// What a provider said the request behind one answer cost, as it is stored on that answer.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The number, the model it was charged for, and the conversation it was counted on travel as one
|
||||
/// value. Each of them is meaningless without the others, and loose fields on the answer could be
|
||||
/// set, cleared, or copied apart.
|
||||
///
|
||||
/// A plain number rather than a TokenUsage: that type only ever comes out of its own factory, which
|
||||
/// is what keeps an impossible usage from existing, while a stored value has to be readable back by
|
||||
/// the serializer. ToTokenUsage is the way back, and it treats a chat file somebody edited by hand
|
||||
/// the same way the reading side treats a provider's JSON.
|
||||
/// </remarks>
|
||||
public sealed record ReportedTokenUsage
|
||||
{
|
||||
/// <summary>
|
||||
/// What everything sent to the model cost.
|
||||
/// </summary>
|
||||
public int PromptTokens { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Which model the number was charged for.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A token count belongs to the tokenizer which produced it. Switch the model of a chat, and
|
||||
/// the same conversation is worth a different number of tokens -- so the reported one stops
|
||||
/// being an answer about the request which is about to be sent, and the estimate, wrong as it
|
||||
/// is, is at least wrong about the right model.
|
||||
/// </remarks>
|
||||
public string ModelId { get; init; } = string.Empty;
|
||||
|
||||
/// <summary>
|
||||
/// How many blocks the conversation had when the request went out, this answer included.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// What tells an outdated report apart without anybody having to remember anything about it.
|
||||
/// Blocks are only ever added at the end, so while this answer is the last block, a thread with
|
||||
/// the same count is the thread the provider saw. A lower count means an earlier message was
|
||||
/// deleted, and that message is still inside the reported number.
|
||||
///
|
||||
/// Taken when the request is sent rather than when the report arrives: whatever is deleted
|
||||
/// while the answer streams in was still part of what the provider counted.
|
||||
/// </remarks>
|
||||
public int BlockCount { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// States the stored number as a usage again.
|
||||
/// </summary>
|
||||
/// <returns>The usage, or TokenUsage.UNKNOWN when the stored number states nothing usable.</returns>
|
||||
public TokenUsage ToTokenUsage() => TokenUsage.OfReported(this.PromptTokens);
|
||||
}
|
||||
Reference in new issue
Block a user