Added the exact token count where the provider reports it (#989)
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions

Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
j-erlerandThorsten Sommer authored and GitHub committed 2026-09-23 20:17:02 +02:00
1 parent 82986afe62
commit be1e6532fb
34 files changed
+1360 -104

No files matched your search

@@ -1118,6 +1118,21 @@ public abstract class BaseProvider : IProvider, ISecretId
continue;
}
//
// The line stating what the request cost carries no content of its own: providers
// send it as the last line of the stream, with no choices at all. It is handled
// before the check below, which would otherwise drop it as an empty response.
//
var usage = providerResponse.GetUsage();
if (usage.IsKnown)
{
yield return providerResponse.ContainsContent()
? providerResponse.GetContent() with { Usage = usage }
: new(string.Empty, [], Usage: usage);
continue;
}
// Skip empty responses:
if (!providerResponse.ContainsContent())
continue;
@@ -1229,6 +1244,7 @@ public abstract class BaseProvider : IProvider, ISecretId
/// <param name="systemPromptRole">The system prompt role to use.</param>
/// <param name="requestPath">The request path, relative to the provider base URL.</param>
/// <param name="headersAction">Optional additional headers to add.</param>
/// <param name="mayAskForSequentialToolCalls">Whether a request which offers tools may ask for one call at a time. False for a provider which rejects the parallel_tool_calls parameter.</param>
/// <param name="token">The cancellation token.</param>
/// <typeparam name="TRequest">The request DTO type.</typeparam>
/// <typeparam name="TDelta">The delta stream line type.</typeparam>
@@ -1245,6 +1261,7 @@ public abstract class BaseProvider : IProvider, ISecretId
string systemPromptRole = "system",
string requestPath = "chat/completions",
Action<HttpRequestHeaders>? headersAction = null,
bool mayAskForSequentialToolCalls = true,
[EnumeratorCancellation] CancellationToken token = default)
where TRequest : ChatCompletionAPIRequest
where TDelta : IResponseStreamLine
@@ -1283,7 +1300,7 @@ public abstract class BaseProvider : IProvider, ISecretId
if (runnableTools.Count > 0)
{
var adapter = new ChatCompletionToolCallingAdapter<TRequest>(requestFactory, systemPrompt, apiParameters,
runnableTools.Select(x => ProviderToolAdapters.ToChatCompletionTool(x.Definition)).ToList(), runnableTools,
runnableTools.Select(x => ProviderToolAdapters.ToChatCompletionTool(x.Definition)).ToList(), mayAskForSequentialToolCalls, runnableTools,
(requestDto, requestToken) => this.StreamChatCompletionRequest(requestDto, providerName, requestPath, requestedSecret, headersAction, requestToken),
ChatCompletionSourceReader.Read<TDelta, TAnnotation>,
this.logger);
@@ -3,9 +3,15 @@ namespace AIStudio.Provider;
/// <summary>
/// A chunk of content from a content stream, along with its associated sources.
/// </summary>
/// <remarks>
/// The usage rides along on the chunk rather than being reported next to the stream, because a
/// provider states it as one more line of that same stream. It is unknown on every chunk but the
/// one which carries it, and unknown on all of them at the providers which report nothing.
/// </remarks>
/// <param name="Content">The text content of the chunk.</param>
/// <param name="Sources">The list of sources associated with the chunk.</param>
public sealed record ContentStreamChunk(string Content, IList<ISource> Sources)
/// <param name="Usage">What the provider said the request cost, where it said anything.</param>
public sealed record ContentStreamChunk(string Content, IList<ISource> Sources, TokenUsage Usage = default)
{
/// <summary>
/// Implicit conversion to string.
@@ -1,3 +1,5 @@
using AIStudio.Provider.OpenAI;
namespace AIStudio.Provider.Fireworks;
/// <summary>
@@ -16,6 +18,19 @@ public readonly record struct ResponseStreamLine(string Id, string Object, uint
/// <inheritdoc />
public ContentStreamChunk GetContent() => new(this.Choices[0].Delta.Content, []);
/// <summary>
/// What Fireworks says the request cost, on the one line which carries it.
/// </summary>
/// <remarks>
/// The same block the OpenAI chat completion API sends, because this is that wire format.
/// Not a positional parameter: the struct is built from JSON, and a further parameter would
/// only be a value nobody passes.
/// </remarks>
public ChatCompletionUsage? Usage { get; init; }
/// <inheritdoc />
public TokenUsage GetUsage() => this.Usage?.ToTokenUsage() ?? TokenUsage.UNKNOWN;
#region Implementation of IAnnotationStreamLine
//
@@ -184,6 +184,15 @@ public sealed class ProviderHuggingFace : BaseProvider
AdditionalApiParameters = apiParameters
};
},
//
// Hugging Face answers parallel_tool_calls=false with a bad request, "feature
// not currently supported", and its specification of the chat completion does
// not list the parameter at all -- read on 2026-09-23 at
// https://huggingface.co/docs/inference-providers/tasks/chat-completion. Asking
// for it would cost every chat which offers tools its answer:
//
mayAskForSequentialToolCalls: false,
token: token))
yield return content;
}
@@ -16,4 +16,18 @@ public interface IResponseStreamLine : IAnnotationStreamLine
/// </summary>
/// <returns>The content of the response line.</returns>
public ContentStreamChunk GetContent();
/// <summary>
/// Gets what the provider said the request cost.
/// </summary>
/// <remarks>
/// Answered here for every wire format which says nothing about it, which is most of them: a
/// provider who reports no usage is the normal case, not a gap somebody has to fill in.
///
/// Unlike content and sources, there is no separate check for whether a line carries it. This
/// never fails on a line without one, and whether the answer means anything is what IsKnown of
/// the returned usage says.
/// </remarks>
/// <returns>The usage, or TokenUsage.UNKNOWN when the line carries none.</returns>
public TokenUsage GetUsage() => TokenUsage.UNKNOWN;
}
@@ -23,6 +23,17 @@ public record ChatCompletionAPIRequest(
[JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)]
public bool? ParallelToolCalls { get; init; }
/// <summary>
/// Asks a streamed request to end with what it cost.
/// </summary>
/// <remarks>
/// Derived rather than set, so that every provider which builds one of these asks for it
/// without having to know that it exists. A request which is not streamed carries no such
/// line, and then the block would only be a field the provider has to ignore.
/// </remarks>
[JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)]
public ChatCompletionStreamOptions? StreamOptions => this.Stream ? ChatCompletionStreamOptions.INCLUDE_USAGE : null;
// Attention: The "required" modifier is not supported for [JsonExtensionData].
[JsonExtensionData]
@@ -15,12 +15,27 @@ public record ChatCompletionDeltaStreamLine(string Id, string Object, uint Creat
{
}
/// <summary>
/// What the provider says the request cost, on the one line which carries it.
/// </summary>
/// <remarks>
/// Not a positional parameter: every provider builds an empty line through the constructor
/// above, and a further parameter would change all of those call sites for a value none of
/// them has. Providers send this block only when the request asked for it, and then on a final
/// line of its own which carries no choices -- which is why the usage is read apart from the
/// content rather than next to it.
/// </remarks>
public ChatCompletionUsage? Usage { get; init; }
/// <inheritdoc />
public bool ContainsContent() => this.Choices.Count > 0;
/// <inheritdoc />
public ContentStreamChunk GetContent() => new(this.Choices[0].Delta.Content, []);
/// <inheritdoc />
public TokenUsage GetUsage() => this.Usage?.ToTokenUsage() ?? TokenUsage.UNKNOWN;
#region Implementation of IAnnotationStreamLine
//
@@ -0,0 +1,18 @@
namespace AIStudio.Provider.OpenAI;
/// <summary>
/// What a streamed chat completion should report beyond its content.
/// </summary>
/// <remarks>
/// An OpenAI-compatible provider says nothing about what a streamed request cost unless it is asked
/// to. Without this block, the stream simply ends and the only token number anybody ever sees is
/// the one AI Studio estimated for itself.
/// </remarks>
/// <param name="IncludeUsage">Whether the stream should end with a line stating the token usage.</param>
public sealed record ChatCompletionStreamOptions(bool IncludeUsage)
{
/// <summary>
/// Asks for the usage line.
/// </summary>
public static readonly ChatCompletionStreamOptions INCLUDE_USAGE = new(true);
}
@@ -5,15 +5,20 @@ namespace AIStudio.Provider.OpenAI;
/// </summary>
/// <param name="TextDelta">The text this line carried, empty when it carried none.</param>
/// <param name="Sources">The sources this line announced, empty when it announced none.</param>
public readonly record struct ChatCompletionStreamPart(string TextDelta, IList<ISource> Sources)
/// <param name="Usage">What the provider said the request cost, unknown on every line but the one which carries it.</param>
public readonly record struct ChatCompletionStreamPart(string TextDelta, IList<ISource> Sources, TokenUsage Usage = default)
{
/// <summary>
/// The part of a line which says nothing to the user, such as a fragment of a tool call.
/// </summary>
public static ChatCompletionStreamPart Nothing => new(string.Empty, []);
/// <summary>
/// Whether this part has anything to show at all.
/// </summary>
/// <remarks>
/// The usage is not part of that: it is nothing to show, and whether it is passed on at all is
/// the adapter's decision, which knows which round this is.
/// </remarks>
public bool HasContent => this.TextDelta.Length > 0 || this.Sources.Count > 0;
}
@@ -60,10 +60,16 @@ public sealed class ChatCompletionToolCallAccumulator(Func<ServerSentEvent, ILis
// list either way.
//
var sources = readSources?.Invoke(serverSentEvent) ?? [];
//
// The usage arrives on a line without choices at most providers, and next to the last
// piece of text at some. Read before the choices are looked at, it is not lost in either.
//
var usage = line?.Usage?.ToTokenUsage() ?? TokenUsage.UNKNOWN;
var delta = line?.Choices?.FirstOrDefault()?.Delta;
if (delta is null)
return WithSources(string.Empty, sources);
return WithSources(string.Empty, sources, usage);
this.hasReadAnything = true;
@@ -80,10 +86,10 @@ public sealed class ChatCompletionToolCallAccumulator(Func<ServerSentEvent, ILis
var textDelta = delta.Content;
if (textDelta.Length is 0)
return WithSources(string.Empty, sources);
return WithSources(string.Empty, sources, usage);
this.text.Append(textDelta);
return new ChatCompletionStreamPart(textDelta, sources);
return new ChatCompletionStreamPart(textDelta, sources, usage);
}
/// <summary>
@@ -183,10 +189,10 @@ public sealed class ChatCompletionToolCallAccumulator(Func<ServerSentEvent, ILis
private static string? Coalesce(string? value) => string.IsNullOrWhiteSpace(value) ? null : value;
/// <summary>
/// A part for a line which brought sources but no text, or nothing at all.
/// A part for a line which brought sources or a usage but no text, or nothing at all.
/// </summary>
private static ChatCompletionStreamPart WithSources(string text, IList<ISource> sources)
=> sources.Count is 0 ? ChatCompletionStreamPart.Nothing : new ChatCompletionStreamPart(text, sources);
private static ChatCompletionStreamPart WithSources(string text, IList<ISource> sources, TokenUsage usage)
=> sources.Count is 0 && !usage.IsKnown ? ChatCompletionStreamPart.Nothing : new ChatCompletionStreamPart(text, sources, usage);
/// <summary>
/// One tool call while its fragments are still arriving.
@@ -16,7 +16,7 @@ namespace AIStudio.Provider.OpenAI;
public sealed class ChatCompletionToolCallingAdapter<TRequest>(
Func<TextMessage, IDictionary<string, object>, IList<object>?, Task<TRequest>> requestFactory,
TextMessage systemPrompt, IDictionary<string, object> apiParameters,
IList<object> providerTools,
IList<object> providerTools, bool mayAskForSequentialToolCalls,
IReadOnlyList<(ToolDefinition Definition, IToolImplementation Implementation)> runnableTools,
Func<ChatCompletionAPIRequest, CancellationToken, IAsyncEnumerable<ServerSentEvent>> streamRequestAsync,
Func<ServerSentEvent, IList<ISource>> readSources,
@@ -49,11 +49,23 @@ public sealed class ChatCompletionToolCallingAdapter<TRequest>(
//
// AI Studio runs tool calls one after another, so asking for parallel calls would
// only produce work it then has to serialize anyway. Requests without tools omit the
// parameter because some providers reject it then.
// parameter because some providers reject it then. So does every request to a provider
// which rejects the parameter altogether: its models may then ask for several calls at
// once, and the loop works through them one by one, checking the limits per call.
//
ParallelToolCalls = requestDtoBase.Tools is null ? null : false,
ParallelToolCalls = requestDtoBase.Tools is null || !mayAskForSequentialToolCalls ? null : false,
};
//
// Only the first round passes on what its request cost. Its prompt is the conversation up
// to the question, which is exactly what the next question will be sent after. Every later
// round carries the tool calls and their results on top, and none of that is sent again
// once the answer stands -- a report of such a round would count a chat far larger than
// the one the next request carries. What the answer adds, all rounds of text together, is
// counted from its text afterwards, cf. ReportedHistory.
//
var passesOnUsage = this.internalMessages.Count is 0;
//
// The text goes out while it is being written; the tool calls are put back together
// behind it, fragment by fragment.
@@ -62,8 +74,9 @@ public sealed class ChatCompletionToolCallingAdapter<TRequest>(
await foreach (var serverSentEvent in streamRequestAsync(requestDto, token))
{
var part = accumulator.Process(serverSentEvent);
if (part.HasContent)
yield return ToolCallingStreamEvent.TextDelta(new ContentStreamChunk(part.TextDelta, part.Sources));
var usage = passesOnUsage ? part.Usage : TokenUsage.UNKNOWN;
if (part.HasContent || usage.IsKnown)
yield return ToolCallingStreamEvent.TextDelta(new ContentStreamChunk(part.TextDelta, part.Sources, Usage: usage));
}
var message = accumulator.Build();
@@ -11,4 +11,15 @@ namespace AIStudio.Provider.OpenAI;
/// </remarks>
/// <param name="Id">The ID of the answer.</param>
/// <param name="Choices">The choices this line adds to.</param>
public sealed record ChatCompletionToolStreamLine(string? Id, IList<ChatCompletionToolStreamChoice?>? Choices);
public sealed record ChatCompletionToolStreamLine(string? Id, IList<ChatCompletionToolStreamChoice?>? Choices)
{
/// <summary>
/// What the provider says the request cost, on the one line which carries it.
/// </summary>
/// <remarks>
/// The same block the plain text path reads, and asked for the same way: every streamed
/// ChatCompletionAPIRequest asks for it, the requests of the tool rounds included. Not a
/// positional parameter, because nobody but the serializer ever builds this line.
/// </remarks>
public ChatCompletionUsage? Usage { get; init; }
}
@@ -0,0 +1,31 @@
// ReSharper disable ClassNeverInstantiated.Global
namespace AIStudio.Provider.OpenAI;
/// <summary>
/// What an OpenAI-compatible provider reports a chat completion cost.
/// </summary>
/// <remarks>
/// The number is optional because this is somebody else's JSON: the block arrives only when the
/// request asked for it, and the providers which follow the shape loosely leave fields out. Reading
/// it is one thing, believing it another -- TokenUsage.OfReported decides that.
///
/// The block states more than this, the completion and its reasoning share among it. Those are left
/// unread on purpose, for the reason given at TokenUsage: no later request carries them.
/// </remarks>
public sealed record ChatCompletionUsage
{
/// <summary>
/// What everything sent to the model cost.
/// </summary>
public int? PromptTokens { get; init; }
/// <summary>
/// States what this block reports, as far as it can be believed.
/// </summary>
/// <remarks>
/// The one way from the wire to a usage, shared by every stream line which carries this block,
/// so that what counts as believable is decided in a single place.
/// </remarks>
/// <returns>The usage, or TokenUsage.UNKNOWN when the block states nothing usable.</returns>
public TokenUsage ToTokenUsage() => TokenUsage.OfReported(this.PromptTokens);
}
@@ -1,3 +1,5 @@
using AIStudio.Provider.OpenAI;
namespace AIStudio.Provider.Perplexity;
/// <summary>
@@ -16,6 +18,19 @@ public readonly record struct ResponseStreamLine(string Id, string Object, uint
/// <inheritdoc />
public ContentStreamChunk GetContent() => new(this.Choices[0].Delta.Content, this.GetSources());
/// <summary>
/// What Perplexity says the request cost, on the one line which carries it.
/// </summary>
/// <remarks>
/// The same block the OpenAI chat completion API sends, because this is that wire format.
/// Not a positional parameter: the struct is built from JSON, and a further parameter would
/// only be a value nobody passes.
/// </remarks>
public ChatCompletionUsage? Usage { get; init; }
/// <inheritdoc />
public TokenUsage GetUsage() => this.Usage?.ToTokenUsage() ?? TokenUsage.UNKNOWN;
/// <inheritdoc />
public bool ContainsSources() => this != default && this.SearchResults.Count > 0;
@@ -0,0 +1,68 @@
namespace AIStudio.Provider;
/// <summary>
/// What a provider said one request actually carried, in tokens.
/// </summary>
/// <remarks>
/// The counterpart to what the app counts for itself: the app estimates what the next request will
/// cost, while this is what the provider counted for the last one. Two different statements, and
/// this one is the only exact one of the two.
///
/// Only the prompt is kept. Providers state what the answer cost as well, but that number includes
/// the model's reasoning and whatever the model wrote between think tags, and neither of them ever
/// becomes part of the answer's text. No later request carries them, so the number has no place in
/// a statement about those requests.
///
/// Nothing here says "unknown" with a zero. The default value of this type is unknown, which is the
/// right answer for every provider which reports nothing, and a counted request can never cost zero
/// prompt tokens because the factory below refuses to build one.
/// </remarks>
public readonly record struct TokenUsage
{
/// <summary>
/// The usage of a request nobody reported anything about.
/// </summary>
public static readonly TokenUsage UNKNOWN = new();
/// <summary>
/// Whether a provider reported anything at all. When false, the number is meaningless.
/// </summary>
public bool IsKnown { get; private init; }
/// <summary>
/// What everything sent to the model cost: the conversation so far, its attachments, the system
/// prompt, and whatever tools were offered.
/// </summary>
public int PromptTokens { get; private init; }
/// <summary>
/// States what a provider reported.
/// </summary>
/// <remarks>
/// A prompt of zero is not a report, because there is no request without one, and a provider
/// sending it means we read the wrong field.
/// </remarks>
/// <param name="promptTokens">What the request carried. Has to be greater than zero.</param>
/// <returns>The usage.</returns>
public static TokenUsage Of(int promptTokens)
{
ArgumentOutOfRangeException.ThrowIfNegativeOrZero(promptTokens);
return new()
{
IsKnown = true,
PromptTokens = promptTokens,
};
}
/// <summary>
/// States what a provider reported, or unknown when it reported nothing usable.
/// </summary>
/// <remarks>
/// For the reading side, where the number comes out of somebody else's JSON: a missing field, a
/// null, or a zero all mean the same thing there, and none of them is worth an exception.
/// </remarks>
/// <param name="promptTokens">What the request carried, as the provider stated it.</param>
/// <returns>The usage, or UNKNOWN.</returns>
public static TokenUsage OfReported(int? promptTokens) => promptTokens is > 0 ? Of(promptTokens.Value) : UNKNOWN;
}