Added the exact token count where the provider reports it (#989)
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions

Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
j-erlerandThorsten Sommer authored and GitHub committed 2026-09-23 20:17:02 +02:00
1 parent 82986afe62
commit be1e6532fb
34 files changed
+1360 -104

No files matched your search

@@ -23,6 +23,17 @@ public record ChatCompletionAPIRequest(
[JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)]
public bool? ParallelToolCalls { get; init; }
/// <summary>
/// Asks a streamed request to end with what it cost.
/// </summary>
/// <remarks>
/// Derived rather than set, so that every provider which builds one of these asks for it
/// without having to know that it exists. A request which is not streamed carries no such
/// line, and then the block would only be a field the provider has to ignore.
/// </remarks>
[JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)]
public ChatCompletionStreamOptions? StreamOptions => this.Stream ? ChatCompletionStreamOptions.INCLUDE_USAGE : null;
// Attention: The "required" modifier is not supported for [JsonExtensionData].
[JsonExtensionData]
@@ -15,12 +15,27 @@ public record ChatCompletionDeltaStreamLine(string Id, string Object, uint Creat
{
}
/// <summary>
/// What the provider says the request cost, on the one line which carries it.
/// </summary>
/// <remarks>
/// Not a positional parameter: every provider builds an empty line through the constructor
/// above, and a further parameter would change all of those call sites for a value none of
/// them has. Providers send this block only when the request asked for it, and then on a final
/// line of its own which carries no choices -- which is why the usage is read apart from the
/// content rather than next to it.
/// </remarks>
public ChatCompletionUsage? Usage { get; init; }
/// <inheritdoc />
public bool ContainsContent() => this.Choices.Count > 0;
/// <inheritdoc />
public ContentStreamChunk GetContent() => new(this.Choices[0].Delta.Content, []);
/// <inheritdoc />
public TokenUsage GetUsage() => this.Usage?.ToTokenUsage() ?? TokenUsage.UNKNOWN;
#region Implementation of IAnnotationStreamLine
//
@@ -0,0 +1,18 @@
namespace AIStudio.Provider.OpenAI;
/// <summary>
/// What a streamed chat completion should report beyond its content.
/// </summary>
/// <remarks>
/// An OpenAI-compatible provider says nothing about what a streamed request cost unless it is asked
/// to. Without this block, the stream simply ends and the only token number anybody ever sees is
/// the one AI Studio estimated for itself.
/// </remarks>
/// <param name="IncludeUsage">Whether the stream should end with a line stating the token usage.</param>
public sealed record ChatCompletionStreamOptions(bool IncludeUsage)
{
/// <summary>
/// Asks for the usage line.
/// </summary>
public static readonly ChatCompletionStreamOptions INCLUDE_USAGE = new(true);
}
@@ -5,15 +5,20 @@ namespace AIStudio.Provider.OpenAI;
/// </summary>
/// <param name="TextDelta">The text this line carried, empty when it carried none.</param>
/// <param name="Sources">The sources this line announced, empty when it announced none.</param>
public readonly record struct ChatCompletionStreamPart(string TextDelta, IList<ISource> Sources)
/// <param name="Usage">What the provider said the request cost, unknown on every line but the one which carries it.</param>
public readonly record struct ChatCompletionStreamPart(string TextDelta, IList<ISource> Sources, TokenUsage Usage = default)
{
/// <summary>
/// The part of a line which says nothing to the user, such as a fragment of a tool call.
/// </summary>
public static ChatCompletionStreamPart Nothing => new(string.Empty, []);
/// <summary>
/// Whether this part has anything to show at all.
/// </summary>
/// <remarks>
/// The usage is not part of that: it is nothing to show, and whether it is passed on at all is
/// the adapter's decision, which knows which round this is.
/// </remarks>
public bool HasContent => this.TextDelta.Length > 0 || this.Sources.Count > 0;
}
@@ -60,10 +60,16 @@ public sealed class ChatCompletionToolCallAccumulator(Func<ServerSentEvent, ILis
// list either way.
//
var sources = readSources?.Invoke(serverSentEvent) ?? [];
//
// The usage arrives on a line without choices at most providers, and next to the last
// piece of text at some. Read before the choices are looked at, it is not lost in either.
//
var usage = line?.Usage?.ToTokenUsage() ?? TokenUsage.UNKNOWN;
var delta = line?.Choices?.FirstOrDefault()?.Delta;
if (delta is null)
return WithSources(string.Empty, sources);
return WithSources(string.Empty, sources, usage);
this.hasReadAnything = true;
@@ -80,10 +86,10 @@ public sealed class ChatCompletionToolCallAccumulator(Func<ServerSentEvent, ILis
var textDelta = delta.Content;
if (textDelta.Length is 0)
return WithSources(string.Empty, sources);
return WithSources(string.Empty, sources, usage);
this.text.Append(textDelta);
return new ChatCompletionStreamPart(textDelta, sources);
return new ChatCompletionStreamPart(textDelta, sources, usage);
}
/// <summary>
@@ -183,10 +189,10 @@ public sealed class ChatCompletionToolCallAccumulator(Func<ServerSentEvent, ILis
private static string? Coalesce(string? value) => string.IsNullOrWhiteSpace(value) ? null : value;
/// <summary>
/// A part for a line which brought sources but no text, or nothing at all.
/// A part for a line which brought sources or a usage but no text, or nothing at all.
/// </summary>
private static ChatCompletionStreamPart WithSources(string text, IList<ISource> sources)
=> sources.Count is 0 ? ChatCompletionStreamPart.Nothing : new ChatCompletionStreamPart(text, sources);
private static ChatCompletionStreamPart WithSources(string text, IList<ISource> sources, TokenUsage usage)
=> sources.Count is 0 && !usage.IsKnown ? ChatCompletionStreamPart.Nothing : new ChatCompletionStreamPart(text, sources, usage);
/// <summary>
/// One tool call while its fragments are still arriving.
@@ -16,7 +16,7 @@ namespace AIStudio.Provider.OpenAI;
public sealed class ChatCompletionToolCallingAdapter<TRequest>(
Func<TextMessage, IDictionary<string, object>, IList<object>?, Task<TRequest>> requestFactory,
TextMessage systemPrompt, IDictionary<string, object> apiParameters,
IList<object> providerTools,
IList<object> providerTools, bool mayAskForSequentialToolCalls,
IReadOnlyList<(ToolDefinition Definition, IToolImplementation Implementation)> runnableTools,
Func<ChatCompletionAPIRequest, CancellationToken, IAsyncEnumerable<ServerSentEvent>> streamRequestAsync,
Func<ServerSentEvent, IList<ISource>> readSources,
@@ -49,11 +49,23 @@ public sealed class ChatCompletionToolCallingAdapter<TRequest>(
//
// AI Studio runs tool calls one after another, so asking for parallel calls would
// only produce work it then has to serialize anyway. Requests without tools omit the
// parameter because some providers reject it then.
// parameter because some providers reject it then. So does every request to a provider
// which rejects the parameter altogether: its models may then ask for several calls at
// once, and the loop works through them one by one, checking the limits per call.
//
ParallelToolCalls = requestDtoBase.Tools is null ? null : false,
ParallelToolCalls = requestDtoBase.Tools is null || !mayAskForSequentialToolCalls ? null : false,
};
//
// Only the first round passes on what its request cost. Its prompt is the conversation up
// to the question, which is exactly what the next question will be sent after. Every later
// round carries the tool calls and their results on top, and none of that is sent again
// once the answer stands -- a report of such a round would count a chat far larger than
// the one the next request carries. What the answer adds, all rounds of text together, is
// counted from its text afterwards, cf. ReportedHistory.
//
var passesOnUsage = this.internalMessages.Count is 0;
//
// The text goes out while it is being written; the tool calls are put back together
// behind it, fragment by fragment.
@@ -62,8 +74,9 @@ public sealed class ChatCompletionToolCallingAdapter<TRequest>(
await foreach (var serverSentEvent in streamRequestAsync(requestDto, token))
{
var part = accumulator.Process(serverSentEvent);
if (part.HasContent)
yield return ToolCallingStreamEvent.TextDelta(new ContentStreamChunk(part.TextDelta, part.Sources));
var usage = passesOnUsage ? part.Usage : TokenUsage.UNKNOWN;
if (part.HasContent || usage.IsKnown)
yield return ToolCallingStreamEvent.TextDelta(new ContentStreamChunk(part.TextDelta, part.Sources, Usage: usage));
}
var message = accumulator.Build();
@@ -11,4 +11,15 @@ namespace AIStudio.Provider.OpenAI;
/// </remarks>
/// <param name="Id">The ID of the answer.</param>
/// <param name="Choices">The choices this line adds to.</param>
public sealed record ChatCompletionToolStreamLine(string? Id, IList<ChatCompletionToolStreamChoice?>? Choices);
public sealed record ChatCompletionToolStreamLine(string? Id, IList<ChatCompletionToolStreamChoice?>? Choices)
{
/// <summary>
/// What the provider says the request cost, on the one line which carries it.
/// </summary>
/// <remarks>
/// The same block the plain text path reads, and asked for the same way: every streamed
/// ChatCompletionAPIRequest asks for it, the requests of the tool rounds included. Not a
/// positional parameter, because nobody but the serializer ever builds this line.
/// </remarks>
public ChatCompletionUsage? Usage { get; init; }
}
@@ -0,0 +1,31 @@
// ReSharper disable ClassNeverInstantiated.Global
namespace AIStudio.Provider.OpenAI;
/// <summary>
/// What an OpenAI-compatible provider reports a chat completion cost.
/// </summary>
/// <remarks>
/// The number is optional because this is somebody else's JSON: the block arrives only when the
/// request asked for it, and the providers which follow the shape loosely leave fields out. Reading
/// it is one thing, believing it another -- TokenUsage.OfReported decides that.
///
/// The block states more than this, the completion and its reasoning share among it. Those are left
/// unread on purpose, for the reason given at TokenUsage: no later request carries them.
/// </remarks>
public sealed record ChatCompletionUsage
{
/// <summary>
/// What everything sent to the model cost.
/// </summary>
public int? PromptTokens { get; init; }
/// <summary>
/// States what this block reports, as far as it can be believed.
/// </summary>
/// <remarks>
/// The one way from the wire to a usage, shared by every stream line which carries this block,
/// so that what counts as believable is decided in a single place.
/// </remarks>
/// <returns>The usage, or TokenUsage.UNKNOWN when the block states nothing usable.</returns>
public TokenUsage ToTokenUsage() => TokenUsage.OfReported(this.PromptTokens);
}