mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-05 08:49:40 +00:00
Added the exact token count where the provider reports it (#989)
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
1 parent
82986afe62
commit
be1e6532fb
34 files changed
+1360
-104
No files matched your search
@@ -0,0 +1,129 @@
|
||||
using System.Text.Json;
|
||||
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
|
||||
namespace AIStudio.Tests.Provider;
|
||||
|
||||
/// <summary>
|
||||
/// Checks that what a provider says a request cost is read off the stream, and asked for.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Both halves matter and neither is visible from the other: an OpenAI-compatible provider says
|
||||
/// nothing about the cost of a streamed request unless the request asks for it, and the line it
|
||||
/// then sends carries no content, so the reading side has to look for it apart from the text.
|
||||
/// Get either half wrong and the app silently keeps estimating, which looks exactly like a
|
||||
/// provider which reports nothing.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ChatCompletionUsageTests
|
||||
{
|
||||
/// <summary>
|
||||
/// The last line of a streamed answer at a provider which was asked for the usage.
|
||||
/// </summary>
|
||||
private const string USAGE_LINE =
|
||||
"""
|
||||
{"id":"chatcmpl-1","object":"chat.completion.chunk","created":1,"model":"gpt-5","choices":[],"usage":{"prompt_tokens":1200,"completion_tokens":345,"total_tokens":1545}}
|
||||
""";
|
||||
|
||||
/// <summary>
|
||||
/// An ordinary line carrying a piece of the answer.
|
||||
/// </summary>
|
||||
private const string CONTENT_LINE =
|
||||
"""
|
||||
{"id":"chatcmpl-1","object":"chat.completion.chunk","created":1,"model":"gpt-5","choices":[{"index":0,"delta":{"content":"Hi"}}]}
|
||||
""";
|
||||
|
||||
[Test]
|
||||
public void TheFinalLineStatesWhatTheRequestCost()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(USAGE_LINE, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.True);
|
||||
Assert.That(line.GetUsage().PromptTokens, Is.EqualTo(1200));
|
||||
|
||||
//
|
||||
// The line which carries the usage carries no answer, which is why it has to be read
|
||||
// before the content check drops it:
|
||||
//
|
||||
Assert.That(line.ContainsContent(), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ALineOfTheAnswerStatesNoCost()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(CONTENT_LINE, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.False);
|
||||
Assert.That(line.ContainsContent(), Is.True);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AStreamedRequestAsksForTheUsage()
|
||||
{
|
||||
var request = new ChatCompletionAPIRequest("gpt-5", [], true);
|
||||
var json = JsonSerializer.Serialize(request, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.That(json, Does.Contain("""
|
||||
"stream_options":{"include_usage":true}
|
||||
"""));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ARequestWhichIsNotStreamedDoesNot()
|
||||
{
|
||||
var request = new ChatCompletionAPIRequest("gpt-5", [], false);
|
||||
var json = JsonSerializer.Serialize(request, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.That(json, Does.Not.Contain("stream_options"));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// A provider which sends the block but fills in nothing usable states nothing.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Several OpenAI-compatible servers send an empty or zeroed usage block on every line while
|
||||
/// streaming and the real numbers only at the end. Reading a zero as a fact would replace an
|
||||
/// estimate with a statement that the conversation costs nothing.
|
||||
/// </remarks>
|
||||
[Test]
|
||||
public void AnEmptyUsageBlockStatesNothing()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(
|
||||
"""
|
||||
{"id":"chatcmpl-1","object":"chat.completion.chunk","created":1,"model":"gpt-5","choices":[],"usage":{"prompt_tokens":0,"completion_tokens":0}}
|
||||
""", ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.False);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// The line a real LM Studio server sends, taken off the wire.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// It carries far more than we read, and it shows why the prompt is all we take: 102 of the 116
|
||||
/// completion tokens are the model's reasoning, which the next request never carries. Counting
|
||||
/// the completion would have put the history at 133 tokens, when what travels on is the prompt
|
||||
/// and an answer of a handful of tokens.
|
||||
/// </remarks>
|
||||
[Test]
|
||||
public void ARealServerLineIsRead()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(
|
||||
"""
|
||||
{"id":"chatcmpl-xb8mn282eff3tiu46xz8t3","object":"chat.completion.chunk","created":1789917744,"model":"google/gemma-4-12b-qat","system_fingerprint":"google/gemma-4-12b-qat","choices":[],"usage":{"prompt_tokens":17,"completion_tokens":116,"total_tokens":133,"completion_tokens_details":{"reasoning_tokens":102}}}
|
||||
""", ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.True);
|
||||
Assert.That(line.GetUsage().PromptTokens, Is.EqualTo(17));
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -183,7 +183,38 @@ public sealed class ChatCompletionToolCallAccumulatorTests
|
||||
|
||||
Assert.That(part.Sources.Select(x => x.URL), Is.EqualTo(new[] { "https://example.org/" }), "Whatever the provider announced on that line reaches the user with it.");
|
||||
}
|
||||
|
||||
|
||||
[Test]
|
||||
public void TheLineWithoutChoicesStatesWhatTheRequestCost()
|
||||
{
|
||||
var accumulator = new ChatCompletionToolCallAccumulator();
|
||||
var part = accumulator.Process(Event("""{"choices":[],"usage":{"prompt_tokens":1200,"completion_tokens":345,"total_tokens":1545}}"""));
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(part.Usage.IsKnown, Is.True, "The last line of the stream has no choices, and it must not be dropped for that.");
|
||||
Assert.That(part.Usage.PromptTokens, Is.EqualTo(1200));
|
||||
Assert.That(part.HasContent, Is.False, "It has nothing to show, though.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AUsageNextToTheLastTextIsReadAsWell()
|
||||
{
|
||||
//
|
||||
// Some providers put the usage on the line which carries the last piece of the answer
|
||||
// rather than on a line of its own.
|
||||
//
|
||||
var accumulator = new ChatCompletionToolCallAccumulator();
|
||||
var part = accumulator.Process(Event("""{"choices":[{"index":0,"delta":{"content":"Bye"}}],"usage":{"prompt_tokens":1200,"completion_tokens":2}}"""));
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(part.TextDelta, Is.EqualTo("Bye"));
|
||||
Assert.That(part.Usage.PromptTokens, Is.EqualTo(1200));
|
||||
});
|
||||
}
|
||||
|
||||
private static ChatCompletionResponseMessage? Read(params string[] data)
|
||||
{
|
||||
var accumulator = new ChatCompletionToolCallAccumulator();
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
using System.Runtime.CompilerServices;
|
||||
using System.Text.Json;
|
||||
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
|
||||
using Microsoft.Extensions.Logging.Abstractions;
|
||||
|
||||
namespace AIStudio.Tests.Provider.ToolCalling;
|
||||
|
||||
/// <summary>
|
||||
/// Checks what a round of a tool calling conversation asks for, and what it passes on.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Every round of a tool conversation is a request of its own, and every one of them reports what
|
||||
/// it cost. Only the first one describes what the next question will be sent after: every later
|
||||
/// round carries the tool calls and their results on top, none of which is sent again once the
|
||||
/// answer stands. Passing on the last report instead would put the chat at the size of everything
|
||||
/// the tools returned, which is the one number a person watching their context window must not
|
||||
/// see as exact.
|
||||
///
|
||||
/// What a round asks for is one tool call at a time, wherever the provider lets it ask: a provider
|
||||
/// which rejects the question fails the whole request, so it is not asked at all.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ChatCompletionToolCallingAdapterTests
|
||||
{
|
||||
private const string FIRST_ROUND_USAGE = """{"choices":[],"usage":{"prompt_tokens":1200,"completion_tokens":20}}""";
|
||||
|
||||
private const string SECOND_ROUND_USAGE = """{"choices":[],"usage":{"prompt_tokens":9800,"completion_tokens":150}}""";
|
||||
|
||||
[Test]
|
||||
public async Task OnlyTheFirstRoundPassesOnWhatItsRequestCost()
|
||||
{
|
||||
var adapter = Adapter(
|
||||
[
|
||||
"""{"choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"id":"call_1","type":"function","function":{"name":"web_search","arguments":"{\"query\":\"weather\"}"}}]}}]}""",
|
||||
FIRST_ROUND_USAGE,
|
||||
"[DONE]",
|
||||
],
|
||||
[
|
||||
"""{"choices":[{"index":0,"delta":{"content":"It is sunny."}}]}""",
|
||||
SECOND_ROUND_USAGE,
|
||||
"[DONE]",
|
||||
]);
|
||||
|
||||
var firstRound = await Usages(adapter);
|
||||
|
||||
//
|
||||
// What the loop does between two rounds: the model's turn and the tool's result become part
|
||||
// of the next request.
|
||||
//
|
||||
adapter.RecordAssistantTurn();
|
||||
adapter.RecordToolResult("call_1", "Sunny, 24 degrees.");
|
||||
|
||||
var secondRound = await Usages(adapter);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(firstRound, Is.EqualTo(new[] { 1200 }), "The first round's prompt is the conversation up to the question.");
|
||||
Assert.That(secondRound, Is.Empty, "The second round's prompt holds the tool result as well, which the next question is not sent with.");
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task ARoundWithoutToolCallsPassesItOnAsWell()
|
||||
{
|
||||
//
|
||||
// Offering tools does not mean the model uses them. Then the first round is the only one,
|
||||
// and its report is as good as the one of a request which offered none.
|
||||
//
|
||||
var adapter = Adapter(
|
||||
[
|
||||
"""{"choices":[{"index":0,"delta":{"content":"Hello."}}]}""",
|
||||
FIRST_ROUND_USAGE,
|
||||
"[DONE]",
|
||||
]);
|
||||
|
||||
Assert.That(await Usages(adapter), Is.EqualTo(new[] { 1200 }));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task ARoundWhichOffersToolsAsksForOneCallAtATime()
|
||||
{
|
||||
Assert.That(await SentRequest(mayAskForSequentialToolCalls: true, includeTools: true), Does.Contain("\"parallel_tool_calls\":false"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task AProviderWhichRejectsTheQuestionIsNotAskedIt()
|
||||
{
|
||||
//
|
||||
// Hugging Face answers the question with a bad request. Its models may then ask for several
|
||||
// calls at once, which the loop works through one by one anyway.
|
||||
//
|
||||
Assert.That(await SentRequest(mayAskForSequentialToolCalls: false, includeTools: true), Does.Not.Contain("parallel_tool_calls"));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public async Task ARoundWithoutToolsDoesNotAskAboutToolCalls()
|
||||
{
|
||||
Assert.That(await SentRequest(mayAskForSequentialToolCalls: true, includeTools: false), Does.Not.Contain("parallel_tool_calls"));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Runs one round and returns the request it sent, as it goes over the wire.
|
||||
/// </summary>
|
||||
private static async Task<string> SentRequest(bool mayAskForSequentialToolCalls, bool includeTools)
|
||||
{
|
||||
ChatCompletionAPIRequest? sent = null;
|
||||
var adapter = Adapter(mayAskForSequentialToolCalls, request => sent = request, ["[DONE]"]);
|
||||
await foreach (var _ in adapter.ExecuteRoundAsync(null, includeTools))
|
||||
{
|
||||
}
|
||||
|
||||
return JsonSerializer.Serialize(sent, ProviderJsonOptions.OPTIONS);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Runs the next round and returns the prompt of every usage it passed on.
|
||||
/// </summary>
|
||||
private static async Task<List<int>> Usages(ChatCompletionToolCallingAdapter<ChatCompletionAPIRequest> adapter)
|
||||
{
|
||||
var usages = new List<int>();
|
||||
await foreach (var streamEvent in adapter.ExecuteRoundAsync(null, true))
|
||||
if (streamEvent.Delta is { Usage.IsKnown: true } delta)
|
||||
usages.Add(delta.Usage.PromptTokens);
|
||||
|
||||
return usages;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Builds an adapter whose requests are answered by the given rounds, one after another.
|
||||
/// </summary>
|
||||
private static ChatCompletionToolCallingAdapter<ChatCompletionAPIRequest> Adapter(params string[][] rounds) => Adapter(true, _ => { }, rounds);
|
||||
|
||||
/// <summary>
|
||||
/// Builds an adapter whose requests are answered by the given rounds, and which hands every
|
||||
/// request it sends to the given observer.
|
||||
/// </summary>
|
||||
private static ChatCompletionToolCallingAdapter<ChatCompletionAPIRequest> Adapter(bool mayAskForSequentialToolCalls, Action<ChatCompletionAPIRequest> sent, params string[][] rounds)
|
||||
{
|
||||
var nextRound = 0;
|
||||
return new(
|
||||
(_, _, tools) => Task.FromResult(new ChatCompletionAPIRequest("model-a", [], true) { Tools = tools }),
|
||||
new TextMessage("You are a helpful assistant.", "system"),
|
||||
new Dictionary<string, object>(),
|
||||
[],
|
||||
mayAskForSequentialToolCalls,
|
||||
[],
|
||||
(request, token) =>
|
||||
{
|
||||
sent(request);
|
||||
return Lines(rounds[nextRound++], token);
|
||||
},
|
||||
_ => [],
|
||||
NullLogger.Instance);
|
||||
}
|
||||
|
||||
private static async IAsyncEnumerable<ServerSentEvent> Lines(string[] data, [EnumeratorCancellation] CancellationToken token = default)
|
||||
{
|
||||
foreach (var line in data)
|
||||
{
|
||||
token.ThrowIfCancellationRequested();
|
||||
yield return new ServerSentEvent($"data: {line}", line);
|
||||
}
|
||||
|
||||
await Task.CompletedTask;
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user