mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-09 00:29:40 +00:00
Added the exact token count where the provider reports it (#989)
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
1 parent
82986afe62
commit
be1e6532fb
34 files changed
+1360
-104
No files matched your search
@@ -0,0 +1,129 @@
|
||||
using System.Text.Json;
|
||||
|
||||
using AIStudio.Provider;
|
||||
using AIStudio.Provider.OpenAI;
|
||||
|
||||
namespace AIStudio.Tests.Provider;
|
||||
|
||||
/// <summary>
|
||||
/// Checks that what a provider says a request cost is read off the stream, and asked for.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Both halves matter and neither is visible from the other: an OpenAI-compatible provider says
|
||||
/// nothing about the cost of a streamed request unless the request asks for it, and the line it
|
||||
/// then sends carries no content, so the reading side has to look for it apart from the text.
|
||||
/// Get either half wrong and the app silently keeps estimating, which looks exactly like a
|
||||
/// provider which reports nothing.
|
||||
/// </remarks>
|
||||
[TestFixture]
|
||||
public sealed class ChatCompletionUsageTests
|
||||
{
|
||||
/// <summary>
|
||||
/// The last line of a streamed answer at a provider which was asked for the usage.
|
||||
/// </summary>
|
||||
private const string USAGE_LINE =
|
||||
"""
|
||||
{"id":"chatcmpl-1","object":"chat.completion.chunk","created":1,"model":"gpt-5","choices":[],"usage":{"prompt_tokens":1200,"completion_tokens":345,"total_tokens":1545}}
|
||||
""";
|
||||
|
||||
/// <summary>
|
||||
/// An ordinary line carrying a piece of the answer.
|
||||
/// </summary>
|
||||
private const string CONTENT_LINE =
|
||||
"""
|
||||
{"id":"chatcmpl-1","object":"chat.completion.chunk","created":1,"model":"gpt-5","choices":[{"index":0,"delta":{"content":"Hi"}}]}
|
||||
""";
|
||||
|
||||
[Test]
|
||||
public void TheFinalLineStatesWhatTheRequestCost()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(USAGE_LINE, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.True);
|
||||
Assert.That(line.GetUsage().PromptTokens, Is.EqualTo(1200));
|
||||
|
||||
//
|
||||
// The line which carries the usage carries no answer, which is why it has to be read
|
||||
// before the content check drops it:
|
||||
//
|
||||
Assert.That(line.ContainsContent(), Is.False);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ALineOfTheAnswerStatesNoCost()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(CONTENT_LINE, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.False);
|
||||
Assert.That(line.ContainsContent(), Is.True);
|
||||
});
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void AStreamedRequestAsksForTheUsage()
|
||||
{
|
||||
var request = new ChatCompletionAPIRequest("gpt-5", [], true);
|
||||
var json = JsonSerializer.Serialize(request, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.That(json, Does.Contain("""
|
||||
"stream_options":{"include_usage":true}
|
||||
"""));
|
||||
}
|
||||
|
||||
[Test]
|
||||
public void ARequestWhichIsNotStreamedDoesNot()
|
||||
{
|
||||
var request = new ChatCompletionAPIRequest("gpt-5", [], false);
|
||||
var json = JsonSerializer.Serialize(request, ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.That(json, Does.Not.Contain("stream_options"));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// A provider which sends the block but fills in nothing usable states nothing.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Several OpenAI-compatible servers send an empty or zeroed usage block on every line while
|
||||
/// streaming and the real numbers only at the end. Reading a zero as a fact would replace an
|
||||
/// estimate with a statement that the conversation costs nothing.
|
||||
/// </remarks>
|
||||
[Test]
|
||||
public void AnEmptyUsageBlockStatesNothing()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(
|
||||
"""
|
||||
{"id":"chatcmpl-1","object":"chat.completion.chunk","created":1,"model":"gpt-5","choices":[],"usage":{"prompt_tokens":0,"completion_tokens":0}}
|
||||
""", ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.False);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// The line a real LM Studio server sends, taken off the wire.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// It carries far more than we read, and it shows why the prompt is all we take: 102 of the 116
|
||||
/// completion tokens are the model's reasoning, which the next request never carries. Counting
|
||||
/// the completion would have put the history at 133 tokens, when what travels on is the prompt
|
||||
/// and an answer of a handful of tokens.
|
||||
/// </remarks>
|
||||
[Test]
|
||||
public void ARealServerLineIsRead()
|
||||
{
|
||||
var line = JsonSerializer.Deserialize<ChatCompletionDeltaStreamLine>(
|
||||
"""
|
||||
{"id":"chatcmpl-xb8mn282eff3tiu46xz8t3","object":"chat.completion.chunk","created":1789917744,"model":"google/gemma-4-12b-qat","system_fingerprint":"google/gemma-4-12b-qat","choices":[],"usage":{"prompt_tokens":17,"completion_tokens":116,"total_tokens":133,"completion_tokens_details":{"reasoning_tokens":102}}}
|
||||
""", ProviderJsonOptions.OPTIONS);
|
||||
|
||||
Assert.Multiple(() =>
|
||||
{
|
||||
Assert.That(line!.GetUsage().IsKnown, Is.True);
|
||||
Assert.That(line.GetUsage().PromptTokens, Is.EqualTo(17));
|
||||
});
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user