Name the tokenizer a model uses, where its vendor does

This commit is contained in:
Thorsten Sommer committed 2026-09-12 20:01:29 +02:00
1 parent 2509fc09d1
commit 5326a0bf6c
16 files changed
+521 -285

No files matched your search

@@ -52,7 +52,8 @@ public sealed class ClaudeFamily : ModelFamily
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://platform.claude.com/docs/en/build-with-claude/vision", new DateOnly(2026, 9, 12), "The vision page gives the image limit as a rule rather than as a number per model: 100 images per request on the API for models with a 200k-token context window, 600 per request for all other models. The 20 it also names belongs to claude.ai, not to the API.")
new("https://platform.claude.com/docs/en/build-with-claude/vision", new DateOnly(2026, 9, 12), "The vision page gives the image limit as a rule rather than as a number per model: 100 images per request on the API for models with a 200k-token context window, 600 per request for all other models. The 20 it also names belongs to claude.ai, not to the API."),
new("https://platform.claude.com/docs/en/build-with-claude/token-counting", new DateOnly(2026, 9, 12), "Anthropic publishes no tokenizer file at all; they count through the /v1/messages/count_tokens endpoint instead. The same page warns that Claude 4.7 and later use a newer tokenizer, on which the same text counts roughly 30 percent higher -- so AI Studio's built-in estimate is further off for those models than for the older ones.")
];
/// <inheritdoc />
@@ -84,7 +85,8 @@ public sealed class ClaudeFamily : ModelFamily
builder.Rule("claude").AsSegment()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT | FUNCTION_CALLING)
.Apis(CHAT_COMPLETION_API)
.ContextWindow(STANDARD_WINDOW);
.ContextWindow(STANDARD_WINDOW)
.Tokenizer(TokenizerKind.PROVIDER_API, "/v1/messages/count_tokens");
//
// The 3.x models say nothing beyond the shape above, so nothing is written for them: the
@@ -29,7 +29,8 @@ public sealed class GeminiFamily : ModelFamily
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://ai.google.dev/gemini-api/docs/image-understanding", new DateOnly(2026, 9, 12), "States one number for the whole family: \"Gemini models support a maximum of 3,600 image files per request.\" The 20 MB it also names is a limit on the request body rather than on the number of images.")
new("https://ai.google.dev/gemini-api/docs/image-understanding", new DateOnly(2026, 9, 12), "States one number for the whole family: \"Gemini models support a maximum of 3,600 image files per request.\" The 20 MB it also names is a limit on the request body rather than on the number of images."),
new("https://ai.google.dev/gemini-api/docs/tokens", new DateOnly(2026, 9, 12), "Google publishes no tokenizer file for Gemini. Counting happens through the countTokens method of the API, which returns the number of tokens of the input alone.")
];
/// <inheritdoc />
@@ -44,7 +45,8 @@ public sealed class GeminiFamily : ModelFamily
builder.Rule("gemini").AsSegment()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | AUDIO_INPUT | SPEECH_INPUT | VIDEO_INPUT | TEXT_OUTPUT | FUNCTION_CALLING)
.Apis(CHAT_COMPLETION_API)
.Images(maxPerRequest: 3_600);
.Images(maxPerRequest: 3_600)
.Tokenizer(TokenizerKind.PROVIDER_API, "countTokens");
// The one Gemini which only ever read text and images:
builder.Rule("gemini-1.0-pro-vision").AsPrefix()
@@ -13,12 +13,19 @@ public sealed class Gpt35Family : ModelFamily
/// <inheritdoc />
public override ModelSource Source => new("https://platform.openai.com/docs/models", new DateOnly(2026, 9, 11), "Ported unchanged from the rules in ProviderExtensions.OpenAI.cs: text in, text out, no tools and no images.");
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://github.com/openai/tiktoken/blob/main/tiktoken/model.py", new DateOnly(2026, 9, 12), "OpenAI's own mapping from model names to encodings. It maps \"gpt-3.5\", \"gpt-3.5-turbo\" and the prefix \"gpt-3.5-turbo-\" to cl100k_base.")
];
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
builder.Rule("gpt-3.5").AsPrefix()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT)
.Apis(CHAT_COMPLETION_API);
.Apis(CHAT_COMPLETION_API)
.Tokenizer(TokenizerKind.TIKTOKEN, "cl100k_base");
//
// The odd one out, and kept odd on purpose: the previous rules put this one model on the
@@ -28,6 +35,7 @@ public sealed class Gpt35Family : ModelFamily
//
builder.Rule("gpt-3.5-turbo").AsExact()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT)
.Apis(RESPONSES_API);
.Apis(RESPONSES_API)
.Tokenizer(TokenizerKind.TIKTOKEN, "cl100k_base");
}
}
@@ -18,13 +18,20 @@ public sealed class Gpt4Family : ModelFamily
/// <inheritdoc />
public override ModelSource Source => new("https://developers.openai.com/api/docs/models/gpt-4-turbo", new DateOnly(2026, 9, 12), "Capabilities ported unchanged from ProviderExtensions.OpenAI.cs: GPT-4 is text only, Turbo adds images and tool calling. The windows are the documented 8,192 tokens of GPT-4 and the 128,000 Turbo raised it to.");
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://github.com/openai/tiktoken/blob/main/tiktoken/model.py", new DateOnly(2026, 9, 12), "OpenAI's own mapping from model names to encodings. It maps \"gpt-4\" and the prefix \"gpt-4-\" to cl100k_base, so Turbo uses it too -- the newer o200k_base begins with the 4o line, which is a family of its own here.")
];
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
builder.Rule("gpt-4").AsPrefix()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT)
.Apis(RESPONSES_API)
.ContextWindow(8_192);
.ContextWindow(8_192)
.Tokenizer(TokenizerKind.TIKTOKEN, "cl100k_base");
builder.Rule("gpt-4-turbo").AsPrefix().Inherits()
.Capabilities(MULTIPLE_IMAGE_INPUT | FUNCTION_CALLING)
@@ -20,13 +20,20 @@ public sealed class Gpt4oFamily : ModelFamily
/// <inheritdoc />
public override ModelSource Source => new("https://developers.openai.com/api/docs/models/gpt-4o", new DateOnly(2026, 9, 12), "The answer the previous rules gave these models through their fallback: images, tool calling, and web search on the Responses API. The model page states a 128,000 token window, which the minis and the search previews share.");
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://github.com/openai/tiktoken/blob/main/tiktoken/model.py", new DateOnly(2026, 9, 12), "OpenAI's own mapping from model names to encodings. It maps the prefix \"gpt-4o-\" to o200k_base, which covers the minis and the search previews as well.")
];
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
builder.Rule("gpt-4o").AsPrefix()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT | FUNCTION_CALLING | WEB_SEARCH)
.Apis(RESPONSES_API)
.ContextWindow(128_000);
.ContextWindow(128_000)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
//
// The search previews are the same generation and almost nothing like it: they search the
@@ -37,11 +44,13 @@ public sealed class Gpt4oFamily : ModelFamily
builder.Rule("gpt-4o-search-preview").AsExact()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT | WEB_SEARCH)
.Apis(CHAT_COMPLETION_API)
.ContextWindow(128_000);
.ContextWindow(128_000)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
builder.Rule("gpt-4o-mini-search-preview").AsExact()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT | WEB_SEARCH)
.Apis(CHAT_COMPLETION_API)
.ContextWindow(128_000);
.ContextWindow(128_000)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
}
}
@@ -26,6 +26,12 @@ public sealed class Gpt5Family : ModelFamily
/// <inheritdoc />
public override ModelSource Source => new("https://developers.openai.com/api/docs/models", new DateOnly(2026, 9, 12), "Capabilities ported unchanged from ProviderExtensions.OpenAI.cs, one rule per generation, except that the chat alias no longer inherits the reasoning it is named for not having. Context windows read per generation from the model pages below that URL.");
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://github.com/openai/tiktoken/blob/main/tiktoken/model.py", new DateOnly(2026, 9, 12), "OpenAI's own mapping from model names to encodings. It maps the prefix \"gpt-5\" to o200k_base, which covers every model of this line.")
];
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
@@ -38,7 +44,8 @@ public sealed class Gpt5Family : ModelFamily
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT | FUNCTION_CALLING | WEB_SEARCH)
.Apis(RESPONSES_API)
.Reasoning(ReasoningSupport.ALWAYS)
.ContextWindow(400_000);
.ContextWindow(400_000)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
//
// The alias for the model of this generation which does not reason. The previous rules had
@@ -22,6 +22,12 @@ public sealed class OSeriesFamily : ModelFamily
/// <inheritdoc />
public override ModelSource Source => new("https://developers.openai.com/api/docs/models", new DateOnly(2026, 9, 12), "Capabilities ported unchanged from ProviderExtensions.OpenAI.cs, one rule per generation and one per mini. The o1 and o3 pages state 200,000 tokens; the two cut-down minis have no page of their own, so no window is stated for them.");
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://github.com/openai/tiktoken/blob/main/tiktoken/model.py", new DateOnly(2026, 9, 12), "OpenAI's own mapping from model names to encodings. It maps \"o1\", \"o3\", \"o4-mini\" and the prefixes \"o1-\", \"o3-\" and \"o4-mini-\" to o200k_base, so the whole series shares one encoding.")
];
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
@@ -29,23 +35,27 @@ public sealed class OSeriesFamily : ModelFamily
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT | FUNCTION_CALLING)
.Apis(RESPONSES_API)
.Reasoning(ReasoningSupport.ALWAYS)
.ContextWindow(200_000);
.ContextWindow(200_000)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
builder.Rule("o1-mini").AsPrefix()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT)
.Apis(CHAT_COMPLETION_API)
.Reasoning(ReasoningSupport.ALWAYS);
.Reasoning(ReasoningSupport.ALWAYS)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
builder.Rule("o3").AsPrefix()
.Capabilities(TEXT_INPUT | MULTIPLE_IMAGE_INPUT | TEXT_OUTPUT | FUNCTION_CALLING | WEB_SEARCH)
.Apis(RESPONSES_API)
.Reasoning(ReasoningSupport.ALWAYS)
.ContextWindow(200_000);
.ContextWindow(200_000)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
builder.Rule("o3-mini").AsPrefix()
.Capabilities(TEXT_INPUT | TEXT_OUTPUT | FUNCTION_CALLING)
.Apis(RESPONSES_API)
.Reasoning(ReasoningSupport.ALWAYS);
.Reasoning(ReasoningSupport.ALWAYS)
.Tokenizer(TokenizerKind.TIKTOKEN, "o200k_base");
// The one mini which is not cut down: it is the o3 generation under another number.
builder.Rule("o4-mini").AsPrefix().InheritsFrom("o3");