Tell embedding providers which tokenizer their model uses

This commit is contained in:
Thorsten Sommer committed 2026-09-12 20:16:27 +02:00
1 parent 5326a0bf6c
commit 672237a225
10 files changed
+116 -64

No files matched your search

@@ -23,12 +23,19 @@ public sealed class OpenAIEmbeddingFamily : ModelFamily
/// <inheritdoc />
public override ModelSource Source => new("https://platform.openai.com/docs/guides/embeddings", new DateOnly(2026, 9, 11), "The app lists these under IProvider.GetEmbeddingModels, which is where the statement that they embed comes from.");
/// <inheritdoc />
public override IReadOnlyList<ModelSource> FurtherSources =>
[
new("https://github.com/openai/tiktoken/blob/main/tiktoken/model.py", new DateOnly(2026, 9, 12), "OpenAI's own mapping from model names to encodings. It names all three of these models -- text-embedding-3-small, text-embedding-3-large and text-embedding-ada-002 -- and maps every one of them to cl100k_base rather than to the newer o200k_base of the chat models.")
];
/// <inheritdoc />
protected override void Declare(ModelFamilyBuilder builder)
{
builder.Rule("text-embedding-3").AsPrefix()
.Capabilities(TEXT_INPUT | EMBEDDING)
.Kind(ModelKind.EMBEDDING);
.Kind(ModelKind.EMBEDDING)
.Tokenizer(TokenizerKind.TIKTOKEN, "cl100k_base");
builder.Rule("text-embedding-ada").AsPrefix().Inherits();
}