mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-10 09:13:47 +00:00
Fixed Hugging Face rejecting chats which offer tools
This commit is contained in:
5 files changed
+73
-9
No files matched your search
@@ -1244,6 +1244,7 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
/// <param name="systemPromptRole">The system prompt role to use.</param>
|
||||
/// <param name="requestPath">The request path, relative to the provider base URL.</param>
|
||||
/// <param name="headersAction">Optional additional headers to add.</param>
|
||||
/// <param name="mayAskForSequentialToolCalls">Whether a request which offers tools may ask for one call at a time. False for a provider which rejects the parallel_tool_calls parameter.</param>
|
||||
/// <param name="token">The cancellation token.</param>
|
||||
/// <typeparam name="TRequest">The request DTO type.</typeparam>
|
||||
/// <typeparam name="TDelta">The delta stream line type.</typeparam>
|
||||
@@ -1260,6 +1261,7 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
string systemPromptRole = "system",
|
||||
string requestPath = "chat/completions",
|
||||
Action<HttpRequestHeaders>? headersAction = null,
|
||||
bool mayAskForSequentialToolCalls = true,
|
||||
[EnumeratorCancellation] CancellationToken token = default)
|
||||
where TRequest : ChatCompletionAPIRequest
|
||||
where TDelta : IResponseStreamLine
|
||||
@@ -1298,7 +1300,7 @@ public abstract class BaseProvider : IProvider, ISecretId
|
||||
if (runnableTools.Count > 0)
|
||||
{
|
||||
var adapter = new ChatCompletionToolCallingAdapter<TRequest>(requestFactory, systemPrompt, apiParameters,
|
||||
runnableTools.Select(x => ProviderToolAdapters.ToChatCompletionTool(x.Definition)).ToList(), runnableTools,
|
||||
runnableTools.Select(x => ProviderToolAdapters.ToChatCompletionTool(x.Definition)).ToList(), mayAskForSequentialToolCalls, runnableTools,
|
||||
(requestDto, requestToken) => this.StreamChatCompletionRequest(requestDto, providerName, requestPath, requestedSecret, headersAction, requestToken),
|
||||
ChatCompletionSourceReader.Read<TDelta, TAnnotation>,
|
||||
this.logger);
|
||||
|
||||
@@ -184,6 +184,15 @@ public sealed class ProviderHuggingFace : BaseProvider
|
||||
AdditionalApiParameters = apiParameters
|
||||
};
|
||||
},
|
||||
|
||||
//
|
||||
// Hugging Face answers parallel_tool_calls=false with a bad request, "feature
|
||||
// not currently supported", and its specification of the chat completion does
|
||||
// not list the parameter at all -- read on 2026-09-23 at
|
||||
// https://huggingface.co/docs/inference-providers/tasks/chat-completion. Asking
|
||||
// for it would cost every chat which offers tools its answer:
|
||||
//
|
||||
mayAskForSequentialToolCalls: false,
|
||||
token: token))
|
||||
yield return content;
|
||||
}
|
||||
|
||||
@@ -16,7 +16,7 @@ namespace AIStudio.Provider.OpenAI;
|
||||
public sealed class ChatCompletionToolCallingAdapter<TRequest>(
|
||||
Func<TextMessage, IDictionary<string, object>, IList<object>?, Task<TRequest>> requestFactory,
|
||||
TextMessage systemPrompt, IDictionary<string, object> apiParameters,
|
||||
IList<object> providerTools,
|
||||
IList<object> providerTools, bool mayAskForSequentialToolCalls,
|
||||
IReadOnlyList<(ToolDefinition Definition, IToolImplementation Implementation)> runnableTools,
|
||||
Func<ChatCompletionAPIRequest, CancellationToken, IAsyncEnumerable<ServerSentEvent>> streamRequestAsync,
|
||||
Func<ServerSentEvent, IList<ISource>> readSources,
|
||||
@@ -49,9 +49,11 @@ public sealed class ChatCompletionToolCallingAdapter<TRequest>(
|
||||
//
|
||||
// AI Studio runs tool calls one after another, so asking for parallel calls would
|
||||
// only produce work it then has to serialize anyway. Requests without tools omit the
|
||||
// parameter because some providers reject it then.
|
||||
// parameter because some providers reject it then. So does every request to a provider
|
||||
// which rejects the parameter altogether: its models may then ask for several calls at
|
||||
// once, and the loop works through them one by one, checking the limits per call.
|
||||
//
|
||||
ParallelToolCalls = requestDtoBase.Tools is null ? null : false,
|
||||
ParallelToolCalls = requestDtoBase.Tools is null || !mayAskForSequentialToolCalls ? null : false,
|
||||
};
|
||||
|
||||
//
|
||||
|
||||
@@ -86,7 +86,7 @@ public sealed class ConversationTokenCounter(RustService rustService, ILogger<Co
|
||||
var growing = new Dictionary<string, int>(StringComparer.Ordinal);
|
||||
var historyTokens = 0;
|
||||
var toolTokens = 0;
|
||||
var draftTokens = 0;
|
||||
int draftTokens;
|
||||
|
||||
try
|
||||
{
|
||||
|
||||
Reference in new issue
Block a user