mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-05 00:49:40 +00:00
Added tool calling support (#731)
Co-authored-by: krut_ni <nils.kruthoff@dlr.de> Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
266 files changed
+11186
-740
No files matched your search
@@ -1,11 +1,16 @@
|
||||
//! The HTTP endpoint for filtering text that does not arrive through a file stream.
|
||||
//! The HTTP endpoints for filtering text that does not arrive through a file stream.
|
||||
//!
|
||||
//! Files are filtered inside `extract_data`, where the runtime already sees every chunk.
|
||||
//! Web pages and retrieval contexts never pass through there — the app fetches and converts
|
||||
//! them itself — so they are handed over here instead. They are small enough that one
|
||||
//! request per text is cheaper than streaming.
|
||||
//!
|
||||
//! A single tool call, however, can produce many texts at once: a web search returns several
|
||||
//! pages, each with its own content, title, description, and authors. Those go through the
|
||||
//! batch endpoint, which filters them in one request instead of one round trip per field.
|
||||
|
||||
use crate::api_token::APIToken;
|
||||
use axum::http::StatusCode;
|
||||
use axum::Json;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
@@ -16,6 +21,11 @@ pub struct SanitizeRequest {
|
||||
pub text: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct SanitizeBatchRequest {
|
||||
pub texts: Vec<String>,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct SanitizeResponse {
|
||||
/// The text with the suspicious passages filtered out. Usable as it stands: filtering
|
||||
@@ -28,6 +38,13 @@ pub struct SanitizeResponse {
|
||||
pub redacted_count: usize,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub struct SanitizeBatchResponse {
|
||||
/// One result per requested text, in request order. The caller matches results to its own
|
||||
/// texts by index, so this list always has the same length as the request's.
|
||||
pub results: Vec<SanitizeResponse>,
|
||||
}
|
||||
|
||||
pub async fn sanitize(_token: APIToken, Json(request): Json<SanitizeRequest>) -> Json<SanitizeResponse> {
|
||||
let (sanitized_text, report) = sanitize_text(&request.text);
|
||||
|
||||
@@ -36,4 +53,38 @@ pub async fn sanitize(_token: APIToken, Json(request): Json<SanitizeRequest>) ->
|
||||
findings: report.findings,
|
||||
redacted_count: report.redacted_count,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn sanitize_batch(
|
||||
_token: APIToken,
|
||||
Json(request): Json<SanitizeBatchRequest>,
|
||||
) -> Result<Json<SanitizeBatchResponse>, (StatusCode, String)> {
|
||||
//
|
||||
// Scanning is CPU-bound, and a batch carries far more text than a single request: an entire
|
||||
// web search instead of one page. Running that on a runtime worker would stall every other
|
||||
// call the app makes meanwhile, so it goes to the blocking pool.
|
||||
//
|
||||
tokio::task::spawn_blocking(move || {
|
||||
let results = request
|
||||
.texts
|
||||
.iter()
|
||||
.map(|text| {
|
||||
let (sanitized_text, report) = sanitize_text(text);
|
||||
SanitizeResponse {
|
||||
sanitized_text,
|
||||
findings: report.findings,
|
||||
redacted_count: report.redacted_count,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
Json(SanitizeBatchResponse { results })
|
||||
})
|
||||
.await
|
||||
.map_err(|error| {
|
||||
(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
format!("The prompt injection filter failed: {error}"),
|
||||
)
|
||||
})
|
||||
}
|
||||
@@ -1,5 +1,6 @@
|
||||
use log::info;
|
||||
use once_cell::sync::Lazy;
|
||||
use axum::extract::DefaultBodyLimit;
|
||||
use axum::routing::{delete, get, post};
|
||||
use axum::Router;
|
||||
use axum_server::tls_rustls::RustlsConfig;
|
||||
@@ -11,6 +12,10 @@ use crate::network::get_available_port;
|
||||
|
||||
static RUSTLS_CRYPTO_PROVIDER_INIT: Once = Once::new();
|
||||
|
||||
/// The request body limit for one batch of prompt injection filtering. The app caps the text
|
||||
/// it returns to a model well below this, so the limit is headroom, not a working constraint.
|
||||
const PROMPT_INJECTION_BATCH_BODY_LIMIT_BYTES: usize = 16 * 1024 * 1024;
|
||||
|
||||
/// The port used for the runtime API server. In the development environment, we use a fixed
|
||||
/// port, in the production environment we use the next available port. This differentiation
|
||||
/// is necessary because we cannot communicate the port to the .NET server in the development
|
||||
@@ -62,6 +67,14 @@ pub fn start_runtime_api() {
|
||||
.route("/system/enterprise/configs", get(crate::environment::read_enterprise_configs))
|
||||
.route("/retrieval/fs/extract", get(crate::file_data::extract_data))
|
||||
.route("/security/prompt-injection/sanitize", post(crate::prompt_injection::api::sanitize))
|
||||
//
|
||||
// A batch carries every text of one tool call, which is far more than Axum's 2 MB
|
||||
// default allows. Exceeding that limit would answer 413, and the app treats a failed
|
||||
// filter call as "cannot filter" and uses the text unfiltered — the protection would
|
||||
// drop out silently on exactly the largest results. Hence the explicit limit.
|
||||
//
|
||||
.route("/security/prompt-injection/sanitize-batch", post(crate::prompt_injection::api::sanitize_batch)
|
||||
.layer(DefaultBodyLimit::max(PROMPT_INJECTION_BATCH_BODY_LIMIT_BYTES)))
|
||||
.route("/media/jobs", post(crate::media::create_job))
|
||||
.route("/media/jobs/{id}/events", get(crate::media::get_job_events))
|
||||
.route("/media/jobs/{id}", delete(crate::media::cancel_job))
|
||||
|
||||
Reference in new issue
Block a user