Allowed reading plain text and JSON from the web, with a stronger prompt injection filter (#1000)
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions

This commit is contained in:
Thorsten Sommer authored and GitHub committed 2026-09-23 22:21:09 +02:00
1 parent 52fdb41c02
commit 40ac215359
18 files changed
+901 -31

No files matched your search

@@ -2,10 +2,24 @@ using AIStudio.Provider;
namespace AIStudio.Tools.Web;
/// <summary>
/// A web page or text document as the retrieval service read it.
/// </summary>
/// <remarks>
/// Nothing here is filtered for prompt injections yet. Everything a caller hands on to a model
/// has to go through the WebPageContentSanitizer or the PromptInjectionGuardService first, after
/// it was cut down to what the model actually gets: only that part needs checking, and a page
/// can be far larger.
/// </remarks>
public sealed class RetrievedWebPage
{
public required HTMLParserWebPage Page { get; init; }
/// <summary>
/// Whether the content was extracted from an HTML page or is the text of a document.
/// </summary>
public required WebContentKind ContentKind { get; init; }
public required ExtractedWebPage ExtractedPage { get; init; }
public required DateTimeOffset RetrievedAtUtc { get; init; }
@@ -0,0 +1,19 @@
namespace AIStudio.Tools.Web;
/// <summary>
/// What a retrieved web resource turned out to be, which decides how its content was read.
/// </summary>
public enum WebContentKind
{
/// <summary>
/// An HTML page. Its readable part was extracted and converted to Markdown, so the content
/// can be shorter than the page when the extraction missed something.
/// </summary>
HTML_PAGE,
/// <summary>
/// A text document such as plain text, JSON, XML, or CSV. The extracted Markdown is its text
/// as the server sent it, so nothing of it was left out.
/// </summary>
TEXT_DOCUMENT,
}
@@ -0,0 +1,53 @@
namespace AIStudio.Tools.Web;
/// <summary>
/// Decides from a response's media type whether AI Studio can read it, and how.
/// </summary>
internal static class WebContentTypeClassifier
{
private static readonly HashSet<string> HTML_MEDIA_TYPES = new(StringComparer.OrdinalIgnoreCase)
{
"text/html", "application/xhtml+xml",
};
/// <summary>
/// The text formats outside of text/* which are worth reading. JSON and XML are covered by
/// their suffixes as well, so problem+json or rss+xml need no entry of their own.
/// </summary>
private static readonly HashSet<string> APPLICATION_TEXT_MEDIA_TYPES = new(StringComparer.OrdinalIgnoreCase)
{
"application/json", "application/xml", "application/x-ndjson", "application/javascript", "application/x-javascript",
"application/yaml", "application/x-yaml", "application/toml", "application/sql",
};
/// <summary>
/// Classifies a media type, such as text/html or application/json.
/// </summary>
/// <remarks>
/// A missing media type counts as HTML, as it always did: servers leaving it out are
/// almost always serving a page.<br/><br/>
/// Binary formats are not readable, even those holding text, such as PDF: their bytes
/// decoded as text are noise, and what a model would make of them is worse than nothing.
/// The suffix rules apply to application/* only, which keeps image/svg+xml out.
/// </remarks>
/// <param name="mediaType">The media type without its parameters.</param>
/// <returns>How the content is read, or null when it cannot be read.</returns>
public static WebContentKind? Classify(string mediaType)
{
mediaType = mediaType.Trim();
if (mediaType.Length is 0 || HTML_MEDIA_TYPES.Contains(mediaType))
return WebContentKind.HTML_PAGE;
if (mediaType.StartsWith("text/", StringComparison.OrdinalIgnoreCase) || APPLICATION_TEXT_MEDIA_TYPES.Contains(mediaType) || HasTextSuffix(mediaType))
return WebContentKind.TEXT_DOCUMENT;
return null;
}
/// <summary>
/// Whether an application/* type states that it is JSON or XML, such as problem+json.
/// </summary>
private static bool HasTextSuffix(string mediaType) =>
mediaType.StartsWith("application/", StringComparison.OrdinalIgnoreCase) &&
(mediaType.EndsWith("+json", StringComparison.OrdinalIgnoreCase) || mediaType.EndsWith("+xml", StringComparison.OrdinalIgnoreCase));
}
@@ -14,7 +14,7 @@ public sealed class WebPageRetrievalOptions
/// hosts named localhost — because those exist to keep a model from reaching into the user's
/// network, and the user is not a model. The network-level protections stay: the connection
/// is still bound to validated addresses, redirects are still checked, the response size is
/// still capped, and only HTML is still accepted.<br/><br/>
/// still capped, and only HTML and text content are still accepted.<br/><br/>
/// Never set this for a URL that reached AI Studio through a model, however plausible it
/// looks.
/// </remarks>
@@ -1,6 +1,7 @@
using System.Net;
using System.Net.Sockets;
using AIStudio.Provider;
using HtmlAgilityPack;
namespace AIStudio.Tools.Web;
@@ -15,6 +16,10 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
{
var triedOsSso = false;
var requiredProviderConfidence = ConfidenceLevel.NONE;
// Always overwritten: the media type is validated before the body is read, so no page
// arrives without the check having decided what it is.
var contentKind = WebContentKind.HTML_PAGE;
HTMLParserWebPage page;
try
{
@@ -37,6 +42,8 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
triedOsSso |= shouldTryOsSso;
return shouldTryOsSso;
},
validateMediaType: mediaType => contentKind = WebContentTypeClassifier.Classify(mediaType) ??
throw new InvalidOperationException($"Unsupported content type '{mediaType}'. Only HTML pages and text formats such as plain text, JSON, XML, or CSV are supported."),
token: token);
}
catch (OperationCanceledException) when (!token.IsCancellationRequested)
@@ -58,18 +65,27 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
throw new InvalidOperationException($"Loading the web page failed: {exception.Message}", exception);
}
if (!IsSupportedHtmlContentType(page.ContentType))
throw new InvalidOperationException($"Unsupported content type '{page.ContentType}'. Only HTML pages are supported.");
return new RetrievedWebPage
{
Page = page,
ExtractedPage = WebPageContentExtractor.Extract(page.Document, page.FinalUrl),
ContentKind = contentKind,
ExtractedPage = contentKind switch
{
WebContentKind.TEXT_DOCUMENT => WebTextContentExtractor.Extract(page.Body, page.ContentType, page.FinalUrl),
_ => WebPageContentExtractor.Extract(ParseHtml(page.Body), page.FinalUrl),
},
RetrievedAtUtc = DateTimeOffset.UtcNow,
RequiredProviderConfidence = requiredProviderConfidence,
};
}
private static HtmlDocument ParseHtml(string html)
{
var document = new HtmlDocument();
document.LoadHtml(html);
return document;
}
private static WebPageAccessBlockedException? FindBlockedException(Exception exception)
{
if (exception is WebPageAccessBlockedException blockedException)
@@ -270,9 +286,4 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
return null;
}
private static bool IsSupportedHtmlContentType(string? contentType) =>
string.IsNullOrWhiteSpace(contentType) ||
contentType.StartsWith("text/html", StringComparison.OrdinalIgnoreCase) ||
contentType.StartsWith("application/xhtml+xml", StringComparison.OrdinalIgnoreCase);
}
@@ -0,0 +1,74 @@
namespace AIStudio.Tools.Web;
/// <summary>
/// Reads a text document fetched from the web, such as plain text, JSON, XML, or CSV.
/// </summary>
/// <remarks>
/// The text is returned as the server sent it. There is no main part to extract from a JSON
/// response, and converting it to Markdown would only take away what the model needs: exact
/// keys, quotes, and indentation. Only the line endings are unified, and the byte order mark
/// is dropped, because it is not part of the text.<br/><br/>
/// The text is not filtered for prompt injections here, just like an extracted HTML page is
/// not. The callers filter it once they have cut it down to what reaches the model. The
/// runtime's filter also decodes the escapes of JSON and XML, which a model reads fluently.
/// </remarks>
internal static class WebTextContentExtractor
{
private const char BYTE_ORDER_MARK = '\uFEFF';
/// <summary>
/// Takes the text of a document. Throws an InvalidOperationException when the body is not
/// text, whatever the server declared.
/// </summary>
/// <param name="body">The response body as text.</param>
/// <param name="mediaType">The media type the server declared, for the error message.</param>
/// <param name="finalUrl">Where the document was found after every redirect, for the error message.</param>
/// <returns>The document, with its text as the content and no metadata.</returns>
public static ExtractedWebPage Extract(string body, string mediaType, Uri finalUrl)
{
//
// Text holds no NUL characters, while nearly every binary format does. A server naming
// a PDF or an archive text/plain is common enough, and its bytes decoded as text would
// reach the model as noise.
//
if (body.Contains('\0'))
throw new InvalidOperationException($"The response of '{finalUrl}' is declared as '{mediaType}' but does not contain readable text.");
var text = body
.TrimStart(BYTE_ORDER_MARK)
.Replace("\r\n", "\n", StringComparison.Ordinal)
.Replace('\r', '\n')
.TrimEnd();
// Only blank lines are dropped at the start. The indentation of the first line carries
// meaning in YAML or in source code:
text = TrimLeadingBlankLines(text);
return new ExtractedWebPage
{
Title = string.Empty,
Description = string.Empty,
Authors = [],
PublishedTime = string.Empty,
ModifiedTime = string.Empty,
Language = string.Empty,
SiteName = string.Empty,
CanonicalUrl = null,
Markdown = text,
Outline = [],
};
}
private static string TrimLeadingBlankLines(string text)
{
var start = 0;
while (true)
{
var lineEnd = text.IndexOf('\n', start);
if (lineEnd < 0 || !string.IsNullOrWhiteSpace(text[start..lineEnd]))
return text[start..];
start = lineEnd + 1;
}
}
}