mirror of
https://github.com/MindWorkAI/AI-Studio.git
synced 2026-10-09 16:53:48 +00:00
Allowed reading plain text and JSON from the web, with a stronger prompt injection filter (#1000)
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
This commit is contained in:
1 parent
52fdb41c02
commit
40ac215359
18 files changed
+901
-31
No files matched your search
@@ -2,10 +2,24 @@ using AIStudio.Provider;
|
||||
|
||||
namespace AIStudio.Tools.Web;
|
||||
|
||||
/// <summary>
|
||||
/// A web page or text document as the retrieval service read it.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// Nothing here is filtered for prompt injections yet. Everything a caller hands on to a model
|
||||
/// has to go through the WebPageContentSanitizer or the PromptInjectionGuardService first, after
|
||||
/// it was cut down to what the model actually gets: only that part needs checking, and a page
|
||||
/// can be far larger.
|
||||
/// </remarks>
|
||||
public sealed class RetrievedWebPage
|
||||
{
|
||||
public required HTMLParserWebPage Page { get; init; }
|
||||
|
||||
/// <summary>
|
||||
/// Whether the content was extracted from an HTML page or is the text of a document.
|
||||
/// </summary>
|
||||
public required WebContentKind ContentKind { get; init; }
|
||||
|
||||
public required ExtractedWebPage ExtractedPage { get; init; }
|
||||
|
||||
public required DateTimeOffset RetrievedAtUtc { get; init; }
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
namespace AIStudio.Tools.Web;
|
||||
|
||||
/// <summary>
|
||||
/// What a retrieved web resource turned out to be, which decides how its content was read.
|
||||
/// </summary>
|
||||
public enum WebContentKind
|
||||
{
|
||||
/// <summary>
|
||||
/// An HTML page. Its readable part was extracted and converted to Markdown, so the content
|
||||
/// can be shorter than the page when the extraction missed something.
|
||||
/// </summary>
|
||||
HTML_PAGE,
|
||||
|
||||
/// <summary>
|
||||
/// A text document such as plain text, JSON, XML, or CSV. The extracted Markdown is its text
|
||||
/// as the server sent it, so nothing of it was left out.
|
||||
/// </summary>
|
||||
TEXT_DOCUMENT,
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
namespace AIStudio.Tools.Web;
|
||||
|
||||
/// <summary>
|
||||
/// Decides from a response's media type whether AI Studio can read it, and how.
|
||||
/// </summary>
|
||||
internal static class WebContentTypeClassifier
|
||||
{
|
||||
private static readonly HashSet<string> HTML_MEDIA_TYPES = new(StringComparer.OrdinalIgnoreCase)
|
||||
{
|
||||
"text/html", "application/xhtml+xml",
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// The text formats outside of text/* which are worth reading. JSON and XML are covered by
|
||||
/// their suffixes as well, so problem+json or rss+xml need no entry of their own.
|
||||
/// </summary>
|
||||
private static readonly HashSet<string> APPLICATION_TEXT_MEDIA_TYPES = new(StringComparer.OrdinalIgnoreCase)
|
||||
{
|
||||
"application/json", "application/xml", "application/x-ndjson", "application/javascript", "application/x-javascript",
|
||||
"application/yaml", "application/x-yaml", "application/toml", "application/sql",
|
||||
};
|
||||
|
||||
/// <summary>
|
||||
/// Classifies a media type, such as text/html or application/json.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// A missing media type counts as HTML, as it always did: servers leaving it out are
|
||||
/// almost always serving a page.<br/><br/>
|
||||
/// Binary formats are not readable, even those holding text, such as PDF: their bytes
|
||||
/// decoded as text are noise, and what a model would make of them is worse than nothing.
|
||||
/// The suffix rules apply to application/* only, which keeps image/svg+xml out.
|
||||
/// </remarks>
|
||||
/// <param name="mediaType">The media type without its parameters.</param>
|
||||
/// <returns>How the content is read, or null when it cannot be read.</returns>
|
||||
public static WebContentKind? Classify(string mediaType)
|
||||
{
|
||||
mediaType = mediaType.Trim();
|
||||
if (mediaType.Length is 0 || HTML_MEDIA_TYPES.Contains(mediaType))
|
||||
return WebContentKind.HTML_PAGE;
|
||||
|
||||
if (mediaType.StartsWith("text/", StringComparison.OrdinalIgnoreCase) || APPLICATION_TEXT_MEDIA_TYPES.Contains(mediaType) || HasTextSuffix(mediaType))
|
||||
return WebContentKind.TEXT_DOCUMENT;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Whether an application/* type states that it is JSON or XML, such as problem+json.
|
||||
/// </summary>
|
||||
private static bool HasTextSuffix(string mediaType) =>
|
||||
mediaType.StartsWith("application/", StringComparison.OrdinalIgnoreCase) &&
|
||||
(mediaType.EndsWith("+json", StringComparison.OrdinalIgnoreCase) || mediaType.EndsWith("+xml", StringComparison.OrdinalIgnoreCase));
|
||||
}
|
||||
@@ -14,7 +14,7 @@ public sealed class WebPageRetrievalOptions
|
||||
/// hosts named localhost — because those exist to keep a model from reaching into the user's
|
||||
/// network, and the user is not a model. The network-level protections stay: the connection
|
||||
/// is still bound to validated addresses, redirects are still checked, the response size is
|
||||
/// still capped, and only HTML is still accepted.<br/><br/>
|
||||
/// still capped, and only HTML and text content are still accepted.<br/><br/>
|
||||
/// Never set this for a URL that reached AI Studio through a model, however plausible it
|
||||
/// looks.
|
||||
/// </remarks>
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
using System.Net;
|
||||
using System.Net.Sockets;
|
||||
using AIStudio.Provider;
|
||||
using HtmlAgilityPack;
|
||||
|
||||
namespace AIStudio.Tools.Web;
|
||||
|
||||
@@ -15,6 +16,10 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
|
||||
{
|
||||
var triedOsSso = false;
|
||||
var requiredProviderConfidence = ConfidenceLevel.NONE;
|
||||
|
||||
// Always overwritten: the media type is validated before the body is read, so no page
|
||||
// arrives without the check having decided what it is.
|
||||
var contentKind = WebContentKind.HTML_PAGE;
|
||||
HTMLParserWebPage page;
|
||||
try
|
||||
{
|
||||
@@ -37,6 +42,8 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
|
||||
triedOsSso |= shouldTryOsSso;
|
||||
return shouldTryOsSso;
|
||||
},
|
||||
validateMediaType: mediaType => contentKind = WebContentTypeClassifier.Classify(mediaType) ??
|
||||
throw new InvalidOperationException($"Unsupported content type '{mediaType}'. Only HTML pages and text formats such as plain text, JSON, XML, or CSV are supported."),
|
||||
token: token);
|
||||
}
|
||||
catch (OperationCanceledException) when (!token.IsCancellationRequested)
|
||||
@@ -58,18 +65,27 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
|
||||
throw new InvalidOperationException($"Loading the web page failed: {exception.Message}", exception);
|
||||
}
|
||||
|
||||
if (!IsSupportedHtmlContentType(page.ContentType))
|
||||
throw new InvalidOperationException($"Unsupported content type '{page.ContentType}'. Only HTML pages are supported.");
|
||||
|
||||
return new RetrievedWebPage
|
||||
{
|
||||
Page = page,
|
||||
ExtractedPage = WebPageContentExtractor.Extract(page.Document, page.FinalUrl),
|
||||
ContentKind = contentKind,
|
||||
ExtractedPage = contentKind switch
|
||||
{
|
||||
WebContentKind.TEXT_DOCUMENT => WebTextContentExtractor.Extract(page.Body, page.ContentType, page.FinalUrl),
|
||||
_ => WebPageContentExtractor.Extract(ParseHtml(page.Body), page.FinalUrl),
|
||||
},
|
||||
RetrievedAtUtc = DateTimeOffset.UtcNow,
|
||||
RequiredProviderConfidence = requiredProviderConfidence,
|
||||
};
|
||||
}
|
||||
|
||||
private static HtmlDocument ParseHtml(string html)
|
||||
{
|
||||
var document = new HtmlDocument();
|
||||
document.LoadHtml(html);
|
||||
return document;
|
||||
}
|
||||
|
||||
private static WebPageAccessBlockedException? FindBlockedException(Exception exception)
|
||||
{
|
||||
if (exception is WebPageAccessBlockedException blockedException)
|
||||
@@ -270,9 +286,4 @@ public sealed class WebPageRetrievalService(HTMLParser htmlParser)
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
private static bool IsSupportedHtmlContentType(string? contentType) =>
|
||||
string.IsNullOrWhiteSpace(contentType) ||
|
||||
contentType.StartsWith("text/html", StringComparison.OrdinalIgnoreCase) ||
|
||||
contentType.StartsWith("application/xhtml+xml", StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
namespace AIStudio.Tools.Web;
|
||||
|
||||
/// <summary>
|
||||
/// Reads a text document fetched from the web, such as plain text, JSON, XML, or CSV.
|
||||
/// </summary>
|
||||
/// <remarks>
|
||||
/// The text is returned as the server sent it. There is no main part to extract from a JSON
|
||||
/// response, and converting it to Markdown would only take away what the model needs: exact
|
||||
/// keys, quotes, and indentation. Only the line endings are unified, and the byte order mark
|
||||
/// is dropped, because it is not part of the text.<br/><br/>
|
||||
/// The text is not filtered for prompt injections here, just like an extracted HTML page is
|
||||
/// not. The callers filter it once they have cut it down to what reaches the model. The
|
||||
/// runtime's filter also decodes the escapes of JSON and XML, which a model reads fluently.
|
||||
/// </remarks>
|
||||
internal static class WebTextContentExtractor
|
||||
{
|
||||
private const char BYTE_ORDER_MARK = '\uFEFF';
|
||||
|
||||
/// <summary>
|
||||
/// Takes the text of a document. Throws an InvalidOperationException when the body is not
|
||||
/// text, whatever the server declared.
|
||||
/// </summary>
|
||||
/// <param name="body">The response body as text.</param>
|
||||
/// <param name="mediaType">The media type the server declared, for the error message.</param>
|
||||
/// <param name="finalUrl">Where the document was found after every redirect, for the error message.</param>
|
||||
/// <returns>The document, with its text as the content and no metadata.</returns>
|
||||
public static ExtractedWebPage Extract(string body, string mediaType, Uri finalUrl)
|
||||
{
|
||||
//
|
||||
// Text holds no NUL characters, while nearly every binary format does. A server naming
|
||||
// a PDF or an archive text/plain is common enough, and its bytes decoded as text would
|
||||
// reach the model as noise.
|
||||
//
|
||||
if (body.Contains('\0'))
|
||||
throw new InvalidOperationException($"The response of '{finalUrl}' is declared as '{mediaType}' but does not contain readable text.");
|
||||
|
||||
var text = body
|
||||
.TrimStart(BYTE_ORDER_MARK)
|
||||
.Replace("\r\n", "\n", StringComparison.Ordinal)
|
||||
.Replace('\r', '\n')
|
||||
.TrimEnd();
|
||||
|
||||
// Only blank lines are dropped at the start. The indentation of the first line carries
|
||||
// meaning in YAML or in source code:
|
||||
text = TrimLeadingBlankLines(text);
|
||||
|
||||
return new ExtractedWebPage
|
||||
{
|
||||
Title = string.Empty,
|
||||
Description = string.Empty,
|
||||
Authors = [],
|
||||
PublishedTime = string.Empty,
|
||||
ModifiedTime = string.Empty,
|
||||
Language = string.Empty,
|
||||
SiteName = string.Empty,
|
||||
CanonicalUrl = null,
|
||||
Markdown = text,
|
||||
Outline = [],
|
||||
};
|
||||
}
|
||||
|
||||
private static string TrimLeadingBlankLines(string text)
|
||||
{
|
||||
var start = 0;
|
||||
while (true)
|
||||
{
|
||||
var lineEnd = text.IndexOf('\n', start);
|
||||
if (lineEnd < 0 || !string.IsNullOrWhiteSpace(text[start..lineEnd]))
|
||||
return text[start..];
|
||||
|
||||
start = lineEnd + 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user