Fixed web searches failing when several pages were read at once (#972)
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions

This commit is contained in:
Thorsten Sommer authored and GitHub committed 2026-09-14 17:45:35 +02:00
1 parent 80eccca999
commit f869122070
4 files changed
+215 -8

No files matched your search

@@ -76,7 +76,7 @@ internal static class WebPageContentExtractor
.Select(x => LimitLength(x, MAX_OUTLINE_ITEM_CHARACTERS))
.Distinct(StringComparer.Ordinal)
.ToList();
var markdown = HTMLParser.ParseToMarkdown(contentRoot.InnerHtml)
var markdown = ConvertToMarkdown(contentRoot.InnerHtml, finalUrl)
.Replace("\r\n", "\n", StringComparison.Ordinal)
.Replace('\r', '\n')
.Trim();
@@ -141,6 +141,30 @@ internal static class WebPageContentExtractor
};
}
/// <summary>
/// Converts the readable part of the page to Markdown.
/// </summary>
/// <remarks>
/// Only the call into the Markdown library is wrapped, not the extraction around it: a fault of
/// our own has to keep surfacing as what it is, instead of being filed away as an unreadable
/// page.<br/><br/>
/// What the library throws depends on the HTML it was handed, and it says nothing beyond "this
/// page could not be converted". Reported as an InvalidOperationException, the retrieval treats
/// it like any other page it could not read, which costs this one page rather than the whole
/// search it belongs to.
/// </remarks>
private static string ConvertToMarkdown(string html, Uri finalUrl)
{
try
{
return HTMLParser.ParseToMarkdown(html);
}
catch (Exception exception) when (exception is not OperationCanceledException)
{
throw new InvalidOperationException($"Converting the HTML of '{finalUrl}' to Markdown failed: {exception.Message}", exception);
}
}
private static JsonLdMetadata ExtractJsonLdMetadata(HtmlDocument document, Uri finalUrl)
{
JsonLdCandidate? bestCandidate = null;