Rework of the file export (#894)

Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
nilskruthoffandThorsten Sommer authored and GitHub committed 2026-08-31 08:40:50 +02:00
1 parent f14c69b938
commit c66a61713d
33 files changed
+1212 -266

No files matched your search

+64
View File
@@ -0,0 +1,64 @@
using System.Globalization;
namespace AIStudio.Tools;
/// <summary>
/// Writes rows of character-separated values. Fields are quoted according to RFC 4180 using the
/// separator of the respective file.
/// </summary>
public static class CsvWriter
{
/// <summary>
/// The separator a spreadsheet expects from a CSV file written for the given language.
/// </summary>
/// <remarks>
/// Wherever a comma separates the decimals of a number, it cannot separate the columns of a
/// file as well: German Excel therefore expects a semicolon and puts a comma-separated file
/// into a single column. This is the same rule Excel itself follows when it writes a CSV, so
/// we ask the culture rather than keeping a list of languages of our own.
/// </remarks>
/// <param name="ietfTag">The IETF tag of the language, for example "de-DE".</param>
/// <returns>The separator to write with.</returns>
public static char SeparatorFor(string ietfTag)
{
if (string.IsNullOrWhiteSpace(ietfTag))
return ',';
try
{
var culture = CultureInfo.GetCultureInfo(ietfTag);
return culture.NumberFormat.NumberDecimalSeparator is "," ? ';' : ',';
}
catch (CultureNotFoundException)
{
return ',';
}
}
/// <summary>
/// Joins the given fields into one row.
/// </summary>
/// <param name="separator">The separator between two fields.</param>
/// <param name="fields">The fields of the row.</param>
/// <returns>The row, without a line ending.</returns>
public static string ToRow(char separator, params string[] fields) => string.Join(separator, fields.Select(field => ToField(field, separator)));
/// <summary>
/// Quotes one field according to RFC 4180.
/// </summary>
private static string ToField(string text, char separator)
{
if (string.IsNullOrEmpty(text))
return string.Empty;
// Quoting the complete field is important for long and multi-line AI
// answers: neither separators nor line breaks within an answer may
// create another column or row.
if (!text.Contains(separator) && !text.Contains('"') && !text.Contains('\n') && !text.Contains('\r'))
return text;
return $"""
"{text.Replace("\"", "\"\"")}"
""";
}
}
@@ -0,0 +1,18 @@
namespace AIStudio.Tools;
/// <summary>
/// The file formats a chat message can be exported to.
/// </summary>
public enum FileExportFormat
{
NONE,
UNKNOWN,
MICROSOFT_WORD,
OPEN_DOCUMENT_TEXT,
LATEX,
MARKDOWN,
HTML,
CSV,
TSV,
}
@@ -0,0 +1,228 @@
using System.Text;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Rust;
namespace AIStudio.Tools;
/// <summary>
/// Everything AI Studio needs to know about an export format: how it is named, how it is shown,
/// which file it produces, and who writes that file.
/// </summary>
/// <remarks>
/// This is the single place where an export format is described. Adding another one means adding
/// an enum member and one line per method here; neither the exporters nor the export menu need
/// to know about it.
/// </remarks>
public static class FileExportFormatExtensions
{
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(FileExportFormatExtensions).Namespace, nameof(FileExportFormatExtensions));
private static readonly Encoding WITH_BYTE_ORDER_MARK = new UTF8Encoding(true);
private static readonly Encoding WITHOUT_BYTE_ORDER_MARK = new UTF8Encoding(false);
/// <summary>
/// The formats which lay the text out as a document you would hand to somebody, in the order
/// the export menu shows them.
/// </summary>
public static readonly IReadOnlyList<FileExportFormat> DOCUMENT_FORMATS =
[
FileExportFormat.MICROSOFT_WORD,
FileExportFormat.OPEN_DOCUMENT_TEXT,
FileExportFormat.LATEX,
];
/// <summary>
/// The formats which keep the text as text, in the order the export menu shows them.
/// </summary>
public static readonly IReadOnlyList<FileExportFormat> TEXT_FORMATS =
[
FileExportFormat.MARKDOWN,
FileExportFormat.HTML,
];
/// <summary>
/// Every format an entire answer can be written as.
/// </summary>
/// <remarks>
/// The tabular formats are missing on purpose: they hold one table out of an answer, never the
/// answer itself. Whoever offers a table adds them.
/// </remarks>
public static readonly IReadOnlyList<FileExportFormat> ANSWER_FORMATS = [..DOCUMENT_FORMATS, ..TEXT_FORMATS];
/// <summary>
/// Returns the name of the format as shown to the user.
/// </summary>
/// <param name="format">The format.</param>
/// <returns>The name of the format.</returns>
public static string ToName(this FileExportFormat format) => format switch
{
FileExportFormat.MICROSOFT_WORD => TB("Microsoft Word (.docx)"),
FileExportFormat.OPEN_DOCUMENT_TEXT => TB("OpenDocument Text (.odt), e.g. LibreOffice"),
FileExportFormat.LATEX => TB("LaTeX (.tex)"),
FileExportFormat.MARKDOWN => TB("Markdown (.md)"),
FileExportFormat.HTML => TB("Webpage (.html)"),
FileExportFormat.CSV => TB("Table (.csv)"),
FileExportFormat.TSV => TB("Table (.tsv)"),
_ => TB("Unknown format"),
};
/// <summary>
/// Returns the icon of the format.
/// </summary>
/// <param name="format">The format.</param>
/// <returns>The icon of the format.</returns>
public static string ToIcon(this FileExportFormat format) => format switch
{
FileExportFormat.MICROSOFT_WORD => Icons.Custom.FileFormats.FileWord,
FileExportFormat.OPEN_DOCUMENT_TEXT => Icons.Custom.FileFormats.FileDocument,
FileExportFormat.LATEX => Icons.Material.Filled.Functions,
FileExportFormat.MARKDOWN => Icons.Material.Filled.TextFields,
FileExportFormat.HTML => Icons.Material.Filled.Html,
FileExportFormat.CSV or FileExportFormat.TSV => Icons.Material.Filled.TableChart,
_ => Icons.Material.Filled.Help,
};
/// <summary>
/// Returns the file extension of the format, including the leading dot.
/// </summary>
/// <param name="format">The format.</param>
/// <returns>The file extension, or an empty string when the format writes no file.</returns>
public static string ToFileExtension(this FileExportFormat format) => format switch
{
FileExportFormat.MICROSOFT_WORD => ".docx",
FileExportFormat.OPEN_DOCUMENT_TEXT => ".odt",
FileExportFormat.LATEX => ".tex",
FileExportFormat.MARKDOWN => ".md",
FileExportFormat.HTML => ".html",
FileExportFormat.CSV => ".csv",
FileExportFormat.TSV => ".tsv",
_ => string.Empty,
};
/// <summary>
/// Returns the file name the save dialog starts with.
/// </summary>
/// <remarks>
/// Without a name, the dialog opens with an empty field and the user easily ends up with a
/// file which carries no extension at all. The fallback name is deliberately not translated:
/// a file name should survive being copied between systems and locales.
/// </remarks>
/// <param name="format">The format.</param>
/// <param name="name">What the file is about, for example the heading above a table. Anything
/// a file name cannot hold is removed. Null or blank falls back to a generic name.</param>
/// <returns>The suggested file name, including its extension.</returns>
public static string ToSuggestedFileName(this FileExportFormat format, string? name = null)
{
var fileName = ToFileNameFragment(name);
return $"{(fileName.Length is 0 ? "export" : fileName)}{format.ToFileExtension()}";
}
/// <summary>
/// Turns arbitrary text into something a file system accepts as a name.
/// </summary>
/// <remarks>
/// We do not ask the runtime which characters are invalid: macOS forbids almost nothing, so a
/// name taken from there would break as soon as the file reaches a Windows share. The fixed
/// set below is what no common file system accepts, plus the length limit which keeps the name
/// readable in a dialog.
/// </remarks>
private static string ToFileNameFragment(string? name)
{
const int MAX_LENGTH = 60;
const string FORBIDDEN_CHARACTERS = @"\/:*?""<>|";
if (string.IsNullOrWhiteSpace(name))
return string.Empty;
var fragment = new StringBuilder(name.Length);
var lastWasSpace = false;
foreach (var character in name)
{
var isSpace = char.IsWhiteSpace(character) || char.IsControl(character) || FORBIDDEN_CHARACTERS.Contains(character);
if (isSpace)
{
// Collapse whatever we dropped into a single space, so "Table 1: People"
// becomes "Table 1 People" instead of "Table 1 People":
if (fragment.Length > 0)
lastWasSpace = true;
continue;
}
if (lastWasSpace)
{
fragment.Append(' ');
lastWasSpace = false;
}
fragment.Append(character);
if (fragment.Length >= MAX_LENGTH)
break;
}
// A trailing dot makes a file invisible on Unix and is dropped by Windows:
return fragment.ToString().TrimEnd('.');
}
/// <summary>
/// Returns the filter which the save dialog offers for the format.
/// </summary>
/// <param name="format">The format.</param>
/// <returns>The filter, or null when the format cannot be written.</returns>
public static FileTypeFilter? ToFileTypeFilter(this FileExportFormat format) => format switch
{
FileExportFormat.MICROSOFT_WORD => FileTypes.MS_WORD,
FileExportFormat.OPEN_DOCUMENT_TEXT => FileTypes.ODT,
FileExportFormat.LATEX => FileTypes.TEX,
FileExportFormat.MARKDOWN => FileTypes.MARKDOWN,
FileExportFormat.HTML => FileTypes.HTML,
FileExportFormat.CSV => FileTypes.CSV,
FileExportFormat.TSV => FileTypes.TSV,
_ => null,
};
/// <summary>
/// Returns the encoding the file gets written with.
/// </summary>
/// <remarks>
/// Everything is UTF-8, the question is only whether the file starts with a byte order mark.
/// Tabular files get one, because Excel otherwise reads them in the local ANSI code page and
/// turns every umlaut into garbage. Text files get none: editors, compilers, and LaTeX have
/// no use for it and some of them stumble over it.
/// </remarks>
/// <param name="format">The format.</param>
/// <returns>The encoding to write the file with.</returns>
public static Encoding ToFileEncoding(this FileExportFormat format) => format switch
{
FileExportFormat.CSV or FileExportFormat.TSV => WITH_BYTE_ORDER_MARK,
_ => WITHOUT_BYTE_ORDER_MARK,
};
/// <summary>
/// Returns the name Pandoc knows the format by.
/// </summary>
/// <param name="format">The format.</param>
/// <returns>The Pandoc output format, or an empty string when AI Studio writes the file itself.</returns>
public static string ToPandocOutputFormat(this FileExportFormat format) => format switch
{
FileExportFormat.MICROSOFT_WORD => "docx",
FileExportFormat.OPEN_DOCUMENT_TEXT => "odt",
FileExportFormat.LATEX => "latex",
FileExportFormat.HTML => "html",
_ => string.Empty,
};
/// <summary>
/// Determines whether writing the format needs Pandoc.
/// </summary>
/// <param name="format">The format.</param>
/// <returns>True, when Pandoc converts the message; false, when AI Studio writes the file itself.</returns>
public static bool UsesPandoc(this FileExportFormat format) => !string.IsNullOrWhiteSpace(format.ToPandocOutputFormat());
}
@@ -0,0 +1,12 @@
namespace AIStudio.Tools;
/// <summary>
/// A table found in a message, ready to be written to a file.
/// </summary>
/// <param name="Ordinal">Which table of the message this is, counting from one. The same table
/// appears once per format we offer for it, so this is what tells two tables apart even when they
/// carry the same heading.</param>
/// <param name="Caption">What the table is about, taken from its first column heading.</param>
/// <param name="Format">The format this content is written as.</param>
/// <param name="Content">The finished file content.</param>
public sealed record MessageTable(int Ordinal, string Caption, FileExportFormat Format, string Content);
+88 -65
View File
@@ -1,77 +1,54 @@
using System.Diagnostics;
using AIStudio.Chat;
using AIStudio.Dialogs;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Rust;
using AIStudio.Tools.Services;
using System.Diagnostics;
using System.Text;
using DialogOptions = AIStudio.Dialogs.DialogOptions;
using AIStudio.Chat;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Services;
namespace AIStudio.Tools;
public static class PandocExport
{
private static readonly ILogger LOGGER = Program.LOGGER_FACTORY.CreateLogger(nameof(PandocExport));
private static string TB(string fallbackEn) => I18N.I.T(fallbackEn, typeof(PandocExport).Namespace, nameof(PandocExport));
public static async Task<bool> ToMicrosoftWord(RustService rustService, IDialogService dialogService, string dialogTitle, IContent markdownContent)
{
var response = await rustService.SaveFile(dialogTitle, [FileTypes.MS_WORD]);
if (response.UserCancelled)
{
LOGGER.LogInformation("User cancelled the save dialog.");
return false;
}
private static readonly ILogger LOGGER = Program.LOGGER_FACTORY.CreateLogger(nameof(PandocExport));
LOGGER.LogInformation($"The user chose the path '{response.SaveFilePath}' for the Microsoft Word export.");
private static string TB(string fallbackEn) => I18N.I.T(fallbackEn, typeof(PandocExport).Namespace, nameof(PandocExport));
/// <summary>
/// Converts the given Markdown text into a document at the given path.
/// </summary>
/// <remarks>
/// This says nothing to the user: it reports what happened and lets the caller decide. A batch
/// run over hundreds of documents would otherwise bury the user under notifications. Pandoc
/// must be available, which PandocAvailabilityService.EnsureAvailabilityAsync takes care of.
/// </remarks>
/// <param name="rustService">The Rust service, used to build the Pandoc call.</param>
/// <param name="markdownText">The Markdown text to convert.</param>
/// <param name="targetFilePath">Where to write the document.</param>
/// <param name="format">The format to write. Must be a format which uses Pandoc.</param>
/// <param name="token">The token to cancel the conversion.</param>
/// <returns>True, when the document was written.</returns>
public static async Task<bool> ConvertAsync(RustService rustService, string markdownText, string targetFilePath, FileExportFormat format, CancellationToken token = default)
{
if (!format.UsesPandoc())
throw new ArgumentOutOfRangeException(nameof(format), format, "Pandoc cannot write this format.");
var tempMarkdownFilePath = string.Empty;
try
{
var tempMarkdownFile = Guid.NewGuid().ToString();
tempMarkdownFilePath = Path.Combine(Path.GetTempPath(), tempMarkdownFile);
// Extract text content from chat:
var markdownText = markdownContent switch
{
ContentText text => text.Text,
ContentImage _ => "Image export to Microsoft Word not yet possible",
_ => "Unknown content type. Cannot export to Word."
};
// Write text content to a temporary file. Pandoc expects UTF-8 without a byte order
// mark; a mark would end up as a stray character at the start of the document:
await File.WriteAllTextAsync(tempMarkdownFilePath, markdownText, new UTF8Encoding(false), token);
// Write text content to a temporary file:
await File.WriteAllTextAsync(tempMarkdownFilePath, markdownText);
// Ensure that Pandoc is installed and ready:
var pandocState = await Pandoc.CheckAvailabilityAsync(rustService, showSuccessMessage: false);
if (!pandocState.IsAvailable)
{
var dialogParameters = new DialogParameters<PandocDialog>
{
{ x => x.ShowInitialResultInSnackbar, false },
};
var dialogReference = await dialogService.ShowAsync<PandocDialog>(TB("Pandoc Installation"), dialogParameters, DialogOptions.FULLSCREEN);
await dialogReference.Result;
pandocState = await Pandoc.CheckAvailabilityAsync(rustService, showSuccessMessage: true);
if (!pandocState.IsAvailable)
{
LOGGER.LogError("Pandoc is not available after installation attempt.");
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("Pandoc is required for Microsoft Word export.")));
return false;
}
}
// Call Pandoc to create the Word file:
// Call Pandoc to create the document:
var pandoc = await PandocProcessBuilder
.Create()
.UseStandaloneMode()
.WithInputFormat("gfm+emoji+tex_math_dollars")
.WithOutputFormat("docx")
.WithOutputFile(response.SaveFilePath)
.WithOutputFormat(format.ToPandocOutputFormat())
.WithOutputFile(targetFilePath)
.WithInputFile(tempMarkdownFilePath)
.BuildAsync(rustService);
@@ -83,30 +60,26 @@ public static class PandocExport
}
// Read output streams asynchronously while the process runs (prevents deadlock):
var outputTask = process.StandardOutput.ReadToEndAsync();
var errorTask = process.StandardError.ReadToEndAsync();
var outputTask = process.StandardOutput.ReadToEndAsync(token);
var errorTask = process.StandardError.ReadToEndAsync(token);
// Wait for the process to exit AND for streams to be fully read:
await process.WaitForExitAsync();
await process.WaitForExitAsync(token);
await outputTask;
var error = await errorTask;
if (process.ExitCode is not 0)
{
LOGGER.LogError("Pandoc failed with exit code {ProcessExitCode}: '{ErrorText}'", process.ExitCode, error);
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("Error during Microsoft Word export")));
return false;
}
LOGGER.LogInformation("Pandoc conversion successful.");
await MessageBus.INSTANCE.SendSuccess(new(Icons.Material.Filled.CheckCircle, TB("Microsoft Word export successful")));
LOGGER.LogInformation("Pandoc conversion to {ExportFormat} successful.", format);
return true;
}
catch (Exception ex)
{
LOGGER.LogError(ex, "Error during Word export.");
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("Error during Microsoft Word export")));
LOGGER.LogError(ex, "Error during {ExportFormat} conversion.", format);
return false;
}
finally
@@ -120,9 +93,59 @@ public static class PandocExport
}
catch
{
LOGGER.LogWarning($"Was not able to delete temporary file: '{tempMarkdownFilePath}'");
LOGGER.LogWarning("Was not able to delete the temporary file '{TempFilePath}'.", tempMarkdownFilePath);
}
}
}
}
/// <summary>
/// Converts the given content to a document using Pandoc and lets the user save it.
/// </summary>
/// <param name="rustService">The Rust service, used for the save dialog and for Pandoc.</param>
/// <param name="pandocAvailability">Makes sure Pandoc is there and offers its installation.</param>
/// <param name="dialogTitle">The title of the save dialog. The caller knows what the user is
/// looking at, a chat message or the result of an assistant, so the caller names it.</param>
/// <param name="format">The format to write. Must be a format which uses Pandoc.</param>
/// <param name="markdownContent">The content to export.</param>
/// <returns>True, when the document was written.</returns>
public static async Task<bool> ToDocument(RustService rustService, PandocAvailabilityService pandocAvailability, string dialogTitle, FileExportFormat format, IContent markdownContent)
{
if (!format.UsesPandoc() || format.ToFileTypeFilter() is not { } fileTypeFilter)
throw new ArgumentOutOfRangeException(nameof(format), format, "Pandoc cannot write this format.");
//
// We read the text before we ask for a path: when there is nothing to convert, the user
// should learn that right away instead of picking a file first and getting an error afterwards.
//
if (!markdownContent.TryGetMarkdownText(out var markdownText))
{
LOGGER.LogWarning("Cannot export the content as {ExportFormat}, because it carries no text.", format);
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("Only text messages can be exported.")));
return false;
}
var response = await rustService.SaveFile(dialogTitle, [fileTypeFilter], format.ToSuggestedFileName());
if (response.UserCancelled)
{
LOGGER.LogInformation("User cancelled the save dialog.");
return false;
}
LOGGER.LogInformation("The user chose the path '{SaveFilePath}' for the {ExportFormat} export.", response.SaveFilePath, format);
// The service reports a missing Pandoc to the user itself, so we only act on the outcome:
var pandocState = await pandocAvailability.EnsureAvailabilityAsync(showSuccessMessage: false, showDialog: true);
if (!pandocState.IsAvailable)
return false;
if (!await ConvertAsync(rustService, markdownText, response.SaveFilePath, format))
{
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("The export failed.")));
return false;
}
await MessageBus.INSTANCE.SendSuccess(new(Icons.Material.Filled.CheckCircle, TB("The export succeeded.")));
return true;
}
}
@@ -0,0 +1,209 @@
using System.Text;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Services;
using Markdig.Extensions.Tables;
using Markdig.Syntax;
using Markdig.Syntax.Inlines;
namespace AIStudio.Tools;
public static class PlainFileExport
{
private static readonly ILogger LOGGER = Program.LOGGER_FACTORY.CreateLogger(nameof(PlainFileExport));
private static string TB(string fallbackEn) => I18N.I.T(fallbackEn, typeof(PlainFileExport).Namespace, nameof(PlainFileExport));
/// <summary>
/// Reads every table a message holds, in the order they appear in it.
/// </summary>
/// <remarks>
/// Two kinds of tables end up in an answer. Almost always it is a Markdown table written with
/// pipes, which is what a model produces on its own; we turn its cells into a file. Rarely a
/// model answers with a fenced code block marked as csv or tsv, which already is the finished
/// file: we hand that through untouched rather than taking it apart and reassembling it.
/// </remarks>
/// <param name="markdown">The Markdown text of the message.</param>
/// <param name="separator">The separator to write a Markdown table with, see CsvWriter.SeparatorFor.</param>
/// <returns>The tables, or an empty list when the message holds none.</returns>
public static IReadOnlyList<MessageTable> ExtractTables(string markdown, char separator)
{
if (string.IsNullOrWhiteSpace(markdown))
return [];
//
// We let Markdig do the reading. It is already part of the app, the pipeline we reuse has
// table support switched on, and it knows every corner of the syntax that a regular
// expression of ours would have to learn one bug at a time.
//
var document = Markdig.Markdown.Parse(markdown, Markdown.SAFE_MARKDOWN_PIPELINE);
//
// What a table is about stands above it, not in it: models introduce their tables with a
// heading. We remember every heading with its line so that each table can take the last
// one before it, and fall back to its own first column heading when there is none.
//
var headings = document.Descendants<HeadingBlock>()
.Select(heading => (heading.Line, Text: ToPlainText(heading)))
.Where(heading => !string.IsNullOrWhiteSpace(heading.Text))
.OrderBy(heading => heading.Line)
.ToList();
var tables = document.Descendants<Table>()
.Select(table => (table.Line, Content: ToContent(table, separator)));
var codeBlocks = document.Descendants<FencedCodeBlock>()
.Select(block => (block.Line, Content: ToContent(block)));
return tables.Concat(codeBlocks)
.Where(entry => entry.Content is not null)
.OrderBy(entry => entry.Line)
.Select((entry, index) => new MessageTable(
index + 1,
Caption: HeadingAbove(entry.Line) is { Length: > 0 } heading ? heading : entry.Content!.Value.Fallback,
entry.Content!.Value.Format,
entry.Content.Value.Text))
.ToList();
string HeadingAbove(int line) => headings.LastOrDefault(heading => heading.Line < line).Text ?? string.Empty;
}
/// <summary>
/// Turns a Markdown table into a file.
/// </summary>
private static (string Fallback, FileExportFormat Format, string Text)? ToContent(Table table, char separator)
{
var rows = table.OfType<TableRow>()
.Select(row => row.OfType<TableCell>().Select(ToPlainText).ToArray())
.Where(fields => fields.Length > 0)
.ToList();
if (rows.Count is 0)
return null;
var text = new StringBuilder();
foreach (var fields in rows)
text.AppendLine(CsvWriter.ToRow(separator, fields));
return (rows[0].FirstOrDefault() ?? string.Empty, FileExportFormat.CSV, text.ToString());
}
/// <summary>
/// Turns a fenced code block into a file, when the model marked it as tabular data.
/// </summary>
private static (string Fallback, FileExportFormat Format, string Text)? ToContent(FencedCodeBlock block)
{
var format = block.Info?.Trim() switch
{
"csv" => FileExportFormat.CSV,
"tsv" => FileExportFormat.TSV,
_ => FileExportFormat.NONE,
};
if (format is FileExportFormat.NONE)
return null;
var content = block.Lines.ToString();
var blockSeparator = format is FileExportFormat.TSV ? '\t' : ',';
var firstLine = content.AsSpan();
var lineEnd = firstLine.IndexOf('\n');
if (lineEnd >= 0)
firstLine = firstLine[..lineEnd];
var separatorPosition = firstLine.IndexOf(blockSeparator);
var fallback = (separatorPosition >= 0 ? firstLine[..separatorPosition] : firstLine).Trim().Trim('"').ToString();
return (fallback, format, content);
}
/// <summary>
/// Reads the text of a table cell or a heading, without the Markdown which decorates it.
/// </summary>
/// <remarks>
/// A spreadsheet has no use for the asterisks around a bold number: they would keep it from
/// being recognized as a number. So we keep what a reader would read and drop the rest.
/// </remarks>
private static string ToPlainText(MarkdownObject container)
{
//
// A leaf block, a heading for example, keeps its text in an inline container of its own.
// Asking the block itself for its descendants walks its child blocks, and a leaf block has
// none, so we would get nothing back. A table cell is a container block and needs the
// opposite: its text sits in the paragraphs below it.
//
var inlines = container is LeafBlock leafBlock
? leafBlock.Inline?.Descendants<LeafInline>() ?? []
: container.Descendants<LeafInline>();
var text = new StringBuilder();
foreach (var inline in inlines)
switch (inline)
{
case CodeInline code:
text.Append(code.Content);
break;
case LiteralInline literal:
text.Append(literal.Content.AsSpan());
break;
case HtmlEntityInline entity:
text.Append(entity.Transcoded.AsSpan());
break;
case AutolinkInline autolink:
text.Append(autolink.Url);
break;
// A cell holds one line in a file, so a line break inside it becomes a space:
case LineBreakInline:
text.Append(' ');
break;
}
return text.ToString().Trim();
}
/// <summary>
/// Writes the given text to a plain text file and lets the user save it.
/// </summary>
/// <param name="rustService">The Rust service, used for the save dialog.</param>
/// <param name="dialogTitle">The title of the save dialog. The caller knows what the user is
/// looking at, a chat message or the result of an assistant, so the caller names it.</param>
/// <param name="format">The format to write. Must be a format which does not use Pandoc.</param>
/// <param name="fileContent">What to write. The caller decides whether that is the entire
/// message or one table out of it.</param>
/// <param name="fileName">What the file is about, used to suggest a name in the save dialog.
/// Null falls back to a generic name.</param>
/// <returns>True, when the file was written.</returns>
public static async Task<bool> ToFile(RustService rustService, string dialogTitle, FileExportFormat format, string fileContent, string? fileName = null)
{
if (format.UsesPandoc() || format.ToFileTypeFilter() is not { } fileTypeFilter)
throw new ArgumentOutOfRangeException(nameof(format), format, "AI Studio cannot write this format itself.");
var response = await rustService.SaveFile(dialogTitle, [fileTypeFilter], format.ToSuggestedFileName(fileName));
if (response.UserCancelled)
{
LOGGER.LogInformation("User cancelled the save dialog.");
return false;
}
LOGGER.LogInformation("The user chose the path '{SaveFilePath}' for the {ExportFormat} export.", response.SaveFilePath, format);
try
{
await File.WriteAllTextAsync(response.SaveFilePath, fileContent, format.ToFileEncoding());
await MessageBus.INSTANCE.SendSuccess(new(Icons.Material.Filled.CheckCircle, TB("The export succeeded.")));
return true;
}
catch (Exception ex)
{
LOGGER.LogError(ex, "Error during {ExportFormat} export.", format);
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("The export failed.")));
return false;
}
}
}
@@ -351,6 +351,7 @@ public sealed class PluginConfiguration(bool isInternal, LuaState state, PluginT
ManagedConfiguration.TryProcessConfiguration(x => x.BatchProcessing, x => x.PromptFilePath, this.Id, settingsTable, dryRun);
ManagedConfiguration.TryProcessConfiguration(x => x.BatchProcessing, x => x.PreselectedPolicyId, this.Id, settingsTable, dryRun);
ManagedConfiguration.TryProcessConfiguration(x => x.BatchProcessing, x => x.OutputMode, this.Id, settingsTable, dryRun);
ManagedConfiguration.TryProcessConfiguration(x => x.BatchProcessing, x => x.ResultFileFormat, this.Id, settingsTable, dryRun);
ManagedConfiguration.TryProcessConfiguration(x => x.BatchProcessing, x => x.CsvFileName, this.Id, settingsTable, dryRun);
ManagedConfiguration.TryProcessConfiguration(x => x.BatchProcessing, x => x.ResultColumnHeader, this.Id, settingsTable, dryRun);
ManagedConfiguration.TryProcessConfiguration(x => x.BatchProcessing, x => x.CsvSeparator, this.Id, settingsTable, dryRun);
@@ -37,6 +37,7 @@ public static class FileTypes
/// Gets the standalone HTML filter used for visual briefing import and export.
/// </summary>
public static readonly FileTypeFilter VISUAL_BRIEFING_HTML = FileTypeFilter.Leaf(TB("Visual briefing"), "html");
public static readonly FileTypeFilter HTML = FileTypeFilter.Leaf("HTML", "html");
public static readonly FileTypeFilter APP = FileTypeFilter.Leaf("Swift/Kotlin", "swift", "kt");
public static readonly FileTypeFilter SHELL = FileTypeFilter.Leaf("Shell", "sh", "bash", "zsh");
public static readonly FileTypeFilter LOG = FileTypeFilter.Leaf("Log", "log");
@@ -53,8 +54,11 @@ public static class FileTypes
public static readonly FileTypeFilter MARKDOWN = FileTypeFilter.Leaf("Markdown", "md");
public static readonly FileTypeFilter TEXT = FileTypeFilter.Leaf(TB("Text"), "txt", "md", "rtf");
public static readonly FileTypeFilter TABULAR = FileTypeFilter.Leaf(TB("Tabular text"), "csv", "tsv");
public static readonly FileTypeFilter CSV = FileTypeFilter.Leaf("CSV", "csv");
public static readonly FileTypeFilter TSV = FileTypeFilter.Leaf("TSV", "tsv");
public static readonly FileTypeFilter MS_WORD = FileTypeFilter.Leaf("Microsoft Word", "docx");
public static readonly FileTypeFilter WORD = FileTypeFilter.Composite("Word", ["odt"], MS_WORD);
public static readonly FileTypeFilter ODT = FileTypeFilter.Leaf("OpenDocument Text", "odt");
public static readonly FileTypeFilter WORD = FileTypeFilter.Parent("Word", ODT, MS_WORD);
public static readonly FileTypeFilter EXCEL = FileTypeFilter.Leaf("Excel", "xls", "xlsx");
// The legacy binary ".ppt" is missing on purpose: AI Studio has no reader for it, so offering
@@ -63,6 +67,10 @@ public static class FileTypes
public static readonly FileTypeFilter MAIL = FileTypeFilter.Leaf(TB("Mail"), "eml", "msg", "mbox");
public static readonly FileTypeFilter LATEX = FileTypeFilter.Leaf("LaTeX", "tex", "bib", "sty", "cls", "log");
// Only the LaTeX document itself, without the auxiliary files of the LaTeX family: this is
// what we write when exporting, whereas the family above is what we accept when reading.
public static readonly FileTypeFilter TEX = FileTypeFilter.Leaf("LaTeX", "tex");
public static readonly FileTypeFilter OFFICE_FILES = FileTypeFilter.Parent(TB("Office Files"),
WORD, EXCEL, POWER_POINT, PDF);
public static readonly FileTypeFilter DOCUMENT = FileTypeFilter.Parent(TB("Document"),
@@ -27,8 +27,11 @@ public sealed class PandocAvailabilityService(RustService rustService, IDialogSe
/// </summary>
/// <param name="showSuccessMessage">Whether to show a success message if Pandoc is available.</param>
/// <param name="showDialog">Whether to show the installation dialog if Pandoc is not available.</param>
/// <param name="showErrorMessage">Whether to report a still missing Pandoc to the user. Turn
/// this off when you can say it better yourself, for example by naming the file which cannot
/// be read; otherwise the user reads two messages about the same thing.</param>
/// <returns>The Pandoc installation state.</returns>
public async Task<PandocInstallation> EnsureAvailabilityAsync(bool showSuccessMessage = false, bool showDialog = true)
public async Task<PandocInstallation> EnsureAvailabilityAsync(bool showSuccessMessage = false, bool showDialog = true, bool showErrorMessage = true)
{
// Check if Pandoc is available:
var pandocState = await Pandoc.CheckAvailabilityAsync(this.RustService, showMessages: false, showSuccessMessage: showSuccessMessage);
@@ -54,7 +57,8 @@ public sealed class PandocAvailabilityService(RustService rustService, IDialogSe
if (!pandocState.IsAvailable)
{
this.Logger.LogError("Pandoc is not available after installation attempt.");
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("Pandoc may be required for importing files.")));
if (showErrorMessage)
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, TB("AI Studio needs Pandoc for this, but it is not available.")));
}
}
+8 -21
View File
@@ -1,8 +1,6 @@
using AIStudio.Dialogs;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Rust;
using AIStudio.Tools.Services;
using DialogOptions = AIStudio.Dialogs.DialogOptions;
namespace AIStudio.Tools;
@@ -21,10 +19,10 @@ public static class UserFile
/// </remarks>
/// <param name="filePath">The full path to the file to be read. Must not be null or empty.</param>
/// <param name="rustService">Rust service used to read file content.</param>
/// <param name="dialogService">Dialogservice used to display the Pandoc installation dialog if needed.</param>
/// <param name="pandocAvailability">Makes sure Pandoc is there and offers its installation.</param>
/// <param name="token">Cancels the extraction when the caller no longer needs the content.</param>
/// <returns>The result of reading the file.</returns>
public static async Task<FileExtractionResult> LoadFileData(string filePath, RustService rustService, IDialogService dialogService, CancellationToken token = default)
public static async Task<FileExtractionResult> LoadFileData(string filePath, RustService rustService, PandocAvailabilityService pandocAvailability, CancellationToken token = default)
{
if (string.IsNullOrEmpty(filePath))
{
@@ -41,24 +39,13 @@ public static class UserFile
//
if (FileTypes.RequiresPandoc(filePath))
{
var pandocState = await Pandoc.CheckAvailabilityAsync(rustService, showSuccessMessage: false);
// We report a missing Pandoc ourselves, because we can name the file which cannot be read:
var pandocState = await pandocAvailability.EnsureAvailabilityAsync(showSuccessMessage: false, showDialog: true, showErrorMessage: false);
if (!pandocState.IsAvailable)
{
var dialogParameters = new DialogParameters<PandocDialog>
{
{ x => x.ShowInitialResultInSnackbar, false },
};
var dialogReference = await dialogService.ShowAsync<PandocDialog>(TB("Pandoc Installation"), dialogParameters, DialogOptions.FULLSCREEN);
await dialogReference.Result;
pandocState = await Pandoc.CheckAvailabilityAsync(rustService, showSuccessMessage: true);
if (!pandocState.IsAvailable)
{
LOGGER.LogError("Pandoc is not available after installation attempt, so '{FilePath}' cannot be read.", filePath);
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, FileExtractionErrorCode.PANDOC_UNAVAILABLE.ToUserMessage(fileName)));
return FileExtractionResult.Failed(FileExtractionErrorCode.PANDOC_UNAVAILABLE, "Pandoc is required to read this file, but it is not available.");
}
LOGGER.LogError("Pandoc is not available after installation attempt, so '{FilePath}' cannot be read.", filePath);
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.Cancel, FileExtractionErrorCode.PANDOC_UNAVAILABLE.ToUserMessage(fileName)));
return FileExtractionResult.Failed(FileExtractionErrorCode.PANDOC_UNAVAILABLE, "Pandoc is required to read this file, but it is not available.");
}
}