diff --git a/app/MindWork AI Studio/Assistants/I18N/allTexts.lua b/app/MindWork AI Studio/Assistants/I18N/allTexts.lua
index 9f0440f8..de185236 100644
--- a/app/MindWork AI Studio/Assistants/I18N/allTexts.lua
+++ b/app/MindWork AI Studio/Assistants/I18N/allTexts.lua
@@ -7708,6 +7708,9 @@ UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T1290340974"] = "Unknown configur
-- Copies the configuration slot to the clipboard
UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T1347508205"] = "Copies the configuration slot to the clipboard"
+-- Once the encoding of a text file is known, encoding_rs turns its content into the text AI Studio works with. Together with chardetng, this lets AI Studio read text, CSV, and similar files no matter which encoding they were saved in.
+UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T1378412877"] = "Once the encoding of a text file is known, encoding_rs turns its content into the text AI Studio works with. Together with chardetng, this lets AI Studio read text, CSV, and similar files no matter which encoding they were saved in."
+
-- This library is used to read PDF files. This is necessary, e.g., for using PDFs as a data source for a chat.
UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T1388816916"] = "This library is used to read PDF files. This is necessary, e.g., for using PDFs as a data source for a chat."
@@ -7819,6 +7822,9 @@ UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T234598990"] = "Linux AppImages b
-- Used PDFium version
UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T2368247719"] = "Used PDFium version"
+-- Text files are not always saved in the same encoding: files written on Windows often use a legacy one. chardetng recognizes which encoding a text file uses, so AI Studio can read it instead of rejecting it.
+UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T236832881"] = "Text files are not always saved in the same encoding: files written on Windows often use a legacy one. chardetng recognizes which encoding a text file uses, so AI Studio can read it instead of rejecting it."
+
-- installation provided by the system
UI_TEXT_CONTENT["AISTUDIO::PAGES::INFORMATION::T2371107659"] = "installation provided by the system"
@@ -9082,9 +9088,6 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::EXTERNALHTTPCLIENTTIMEOUT::T599774443"] = "The
-- policy files
UI_TEXT_CONTENT["AISTUDIO::TOOLS::EXTERNALHTTPCLIENTTIMEOUT::T632340680"] = "policy files"
--- No text could be read from the file '{0}', so it was not sent. The file might consist of scanned images without a text layer.
-UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T1268231046"] = "No text could be read from the file '{0}', so it was not sent. The file might consist of scanned images without a text layer."
-
-- The file type of '{0}' could not be determined, so the file was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T1459702734"] = "The file type of '{0}' could not be determined, so the file was not sent."
@@ -9115,6 +9118,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2793077828"]
-- The file '{0}' is not a readable PDF and was not sent. It might be damaged or transferred incompletely.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2891768359"] = "The file '{0}' is not a readable PDF and was not sent. It might be damaged or transferred incompletely."
+-- No text could be read from the file '{0}', so it was not sent. It might contain images only, such as a scanned PDF without a text layer, or no readable text at all.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2897122009"] = "No text could be read from the file '{0}', so it was not sent. It might contain images only, such as a scanned PDF without a text layer, or no readable text at all."
+
-- The file '{0}' is a {1}, which AI Studio cannot read, so it was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T3262447403"] = "The file '{0}' is a {1}, which AI Studio cannot read, so it was not sent."
diff --git a/app/MindWork AI Studio/Pages/Information.razor b/app/MindWork AI Studio/Pages/Information.razor
index 67ac1bdb..f0c8b60b 100644
--- a/app/MindWork AI Studio/Pages/Information.razor
+++ b/app/MindWork AI Studio/Pages/Information.razor
@@ -340,6 +340,8 @@
+
+
diff --git a/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua b/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua
index 29f48ce9..79a50468 100644
--- a/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua
+++ b/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua
@@ -9084,9 +9084,6 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::EXTERNALHTTPCLIENTTIMEOUT::T599774443"] = "Das
-- policy files
UI_TEXT_CONTENT["AISTUDIO::TOOLS::EXTERNALHTTPCLIENTTIMEOUT::T632340680"] = "Richtliniendateien"
--- No text could be read from the file '{0}', so it was not sent. The file might consist of scanned images without a text layer.
-UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T1268231046"] = "Aus der Datei „{0}“ konnte kein Text gelesen werden, daher wurde sie nicht gesendet. Die Datei könnte aus gescannten Bildern ohne Textebene bestehen."
-
-- The file type of '{0}' could not be determined, so the file was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T1459702734"] = "Der Dateityp von „{0}“ konnte nicht bestimmt werden. Daher wurde die Datei nicht gesendet."
@@ -9117,6 +9114,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2793077828"]
-- The file '{0}' is not a readable PDF and was not sent. It might be damaged or transferred incompletely.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2891768359"] = "Die Datei „{0}“ ist keine lesbare PDF-Datei und wurde nicht gesendet. Sie ist möglicherweise beschädigt oder wurde unvollständig übertragen."
+-- No text could be read from the file '{0}', so it was not sent. It might contain images only, such as a scanned PDF without a text layer, or no readable text at all.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2897122009"] = "Aus der Datei „{0}“ konnte kein Text gelesen werden, daher wurde sie nicht gesendet. Möglicherweise enthält sie nur Bilder, etwa ein gescanntes PDF ohne Textebene, oder gar keinen lesbaren Text."
+
-- The file '{0}' is a {1}, which AI Studio cannot read, so it was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T3262447403"] = "Die Datei „{0}“ ist eine {1}, die AI Studio nicht lesen kann. Daher wurde sie nicht gesendet."
diff --git a/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua b/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua
index f99b1e34..7ba82693 100644
--- a/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua
+++ b/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua
@@ -9084,9 +9084,6 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::EXTERNALHTTPCLIENTTIMEOUT::T599774443"] = "The
-- policy files
UI_TEXT_CONTENT["AISTUDIO::TOOLS::EXTERNALHTTPCLIENTTIMEOUT::T632340680"] = "policy files"
--- No text could be read from the file '{0}', so it was not sent. The file might consist of scanned images without a text layer.
-UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T1268231046"] = "No text could be read from the file '{0}', so it was not sent. The file might consist of scanned images without a text layer."
-
-- The file type of '{0}' could not be determined, so the file was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T1459702734"] = "The file type of '{0}' could not be determined, so the file was not sent."
@@ -9117,6 +9114,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2793077828"]
-- The file '{0}' is not a readable PDF and was not sent. It might be damaged or transferred incompletely.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2891768359"] = "The file '{0}' is not a readable PDF and was not sent. It might be damaged or transferred incompletely."
+-- No text could be read from the file '{0}', so it was not sent. It might contain images only, such as a scanned PDF without a text layer, or no readable text at all.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T2897122009"] = "No text could be read from the file '{0}', so it was not sent. It might contain images only, such as a scanned PDF without a text layer, or no readable text at all."
+
-- The file '{0}' is a {1}, which AI Studio cannot read, so it was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T3262447403"] = "The file '{0}' is a {1}, which AI Studio cannot read, so it was not sent."
diff --git a/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs b/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs
index acfa22f1..b903cbdd 100644
--- a/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs
+++ b/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs
@@ -81,7 +81,7 @@ internal static class FileExtractionResultExtensions
FileExtractionErrorCode.PDF_ENCRYPTED => TB("The file '{0}' is protected and could not be opened, so it was not sent."),
FileExtractionErrorCode.PDFIUM_UNAVAILABLE => TB("AI Studio was not able to start its PDF engine, so the file '{0}' was not sent."),
FileExtractionErrorCode.PANDOC_UNAVAILABLE => TB("Reading the file '{0}' needs Pandoc, which is not available, so the file was not sent."),
- FileExtractionErrorCode.NO_TEXT_EXTRACTED => TB("No text could be read from the file '{0}', so it was not sent. The file might consist of scanned images without a text layer."),
+ FileExtractionErrorCode.NO_TEXT_EXTRACTED => TB("No text could be read from the file '{0}', so it was not sent. It might contain images only, such as a scanned PDF without a text layer, or no readable text at all."),
FileExtractionErrorCode.NO_CONTENT => TB("The file '{0}' did not provide any content and was not sent."),
FileExtractionErrorCode.NOT_TEXT_CONTENT => TB("The file '{0}' is not a text file and was not sent. Its content could not be read as text, so it might have a wrong file extension."),
diff --git a/runtime/Cargo.lock b/runtime/Cargo.lock
index ea44a294..136f7cbb 100644
--- a/runtime/Cargo.lock
+++ b/runtime/Cargo.lock
@@ -1251,6 +1251,17 @@ dependencies = [
"whatlang",
]
+[[package]]
+name = "chardetng"
+version = "1.0.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "13de944a44b5064ee5d3a5ceccc49a41bfec50f2580e66f82e87703acdb88b53"
+dependencies = [
+ "cfg-if",
+ "encoding_rs",
+ "memchr",
+]
+
[[package]]
name = "chrono"
version = "0.4.44"
@@ -2160,9 +2171,9 @@ checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0"
[[package]]
name = "encoding_rs"
-version = "0.8.34"
+version = "0.8.35"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "b45de904aa0b010bce2ab45264d0631681847fa7b6f2eaa7dab7619943bc4f59"
+checksum = "75030f3c4f45dafd7586dd6780965a8c7e8e285a5ecb86713e63a79c5b2766f3"
dependencies = [
"cfg-if",
]
@@ -4062,7 +4073,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fc2f4eb4bc735547cfed7c0a4922cbd04a4655978c09b54f1f7b228750664c34"
dependencies = [
"cfg-if",
- "windows-targets 0.48.5",
+ "windows-targets 0.52.6",
]
[[package]]
@@ -4249,9 +4260,11 @@ dependencies = [
"calamine",
"cbc 0.2.1",
"cfg-if",
+ "chardetng",
"dbus-secret-service",
"dbus-secret-service-keyring-store",
"dirs",
+ "encoding_rs",
"file-format",
"flexi_logger",
"futures",
diff --git a/runtime/Cargo.toml b/runtime/Cargo.toml
index 0ec6b91a..b4ecbdda 100644
--- a/runtime/Cargo.toml
+++ b/runtime/Cargo.toml
@@ -46,6 +46,12 @@ rcgen = { version = "0.14.8", features = ["pem"] }
# is only detected as a plain archive.
file-format = { version = "0.29.0", features = ["reader-zip", "reader-cfb", "reader-txt", "reader-exe"] }
+# Text files are not always UTF-8: on Windows they are frequently encoded in Windows-1252, whose
+# umlauts are single bytes and therefore invalid UTF-8. chardetng guesses the encoding, encoding_rs
+# decodes it.
+chardetng = "1.0.0"
+encoding_rs = "0.8.35"
+
symphonia = { version = "0.6", default-features = false, features = ["aac", "aiff", "alac", "caf", "flac", "isomp4", "mkv", "mp1", "mp2", "mp3", "ogg", "pcm", "vorbis", "wav"] }
ropus = "=0.12.18"
rubato = { version = "4", default-features = false, features = ["fft_resampler"] }
diff --git a/runtime/src/file_data.rs b/runtime/src/file_data.rs
index 6fb44447..e0a59549 100644
--- a/runtime/src/file_data.rs
+++ b/runtime/src/file_data.rs
@@ -9,6 +9,8 @@ use axum::extract::rejection::QueryRejection;
use axum::response::sse::{Event, Sse};
use base64::{engine::general_purpose, Engine as _};
use calamine::{open_workbook_auto, Error as CalamineError, Reader};
+use chardetng::EncodingDetector;
+use encoding_rs::Encoding;
use file_format::{FileFormat, Kind};
use futures::{Stream, StreamExt};
use pdfium_render::prelude::{Pdfium, PdfiumError, PdfiumInternalError};
@@ -19,7 +21,7 @@ use std::path::Path;
use std::pin::Pin;
use std::fmt;
use log::{debug, error, warn};
-use tokio::io::{AsyncBufReadExt, AsyncReadExt};
+use tokio::io::AsyncReadExt;
use tokio::sync::mpsc;
use tokio_stream::wrappers::ReceiverStream;
@@ -554,14 +556,59 @@ async fn stream_data(file_path: &str, extract_images: bool) -> Result) -> Result {
- let file = tokio::fs::File::open(file_path).await?;
- let reader = tokio::io::BufReader::new(file);
- let mut lines = reader.lines();
- let mut line_number = 0;
+/// How many bytes we inspect for NUL bytes to tell binary content from text.
+const BINARY_PROBE_SIZE: usize = 8_192;
- // The stream outlives this call, so it needs its own copy of the path for diagnostics:
- let path = file_path.to_owned();
+/// Reads a text file and decodes it, no matter which encoding it uses.
+///
+/// Insisting on UTF-8 is not enough in practice: text files written on Windows are frequently
+/// encoded in Windows-1252, where umlauts are single bytes which UTF-8 rejects. Such a file used
+/// to look like it was not text at all.
+async fn read_text_file(file_path: &str) -> Result {
+ let bytes = tokio::fs::read(file_path).await.map_err(|error| ExtractionError::new(
+ classify_io_error(&error),
+ format!("The file could not be read: {error}"),
+ ))?;
+
+ //
+ // A byte order mark is authoritative and also covers UTF-16, which the detector below does not
+ // recognize. We therefore check it first and let `decode` act on it.
+ //
+ if let Some((encoding, _)) = Encoding::for_bom(&bytes) {
+ let (text, _, _) = encoding.decode(&bytes);
+ debug!("Decoded '{file_path}' as {name}, chosen by its byte order mark.", name = encoding.name());
+ return Ok(text.into_owned());
+ }
+
+ //
+ // Without a byte order mark, every byte sequence decodes into *something*, so the decoder can
+ // no longer tell us that a file is binary. NUL bytes do: they do not occur in text, and after
+ // the check above no UTF-16 file can reach this point.
+ //
+ let probe_length = min(bytes.len(), BINARY_PROBE_SIZE);
+ if bytes[..probe_length].contains(&0) {
+ return Err(ExtractionError::new(
+ ExtractionErrorCode::NotTextContent,
+ "The file contains binary data and is not a text file.",
+ ).into());
+ }
+
+ let mut detector = EncodingDetector::new();
+ detector.feed(&bytes, true);
+
+ let (text, encoding, had_errors) = detector.guess(None, true).decode(&bytes);
+ if had_errors {
+ warn!("Decoding '{file_path}' as {name} replaced malformed sequences.", name = encoding.name());
+ } else {
+ debug!("Decoded '{file_path}' as {name}.", name = encoding.name());
+ }
+
+ Ok(text.into_owned())
+}
+
+async fn stream_text_file(file_path: &str, use_md_fences: bool, fence_language: Option) -> Result {
+ let text = read_text_file(file_path).await?;
+ let mut line_number = 0;
let stream = stream! {
@@ -581,35 +628,12 @@ async fn stream_text_file(file_path: &str, use_md_fences: bool, fence_language:
};
}
- loop {
- match lines.next_line().await {
- Ok(Some(line)) => {
- line_number += 1;
- yield Ok(Chunk::new(
- line,
- Metadata::Text { line_number }
- ));
- },
-
- Ok(None) => break,
-
- //
- // Reading a line fails when the bytes are not valid UTF-8, which means the file is
- // not text at all. Treating that like the end of the file, as this loop did before,
- // turned every binary file into an empty document without a word.
- //
- Err(e) => {
- let code = if e.kind() == std::io::ErrorKind::InvalidData {
- ExtractionErrorCode::NotTextContent
- } else {
- classify_io_error(&e)
- };
-
- error!("Reading '{path}' as text failed after {line_number} line(s) ({code:?}): {e}");
- yield Err(ExtractionError::new(code, format!("The file could not be read as text: {e}")).into());
- break;
- },
- }
+ for line in text.lines() {
+ line_number += 1;
+ yield Ok(Chunk::new(
+ line.to_string(),
+ Metadata::Text { line_number }
+ ));
}
if use_md_fences {
@@ -864,21 +888,44 @@ async fn convert_with_pandoc(
.with_output_format(to)
.build()
.command.output().await?;
-
+
+ let exit_code = output.status.code();
+ let stderr_text = String::from_utf8_lossy(&output.stderr).trim().to_string();
+ debug!("Pandoc converted '{file_path}' from '{from}' to '{to}': exit={exit_code:?}, {stdout_length} byte(s) of output.", stdout_length = output.stdout.len());
+
+ if !stderr_text.is_empty() {
+ warn!("Pandoc reported while converting '{file_path}': {stderr_text}");
+ }
+
let stream = stream! {
- if output.status.success() {
- match String::from_utf8(output.stdout.clone()) {
+ if !output.status.success() {
+ yield Err(ExtractionError::new(
+ ExtractionErrorCode::Internal,
+ format!("Pandoc failed with exit code {exit_code:?}: {stderr_text}"),
+ ).into());
+ } else {
+ match String::from_utf8(output.stdout) {
+ //
+ // Pandoc succeeded, yet nothing came out. Passing that on as content would hand an
+ // empty document to the AI, which is exactly what this whole path must not do.
+ //
+ Ok(content) if content.trim().is_empty() => {
+ yield Err(ExtractionError::new(
+ ExtractionErrorCode::NoTextExtracted,
+ format!("Pandoc read the file without finding any text{separator}{stderr_text}", separator = if stderr_text.is_empty() { "." } else { ": " }),
+ ).into());
+ },
+
Ok(content) => yield Ok(Chunk::new(
content,
Metadata::Document {}
)),
- Err(e) => yield Err(e.into()),
+
+ Err(e) => yield Err(ExtractionError::new(
+ ExtractionErrorCode::Internal,
+ format!("The output of Pandoc was not valid UTF-8: {e}"),
+ ).into()),
}
- } else {
- yield Err(format!(
- "Pandoc error: {}",
- String::from_utf8_lossy(&output.stderr)
- ).into());
}
};