diff --git a/app/MindWork AI Studio/Assistants/I18N/allTexts.lua b/app/MindWork AI Studio/Assistants/I18N/allTexts.lua
index b691858b..a6fa350e 100644
--- a/app/MindWork AI Studio/Assistants/I18N/allTexts.lua
+++ b/app/MindWork AI Studio/Assistants/I18N/allTexts.lua
@@ -10723,6 +10723,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1465212038
-- The file '{0}' does not exist anymore and was not indexed.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1553912802"] = "The file '{0}' does not exist anymore and was not indexed."
+-- Not a readable document
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1671731444"] = "Not a readable document"
+
-- No text could be read from the file '{0}', so it was not indexed. It might contain images only, such as a scanned PDF without a text layer. AI Studio reads it again as soon as the file changes.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1675617688"] = "No text could be read from the file '{0}', so it was not indexed. It might contain images only, such as a scanned PDF without a text layer. AI Studio reads it again as soon as the file changes."
@@ -10804,6 +10807,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T4041351522
-- File is open elsewhere
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T4201096587"] = "File is open elsewhere"
+-- The file '{0}' is not a readable document, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T564482210"] = "The file '{0}' is not a readable document, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes."
+
-- PDF system unavailable
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T800475300"] = "PDF system unavailable"
@@ -10876,6 +10882,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T4291141931"]
-- Reading the file '{0}' needs Pandoc, which is not available, so the file was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T594894810"] = "Reading the file '{0}' needs Pandoc, which is not available, so the file was not sent."
+-- The file '{0}' is not a readable document and was not sent. It might be damaged or transferred incompletely.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T985448614"] = "The file '{0}' is not a readable document and was not sent. It might be damaged or transferred incompletely."
+
-- AI Studio couldn't install Pandoc because the archive was not found.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::PANDOC::T1059477764"] = "AI Studio couldn't install Pandoc because the archive was not found."
diff --git a/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua b/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua
index 515797b3..180a0c33 100644
--- a/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua
+++ b/app/MindWork AI Studio/Plugins/languages/de-de-43065dbc-78d0-45b7-92be-f14c2926e2dc/plugin.lua
@@ -10725,6 +10725,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1465212038
-- The file '{0}' does not exist anymore and was not indexed.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1553912802"] = "Die Datei „{0}“ existiert nicht mehr und wurde nicht indexiert."
+-- Not a readable document
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1671731444"] = "Kein lesbares Dokument"
+
-- No text could be read from the file '{0}', so it was not indexed. It might contain images only, such as a scanned PDF without a text layer. AI Studio reads it again as soon as the file changes.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1675617688"] = "Aus der Datei „{0}“ konnte kein Text gelesen werden, daher wurde sie nicht indexiert. Möglicherweise enthält sie nur Bilder, etwa ein gescanntes PDF ohne Textebene. AI Studio liest sie erneut ein, sobald sich die Datei ändert."
@@ -10806,6 +10809,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T4041351522
-- File is open elsewhere
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T4201096587"] = "Die Datei ist an anderer Stelle geöffnet."
+-- The file '{0}' is not a readable document, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T564482210"] = "Die Datei „{0}“ ist kein lesbares Dokument und wurde nicht indexiert. Möglicherweise ist sie beschädigt oder unvollständig übertragen worden. AI Studio liest sie erneut ein, sobald sich die Datei ändert."
+
-- PDF system unavailable
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T800475300"] = "PDF-System nicht verfügbar"
@@ -10878,6 +10884,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T4291141931"]
-- Reading the file '{0}' needs Pandoc, which is not available, so the file was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T594894810"] = "Zum Lesen der Datei „{0}“ wird Pandoc benötigt. Da Pandoc nicht verfügbar ist, wurde die Datei nicht gesendet."
+-- The file '{0}' is not a readable document and was not sent. It might be damaged or transferred incompletely.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T985448614"] = "Die Datei „{0}“ ist kein lesbares Dokument und wurde nicht gesendet. Möglicherweise ist sie beschädigt oder unvollständig übertragen worden."
+
-- AI Studio couldn't install Pandoc because the archive was not found.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::PANDOC::T1059477764"] = "AI Studio konnte Pandoc nicht installieren, da das Archiv nicht gefunden wurde."
diff --git a/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua b/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua
index 455699e6..31b9f0bd 100644
--- a/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua
+++ b/app/MindWork AI Studio/Plugins/languages/en-us-97dfb1ba-50c4-4440-8dfa-6575daf543c8/plugin.lua
@@ -10725,6 +10725,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1465212038
-- The file '{0}' does not exist anymore and was not indexed.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1553912802"] = "The file '{0}' does not exist anymore and was not indexed."
+-- Not a readable document
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1671731444"] = "Not a readable document"
+
-- No text could be read from the file '{0}', so it was not indexed. It might contain images only, such as a scanned PDF without a text layer. AI Studio reads it again as soon as the file changes.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T1675617688"] = "No text could be read from the file '{0}', so it was not indexed. It might contain images only, such as a scanned PDF without a text layer. AI Studio reads it again as soon as the file changes."
@@ -10806,6 +10809,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T4041351522
-- File is open elsewhere
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T4201096587"] = "File is open elsewhere"
+-- The file '{0}' is not a readable document, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T564482210"] = "The file '{0}' is not a readable document, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes."
+
-- PDF system unavailable
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONERRORCODEEXTENSIONS::T800475300"] = "PDF system unavailable"
@@ -10878,6 +10884,9 @@ UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T4291141931"]
-- Reading the file '{0}' needs Pandoc, which is not available, so the file was not sent.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T594894810"] = "Reading the file '{0}' needs Pandoc, which is not available, so the file was not sent."
+-- The file '{0}' is not a readable document and was not sent. It might be damaged or transferred incompletely.
+UI_TEXT_CONTENT["AISTUDIO::TOOLS::FILEEXTRACTIONRESULTEXTENSIONS::T985448614"] = "The file '{0}' is not a readable document and was not sent. It might be damaged or transferred incompletely."
+
-- AI Studio couldn't install Pandoc because the archive was not found.
UI_TEXT_CONTENT["AISTUDIO::TOOLS::PANDOC::T1059477764"] = "AI Studio couldn't install Pandoc because the archive was not found."
diff --git a/app/MindWork AI Studio/Tools/FileExtractionErrorCode.cs b/app/MindWork AI Studio/Tools/FileExtractionErrorCode.cs
index bb7c6a63..6b69d8a2 100644
--- a/app/MindWork AI Studio/Tools/FileExtractionErrorCode.cs
+++ b/app/MindWork AI Studio/Tools/FileExtractionErrorCode.cs
@@ -32,6 +32,13 @@ public enum FileExtractionErrorCode
FORMAT_DETECTION_FAILED,
NOT_A_VALID_PDF,
NOT_A_VALID_SPREADSHEET,
+
+ ///
+ /// The package of a Word, OpenDocument, or presentation file is broken, e.g. a damaged
+ /// archive or a missing part inside it.
+ ///
+ NOT_A_VALID_DOCUMENT,
+
PDFIUM_UNAVAILABLE,
PDF_ENCRYPTED,
PAGE_EXTRACTION_FAILED,
diff --git a/app/MindWork AI Studio/Tools/FileExtractionErrorCodeExtensions.cs b/app/MindWork AI Studio/Tools/FileExtractionErrorCodeExtensions.cs
index a5366827..ee07c00c 100644
--- a/app/MindWork AI Studio/Tools/FileExtractionErrorCodeExtensions.cs
+++ b/app/MindWork AI Studio/Tools/FileExtractionErrorCodeExtensions.cs
@@ -32,6 +32,7 @@ internal static class FileExtractionErrorCodeExtensions
FileExtractionErrorCode.NOT_TEXT_CONTENT => true,
FileExtractionErrorCode.NOT_A_VALID_PDF => true,
FileExtractionErrorCode.NOT_A_VALID_SPREADSHEET => true,
+ FileExtractionErrorCode.NOT_A_VALID_DOCUMENT => true,
FileExtractionErrorCode.PDF_ENCRYPTED => true,
FileExtractionErrorCode.FORMAT_DETECTION_FAILED => true,
FileExtractionErrorCode.EXECUTABLE_REJECTED => true,
@@ -65,6 +66,7 @@ internal static class FileExtractionErrorCodeExtensions
FileExtractionErrorCode.NOT_TEXT_CONTENT => TB("Not a text file"),
FileExtractionErrorCode.NOT_A_VALID_PDF => TB("Not a readable PDF"),
FileExtractionErrorCode.NOT_A_VALID_SPREADSHEET => TB("Not a readable spreadsheet"),
+ FileExtractionErrorCode.NOT_A_VALID_DOCUMENT => TB("Not a readable document"),
FileExtractionErrorCode.PDF_ENCRYPTED => TB("Protected PDF"),
FileExtractionErrorCode.FORMAT_DETECTION_FAILED => TB("Unknown file type"),
FileExtractionErrorCode.EXECUTABLE_REJECTED => TB("Executable program"),
@@ -115,6 +117,7 @@ internal static class FileExtractionErrorCodeExtensions
FileExtractionErrorCode.NOT_TEXT_CONTENT => TB("The file '{0}' is not a text file, so it was not indexed. Its content could not be read as text, which means it might have a wrong file extension. AI Studio reads it again as soon as the file changes."),
FileExtractionErrorCode.NOT_A_VALID_PDF => TB("The file '{0}' is not a readable PDF, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes."),
FileExtractionErrorCode.NOT_A_VALID_SPREADSHEET => TB("The file '{0}' is not a readable spreadsheet, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes."),
+ FileExtractionErrorCode.NOT_A_VALID_DOCUMENT => TB("The file '{0}' is not a readable document, so it was not indexed. It might be damaged or transferred incompletely. AI Studio reads it again as soon as the file changes."),
FileExtractionErrorCode.PDF_ENCRYPTED => TB("The file '{0}' is protected and could not be opened, so it was not indexed. AI Studio reads it again as soon as the file changes."),
FileExtractionErrorCode.FORMAT_DETECTION_FAILED => TB("The file type of '{0}' could not be determined, so the file was not indexed. AI Studio reads it again as soon as the file changes."),
FileExtractionErrorCode.EXECUTABLE_REJECTED => TB("The file '{0}' is an executable program and was not indexed, regardless of its file extension."),
diff --git a/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs b/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs
index b903cbdd..82d82771 100644
--- a/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs
+++ b/app/MindWork AI Studio/Tools/FileExtractionResultExtensions.cs
@@ -78,6 +78,7 @@ internal static class FileExtractionResultExtensions
FileExtractionErrorCode.TIMEOUT => TB("Reading the file '{0}' took too long and was stopped, so the file was not sent. When the file is stored on a network drive, the connection might be slow or interrupted."),
FileExtractionErrorCode.NOT_A_VALID_PDF => TB("The file '{0}' is not a readable PDF and was not sent. It might be damaged or transferred incompletely."),
FileExtractionErrorCode.NOT_A_VALID_SPREADSHEET => TB("The file '{0}' is not a readable spreadsheet and was not sent. It might be damaged or transferred incompletely."),
+ FileExtractionErrorCode.NOT_A_VALID_DOCUMENT => TB("The file '{0}' is not a readable document and was not sent. It might be damaged or transferred incompletely."),
FileExtractionErrorCode.PDF_ENCRYPTED => TB("The file '{0}' is protected and could not be opened, so it was not sent."),
FileExtractionErrorCode.PDFIUM_UNAVAILABLE => TB("AI Studio was not able to start its PDF engine, so the file '{0}' was not sent."),
FileExtractionErrorCode.PANDOC_UNAVAILABLE => TB("Reading the file '{0}' needs Pandoc, which is not available, so the file was not sent."),
diff --git a/runtime/src/file_data.rs b/runtime/src/file_data.rs
index 82018eae..cdf0de63 100644
--- a/runtime/src/file_data.rs
+++ b/runtime/src/file_data.rs
@@ -12,12 +12,12 @@ use axum::response::sse::{Event, Sse};
use base64::{engine::general_purpose, Engine as _};
use calamine::{open_workbook_auto, Error as CalamineError, Reader};
use chardetng::{EncodingDetector, Iso2022JpDetection, Utf8Detection};
-use docx_to_md::{DocumentContainer, ImageHandlingMode as DocumentImageHandlingMode, Metadata as DocumentMetadata, ParserConfig as DocumentParserConfig};
+use docx_to_md::{DocumentContainer, Error as DocumentError, ImageHandlingMode as DocumentImageHandlingMode, Metadata as DocumentMetadata, ParserConfig as DocumentParserConfig};
use encoding_rs::Encoding;
use file_format::{FileFormat, Kind};
use futures::{Stream, StreamExt};
use pdfium_render::prelude::{Pdfium, PdfiumError, PdfiumInternalError};
-use pptx_to_md::{DiagnosticSeverity, ImageHandlingMode, MarkdownOptions, ParserConfig, PresentationContainer, PresentationFormat, PresentationMetadata, ReadingOrder};
+use pptx_to_md::{DiagnosticSeverity, Error as PresentationError, ImageHandlingMode, MarkdownOptions, ParserConfig, PresentationContainer, PresentationFormat, PresentationMetadata, ReadingOrder};
use serde::{Deserialize, Deserializer, Serialize};
use serde::de::{Error as SerdeError, Visitor};
use std::path::Path;
@@ -175,6 +175,11 @@ pub enum ExtractionErrorCode {
FormatDetectionFailed,
NotAValidPdf,
NotAValidSpreadsheet,
+
+ /// The package of a document or presentation is broken, e.g. a damaged ZIP or a missing part.
+ /// The counterpart of `NotAValidPdf` and `NotAValidSpreadsheet` for the OOXML and ODF formats.
+ NotAValidDocument,
+
PdfiumUnavailable,
PdfEncrypted,
PageExtractionFailed,
@@ -1219,6 +1224,48 @@ fn classify_spreadsheet_error_code(error: &CalamineError) -> ExtractionErrorCode
}
}
+/// Classifies a failure of the document reader, so a broken package is told apart from a file
+/// which is merely out of reach right now.
+///
+/// The distinction decides how long a file stays out of the index: a damaged ZIP or a missing
+/// `content.xml` is a property of the file and will fail the same way on every run, while a
+/// network share which went away is worth another attempt. Without this, both arrived as an
+/// unclassified failure and every run read the broken file again.
+///
+/// What remains uncoded are the failures of our own image handling. They say nothing about the
+/// document, so they keep the generic code.
+fn classify_document_error(error: &DocumentError) -> ExtractionErrorCode {
+ match error {
+ DocumentError::Io(io_error) => classify_io_error(io_error),
+
+ DocumentError::Zip(_)
+ | DocumentError::Xml { .. }
+ | DocumentError::Utf8 { .. }
+ | DocumentError::UnknownFormat
+ | DocumentError::FormatMismatch { .. }
+ | DocumentError::MissingPart(_)
+ | DocumentError::InvalidRelationship { .. } => ExtractionErrorCode::NotAValidDocument,
+
+ _ => ExtractionErrorCode::Internal,
+ }
+}
+
+/// Classifies a failure of the presentation reader. Same reasoning as for documents above.
+fn classify_presentation_error(error: &PresentationError) -> ExtractionErrorCode {
+ match error {
+ PresentationError::Io(io_error) => classify_io_error(io_error),
+
+ PresentationError::Zip(_)
+ | PresentationError::Xml { .. }
+ | PresentationError::Utf8(_)
+ | PresentationError::ParseError(_)
+ | PresentationError::SlideNotFound
+ | PresentationError::RelationshipNotFound => ExtractionErrorCode::NotAValidDocument,
+
+ _ => ExtractionErrorCode::Internal,
+ }
+}
+
async fn stream_spreadsheet_as_csv(file_path: &str) -> Result {
let path = file_path.to_owned();
let (tx, rx) = mpsc::channel(10);
@@ -1465,7 +1512,7 @@ async fn stream_document(file_path: &str, extract_images: bool, stream_id: &str)
Ok(document) => document,
Err(e) => {
let _ = tx.blocking_send(Err(ExtractionError::new(
- ExtractionErrorCode::FileNotReadable,
+ classify_document_error(&e),
format!("The document could not be read: {e}"),
).into()));
return;
@@ -1476,7 +1523,7 @@ async fn stream_document(file_path: &str, extract_images: bool, stream_id: &str)
Ok(pages) => pages,
Err(e) => {
let _ = tx.blocking_send(Err(ExtractionError::new(
- ExtractionErrorCode::FileNotReadable,
+ classify_document_error(&e),
format!("The pages of the document could not be read: {e}"),
).into()));
return;
@@ -1498,7 +1545,7 @@ async fn stream_document(file_path: &str, extract_images: bool, stream_id: &str)
Ok(page) => page,
Err(e) => {
let _ = tx.blocking_send(Err(ExtractionError::new(
- ExtractionErrorCode::Internal,
+ classify_document_error(&e),
format!("A page of the document could not be read: {e}"),
).into()));
return;
@@ -1508,7 +1555,7 @@ async fn stream_document(file_path: &str, extract_images: bool, stream_id: &str)
Ok(content) => content,
Err(e) => {
let _ = tx.blocking_send(Err(ExtractionError::new(
- ExtractionErrorCode::Internal,
+ classify_document_error(&e),
format!("Page {page_number} of the document could not be converted: {e}", page_number = page.page_number),
).into()));
return;
@@ -1601,7 +1648,10 @@ async fn stream_presentation(file_path: &str, extract_images: bool, format: Pres
};
let mut streamer = tokio::task::spawn_blocking(move || {
- PresentationContainer::open_as(&path, parser_config, format).map_err(|e| Box::new(e) as Box)
+ PresentationContainer::open_as(&path, parser_config, format).map_err(|e| Box::new(ExtractionError::new(
+ classify_presentation_error(&e),
+ format!("The presentation could not be read: {e}"),
+ )) as Box)
}).await??;
let (tx, rx) = mpsc::channel(32);
@@ -1618,7 +1668,10 @@ async fn stream_presentation(file_path: &str, extract_images: bool, format: Pres
let slide = match slide_result {
Ok(slide) => slide,
Err(e) => {
- let _ = tx.blocking_send(Err(Box::new(e) as Box));
+ let _ = tx.blocking_send(Err(ExtractionError::new(
+ classify_presentation_error(&e),
+ format!("A slide of the presentation could not be read: {e}"),
+ ).into()));
return;
},
};
@@ -1644,7 +1697,10 @@ async fn stream_presentation(file_path: &str, extract_images: bool, format: Pres
let mut content = match slide.to_markdown(&markdown_options) {
Ok(content) => content,
Err(e) => {
- let _ = tx.blocking_send(Err(Box::new(e) as Box));
+ let _ = tx.blocking_send(Err(ExtractionError::new(
+ classify_presentation_error(&e),
+ format!("Slide {slide_number} of the presentation could not be converted: {e}", slide_number = slide.slide_number),
+ ).into()));
return;
},
};