using AIStudio.Provider; using AIStudio.Settings.DataModel; using AIStudio.Tools.Databases.IndexStore; using AIStudio.Tools.PluginSystem; using static AIStudio.Tools.Services.Indexing.IndexingLogFormat; namespace AIStudio.Tools.Services.Indexing; /// /// How far an indexing run got through the documents of its data source, and what became of each. /// /// /// Counts the documents, keeps the list of failures and turns both into the status the embedding /// page shows. What happens to a document after it was indexed, or after it failed, is the same for /// every kind of data source, so that is decided here as well: the stores learn about it, the /// manifest follows, and the user hears about it the way the failure deserves. /// /// The run the documents are indexed in. /// How many documents the data source has, readable or not. /// How many of them could not even be looked at. /// Why the last of those could not be looked at, or an empty string. /// The failures of those which could not be looked at. /// The logger of the embedding service, so the log reads the same whoever writes it. internal sealed class DocumentRunProgress(IndexedRunContext context, int totalDocuments, int failedInputs, string lastInputError, IEnumerable inputFailures, ILogger logger) { /// /// How often the block progress within one document is reported to the user interface at most. /// private static readonly TimeSpan BLOCK_PROGRESS_INTERVAL = TimeSpan.FromSeconds(3); private readonly List failures = inputFailures.ToList(); // // Which kinds of provider failure the user was already told about in this run. A rejected // API key is the same problem for every one of a few thousand documents, and one message // is what it takes to send the user to the settings. // private readonly HashSet reportedFailureReasons = []; private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(DocumentRunProgress).Namespace, nameof(DocumentRunProgress)); public int TotalDocuments => totalDocuments; /// /// Documents which were left alone because nothing changed since they were indexed. /// public int UnchangedDocuments { get; private set; } /// /// Documents which were not read because they failed for a reason of their own before. /// public int PermanentlySkippedDocuments { get; private set; } /// /// Documents which were indexed in this run. /// public int IndexedDocuments { get; private set; } public int NewDocuments { get; private set; } public int ChangedDocuments { get; private set; } public int FailedDocuments { get; private set; } = failedInputs; public string LastError { get; private set; } = lastInputError; /// /// When the data source was last worked through as a whole, for those which are synced rather /// than watched. Every status of the run carries it. /// public DateTimeOffset? LastSyncUtc { get; set; } /// /// Documents which need nothing more in this run, whether they were indexed now or before. /// public int DoneDocuments => this.UnchangedDocuments + this.IndexedDocuments; /// /// Counts documents which nothing changed about since they were indexed. /// /// How many of them. public void RecordUnchanged(int count = 1) => this.UnchangedDocuments += count; /// /// Counts a document which is not read again, because it failed for a reason of its own before /// and has not changed since. /// /// The key of the document. /// Why it failed back then. public void RecordStillUnreadable(string documentKey, PermanentIndexingFailureRecord failure) { this.PermanentlySkippedDocuments++; // The stored reason keeps its place in the list, so the user still sees why: this.failures.Add(new DataSourceEmbeddingFailure(documentKey, failure.Message, failure.OccurredAtUtc, ExtractionCode: failure.Code, IsPermanent: true)); } /// /// Tells the user interface where the run stands. /// /// The name of the document being worked on, or an empty string. /// The block of that document being worked on, when known. /// The page that block is on, when known. public void Publish(string currentDocument = "", int? currentBlock = null, int? currentPage = null) => context.PublishStatus(this.CreateStatus(DataSourceEmbeddingState.RUNNING, currentDocument, this.LastError, currentBlock, currentPage)); /// /// Announces that work on a document starts, and hands out what reports its progress. /// /// The document. /// Told about every block of the document, with its number and its page. public Action BeginDocument(EmbeddingDocument document) { this.Publish(document.DisplayName); // // What the page says while one document is being worked on. Without it, a document of // several thousand pages leaves the same sentence standing for hours, and a progress // which never moves cannot be told apart from one which is stuck. // var lastBlockReportUtc = DateTimeOffset.MinValue; return (blockNumber, pageNumber) => { // // The first block goes out at once, so the line is there instead of blank. After // that, at most one message every BLOCK_PROGRESS_INTERVAL: each one re-renders the // embedding page, the navigation bar and the table in the settings, and the blocks // of a large document arrive far faster than anybody can read them. // var nowUtc = DateTimeOffset.UtcNow; if (blockNumber > 1 && nowUtc - lastBlockReportUtc < BLOCK_PROGRESS_INTERVAL) return; lastBlockReportUtc = nowUtc; this.Publish(document.DisplayName, blockNumber, pageNumber); }; } /// /// Records a document as indexed, in the index store as well as in the manifest. /// /// /// Called once the kind of data source is sure the document did not change while it was read. /// Until then, its index row says it has no chunks. /// /// The document. /// How many chunks were stored for it. /// Whether the index knew nothing about it before. /// The cancellation token. public async Task RecordDocumentIndexedAsync(EmbeddingDocument document, int chunkCount, bool isNew, CancellationToken token) { var embeddedAtUtc = DateTimeOffset.UtcNow; var state = document.State with { ChunkCount = chunkCount, EmbeddedAtUtc = embeddedAtUtc }; await context.IndexStore.UpsertFileAsync(context.DataSource.Id, state, token); context.Manifest.Files[document.Key] = new EmbeddedFileRecord(state.Fingerprint, state.FileSize, state.LastWriteUtc, embeddedAtUtc, chunkCount); await context.ForgetPermanentFailureAsync(document.Key, token); this.IndexedDocuments++; if (isNew) this.NewDocuments++; else this.ChangedDocuments++; } /// /// Records why a document could not be indexed, and removes what the attempt left behind. /// /// /// Never called for a cancelled run, nor for a vector store which cannot be read at all: those /// are not about one document, and whoever indexes has to let them through. /// /// The document. /// What went wrong. /// The cancellation token. public async Task RecordDocumentFailureAsync(EmbeddingDocument document, Exception exception, CancellationToken token) { var dataSource = context.DataSource; switch (exception) { case ProviderRequestException providerFailure: { // // The provider said what went wrong and what the user can do about it. That // sentence is what goes into the status, together with the classification the UI // needs to offer the matching way out. // this.FailedDocuments++; this.LastError = providerFailure.UserMessage; this.failures.Add(new DataSourceEmbeddingFailure(document.Key, providerFailure.UserMessage, DateTimeOffset.UtcNow, providerFailure.FailureReason, providerFailure.StatusCode, context.EmbeddingProvider.Name, DisplayName: document.DisplayName)); context.Manifest.Files.Remove(document.Key); await context.ForgetPermanentFailureAsync(document.Key, token); await context.CleanupFailedDocumentAsync(document.Key, token); logger.LogWarning( providerFailure, "Failed to embed file '{FilePath}' for data source '{DataSourceName}' because the embedding provider '{EmbeddingProviderName}' failed. FailureReason={FailureReason}, StatusCode={StatusCode}.", document.Key, dataSource.Name, context.EmbeddingProvider.Name, providerFailure.FailureReason, providerFailure.StatusCode); this.Publish(document.DisplayName); // Once per kind of failure, not once per document: if (this.reportedFailureReasons.Add(providerFailure.FailureReason)) await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.CloudOff, providerFailure.UserMessage)); break; } case FileExtractionException extractionFailure when extractionFailure.Code.IsPermanentIndexingFailure(): { // // The document itself is why this failed, so trying it again changes nothing until // the document does. The reason is written into the index, and the fingerprint next // to it decides when to come back: an OCR run over a scanned PDF changes both size // and write time, which is exactly the moment the file deserves another attempt. // this.PermanentlySkippedDocuments++; var occurredAtUtc = DateTimeOffset.UtcNow; var indexingMessage = this.GetFailureMessage(extractionFailure.Code, document); this.failures.Add(new DataSourceEmbeddingFailure(document.Key, indexingMessage, occurredAtUtc, ExtractionCode: extractionFailure.Code, IsPermanent: true, DisplayName: document.DisplayName)); context.Manifest.Files.Remove(document.Key); await context.CleanupFailedDocumentAsync(document.Key, token); var state = document.State; context.Manifest.PermanentFailures[state.AbsolutePath] = new PermanentIndexingFailureRecord(state.Fingerprint, extractionFailure.Code, indexingMessage, occurredAtUtc); await context.IndexStore.UpsertPermanentFailureAsync( dataSource.Id, new PermanentIndexingFailure(state.ParentFileId, state.AbsolutePath, state.Fingerprint, extractionFailure.Code, indexingMessage, occurredAtUtc), token); logger.LogInformation( extractionFailure, "Skipping file '{FilePath}' of data source '{DataSourceName}' ({DataSourceId}) from now on because reading it failed for a reason which lies in the file. FailureCode={FailureCode}, MetadataHashPrefix={MetadataHashPrefix}.", document.Key, dataSource.Name, dataSource.Id, extractionFailure.Code, ShortHash(state.Fingerprint)); this.Publish(document.DisplayName); break; } default: { // // Everything which is not the provider's doing: a document which changed while it // was read, one which yielded no text, a vector store which refused to store. These // are about this one document, so they go into the list and not into a message // which would interrupt whatever the user is doing right now. // this.FailedDocuments++; var extractionCode = exception is FileExtractionException extractionFailure ? extractionFailure.Code : FileExtractionErrorCode.NONE; // // Deliberately not the message of the exception: that one is written for the log // file, in English, and repeats the path which the list shows anyway. // var failureMessage = this.GetFailureMessage(extractionCode, document); this.LastError = failureMessage; this.failures.Add(new DataSourceEmbeddingFailure(document.Key, failureMessage, DateTimeOffset.UtcNow, EmbeddingProviderName: context.EmbeddingProvider.Name, ExtractionCode: extractionCode, DisplayName: document.DisplayName)); context.Manifest.Files.Remove(document.Key); await context.ForgetPermanentFailureAsync(document.Key, token); await context.CleanupFailedDocumentAsync(document.Key, token); logger.LogWarning(exception, "Failed to embed file '{FilePath}' for data source '{DataSourceName}'.", document.Key, dataSource.Name); this.Publish(document.DisplayName); break; } } } /// /// Finishes the run: the collection is tidied up, the data source is marked as worked through, /// and the user interface learns how it went. /// /// /// The hash is written last on purpose. It is what says that a run got through the whole data /// source, so a run which stops before this point leaves the data source marked as unfinished. /// /// The hash of the data source as this run found it. /// Why the collection is optimized now, for the log. /// The cancellation token. public async Task CompleteRunAsync(string sourceHash, string reason, CancellationToken token) { await this.StoreSourceHashAsync(sourceHash, reason, token); // // Documents which were skipped for good do not make a run unsuccessful: nothing is left to // try, and a data source made of nothing but scanned images would otherwise ask for // attention forever. // var hasFailures = this.FailedDocuments > 0; var lastError = hasFailures ? string.IsNullOrWhiteSpace(this.LastError) ? context.DataSource is DataSourceMailbox ? TB("Some mails could not be indexed. The list below says which ones and why.") : TB("Some files could not be indexed. The list below says which ones and why.") : this.LastError : string.Empty; context.PublishStatus(this.CreateStatus(hasFailures ? DataSourceEmbeddingState.FAILED : DataSourceEmbeddingState.COMPLETED, string.Empty, lastError, null, null)); } /// /// Ends a run before it got through the whole data source, keeping what it did for the next one. /// /// /// Stores everything CompleteRunAsync stores, the hash included. For a data source which is /// worked through in several runs, the hash therefore says that its index can be searched, not /// that the data source was worked through as a whole: that is what LastSyncUtc tells. Without /// it, a mailbox would stay out of every search for the hours its first sync takes. /// /// The user interface learns nothing here. Whatever comes next tells it: the run being queued /// again, or the question the user has to answer first. /// /// The hash of the data source as this run found it. /// Why the collection is optimized now, for the log. /// The cancellation token. public Task PauseRunAsync(string sourceHash, string reason, CancellationToken token) => this.StoreSourceHashAsync(sourceHash, reason, token); /// /// Tells the user interface how the data source stands as the index holds it, without a run. /// /// /// For a data source which is not looked at right now, e.g. a mailbox while AI Studio starts: /// its server is asked nothing then. Nothing is written. /// /// Whether a run got through the whole data source at some point, which decides between completed and idle. public void PublishStoredState(bool workedThrough) => context.PublishStatus(this.CreateStatus(workedThrough ? DataSourceEmbeddingState.COMPLETED : DataSourceEmbeddingState.IDLE, string.Empty, string.Empty, null, null)); /// /// Ends a run which cannot go on, for a reason which is not about any one document. /// /// /// E.g. the server of a mailbox which cannot be reached or refuses the sign-in, or a removal /// the user has to agree to first. Nothing is written: whatever the run indexed so far stays, /// and the data source keeps counting as not worked through, so the next run picks up where /// this one stopped. /// /// Why the run stopped, and what the user can do about it. /// What the user has to decide, if anything. /// The number of documents a held back removal asks about. public void PublishRunFailure(string message, DataSourceAttention attention = DataSourceAttention.NONE, int? pendingRemovalCount = null) { this.LastError = message; this.failures.Add(new DataSourceEmbeddingFailure(context.DataSource.Name, message, DateTimeOffset.UtcNow)); context.PublishStatus(this.CreateStatus(DataSourceEmbeddingState.FAILED, string.Empty, message, null, null, attention, pendingRemovalCount)); } /// /// Stores what a run found about the data source as a whole, and tidies up the collection. /// private async Task StoreSourceHashAsync(string sourceHash, string reason, CancellationToken token) { context.Manifest.SourceHash = sourceHash; token.ThrowIfCancellationRequested(); await context.OptimizeCollectionIfNeededAsync(reason, token); token.ThrowIfCancellationRequested(); await context.IndexStore.UpdateDataSourceHashAsync(context.DataSource.Id, sourceHash, token); token.ThrowIfCancellationRequested(); } /// /// What the user reads about a document which could not be indexed. /// /// /// A mail only ever fails as a whole: its text is not read from a file, and an attachment which /// cannot be read costs the mail nothing but that attachment. So one sentence serves every /// reason, and the log holds the details. The sentences about files would speak of a file, and /// of a change which never comes to a mail. /// /// Why reading the document failed, NONE when it was not about reading it. /// The document. /// The message, ready to show. private string GetFailureMessage(FileExtractionErrorCode code, EmbeddingDocument document) => context.DataSource is DataSourceMailbox ? string.Format(TB("The mail '{0}' could not be indexed. AI Studio tries again during the next sync."), document.DisplayName) : code.ToIndexingUserMessage(document.DisplayName); private DataSourceEmbeddingStatus CreateStatus(DataSourceEmbeddingState state, string currentDocument, string lastError, int? currentBlock, int? currentPage, DataSourceAttention attention = DataSourceAttention.NONE, int? pendingRemovalCount = null) => new( context.DataSource.Id, context.DataSource.Name, context.DataSource.Type, state, totalDocuments, this.DoneDocuments, this.FailedDocuments, currentDocument, lastError, this.failures.ToList(), this.PermanentlySkippedDocuments, currentBlock, currentPage, Attention: attention, PendingRemovalCount: pendingRemovalCount, LastSyncUtc: this.LastSyncUtc); }