using AIStudio.Provider;
using AIStudio.Settings.DataModel;
using AIStudio.Tools.Databases.IndexStore;
using AIStudio.Tools.PluginSystem;
using static AIStudio.Tools.Services.Indexing.IndexingLogFormat;
namespace AIStudio.Tools.Services.Indexing;
///
/// How far an indexing run got through the documents of its data source, and what became of each.
///
///
/// Counts the documents, keeps the list of failures and turns both into the status the embedding
/// page shows. What happens to a document after it was indexed, or after it failed, is the same for
/// every kind of data source, so that is decided here as well: the stores learn about it, the
/// manifest follows, and the user hears about it the way the failure deserves.
///
/// The run the documents are indexed in.
/// How many documents the data source has, readable or not.
/// How many of them could not even be looked at.
/// Why the last of those could not be looked at, or an empty string.
/// The failures of those which could not be looked at.
/// The logger of the embedding service, so the log reads the same whoever writes it.
internal sealed class DocumentRunProgress(IndexedRunContext context, int totalDocuments, int failedInputs, string lastInputError, IEnumerable inputFailures, ILogger logger)
{
///
/// How often the block progress within one document is reported to the user interface at most.
///
private static readonly TimeSpan BLOCK_PROGRESS_INTERVAL = TimeSpan.FromSeconds(3);
private readonly List failures = inputFailures.ToList();
//
// Which kinds of provider failure the user was already told about in this run. A rejected
// API key is the same problem for every one of a few thousand documents, and one message
// is what it takes to send the user to the settings.
//
private readonly HashSet reportedFailureReasons = [];
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(DocumentRunProgress).Namespace, nameof(DocumentRunProgress));
public int TotalDocuments => totalDocuments;
///
/// Documents which were left alone because nothing changed since they were indexed.
///
public int UnchangedDocuments { get; private set; }
///
/// Documents which were not read because they failed for a reason of their own before.
///
public int PermanentlySkippedDocuments { get; private set; }
///
/// Documents which were indexed in this run.
///
public int IndexedDocuments { get; private set; }
public int NewDocuments { get; private set; }
public int ChangedDocuments { get; private set; }
public int FailedDocuments { get; private set; } = failedInputs;
public string LastError { get; private set; } = lastInputError;
///
/// When the data source was last worked through as a whole, for those which are synced rather
/// than watched. Every status of the run carries it.
///
public DateTimeOffset? LastSyncUtc { get; set; }
///
/// Documents which need nothing more in this run, whether they were indexed now or before.
///
public int DoneDocuments => this.UnchangedDocuments + this.IndexedDocuments;
///
/// Counts documents which nothing changed about since they were indexed.
///
/// How many of them.
public void RecordUnchanged(int count = 1) => this.UnchangedDocuments += count;
///
/// Counts a document which is not read again, because it failed for a reason of its own before
/// and has not changed since.
///
/// The key of the document.
/// Why it failed back then.
public void RecordStillUnreadable(string documentKey, PermanentIndexingFailureRecord failure)
{
this.PermanentlySkippedDocuments++;
// The stored reason keeps its place in the list, so the user still sees why:
this.failures.Add(new DataSourceEmbeddingFailure(documentKey, failure.Message, failure.OccurredAtUtc, ExtractionCode: failure.Code, IsPermanent: true));
}
///
/// Tells the user interface where the run stands.
///
/// The name of the document being worked on, or an empty string.
/// The block of that document being worked on, when known.
/// The page that block is on, when known.
public void Publish(string currentDocument = "", int? currentBlock = null, int? currentPage = null) =>
context.PublishStatus(this.CreateStatus(DataSourceEmbeddingState.RUNNING, currentDocument, this.LastError, currentBlock, currentPage));
///
/// Announces that work on a document starts, and hands out what reports its progress.
///
/// The document.
/// Told about every block of the document, with its number and its page.
public Action BeginDocument(EmbeddingDocument document)
{
this.Publish(document.DisplayName);
//
// What the page says while one document is being worked on. Without it, a document of
// several thousand pages leaves the same sentence standing for hours, and a progress
// which never moves cannot be told apart from one which is stuck.
//
var lastBlockReportUtc = DateTimeOffset.MinValue;
return (blockNumber, pageNumber) =>
{
//
// The first block goes out at once, so the line is there instead of blank. After
// that, at most one message every BLOCK_PROGRESS_INTERVAL: each one re-renders the
// embedding page, the navigation bar and the table in the settings, and the blocks
// of a large document arrive far faster than anybody can read them.
//
var nowUtc = DateTimeOffset.UtcNow;
if (blockNumber > 1 && nowUtc - lastBlockReportUtc < BLOCK_PROGRESS_INTERVAL)
return;
lastBlockReportUtc = nowUtc;
this.Publish(document.DisplayName, blockNumber, pageNumber);
};
}
///
/// Records a document as indexed, in the index store as well as in the manifest.
///
///
/// Called once the kind of data source is sure the document did not change while it was read.
/// Until then, its index row says it has no chunks.
///
/// The document.
/// How many chunks were stored for it.
/// Whether the index knew nothing about it before.
/// The cancellation token.
public async Task RecordDocumentIndexedAsync(EmbeddingDocument document, int chunkCount, bool isNew, CancellationToken token)
{
var embeddedAtUtc = DateTimeOffset.UtcNow;
var state = document.State with { ChunkCount = chunkCount, EmbeddedAtUtc = embeddedAtUtc };
await context.IndexStore.UpsertFileAsync(context.DataSource.Id, state, token);
context.Manifest.Files[document.Key] = new EmbeddedFileRecord(state.Fingerprint, state.FileSize, state.LastWriteUtc, embeddedAtUtc, chunkCount);
await context.ForgetPermanentFailureAsync(document.Key, token);
this.IndexedDocuments++;
if (isNew)
this.NewDocuments++;
else
this.ChangedDocuments++;
}
///
/// Records why a document could not be indexed, and removes what the attempt left behind.
///
///
/// Never called for a cancelled run, nor for a vector store which cannot be read at all: those
/// are not about one document, and whoever indexes has to let them through.
///
/// The document.
/// What went wrong.
/// The cancellation token.
public async Task RecordDocumentFailureAsync(EmbeddingDocument document, Exception exception, CancellationToken token)
{
var dataSource = context.DataSource;
switch (exception)
{
case ProviderRequestException providerFailure:
{
//
// The provider said what went wrong and what the user can do about it. That
// sentence is what goes into the status, together with the classification the UI
// needs to offer the matching way out.
//
this.FailedDocuments++;
this.LastError = providerFailure.UserMessage;
this.failures.Add(new DataSourceEmbeddingFailure(document.Key, providerFailure.UserMessage, DateTimeOffset.UtcNow, providerFailure.FailureReason, providerFailure.StatusCode, context.EmbeddingProvider.Name, DisplayName: document.DisplayName));
context.Manifest.Files.Remove(document.Key);
await context.ForgetPermanentFailureAsync(document.Key, token);
await context.CleanupFailedDocumentAsync(document.Key, token);
logger.LogWarning(
providerFailure,
"Failed to embed file '{FilePath}' for data source '{DataSourceName}' because the embedding provider '{EmbeddingProviderName}' failed. FailureReason={FailureReason}, StatusCode={StatusCode}.",
document.Key,
dataSource.Name,
context.EmbeddingProvider.Name,
providerFailure.FailureReason,
providerFailure.StatusCode);
this.Publish(document.DisplayName);
// Once per kind of failure, not once per document:
if (this.reportedFailureReasons.Add(providerFailure.FailureReason))
await MessageBus.INSTANCE.SendError(new(Icons.Material.Filled.CloudOff, providerFailure.UserMessage));
break;
}
case FileExtractionException extractionFailure when extractionFailure.Code.IsPermanentIndexingFailure():
{
//
// The document itself is why this failed, so trying it again changes nothing until
// the document does. The reason is written into the index, and the fingerprint next
// to it decides when to come back: an OCR run over a scanned PDF changes both size
// and write time, which is exactly the moment the file deserves another attempt.
//
this.PermanentlySkippedDocuments++;
var occurredAtUtc = DateTimeOffset.UtcNow;
var indexingMessage = this.GetFailureMessage(extractionFailure.Code, document);
this.failures.Add(new DataSourceEmbeddingFailure(document.Key, indexingMessage, occurredAtUtc, ExtractionCode: extractionFailure.Code, IsPermanent: true, DisplayName: document.DisplayName));
context.Manifest.Files.Remove(document.Key);
await context.CleanupFailedDocumentAsync(document.Key, token);
var state = document.State;
context.Manifest.PermanentFailures[state.AbsolutePath] = new PermanentIndexingFailureRecord(state.Fingerprint, extractionFailure.Code, indexingMessage, occurredAtUtc);
await context.IndexStore.UpsertPermanentFailureAsync(
dataSource.Id,
new PermanentIndexingFailure(state.ParentFileId, state.AbsolutePath, state.Fingerprint, extractionFailure.Code, indexingMessage, occurredAtUtc),
token);
logger.LogInformation(
extractionFailure,
"Skipping file '{FilePath}' of data source '{DataSourceName}' ({DataSourceId}) from now on because reading it failed for a reason which lies in the file. FailureCode={FailureCode}, MetadataHashPrefix={MetadataHashPrefix}.",
document.Key,
dataSource.Name,
dataSource.Id,
extractionFailure.Code,
ShortHash(state.Fingerprint));
this.Publish(document.DisplayName);
break;
}
default:
{
//
// Everything which is not the provider's doing: a document which changed while it
// was read, one which yielded no text, a vector store which refused to store. These
// are about this one document, so they go into the list and not into a message
// which would interrupt whatever the user is doing right now.
//
this.FailedDocuments++;
var extractionCode = exception is FileExtractionException extractionFailure ? extractionFailure.Code : FileExtractionErrorCode.NONE;
//
// Deliberately not the message of the exception: that one is written for the log
// file, in English, and repeats the path which the list shows anyway.
//
var failureMessage = this.GetFailureMessage(extractionCode, document);
this.LastError = failureMessage;
this.failures.Add(new DataSourceEmbeddingFailure(document.Key, failureMessage, DateTimeOffset.UtcNow, EmbeddingProviderName: context.EmbeddingProvider.Name, ExtractionCode: extractionCode, DisplayName: document.DisplayName));
context.Manifest.Files.Remove(document.Key);
await context.ForgetPermanentFailureAsync(document.Key, token);
await context.CleanupFailedDocumentAsync(document.Key, token);
logger.LogWarning(exception, "Failed to embed file '{FilePath}' for data source '{DataSourceName}'.", document.Key, dataSource.Name);
this.Publish(document.DisplayName);
break;
}
}
}
///
/// Finishes the run: the collection is tidied up, the data source is marked as worked through,
/// and the user interface learns how it went.
///
///
/// The hash is written last on purpose. It is what says that a run got through the whole data
/// source, so a run which stops before this point leaves the data source marked as unfinished.
///
/// The hash of the data source as this run found it.
/// Why the collection is optimized now, for the log.
/// The cancellation token.
public async Task CompleteRunAsync(string sourceHash, string reason, CancellationToken token)
{
await this.StoreSourceHashAsync(sourceHash, reason, token);
//
// Documents which were skipped for good do not make a run unsuccessful: nothing is left to
// try, and a data source made of nothing but scanned images would otherwise ask for
// attention forever.
//
var hasFailures = this.FailedDocuments > 0;
var lastError = hasFailures
? string.IsNullOrWhiteSpace(this.LastError)
? context.DataSource is DataSourceMailbox
? TB("Some mails could not be indexed. The list below says which ones and why.")
: TB("Some files could not be indexed. The list below says which ones and why.")
: this.LastError
: string.Empty;
context.PublishStatus(this.CreateStatus(hasFailures ? DataSourceEmbeddingState.FAILED : DataSourceEmbeddingState.COMPLETED, string.Empty, lastError, null, null));
}
///
/// Ends a run before it got through the whole data source, keeping what it did for the next one.
///
///
/// Stores everything CompleteRunAsync stores, the hash included. For a data source which is
/// worked through in several runs, the hash therefore says that its index can be searched, not
/// that the data source was worked through as a whole: that is what LastSyncUtc tells. Without
/// it, a mailbox would stay out of every search for the hours its first sync takes.
///
/// The user interface learns nothing here. Whatever comes next tells it: the run being queued
/// again, or the question the user has to answer first.
///
/// The hash of the data source as this run found it.
/// Why the collection is optimized now, for the log.
/// The cancellation token.
public Task PauseRunAsync(string sourceHash, string reason, CancellationToken token) => this.StoreSourceHashAsync(sourceHash, reason, token);
///
/// Tells the user interface how the data source stands as the index holds it, without a run.
///
///
/// For a data source which is not looked at right now, e.g. a mailbox while AI Studio starts:
/// its server is asked nothing then. Nothing is written.
///
/// Whether a run got through the whole data source at some point, which decides between completed and idle.
public void PublishStoredState(bool workedThrough) =>
context.PublishStatus(this.CreateStatus(workedThrough ? DataSourceEmbeddingState.COMPLETED : DataSourceEmbeddingState.IDLE, string.Empty, string.Empty, null, null));
///
/// Ends a run which cannot go on, for a reason which is not about any one document.
///
///
/// E.g. the server of a mailbox which cannot be reached or refuses the sign-in, or a removal
/// the user has to agree to first. Nothing is written: whatever the run indexed so far stays,
/// and the data source keeps counting as not worked through, so the next run picks up where
/// this one stopped.
///
/// Why the run stopped, and what the user can do about it.
/// What the user has to decide, if anything.
/// The number of documents a held back removal asks about.
public void PublishRunFailure(string message, DataSourceAttention attention = DataSourceAttention.NONE, int? pendingRemovalCount = null)
{
this.LastError = message;
this.failures.Add(new DataSourceEmbeddingFailure(context.DataSource.Name, message, DateTimeOffset.UtcNow));
context.PublishStatus(this.CreateStatus(DataSourceEmbeddingState.FAILED, string.Empty, message, null, null, attention, pendingRemovalCount));
}
///
/// Stores what a run found about the data source as a whole, and tidies up the collection.
///
private async Task StoreSourceHashAsync(string sourceHash, string reason, CancellationToken token)
{
context.Manifest.SourceHash = sourceHash;
token.ThrowIfCancellationRequested();
await context.OptimizeCollectionIfNeededAsync(reason, token);
token.ThrowIfCancellationRequested();
await context.IndexStore.UpdateDataSourceHashAsync(context.DataSource.Id, sourceHash, token);
token.ThrowIfCancellationRequested();
}
///
/// What the user reads about a document which could not be indexed.
///
///
/// A mail only ever fails as a whole: its text is not read from a file, and an attachment which
/// cannot be read costs the mail nothing but that attachment. So one sentence serves every
/// reason, and the log holds the details. The sentences about files would speak of a file, and
/// of a change which never comes to a mail.
///
/// Why reading the document failed, NONE when it was not about reading it.
/// The document.
/// The message, ready to show.
private string GetFailureMessage(FileExtractionErrorCode code, EmbeddingDocument document) => context.DataSource is DataSourceMailbox
? string.Format(TB("The mail '{0}' could not be indexed. AI Studio tries again during the next sync."), document.DisplayName)
: code.ToIndexingUserMessage(document.DisplayName);
private DataSourceEmbeddingStatus CreateStatus(DataSourceEmbeddingState state, string currentDocument, string lastError, int? currentBlock, int? currentPage, DataSourceAttention attention = DataSourceAttention.NONE, int? pendingRemovalCount = null) => new(
context.DataSource.Id,
context.DataSource.Name,
context.DataSource.Type,
state,
totalDocuments,
this.DoneDocuments,
this.FailedDocuments,
currentDocument,
lastError,
this.failures.ToList(),
this.PermanentlySkippedDocuments,
currentBlock,
currentPage,
Attention: attention,
PendingRemovalCount: pendingRemovalCount,
LastSyncUtc: this.LastSyncUtc);
}