using System.Text; using AIStudio.Settings.DataModel; using AIStudio.Tools.Databases.IndexStore; using AIStudio.Tools.Databases.VectorStore; using AIStudio.Tools.Mail; using AIStudio.Tools.Security; using MailKit; using MimeKit; namespace AIStudio.Tools.Services.Indexing; /// /// One mail at a time: what it is called in the index, how its text is read, and what the index /// keeps of it beyond its chunks. /// internal sealed partial class MailboxIndexer { /// /// What the index stores as the type of a mail, where a file has its extension. /// private const string MAIL_FILE_TYPE = "mail"; /// /// The content type of a header block on its own, cf. RFC 6522. /// private const string HEADER_BLOCK_CONTENT_TYPE = "text/rfc822-headers"; /// /// How many fields of a mail pass the filter ahead of the names of its attachments: the header /// block, the body and the subject. /// private const int FILTERED_MAIL_FIELDS = 3; /// /// Links a mail the index holds already to its place in this folder, or reads and indexes it. /// private async Task SyncNewMailAsync(IndexedRunContext context, DataSourceMailbox mailbox, ImapMailboxConnector connector, string folderPath, IMessageSummary summary, DocumentRunProgress progress, ISet encounteredKeys, CancellationToken token) { if (summary.Headers is not { } headers) { logger.LogWarning("The server delivered a mail of mailbox '{MailboxId}' without its header block. The mail is skipped.", mailbox.Id); return; } var key = MailContentKey.Create(MailSummaryReader.ReadIdentity(summary)); var mailId = IndexedDocumentIds.CreateParentId(mailbox.Id, key); var location = new MailLocationRecord(folderPath, summary.UniqueId.Id, MailSummaryReader.ReadFlags(summary.Flags)); encounteredKeys.Add(key); // // A mail the index holds already, which was moved or copied here, or numbered anew by the // server. It keeps its chunks and only gains a location. // if (await context.IndexStore.AddMailLocationAsync(mailbox.Id, mailId, location, token)) { progress.RecordUnchanged(); progress.Publish(); return; } var mailHash = MailSummaryReader.ComputeMailHash(summary); if (context.Manifest.PermanentFailures.TryGetValue(key, out var permanentFailure) && string.Equals(permanentFailure.Fingerprint, mailHash, StringComparison.Ordinal)) { progress.RecordStillUnreadable(key, permanentFailure); progress.Publish(); return; } // // Named after the subject as the server delivered it. The name is for the user alone, who // reads it on the embeddings page as in any mail program. A model only ever gets to read // the filtered subject, which the index stores. // var subject = MailTextNormalization.NormalizeHeaderValue(headers[HeaderId.Subject]); var displayName = subject.Length > 0 ? subject : TB("(no subject)"); var foundAtUtc = DateTimeOffset.UtcNow; var isNew = !context.Manifest.Files.ContainsKey(key); var document = this.CreateMailDocument(context, mailbox, summary, key, mailId, mailHash, displayName, foundAtUtc, null, []); try { var (text, attachments) = await this.ReadMailAsync(context, connector, mailbox, summary, token); document = this.CreateMailDocument(context, mailbox, summary, key, mailId, mailHash, displayName, foundAtUtc, text, attachments); var reportBlockProgress = progress.BeginDocument(document); var chunkCount = await context.IndexDocumentAsync(document, reportBlockProgress, token); await progress.RecordDocumentIndexedAsync(document, chunkCount, isNew, token); // Only after the last chunk: indexing the document deleted whatever the index kept of the mail. await context.IndexStore.UpsertMailAsync(mailbox.Id, CreateMailRecord(mailId, summary, text, attachments, mailHash, location, foundAtUtc), token); } catch (OperationCanceledException) when (token.IsCancellationRequested) { throw; } catch (MailboxConnectionException) { // Not about this one mail: the connection is gone, and every other mail would fail the same way. throw; } catch (VectorStoreUnreadableException) { // Not about this one mail either: the store of the whole mailbox cannot be opened. throw; } catch (Exception exception) { await progress.RecordDocumentFailureAsync(document, exception, token); } } /// /// Reads the text of a mail and of its attachments, filtered for prompt injections. /// /// /// Of an encrypted mail, nothing but the header block is read: its text parts and its /// attachments stay on the server. The subject and the names of the attachments are filtered /// on their own as well, since the index stores them apart from the text, and whatever reads /// them there may hand them on to a model. A passage the filter removes from them is therefore /// reported twice. /// private async Task<(MailText Text, IReadOnlyList Attachments)> ReadMailAsync(IndexedRunContext context, ImapMailboxConnector connector, DataSourceMailbox mailbox, IMessageSummary summary, CancellationToken token) { var textParts = MailEncryptionDetection.Detect(summary.Body) is MailEncryptionKind.NONE ? await connector.FetchTextPartsAsync(summary, token) : null; var text = MailTextBuilder.Build(MailSummaryReader.ReadTextSource(summary, textParts)); // The text itself may reveal an encryption the structure did not, cf. MailTextBuilder: var attachmentParts = text.EncryptionKind is MailEncryptionKind.NONE ? MailSummaryReader.ReadAttachments(summary) : []; var source = PromptInjectionSource.MailContent(mailbox.Name); var filtered = await guardService.SanitizeAsync([ new(text.HeaderBlock, source), new(text.Body, source), new(text.Subject, source), ..attachmentParts.Select(part => new PromptInjectionText(MailTextNormalization.NormalizeHeaderValue(part.FileName), source)), ]); var attachments = new List(attachmentParts.Count); for (var index = 0; index < attachmentParts.Count; index++) attachments.Add(await this.ReadAttachmentAsync(context, mailbox, connector, summary.UniqueId, attachmentParts[index], filtered[FILTERED_MAIL_FIELDS + index], token)); return (text with { HeaderBlock = filtered[0], Body = filtered[1], Subject = filtered[2] }, attachments); } /// /// Describes a mail as a document for the shared part of an indexing run. /// /// /// A mail has no path of its own. Where it lies is kept as its locations, which change without /// the mail being embedded again, so a folder stored here would soon name the wrong one. /// /// The run the mail is indexed in. /// The mailbox the mail belongs to. /// The summary of the mail. /// The content key of the mail. /// The id of the mail, which is the id of its document. /// The hash of the mail, which is the fingerprint of its document. /// How the mail is called in messages for the user. /// When AI Studio found the mail. /// The filtered text of the mail, or null while it is not read yet. Such a document has no chunks and only serves to record a failure. /// The attachments of the mail, empty while it is not read yet. /// The document. private EmbeddingDocument CreateMailDocument(IndexedRunContext context, DataSourceMailbox mailbox, IMessageSummary summary, string key, string mailId, string mailHash, string displayName, DateTimeOffset foundAtUtc, MailText? text, IReadOnlyList attachments) { var (sentAtUtc, receivedAtUtc) = ReadDates(summary, foundAtUtc); var state = new EmbeddingStateFile( mailId, key, text?.Subject ?? string.Empty, string.Empty, MAIL_FILE_TYPE, mailHash, summary.Size ?? 0, sentAtUtc ?? receivedAtUtc, receivedAtUtc, DateTimeOffset.UtcNow, 0); // // The mail itself is one piece of text: the first chunk starts with the header block, so a // search for a sender or a subject finds the mail. The attachments follow it. // var fullText = text?.FullText ?? string.Empty; var content = new SegmentedText(fullText, fullText.Length is 0 ? [] : [new TextSegment(fullText, null, null)]); var chunkingOptions = DataSourceEmbeddingService.GetChunkingOptions(mailbox, context.EmbeddingProvider); return new(key, state, displayName, chunkToken => this.StreamMailChunksAsync(content, attachments, chunkingOptions, context.EmbeddingProvider, chunkToken)); } /// /// Puts together what the index keeps about a mail beyond its chunks. /// /// The id of the mail, which is the id of its document. /// The summary of the mail, with its header block and its structure. /// The filtered text of the mail. /// The attachments of the mail. Each one is kept, with its text or with the reason why there is none. /// The hash of the mail, cf. MailSummaryReader.ComputeMailHash. /// Where the mail was found. /// When AI Studio found the mail. The index keeps the earliest time it found the mail. /// The mail as the index keeps it. internal static MailRecord CreateMailRecord(string mailId, IMessageSummary summary, MailText text, IReadOnlyList attachments, string mailHash, MailLocationRecord location, DateTimeOffset foundAtUtc) { var headers = summary.Headers ?? throw new ArgumentException("The mail was fetched without its header block.", nameof(summary)); var (sentAtUtc, receivedAtUtc) = ReadDates(summary, foundAtUtc); // // The header block as the server delivered it, encoded words and all, for whoever has to // judge the mail later. It is never cut into chunks, and it is filtered when it is read. // var headerBlock = MailSummaryReader.ReadHeaderBlock(headers); List parts = [new(MailPartKind.HEADERS, string.Empty, HEADER_BLOCK_CONTENT_TYPE, Encoding.UTF8.GetByteCount(headerBlock), headerBlock, MailPartTextState.EXTRACTED)]; var bodyPart = text.BodySource switch { MailBodySource.HTML => summary.HtmlBody, MailBodySource.PLAIN_TEXT => summary.TextBody, _ => null, }; if (bodyPart is not null && text.Body.Length > 0) parts.Add(new(MailPartKind.BODY, string.Empty, bodyPart.ContentType.MimeType, bodyPart.Octets, text.Body, MailPartTextState.EXTRACTED)); // // Every attachment, whether its text was read or not: that a mail has attachments, and // what they are called, is worth searching for either way. // parts.AddRange(attachments.Select(attachment => new MailPartRecord( MailPartKind.ATTACHMENT, attachment.Name, attachment.Part.ContentType.MimeType, attachment.Part.Octets, attachment.TextState is MailPartTextState.EXTRACTED ? attachment.Content.Text : null, attachment.TextState))); return new( mailId, MailHeaders.ReadMessageIds(headers, HeaderId.MessageId).FirstOrDefault(IsStorableMessageId) ?? string.Empty, MailHeaders.ReadMessageIds(headers, HeaderId.InReplyTo).FirstOrDefault(IsStorableMessageId) ?? string.Empty, MailHeaders.ReadMessageIds(headers, HeaderId.References).Where(IsStorableMessageId).ToList(), sentAtUtc, receivedAtUtc, text.Importance, text.EncryptionKind, mailHash, foundAtUtc, MailSummaryReader.ReadAddresses(headers), parts, [location]); } /// /// When the sender says the mail was written, and when it arrived at the server. /// /// /// A server has to report when a mail arrived. Should one not do so, the date the sender gives /// comes closest, and after that the moment AI Studio found the mail. /// private static (DateTimeOffset? SentAtUtc, DateTimeOffset ReceivedAtUtc) ReadDates(IMessageSummary summary, DateTimeOffset foundAtUtc) { var sentAtUtc = summary.Headers is { } headers ? MailHeaders.ReadDate(headers)?.ToUniversalTime() : null; return (sentAtUtc, summary.InternalDate?.ToUniversalTime() ?? sentAtUtc ?? foundAtUtc); } /// /// Whether the index can keep a Message-ID, which it stores separated by spaces, cf. UpsertMailAsync. /// private static bool IsStorableMessageId(string messageId) => messageId.Length > 0 && !messageId.Any(char.IsWhiteSpace); }