Merge branch 'main' into feature/configurable-transcription-bitrate

This commit is contained in:
Thorsten Sommer authored and GitHub committed 2026-09-18 17:37:31 +02:00
commit 28ff3ba625
26 files changed
+750 -571

No files matched your search

@@ -3,7 +3,7 @@ using AIStudio.Settings;
namespace AIStudio.Tools;
/// <summary>
/// Contains the allowed and selected data sources, plus the ones waiting for their index.
/// Contains the allowed and selected data sources, plus the ones which cannot be searched right now.
/// </summary>
/// <remarks>
/// The selected data sources are a subset of the allowed data sources.
@@ -13,8 +13,13 @@ namespace AIStudio.Tools;
/// -- takes it to mean "may be used to answer with", and a source whose index is being rebuilt
/// cannot answer anything. It is listed separately so the user interface can still show it and say
/// why it is greyed out, instead of letting it vanish without a word.
///
/// The same holds for the ones waiting for a repair, and they are a list of their own because the
/// two reasons call for different words: one passes by itself, the other one waits for the user.
/// A data source is in at most one of the two lists.
/// </remarks>
/// <param name="AllowedDataSources">The allowed data sources.</param>
/// <param name="SelectedDataSources">The selected data sources, which are a subset of the allowed data sources.</param>
/// <param name="DataSourcesAwaitingReindex">The data sources which passed every check but cannot be searched until their index has been rebuilt.</param>
public readonly record struct AllowedSelectedDataSources(IReadOnlyList<IDataSource> AllowedDataSources, IReadOnlyList<IDataSource> SelectedDataSources, IReadOnlyList<IDataSource> DataSourcesAwaitingReindex);
/// <param name="DataSourcesNeedingRepair">The data sources which passed every check but whose index cannot be read anymore, so that only the user can get them back.</param>
public readonly record struct AllowedSelectedDataSources(IReadOnlyList<IDataSource> AllowedDataSources, IReadOnlyList<IDataSource> SelectedDataSources, IReadOnlyList<IDataSource> DataSourcesAwaitingReindex, IReadOnlyList<IDataSource> DataSourcesNeedingRepair);
@@ -0,0 +1,42 @@
using AIStudio.Dialogs;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Services;
namespace AIStudio.Tools;
/// <summary>
/// Asks whether a data source should be indexed anew, and starts the rebuild when the user agrees.
/// </summary>
/// <remarks>
/// Kept here rather than in the two places which offer the repair -- the background embeddings page
/// and the data source table -- so the sentence naming what a rebuild costs cannot drift apart
/// between them. Naming both costs is the whole reason for asking at all.
/// </remarks>
public static class DataSourceRepair
{
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(DataSourceRepair).Namespace, nameof(DataSourceRepair));
/// <summary>
/// Asks the user, and rebuilds the index of the data source when they agree.
/// </summary>
/// <param name="dialogService">The dialog service to ask with.</param>
/// <param name="embeddingService">The service which does the rebuild.</param>
/// <param name="dataSourceId">The data source to repair.</param>
/// <param name="dataSourceName">The name of that data source, as the question names it.</param>
/// <returns>True when the rebuild was started.</returns>
public static async Task<bool> ConfirmAndRepairAsync(IDialogService dialogService, DataSourceEmbeddingService embeddingService, string dataSourceId, string dataSourceName)
{
var dialogParameters = new DialogParameters<ConfirmDialog>
{
{ x => x.Message, string.Format(TB("The index of the data source '{0}' cannot be read anymore. Repairing it means building the index from scratch: everything indexed so far is thrown away, and every document of this data source is sent to your embedding provider once more. With a cloud provider, this costs money, and with a large data source it takes a while. Do you want to repair this data source now?"), dataSourceName) },
};
var dialogReference = await dialogService.ShowAsync<ConfirmDialog>(TB("Repair Data Source"), dialogParameters, Dialogs.DialogOptions.FULLSCREEN);
var dialogResult = await dialogReference.Result;
if (dialogResult is null || dialogResult.Canceled)
return false;
await embeddingService.RepairDataSourceAsync(dataSourceId);
return true;
}
}
@@ -0,0 +1,13 @@
namespace AIStudio.Tools.Databases.VectorStore;
/// <summary>
/// Thrown when a vector store is there on disk, but cannot be opened.
/// </summary>
/// <remarks>
/// Separate from every other database failure, because it is the one which no retry heals and which
/// the app must not heal on its own: building the index anew sends every document to the embedding
/// provider once more, which costs real money and, for a large data source, hours. So this failure
/// travels as its own type up to the places which can say so and offer the rebuild, and the decision
/// stays with the user.
/// </remarks>
public sealed class VectorStoreUnreadableException(string message) : Exception(message);
@@ -5,6 +5,43 @@ namespace AIStudio.Tools.Services;
public sealed partial class DataSourceEmbeddingService
{
/// <summary>
/// Throws away everything stored for one data source and starts a fresh indexing run.
/// </summary>
/// <remarks>
/// The one way out of an index which cannot be read, and nothing in the app takes it by itself:
/// a rebuild sends every document of the data source to the embedding provider once more, which
/// costs money with a cloud provider and hours with a large data source. It happens because the
/// user asked for it, after being told both.
///
/// An active run is stopped first, the same way deleting a data source does it. The repair is
/// offered for a failed data source only, so there should be none -- but a file watcher may
/// well have queued one between the click and this call, and discarding the index next to a
/// live run would leave it half thrown away.
/// </remarks>
/// <param name="dataSourceId">The data source to build anew.</param>
public async Task RepairDataSourceAsync(string dataSourceId)
{
if (!this.TryGetConfiguredDataSource(dataSourceId, out var dataSource) || !this.IsSupportedInternalDataSource(dataSource))
return;
logger.LogWarning(
"Repairing data source '{DataSourceName}' ({DataSourceId}) on the user's request: the stored index is discarded and built anew.",
dataSource.Name,
dataSource.Id);
var activeRun = this.CancelActiveDataSourceRun(dataSource);
this.ClearQueuedDataSourceState(dataSourceId);
if (activeRun is not null)
await activeRun.Completion.Task;
await this.ResetPersistedStateAsync(dataSourceId, null, null, CancellationToken.None);
this.statuses.TryRemove(dataSourceId, out _);
this.PublishStatusChanged();
await this.QueueDataSourceAsync(dataSource, true, DataSourceEmbeddingRefreshMode.MANUAL_RETRY);
}
private async Task ResetPersistedStateAsync(
string dataSourceId,
VectorStoreClient? vectorStore,
@@ -317,6 +317,22 @@ public sealed partial class DataSourceEmbeddingService(SettingsManager settingsM
return runState is not DataSourceEmbeddingState.FAILED;
}
/// <summary>
/// Whether a data source cannot be searched because its vector store cannot be read anymore.
/// </summary>
/// <remarks>
/// Unlike the re-index check above, this reads no database at all: the state comes from the run
/// or the search which ran into the unreadable store, and is kept in memory only. That it does
/// not survive a restart is deliberate. The very same store may well open on the next start,
/// and a mark written to disk would then be wrong with nobody noticing. Until something touches
/// the store again, the data source counts as usable, and a failing search says so on its own.
/// </remarks>
/// <param name="dataSource">The data source to ask about.</param>
/// <returns>True when the data source waits for the user to have its index rebuilt.</returns>
public bool NeedsIndexRepair(IDataSource dataSource) =>
this.statuses.TryGetValue(dataSource.Id, out var status) &&
status is { State: DataSourceEmbeddingState.FAILED, VectorStoreUnreadable: true };
public Task QueueDataSourceAsync(IDataSource dataSource)
{
return this.QueueDataSourceAsync(dataSource, true, DataSourceEmbeddingRefreshMode.HASH_CHECK);
@@ -430,6 +446,20 @@ public sealed partial class DataSourceEmbeddingService(SettingsManager settingsM
{
break;
}
catch (VectorStoreUnreadableException exception) when (dataSource is not null)
{
//
// Nothing is deleted and nothing is rebuilt here. The data source says what is
// wrong with it, stays out of the selection while it says so, and waits for the
// user to ask for the repair.
//
logger.LogError(
exception,
"The vector store of data source '{DataSourceName}' ({DataSourceId}) cannot be read. The data source is waiting for a repair.",
dataSource.Name,
dataSource.Id);
this.UpsertStatus(this.GetUnreadableVectorStoreStatus(dataSource));
}
catch (Exception exception)
{
if (dataSource is null)
@@ -873,6 +903,15 @@ public sealed partial class DataSourceEmbeddingService(SettingsManager settingsM
ShortHash(fingerprint));
this.UpsertStatus(this.CreateStatus(dataSource, DataSourceEmbeddingState.RUNNING, totalFiles, skippedFiles + completedFiles, failedFiles, file.Name, lastError, failureDetails, permanentlySkippedFiles));
}
catch (VectorStoreUnreadableException)
{
//
// Not about this one file: the store of the whole data source cannot be opened, so
// every remaining file would fail the same way. Carrying on would fill the list
// with one entry per file and hide the single cause behind them.
//
throw;
}
catch (Exception exception)
{
//
@@ -1338,6 +1377,15 @@ public sealed partial class DataSourceEmbeddingService(SettingsManager settingsM
{
throw;
}
catch (VectorStoreUnreadableException exception)
{
logger.LogError(
exception,
"The vector store of data source '{DataSourceName}' ({DataSourceId}) cannot be read. The data source is waiting for a repair.",
dataSource.Name,
dataSource.Id);
this.UpsertStatus(this.GetUnreadableVectorStoreStatus(dataSource));
}
catch (Exception exception)
{
logger.LogError(exception, "Initial embedding hash check failed for data source '{DataSourceName}' ({DataSourceId}).", dataSource.Name, dataSource.Id);
@@ -1583,7 +1631,8 @@ public sealed partial class DataSourceEmbeddingService(SettingsManager settingsM
IReadOnlyList<DataSourceEmbeddingFailure>? failures = null,
int permanentlySkippedFiles = 0,
int? currentFileBlock = null,
int? currentFilePage = null)
int? currentFilePage = null,
bool vectorStoreUnreadable = false)
{
return new DataSourceEmbeddingStatus(
dataSource.Id,
@@ -1598,7 +1647,8 @@ public sealed partial class DataSourceEmbeddingService(SettingsManager settingsM
failures?.ToList() ?? [],
permanentlySkippedFiles,
currentFileBlock,
currentFilePage);
currentFilePage,
vectorStoreUnreadable);
}
/// <remarks>
@@ -1635,6 +1685,25 @@ public sealed partial class DataSourceEmbeddingService(SettingsManager settingsM
failures: [new DataSourceEmbeddingFailure(dataSource.Name, errorMessage, DateTimeOffset.UtcNow)]);
}
/// <remarks>
/// Deliberately not the message which came from the runtime: that one names a store name and a
/// path, is written in English for the log file, and says nothing about what happens next. What
/// the user needs to read is what this means for their chats and where the way out is.
/// </remarks>
private DataSourceEmbeddingStatus GetUnreadableVectorStoreStatus(IDataSource dataSource)
{
var errorMessage = string.Format(TB("The index of the data source '{0}' cannot be read anymore. The data source stays out of your chats until its index was built anew. Use the repair action to start that."), dataSource.Name);
return this.CreateStatus(
dataSource,
DataSourceEmbeddingState.FAILED,
0,
0,
1,
lastError: errorMessage,
failures: [new DataSourceEmbeddingFailure(dataSource.Name, errorMessage, DateTimeOffset.UtcNow)],
vectorStoreUnreadable: true);
}
private DataSourceQueueRequestResult TryReserveDataSourceQueueSlot(string dataSourceId, bool queueAfterCurrentRun)
{
lock (this.queueStateLock)
@@ -7,6 +7,10 @@ namespace AIStudio.Tools.Services;
/// CurrentFileBlock and CurrentFilePage are null rather than zero while nothing is known about
/// them: a file which is only about to start has no first block, and not every kind of document
/// has pages to count. Block numbers start at one, the way the page states them.
///
/// VectorStoreUnreadable says why a data source failed, not only that it did. The UI needs that
/// difference to offer the repair for this one case, and it is carried as its own flag so nothing
/// has to read it back out of the message in LastError.
/// </remarks>
public sealed record DataSourceEmbeddingStatus(
string DataSourceId,
@@ -21,7 +25,8 @@ public sealed record DataSourceEmbeddingStatus(
IReadOnlyList<DataSourceEmbeddingFailure> Failures,
int PermanentlySkippedFiles = 0,
int? CurrentFileBlock = null,
int? CurrentFilePage = null)
int? CurrentFilePage = null,
bool VectorStoreUnreadable = false)
{
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(DataSourceEmbeddingStatus).Namespace, nameof(DataSourceEmbeddingStatus));
@@ -180,6 +180,17 @@ public sealed class DataSourceLocalRetrievalService(
await this.ReportRetrievalGapAsync(dataSource, $"provider-{exception.FailureReason}", string.Format(TB("The data source '{0}' was left out of the answer. {1}"), dataSource.Name, exception.UserMessage));
return [];
}
catch (VectorStoreUnreadableException exception)
{
//
// Its own gap key, because this is not a search which went wrong but an index which has
// to be built anew. Saying that once per session is what turns a silently shortened
// answer into one the user can do something about.
//
logger.LogWarning(exception, "Vector retrieval failed for data source '{DataSourceName}' ({DataSourceId}) because its vector store cannot be read.", dataSource.Name, dataSource.Id);
await this.ReportRetrievalGapAsync(dataSource, "vector-store-unreadable", string.Format(TB("The data source '{0}' was left out of the answer: its index cannot be read anymore. You can repair it in your data source settings."), dataSource.Name));
return [];
}
catch (Exception exception)
{
logger.LogWarning(exception, "Vector retrieval failed for data source '{DataSourceName}' ({DataSourceId}).", dataSource.Name, dataSource.Id);
@@ -51,7 +51,7 @@ public sealed class DataSourceService
if (selectedLLMProvider == Settings.Provider.NONE)
{
this.logger.LogWarning("The selected LLM provider is not set. We cannot filter the data sources by any means.");
return new([], [], []);
return new([], [], [], []);
}
var usingTrustedProvider = selectedLLMProvider.IsTrustedForDataSourceSecurityChecks(this.settingsManager);
@@ -83,12 +83,13 @@ public sealed class DataSourceService
var allowedDataSources = await this.GetAllowedDataSources(usingTrustedProvider, participatingProviders, requestedDataSources);
//
// Whoever asks this way has no list to show, so a data source waiting for its index is
// Whoever asks this way has no list to show, so a data source which cannot be searched is
// dropped rather than marked. Handing it back would start a chat with a data source which
// finds nothing -- the very thing being greyed out elsewhere is meant to prevent.
//
var awaitingReindexIds = (await this.GetDataSourcesAwaitingReindex(allowedDataSources)).Select(source => source.Id).ToHashSet(StringComparer.Ordinal);
return allowedDataSources.Where(source => !awaitingReindexIds.Contains(source.Id)).ToList();
var unsearchableIds = (await this.GetDataSourcesAwaitingReindex(allowedDataSources)).Select(source => source.Id).ToHashSet(StringComparer.Ordinal);
unsearchableIds.UnionWith(this.GetDataSourcesNeedingRepair(allowedDataSources).Select(source => source.Id));
return allowedDataSources.Where(source => !unsearchableIds.Contains(source.Id)).ToList();
}
/// <summary>
@@ -109,7 +110,7 @@ public sealed class DataSourceService
if (selectedLLMProvider is NoProvider)
{
this.logger.LogWarning("The selected LLM provider is the default provider. We cannot filter the data sources by any means.");
return new([], [], []);
return new([], [], [], []);
}
var usingTrustedProvider = selectedLLMProvider.IsTrustedForDataSourceSecurityChecks(this.settingsManager);
@@ -159,12 +160,21 @@ public sealed class DataSourceService
// being rebuilt is usable again in a while, and saying so on its own row beats letting it
// disappear from the selection without a word.
//
var awaitingReindex = await this.GetDataSourcesAwaitingReindex(filteredDataSources);
var awaitingReindexIds = awaitingReindex.Select(source => source.Id).ToHashSet(StringComparer.Ordinal);
var usableDataSources = filteredDataSources.Where(source => !awaitingReindexIds.Contains(source.Id)).ToList();
// A source whose index cannot be read is asked about first and then kept out of the other
// list: both reasons can be true at once, and of the two it is the only one the user can do
// anything about. Telling them to wait instead would be telling them to wait forever.
//
var needingRepair = this.GetDataSourcesNeedingRepair(filteredDataSources);
var needingRepairIds = needingRepair.Select(source => source.Id).ToHashSet(StringComparer.Ordinal);
var awaitingReindex = (await this.GetDataSourcesAwaitingReindex(filteredDataSources)).Where(source => !needingRepairIds.Contains(source.Id)).ToList();
var blockedIds = awaitingReindex.Select(source => source.Id).ToHashSet(StringComparer.Ordinal);
blockedIds.UnionWith(needingRepairIds);
var usableDataSources = filteredDataSources.Where(source => !blockedIds.Contains(source.Id)).ToList();
var filteredSelectedDataSources = usableDataSources.Where(source => previousSelectedDataSourceIds.Contains(source.Id)).ToList();
return new(usableDataSources, filteredSelectedDataSources, awaitingReindex);
return new(usableDataSources, filteredSelectedDataSources, awaitingReindex, needingRepair);
}
/// <summary>
@@ -195,6 +205,30 @@ public sealed class DataSourceService
return awaitingReindex;
}
/// <summary>
/// Picks out the data sources whose index cannot be read anymore, so that they wait for a repair.
/// </summary>
/// <remarks>
/// Reads nothing from a database, unlike the re-index check above: the state is held in memory
/// by the embedding service, which is why this one needs no parallelism and no timeout.
/// </remarks>
/// <param name="dataSources">The data sources which passed every other check.</param>
/// <returns>Those of them which wait for a repair, in the order they came in.</returns>
private IReadOnlyList<IDataSource> GetDataSourcesNeedingRepair(IReadOnlyList<IDataSource> dataSources)
{
var needingRepair = new List<IDataSource>();
foreach (var dataSource in dataSources)
{
if (!this.embeddingService.NeedsIndexRepair(dataSource))
continue;
this.logger.LogInformation("The index of data source '{DataSourceName}' ({DataSourceId}) cannot be read. It is shown, but cannot be selected until it was repaired.", dataSource.Name, dataSource.Id);
needingRepair.Add(dataSource);
}
return needingRepair;
}
private async Task<IReadOnlyList<IDataSource>> GetAllowedDataSources(bool usingTrustedProvider, IReadOnlyList<ParticipatingProvider> participatingProviders, IReadOnlyCollection<IDataSource> requestedDataSources)
{
var filteredDataSources = new List<IDataSource>(requestedDataSources.Count);
@@ -1,7 +1,18 @@
using AIStudio.Tools.Databases.VectorStore;
namespace AIStudio.Tools.Services;
public sealed partial class RustService
{
/// <summary>
/// The issue code the Rust runtime sends when a vector store is there, but cannot be opened.
/// </summary>
/// <remarks>
/// Mirrors ISSUE_CODE_STORE_UNREADABLE in runtime/src/qdrant_edge_database.rs. Reading the code
/// rather than the message is what keeps a reworded message on the Rust side harmless here.
/// </remarks>
private const string ISSUE_CODE_STORE_UNREADABLE = "store-unreadable";
public async Task<TDatabaseInfo> GetDatabaseInfo<TDatabaseInfo>(
string databaseName,
string infoPath,
@@ -46,7 +57,7 @@ public sealed partial class RustService
var operation = await response.Content.ReadFromJsonAsync<DatabaseOperationResponse>(this.jsonRustSerializerOptions, cts.Token);
if (operation is not { Success: true })
throw new InvalidOperationException(operation?.Issue ?? $"The {databaseName} operation failed.");
throw CreateDatabaseException(operation?.Issue, operation?.IssueCode, $"The {databaseName} operation failed.");
}
public async Task<TResult?> ExecuteDatabaseQuery<TRequest, TResult>(string databaseName, string path, TRequest request, CancellationToken cancellationToken = default)
@@ -59,12 +70,31 @@ public sealed partial class RustService
var operation = await response.Content.ReadFromJsonAsync<DatabaseQueryResponse<TResult>>(this.jsonRustSerializerOptions, cts.Token);
if (operation is not { Success: true })
throw new InvalidOperationException(operation?.Issue ?? $"The {databaseName} query failed.");
throw CreateDatabaseException(operation?.Issue, operation?.IssueCode, $"The {databaseName} query failed.");
return operation.Data;
}
private sealed record DatabaseOperationResponse(bool Success, string Issue);
/// <summary>
/// Turns a failed database response into the exception which fits its issue code.
/// </summary>
/// <remarks>
/// Almost every failure says all it has to say in its message. A store which cannot be opened is
/// the exception: the only way out of it is a rebuild which costs the user money and time, so it
/// gets a type of its own and reaches the places which can offer that rebuild instead of
/// starting it unasked.
/// </remarks>
private static Exception CreateDatabaseException(string? issue, string? issueCode, string fallbackMessage)
{
var message = string.IsNullOrWhiteSpace(issue) ? fallbackMessage : issue;
return issueCode switch
{
ISSUE_CODE_STORE_UNREADABLE => new VectorStoreUnreadableException(message),
_ => new InvalidOperationException(message),
};
}
private sealed record DatabaseQueryResponse<TResult>(bool Success, string Issue, TResult? Data);
private sealed record DatabaseOperationResponse(bool Success, string Issue, string IssueCode);
private sealed record DatabaseQueryResponse<TResult>(bool Success, string Issue, string IssueCode, TResult? Data);
}