Added a configurable Opus bitrate for audio transcription (#981)
Build and Release / Read metadata (push) Blocked by required conditions
Build and Release / Sync Flatpak repo (push) Blocked by required conditions
Build and Release / Collect Flatpak artifacts (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-apple-darwin, osx-arm64, macos-latest, aarch64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-pc-windows-msvc.exe, win-arm64, windows-latest, aarch64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-apple-darwin, osx-x64, macos-latest, x86_64-apple-darwin, dmg,app,updater, dmg) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-pc-windows-msvc.exe, win-x64, windows-latest, x86_64-pc-windows-msvc, nsis,updater, nsis) (push) Blocked by required conditions
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-x86_64-unknown-linux-gnu, linux-x64, ubuntu-22.04, x86_64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions
Build and Release / Prepare & create release (push) Blocked by required conditions
Build and Release / Publish release (push) Blocked by required conditions
Build and Release / Verify (push) Waiting to run
Build and Release / Determine run mode (push) Waiting to run
Build and Release / Build app (${{ matrix.dotnet_runtime }}) (-aarch64-unknown-linux-gnu, linux-arm64, ubuntu-22.04-arm, aarch64-unknown-linux-gnu, appimage,updater, appimage) (push) Blocked by required conditions

Co-authored-by: Thorsten Sommer <SommerEngineering@users.noreply.github.com>
This commit is contained in:
Dominic NeuburgandThorsten Sommer authored and GitHub committed 2026-09-19 19:13:33 +02:00
1 parent f9c6075c50
commit 1379ff6aab
19 files changed
+304 -23

No files matched your search

@@ -4561,6 +4561,9 @@ UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1723256298"]
-- Select a transcription provider for transcribing your voice. Without a selected provider, dictation and transcription features will be disabled.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1834486728"] = "Select a transcription provider for transcribing your voice. Without a selected provider, dictation and transcription features will be disabled."
-- Higher bitrates can improve transcription accuracy, especially for quiet or noisy recordings, at the cost of a larger upload to the transcription provider. 128 kbps is recommended.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1859657826"] = "Higher bitrates can improve transcription accuracy, especially for quiet or noisy recordings, at the cost of a larger upload to the transcription provider. 128 kbps is recommended."
-- Select the language behavior for the app. The default is to use the system language. You might want to choose a language manually?
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T186780842"] = "Select the language behavior for the app. The default is to use the system language. You might want to choose a language manually?"
@@ -4627,6 +4630,9 @@ UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T2960110864"]
-- Save energy?
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3100928009"] = "Save energy?"
-- Transcription audio quality
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3103106744"] = "Transcription audio quality"
-- Development builds do not install updates.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3138812562"] = "Development builds do not install updates."
@@ -10489,6 +10495,21 @@ UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::THEMESEXTENSIONS::T4107955313"]
-- Always use light theme
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::THEMESEXTENSIONS::T534715610"] = "Always use light theme"
-- 128 kbps (recommended)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T2152168180"] = "128 kbps (recommended)"
-- 256 kbps (largest upload, highest accuracy)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3092489829"] = "256 kbps (largest upload, highest accuracy)"
-- Unknown
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3424652889"] = "Unknown"
-- 64 kbps
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3501477553"] = "64 kbps"
-- 32 kbps (smallest upload, lowest accuracy)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T767394292"] = "32 kbps (smallest upload, lowest accuracy)"
-- Use no profile
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::PROFILE::T2205839602"] = "Use no profile"
@@ -50,6 +50,7 @@
</ItemTemplate>
</ConfigurationSelect>
<ConfigurationShortcut Data="@this.VoiceRecordingShortcut" OptionDescription="@T("Voice recording shortcut")" OptionHelp="@T("The global keyboard shortcut for toggling voice recording. This shortcut works system-wide, even when the app is not focused.")" IsLocked="() => ManagedConfiguration.TryGet(x => x.App, x => x.ShortcutVoiceRecording, out var meta) && meta.IsLocked"/>
<ConfigurationSelect OptionDescription="@T("Transcription audio quality")" SelectedValue="@(() => this.SettingsManager.ConfigurationData.App.OpusBitrate)" Data="@ConfigurationSelectDataFactory.GetTranscriptionOpusBitrateData()" SelectionUpdate="@(selectedValue => this.SettingsManager.ConfigurationData.App.OpusBitrate = selectedValue)" OptionHelp="@this.OpusBitrateHelp" IsLocked="@IsOpusBitrateLocked"/>
}
@if (this.SettingsManager.ConfigurationData.App.ShowAdminSettings)
@@ -79,6 +79,10 @@ public partial class SettingsPanelApp : SettingsPanelBase
DisplayUpdate = this.UpdateShortcutVoiceRecordingDisplay,
};
private string OpusBitrateHelp => T("Higher bitrates can improve transcription accuracy, especially for quiet or noisy recordings, at the cost of a larger upload to the transcription provider. 128 kbps is recommended.");
private static bool IsOpusBitrateLocked() => ManagedConfiguration.TryGet(x => x.App, x => x.OpusBitrate, out var meta) && meta.IsLocked;
private async Task GenerateEncryptionSecret()
{
var secret = EnterpriseEncryption.GenerateSecret();
@@ -12,7 +12,7 @@
<MudJustifiedText Typo="Typo.body1" Class="mb-3">
@T("With the support of transcription models, MindWork AI Studio can convert human speech into text. This is useful, for example, when you need to dictate text. You can choose from dedicated transcription models, but not multimodal LLMs (large language models) that can handle both speech and text. The configuration of multimodal models is done in the 'Configure providers' section.")
</MudJustifiedText>
<MudTable Items="@this.SettingsManager.GetAllTranscriptionProviders()" Hover="@true" GroupBy="@GROUP_CONFIG" Class="border-dashed border rounded-lg">
<ColGroup>
<col style="width: 12em;"/>
@@ -36,7 +36,7 @@ public partial class SettingsPanelTranscription : SettingsPanelProviderBase
var modelName = provider.Model.ToString();
return modelName.Length > MAX_LENGTH ? "[...] " + modelName[^Math.Min(MAX_LENGTH, modelName.Length)..] : modelName;
}
#region Overrides of ComponentBase
protected override async Task OnInitializedAsync()
@@ -578,6 +578,18 @@ CONFIG["SETTINGS"] = {}
-- Please note: using an empty string ("") will lock the selection and disable dictation/transcription.
-- CONFIG["SETTINGS"]["DataApp.UseTranscriptionProvider"] = "00000000-0000-0000-0000-000000000000"
-- Configure the Opus bitrate used when normalizing uploaded audio/video for transcription.
-- Allowed values are: KBPS_32, KBPS_64, KBPS_128, KBPS_256
-- Higher bitrates improve transcription accuracy on noisy or quiet recordings, at the cost of
-- a larger upload to the transcription provider. KBPS_128 is recommended.
-- Please note: this bitrate applies whenever a recording has to be re-encoded. A file which already
-- is a single mono 48 kHz Opus track in a WebM container and stays below 25 MiB is forwarded to the
-- transcription provider unchanged, keeping the bitrate it was created with.
-- CONFIG["SETTINGS"]["DataApp.OpusBitrate"] = "KBPS_32"
--
-- Allow the user to change the Opus bitrate even though your organization set a default above:
-- CONFIG["SETTINGS"]["DataApp.OpusBitrate.AllowUserOverride"] = true
-- Configure which assistants should be hidden from the UI.
-- Allowed values are:
-- GRAMMAR_SPELLING_ASSISTANT, ICON_FINDER_ASSISTANT, REWRITE_ASSISTANT,
@@ -4563,6 +4563,9 @@ UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1723256298"]
-- Select a transcription provider for transcribing your voice. Without a selected provider, dictation and transcription features will be disabled.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1834486728"] = "Wählen Sie für die Transkription Ihrer Stimme einen Anbieter für Transkriptionen aus. Ohne einen ausgewählten Anbieter wird die Diktier- und Transkriptions-Funktion deaktiviert."
-- Higher bitrates can improve transcription accuracy, especially for quiet or noisy recordings, at the cost of a larger upload to the transcription provider. 128 kbps is recommended.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1859657826"] = "Höhere Bitraten können die Genauigkeit der Transkription verbessern, insbesondere bei leisen oder verrauschten Aufnahmen. Dafür wird eine größere Datei an den Anbieter der Transkription übertragen. Empfohlen werden 128 kbit/s."
-- Select the language behavior for the app. The default is to use the system language. You might want to choose a language manually?
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T186780842"] = "Wählen Sie das Sprachverhalten für die App aus. Standardmäßig wird die Systemsprache verwendet. Möchten Sie die Sprache manuell einstellen?"
@@ -4629,6 +4632,9 @@ UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T2960110864"]
-- Save energy?
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3100928009"] = "Energie sparen?"
-- Transcription audio quality
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3103106744"] = "Audioqualität der Transkription"
-- Development builds do not install updates.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3138812562"] = "Entwicklerversionen installieren keine Updates."
@@ -10491,6 +10497,21 @@ UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::THEMESEXTENSIONS::T4107955313"]
-- Always use light theme
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::THEMESEXTENSIONS::T534715610"] = "Immer das helle Design verwenden"
-- 128 kbps (recommended)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T2152168180"] = "128 kbit/s (empfohlen)"
-- 256 kbps (largest upload, highest accuracy)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3092489829"] = "256 kbit/s (größte Datei, höchste Genauigkeit)"
-- Unknown
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3424652889"] = "Unbekannt"
-- 64 kbps
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3501477553"] = "64 kbit/s"
-- 32 kbps (smallest upload, lowest accuracy)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T767394292"] = "32 kbit/s (kleinste Datei, geringste Genauigkeit)"
-- Use no profile
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::PROFILE::T2205839602"] = "Kein Profil verwenden"
@@ -4563,6 +4563,9 @@ UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1723256298"]
-- Select a transcription provider for transcribing your voice. Without a selected provider, dictation and transcription features will be disabled.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1834486728"] = "Select a transcription provider for transcribing your voice. Without a selected provider, dictation and transcription features will be disabled."
-- Higher bitrates can improve transcription accuracy, especially for quiet or noisy recordings, at the cost of a larger upload to the transcription provider. 128 kbps is recommended.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T1859657826"] = "Higher bitrates can improve transcription accuracy, especially for quiet or noisy recordings, at the cost of a larger upload to the transcription provider. 128 kbps is recommended."
-- Select the language behavior for the app. The default is to use the system language. You might want to choose a language manually?
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T186780842"] = "Select the language behavior for the app. The default is to use the system language. You might want to choose a language manually?"
@@ -4629,6 +4632,9 @@ UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T2960110864"]
-- Save energy?
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3100928009"] = "Save energy?"
-- Transcription audio quality
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3103106744"] = "Transcription audio quality"
-- Development builds do not install updates.
UI_TEXT_CONTENT["AISTUDIO::COMPONENTS::SETTINGS::SETTINGSPANELAPP::T3138812562"] = "Development builds do not install updates."
@@ -10491,6 +10497,21 @@ UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::THEMESEXTENSIONS::T4107955313"]
-- Always use light theme
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::THEMESEXTENSIONS::T534715610"] = "Always use light theme"
-- 128 kbps (recommended)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T2152168180"] = "128 kbps (recommended)"
-- 256 kbps (largest upload, highest accuracy)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3092489829"] = "256 kbps (largest upload, highest accuracy)"
-- Unknown
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3424652889"] = "Unknown"
-- 64 kbps
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T3501477553"] = "64 kbps"
-- 32 kbps (smallest upload, lowest accuracy)
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::DATAMODEL::TRANSCRIPTIONOPUSBITRATEEXTENSIONS::T767394292"] = "32 kbps (smallest upload, lowest accuracy)"
-- Use no profile
UI_TEXT_CONTENT["AISTUDIO::SETTINGS::PROFILE::T2205839602"] = "Use no profile"
@@ -323,4 +323,12 @@ public static class ConfigurationSelectDataFactory
yield return new(level.GetName(), level);
}
}
public static IEnumerable<ConfigurationSelectData<TranscriptionOpusBitrate>> GetTranscriptionOpusBitrateData()
{
foreach (var bitrate in Enum.GetValues<TranscriptionOpusBitrate>())
{
yield return new(bitrate.GetName(), bitrate);
}
}
}
@@ -112,6 +112,18 @@ public sealed class DataApp(Expression<Func<Data, DataApp>>? configSelection = n
/// </summary>
public string UseTranscriptionProvider { get; set; } = ManagedConfiguration.Register(configSelection, n => n.UseTranscriptionProvider, string.Empty);
/// <summary>
/// The Opus bitrate used when normalizing uploaded audio/video for transcription.
/// </summary>
/// <remarks>
/// Every recording is re-encoded to mono Opus before it goes to the transcription provider. That
/// encoding used to be fixed at 32 kbps, which cost the transcription models entire quiet passages:
/// a greeting spoken softly at the start of a recording was simply missing from the transcript. The
/// same recording compared at 64 and 128 kbps came back complete, which is why 128 kbps is the
/// default here. Users trading accuracy for a smaller upload can still pick a lower bitrate.
/// </remarks>
public TranscriptionOpusBitrate OpusBitrate { get; set; } = ManagedConfiguration.Register(configSelection, n => n.OpusBitrate, TranscriptionOpusBitrate.KBPS_128);
/// <summary>
/// The global keyboard shortcut for toggling voice recording.
/// Uses Tauri's shortcut format, e.g., "CmdOrControl+1" (Cmd+1 on macOS, Ctrl+1 on Windows/Linux).
@@ -0,0 +1,14 @@
namespace AIStudio.Settings.DataModel;
public enum TranscriptionOpusBitrate
{
// The recommended bitrate is deliberately the member with the underlying value 0: when the
// settings file holds a value TolerantEnumConverter cannot read, it falls back to that member.
// Landing on the lowest bitrate there would silently reintroduce the very defect this setting
// exists to prevent -- transcripts losing what was said quietly.
KBPS_128 = 0,
KBPS_32,
KBPS_64,
KBPS_256,
}
@@ -0,0 +1,29 @@
using AIStudio.Tools.PluginSystem;
namespace AIStudio.Settings.DataModel;
public static class TranscriptionOpusBitrateExtensions
{
private static string TB(string fallbackEN) => I18N.I.T(fallbackEN, typeof(TranscriptionOpusBitrateExtensions).Namespace, nameof(TranscriptionOpusBitrateExtensions));
public static string GetName(this TranscriptionOpusBitrate bitrate) => bitrate switch
{
TranscriptionOpusBitrate.KBPS_32 => TB("32 kbps (smallest upload, lowest accuracy)"),
TranscriptionOpusBitrate.KBPS_64 => TB("64 kbps"),
TranscriptionOpusBitrate.KBPS_128 => TB("128 kbps (recommended)"),
TranscriptionOpusBitrate.KBPS_256 => TB("256 kbps (largest upload, highest accuracy)"),
_ => TB("Unknown"),
};
public static uint GetBitsPerSecond(this TranscriptionOpusBitrate bitrate) => bitrate switch
{
TranscriptionOpusBitrate.KBPS_32 => 32_000,
TranscriptionOpusBitrate.KBPS_64 => 64_000,
TranscriptionOpusBitrate.KBPS_128 => 128_000,
TranscriptionOpusBitrate.KBPS_256 => 256_000,
// A value we do not know must never mean the lowest quality: that is how quiet passages
// went missing from transcripts in the first place. Fall back to the recommended bitrate.
_ => 128_000,
};
}
@@ -398,6 +398,9 @@ public sealed class PluginConfiguration(bool isInternal, LuaState state, PluginT
// Config: transcription provider?
ManagedConfiguration.TryProcessConfiguration(x => x.App, x => x.UseTranscriptionProvider, Guid.Empty, this.Id, settingsTable, dryRun);
// Config: transcription Opus bitrate?
ManagedConfiguration.TryProcessConfiguration(x => x.App, x => x.OpusBitrate, this.Id, settingsTable, dryRun);
message = string.Empty;
return true;
}
@@ -4,4 +4,5 @@ namespace AIStudio.Tools.Rust;
/// <param name="InputPath">Absolute source media path.</param>
/// <param name="OutputPath">Absolute operation-owned output path.</param>
/// <param name="MaxPassThroughBytes">Optional pass-through size ceiling.</param>
public sealed record CreateMediaJobRequest(string InputPath, string OutputPath, ulong? MaxPassThroughBytes = null);
/// <param name="OpusBitrateBps">Optional target Opus encoder bitrate in bits per second.</param>
public sealed record CreateMediaJobRequest(string InputPath, string OutputPath, ulong? MaxPassThroughBytes = null, uint? OpusBitrateBps = null);
@@ -1,6 +1,7 @@
using AIStudio.Chat;
using AIStudio.Provider;
using AIStudio.Settings;
using AIStudio.Settings.DataModel;
using AIStudio.Tools.Media;
using AIStudio.Tools.PluginSystem;
using AIStudio.Tools.Rust;
@@ -646,9 +647,23 @@ public sealed class MediaTranscriptionService(
MediaOperation operation,
bool updateImportState)
{
// The bitrate governs re-encoding, not every upload: the runtime hands a file through
// unchanged when it already is a single mono 48 kHz Opus track in a WebM container and small
// enough. Such a file keeps whatever bitrate it was made with, which is the better outcome --
// re-encoding it could only take quality away, never add any.
var opusBitrateBps = settingsManager.ConfigurationData.App.OpusBitrate.GetBitsPerSecond();
// Which quality an upload was produced with is the first question to ask when a transcript
// comes back missing something, so it has to be in the log of the job it belongs to:
logger.LogInformation(
"Normalizing media for operation {OperationId}; re-encoding uses the Opus bitrate {OpusBitrate} ({OpusBitrateBps} bps).",
operation.Id,
settingsManager.ConfigurationData.App.OpusBitrate,
opusBitrateBps);
// The quick POST is intentionally not cancelled: losing its response could orphan a job
// whose ID the client never received. Cancellation is applied immediately after ownership.
var jobId = await rustService.StartMediaJobAsync(mediaPath, normalizedPath, CancellationToken.None);
var jobId = await rustService.StartMediaJobAsync(mediaPath, normalizedPath, opusBitrateBps, CancellationToken.None);
operation.JobId = jobId;
try
@@ -10,13 +10,14 @@ public partial class RustService
/// <summary>Starts a Rust media normalization job.</summary>
/// <param name="inputPath">Absolute source path.</param>
/// <param name="outputPath">Absolute operation-owned output path.</param>
/// <param name="opusBitrateBps">Target Opus encoder bitrate in bits per second.</param>
/// <param name="token">Request cancellation token.</param>
/// <returns>The opaque runtime job identifier.</returns>
public async Task<string> StartMediaJobAsync(string inputPath, string outputPath, CancellationToken token = default)
public async Task<string> StartMediaJobAsync(string inputPath, string outputPath, uint opusBitrateBps, CancellationToken token = default)
{
using var response = await this.http.PostAsJsonAsync(
"/media/jobs",
new CreateMediaJobRequest(inputPath, outputPath),
new CreateMediaJobRequest(inputPath, outputPath, OpusBitrateBps: opusBitrateBps),
this.jsonRustSerializerOptions,
token);
@@ -28,6 +28,8 @@
- Added drag and drop to the input and output folder of the Batch Processing assistant: drop a folder onto either field to choose it.
- Added ways to load text from a file and drop zones for them, throughout the assistants and dialogs. We went through them one by one, so many fields that used to accept typed text only now take the content of a file as well.
- Added an optional API key to every server you host yourself, among them LM Studio, llama.cpp, and whisper.cpp. Such a server may ask for one itself or sit behind a login your organization placed in front of it. So far, only Ollama and vLLM could be given a key.
- Added a setting for the audio quality used when your speech and your audio and video files are transcribed. AI Studio prepares every recording before it goes to your transcription provider, and you now decide how much detail it keeps: a lower quality travels faster, a higher one gives the transcription model more to work with. You find it in the app settings, right below your transcription provider. Thanks, Dominic Neuburg (`donework`), for this contribution.
- Added organization-wide management for the audio quality used when transcribing. IT departments can set the quality their organization works with and lock it, or leave it as a default their colleagues are free to change.
- Improved loading web content in the assistants: it now uses the same reader as the Read Web Page tool, which extracts the main content of a page more reliably and skips navigation and boilerplate. Pages from your own network, including local servers, keep working as before. When a page cannot be read, AI Studio now says why instead of leaving the field empty.
- Improved the app icon. The previous one was generated by an image model; the new one was created based on it and keeps the familiar green landscape with the chat bubble. Because it is now a vector drawing, it stays sharp everywhere it appears: in your taskbar or dock, in the window list, and on the start screen while AI Studio is loading.
- Improved how AI Studio works out what a model can do. Every model family now stands on its own, together with the page it was read from, and our build refuses rules which contradict each other or name no source. That way, mistakes are caught before they ever reach you.
@@ -69,5 +71,6 @@
- Fixed the list of models staying empty at a server you host yourself, which made the model you had picked look as if it had vanished. Your key was there all along, it just was not read when the settings opened.
- Fixed AI Studio asking such a server for its models with an empty key attached when you had stored none at all. Servers behind a login turn those requests down.
- Fixed a key that could not be saved going unmentioned for the servers you host yourself. You are now told what went wrong, instead of the settings simply staying open.
- Fixed transcripts are quietly losing what was said softly, such as a greeting at the very beginning of a recording. AI Studio compressed recordings so far before sending them to your transcription provider that the model could no longer make out those passages. Recordings now keep enough details for the whole of what you said to arrive.
- Upgraded the Visual Briefing Assistant (in preview) from the prototype to the beta state. The assistant is now completely implemented and is undergoing a deeper testing phase in preparation for release. To try it, open the app settings, allow preview features down to beta, and then enable the Visual Briefing Assistant there.
- Upgraded the vector database behind local RAG (Qdrant Edge) to version 0.8.0.
@@ -0,0 +1,54 @@
using AIStudio.Settings.DataModel;
namespace AIStudio.Tests.Settings;
/// <summary>
/// Checks which bitrate the transcription pipeline ends up asking the Opus encoder for.
/// </summary>
/// <remarks>
/// This number is the whole reason the setting exists. Until v26.9.1 the encoder was fixed at
/// 32 kbps, and transcription models silently dropped what had been spoken quietly -- a greeting at
/// the start of a recording never reached the transcript. Two ways back to that value have to stay
/// closed: the arm catching a bitrate nobody declared, and the member a settings file the app cannot
/// read falls back to. Both are one edit away from pointing at the lowest quality again.
/// </remarks>
[TestFixture]
public sealed class TranscriptionOpusBitrateTests
{
private static readonly Dictionary<TranscriptionOpusBitrate, uint> EXPECTED_BITS_PER_SECOND = new()
{
[TranscriptionOpusBitrate.KBPS_32] = 32_000,
[TranscriptionOpusBitrate.KBPS_64] = 64_000,
[TranscriptionOpusBitrate.KBPS_128] = 128_000,
[TranscriptionOpusBitrate.KBPS_256] = 256_000,
};
[Test]
public void EveryOfferedBitrateAsksForWhatItsNameSays()
{
foreach (var bitrate in Enum.GetValues<TranscriptionOpusBitrate>())
{
Assert.That(EXPECTED_BITS_PER_SECOND.ContainsKey(bitrate), Is.True, $"The selection offers {bitrate}, so this test has to state what that is worth in bits per second.");
Assert.That(bitrate.GetBitsPerSecond(), Is.EqualTo(EXPECTED_BITS_PER_SECOND[bitrate]), $"{bitrate} is what the user picked; anything else travels to the encoder behind their back.");
}
}
[Test]
public void ABitrateNobodyDeclaredFallsBackToTheRecommendedOne()
{
var undeclared = (TranscriptionOpusBitrate)999;
Assert.That(Enum.IsDefined(undeclared), Is.False, "The point of this test is a value outside the enum; a declared one would prove nothing.");
Assert.That(undeclared.GetBitsPerSecond(), Is.EqualTo(128_000u), "Not knowing which quality was meant is no reason to pick the worst one available.");
}
[Test]
public void AnUnreadableSettingsValueLandsOnTheRecommendedBitrate()
{
//
// TolerantEnumConverter answers a value it cannot parse with the member whose underlying
// value is zero. Which member that is decides what a damaged settings file transcribes with:
//
Assert.That(default(TranscriptionOpusBitrate), Is.EqualTo(TranscriptionOpusBitrate.KBPS_128), "The member with the underlying value zero is what a settings file the app cannot read falls back to, so it has to be the recommended bitrate.");
}
}