diff --git a/MeetingAssistant.Tests/HealthEndpointTests.cs b/MeetingAssistant.Tests/HealthEndpointTests.cs index 2ae66de..718d86a 100644 --- a/MeetingAssistant.Tests/HealthEndpointTests.cs +++ b/MeetingAssistant.Tests/HealthEndpointTests.cs @@ -1,6 +1,7 @@ using System.Net; using System.Net.Http.Json; using MeetingAssistant.Recording; +using MeetingAssistant.Workflow; using Microsoft.AspNetCore.Hosting; using Microsoft.AspNetCore.Mvc.Testing; using Microsoft.AspNetCore.TestHost; @@ -54,6 +55,17 @@ public sealed class HealthEndpointTests : IClassFixture(); + + var result = await playback.QueueAsync(1, [1, 2]); + + Assert.Contains("Windows build", result); + Assert.DoesNotContain("Queued", result); + } + [Fact] public async Task ApplicationStartupDeletesStaleTemporaryRecordings() { diff --git a/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs b/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs index a39d570..bfa622d 100644 --- a/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs +++ b/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs @@ -58,6 +58,64 @@ public sealed class LaunchProfileOptionsProviderTests Assert.Equal("Ctrl+Alt+L", profile.Options.Hotkey.Toggle); } + [Fact] + public void CheckedInConfigurationKeepsResemblyzerOptInWithDocumentedDefaults() + { + var profile = CreateProviderFromAppsettings().GetRequiredProfile(null); + + Assert.False(profile.Options.SpeakerIdentification.Resemblyzer.Enabled); + Assert.Equal(5, profile.Options.SpeakerIdentification.Resemblyzer.RequiredVectorsPerSpeaker); + Assert.Equal(1000, profile.Options.SpeakerIdentification.Resemblyzer.MaxVectorsPerIdentity); + Assert.Equal(20, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningMinimumVectors); + Assert.Equal(0.75, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningNeighborSimilarity); + Assert.Equal(3, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningMinimumNeighbors); + Assert.Equal(0.60, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningMinimumClusterRatio); + } + + [Fact] + public void InvalidResemblyzerOutlierPruningSettingsAreRejected() + { + var provider = CreateProvider(new Dictionary + { + ["MeetingAssistant:SpeakerIdentification:Resemblyzer:OutlierPruningMinimumClusterRatio"] = "1.1" + }); + + var exception = Assert.Throws( + () => provider.GetRequiredProfile(null)); + + Assert.Contains("OutlierPruningMinimumClusterRatio", exception.Message); + } + + [Fact] + public void NonFiniteResemblyzerSimilaritySettingsAreRejected() + { + var provider = CreateProvider(new Dictionary + { + ["MeetingAssistant:SpeakerIdentification:Resemblyzer:MinimumIdentitySimilarity"] = "NaN" + }); + + var exception = Assert.Throws( + () => provider.GetRequiredProfile(null)); + + Assert.Contains("MinimumIdentitySimilarity", exception.Message); + } + + [Fact] + public void ContradictorySpeakerSampleDurationsAreRejected() + { + var provider = CreateProvider(new Dictionary + { + ["MeetingAssistant:SpeakerIdentification:MinimumSampleSpeechDuration"] = "00:01:01", + ["MeetingAssistant:SpeakerIdentification:MaximumSampleDuration"] = "00:01:00" + }); + + var exception = Assert.Throws( + () => provider.GetRequiredProfile(null)); + + Assert.Contains("MinimumSampleSpeechDuration", exception.Message); + Assert.Contains("MaximumSampleDuration", exception.Message); + } + [Fact] public void DuplicateProfileHotkeysAreRejected() { diff --git a/MeetingAssistant.Tests/MacOsMeetingAudioSourceTests.cs b/MeetingAssistant.Tests/MacOsMeetingAudioSourceTests.cs index c2f5801..0b2bb65 100644 --- a/MeetingAssistant.Tests/MacOsMeetingAudioSourceTests.cs +++ b/MeetingAssistant.Tests/MacOsMeetingAudioSourceTests.cs @@ -5,6 +5,7 @@ using MeetingAssistant.Transcription; using Microsoft.AspNetCore.Hosting; using Microsoft.AspNetCore.Mvc.Testing; using Microsoft.AspNetCore.TestHost; +using Microsoft.Data.Sqlite; using Microsoft.Extensions.Configuration; using Microsoft.Extensions.DependencyInjection; using Microsoft.Extensions.DependencyInjection.Extensions; @@ -113,7 +114,7 @@ public sealed class MacOsMeetingAudioSourceTests microphonePcm: Pcm16(2_000), systemPcm: Pcm16(10_000)); var speechPipelines = new CapturingSpeechRecognitionPipelineFactory(); - await using var factory = new WebApplicationFactory().WithWebHostBuilder(builder => + var factory = new WebApplicationFactory().WithWebHostBuilder(builder => { builder.ConfigureAppConfiguration((_, configuration) => { @@ -168,6 +169,9 @@ public sealed class MacOsMeetingAudioSourceTests } finally { + await factory.DisposeAsync(); + using var connection = new SqliteConnection($"Data Source={Path.Combine(testRoot, "speakers.db")}"); + SqliteConnection.ClearPool(connection); if (Directory.Exists(testRoot)) { Directory.Delete(testRoot, recursive: true); diff --git a/MeetingAssistant.Tests/MeetingAssistant.Tests.csproj b/MeetingAssistant.Tests/MeetingAssistant.Tests.csproj index a0db135..8cc0182 100644 --- a/MeetingAssistant.Tests/MeetingAssistant.Tests.csproj +++ b/MeetingAssistant.Tests/MeetingAssistant.Tests.csproj @@ -9,10 +9,10 @@ - - + + - + diff --git a/MeetingAssistant.Tests/NotificationActivationArgumentsTests.cs b/MeetingAssistant.Tests/NotificationActivationArgumentsTests.cs index 6fe719e..f1d44f0 100644 --- a/MeetingAssistant.Tests/NotificationActivationArgumentsTests.cs +++ b/MeetingAssistant.Tests/NotificationActivationArgumentsTests.cs @@ -25,4 +25,16 @@ public sealed class NotificationActivationArgumentsTests Assert.Equal("abc123", arguments["promptId"]); Assert.Equal("continue", arguments["response"]); } + + [Fact] + public void InactivityPromptActionsIncludePauseAndParseItsResponse() + { + Assert.Contains( + MeetingInactivityPromptActions.All, + action => action.Content == "Pause transcription" && + action.ResponseArgument == "pause"); + Assert.Equal( + MeetingInactivityPromptResponse.Pause, + MeetingInactivityPromptActions.ParseResponse("pause")); + } } diff --git a/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs b/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs index f7c3baf..a7306a1 100644 --- a/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs +++ b/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs @@ -18,31 +18,8 @@ public sealed class PyannoteDiarizationWarmupHostedServiceTests NullLogger.Instance); var service = new PyannoteDiarizationWarmupHostedService( finalizer, - new FakeLaunchProfileOptionsProvider(new MeetingAssistantOptions - { - Recording = { TranscriptionProvider = "azure-speech" }, - SpeakerIdentification = - { - PyannoteValidation = - { - Enabled = true, - Diarization = - { - Enabled = true, - DockerCommand = "docker", - Image = "meeting-assistant-pyannote-validation:local", - ModelsFolder = Path.Combine( - Path.GetTempPath(), - "meeting-assistant-tests", - Guid.NewGuid().ToString("N"), - "models"), - Token = "hf_test", - TokenEnv = "", - CommandTimeout = TimeSpan.FromMinutes(1) - } - } - } - }), + new FakeLaunchProfileOptionsProvider( + CreateValidationOptions("meeting-assistant-pyannote-validation:local")), NullLogger.Instance); await service.StartAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(1)); @@ -54,23 +31,109 @@ public sealed class PyannoteDiarizationWarmupHostedServiceTests Assert.True(commandRunner.RunCancellationWasObserved); } + [Fact] + public async Task HostedServiceUsesApplicationValidationRuntimeInsteadOfNamedProfileOverride() + { + var commandRunner = new BlockingCommandRunner(); + var finalizer = new PyannoteTranscriptFinalizer( + commandRunner, + Options.Create(new MeetingAssistantOptions()), + NullLogger.Instance); + var defaultOptions = CreateValidationOptions("meeting-assistant-pyannote-default:local"); + var namedOptions = CreateValidationOptions("meeting-assistant-pyannote-named:local"); + var service = new PyannoteDiarizationWarmupHostedService( + finalizer, + new FakeLaunchProfileOptionsProvider( + [ + new LaunchProfile("english", namedOptions), + new LaunchProfile(ConfigurationLaunchProfileOptionsProvider.DefaultProfileName, defaultOptions) + ]), + NullLogger.Instance); + + await service.StartAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(1)); + await commandRunner.WaitForRunAsync(); + + await service.StopAsync(CancellationToken.None); + + Assert.Contains(commandRunner.Commands, command => command.Arguments.Contains("meeting-assistant-pyannote-default:local")); + Assert.DoesNotContain(commandRunner.Commands, command => command.Arguments.Contains("meeting-assistant-pyannote-named:local")); + } + + [Fact] + public async Task HostedServiceDoesNotWarmIdentityValidationWhenResemblyzerIsSelected() + { + var commandRunner = new BlockingCommandRunner(); + var finalizer = new PyannoteTranscriptFinalizer( + commandRunner, + Options.Create(new MeetingAssistantOptions()), + NullLogger.Instance); + var options = CreateValidationOptions("meeting-assistant-pyannote-validation:local"); + options.SpeakerIdentification.Resemblyzer.Enabled = true; + var service = new PyannoteDiarizationWarmupHostedService( + finalizer, + new FakeLaunchProfileOptionsProvider(options), + NullLogger.Instance); + + await service.StartAsync(CancellationToken.None); + await Task.Delay(TimeSpan.FromMilliseconds(50)); + await service.StopAsync(CancellationToken.None); + + Assert.Empty(commandRunner.Commands); + } + + private static MeetingAssistantOptions CreateValidationOptions(string image) + { + return new MeetingAssistantOptions + { + Recording = { TranscriptionProvider = "azure-speech" }, + SpeakerIdentification = + { + PyannoteValidation = + { + Enabled = true, + Diarization = + { + DockerCommand = "docker", + Image = image, + ModelsFolder = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "models"), + Token = "hf_test", + TokenEnv = "", + CommandTimeout = TimeSpan.FromMinutes(1) + } + } + } + }; + } + private sealed class FakeLaunchProfileOptionsProvider : ILaunchProfileOptionsProvider { - private readonly MeetingAssistantOptions options; + private readonly IReadOnlyList profiles; public FakeLaunchProfileOptionsProvider(MeetingAssistantOptions options) + : this([new LaunchProfile(ConfigurationLaunchProfileOptionsProvider.DefaultProfileName, options)]) { - this.options = options; + } + + public FakeLaunchProfileOptionsProvider(IReadOnlyList profiles) + { + this.profiles = profiles; } public LaunchProfile GetRequiredProfile(string? name) { - return new LaunchProfile(ConfigurationLaunchProfileOptionsProvider.DefaultProfileName, options); + var profileName = string.IsNullOrWhiteSpace(name) + ? ConfigurationLaunchProfileOptionsProvider.DefaultProfileName + : name; + return profiles.Single(profile => profile.Name.Equals(profileName, StringComparison.OrdinalIgnoreCase)); } public IReadOnlyList GetProfiles() { - return [GetRequiredProfile(null)]; + return profiles; } public IReadOnlyList GetHotkeys() diff --git a/MeetingAssistant.Tests/PyannoteSpeakerIdentityMatchValidatorTests.cs b/MeetingAssistant.Tests/PyannoteSpeakerIdentityMatchValidatorTests.cs index b025d58..5263b37 100644 --- a/MeetingAssistant.Tests/PyannoteSpeakerIdentityMatchValidatorTests.cs +++ b/MeetingAssistant.Tests/PyannoteSpeakerIdentityMatchValidatorTests.cs @@ -2,6 +2,7 @@ using MeetingAssistant; using MeetingAssistant.Recording; using MeetingAssistant.Speakers; using MeetingAssistant.Transcription; +using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging.Abstractions; using Microsoft.Extensions.Options; using NAudio.Wave; @@ -57,17 +58,49 @@ public sealed class PyannoteSpeakerIdentityMatchValidatorTests Assert.Empty(commandRunner.Commands); } + [Fact] + public async Task OuterToggleControlsValidationWhenLegacyNestedToggleIsFalse() + { + var commandRunner = new CapturingCommandRunner( + """ + __MEETING_ASSISTANT_PYANNOTE_JSON_START__ + [{"start":0.0,"end":20.0,"speaker":"SPEAKER_00"}] + __MEETING_ASSISTANT_PYANNOTE_JSON_END__ + """); + var modelsFolder = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "models"); + var configuration = new ConfigurationBuilder() + .AddInMemoryCollection(new Dictionary + { + ["SpeakerIdentification:PyannoteValidation:Enabled"] = "true", + ["SpeakerIdentification:PyannoteValidation:Diarization:Enabled"] = "false", + ["SpeakerIdentification:PyannoteValidation:Diarization:BuildImage"] = "false", + ["SpeakerIdentification:PyannoteValidation:Diarization:ModelsFolder"] = modelsFolder, + ["SpeakerIdentification:PyannoteValidation:Diarization:Token"] = "hf_test", + ["SpeakerIdentification:PyannoteValidation:Diarization:TokenEnv"] = "" + }) + .Build(); + var configuredOptions = configuration.Get()!; + var validator = CreateValidator(commandRunner, configuredOptions); + + var valid = await validator.ValidateSampleAsync( + CreateWav(TimeSpan.FromSeconds(20)), + CancellationToken.None); + + Assert.True(valid); + Assert.Contains(commandRunner.Commands, command => command.Arguments.Contains("run")); + } + private static PyannoteSpeakerIdentityMatchValidator CreateValidator( CapturingCommandRunner commandRunner, bool enabled = true) { - var finalizer = new PyannoteTranscriptFinalizer( + return CreateValidator( commandRunner, - Options.Create(new MeetingAssistantOptions()), - NullLogger.Instance); - return new PyannoteSpeakerIdentityMatchValidator( - finalizer, - Options.Create(new MeetingAssistantOptions + new MeetingAssistantOptions { SpeakerIdentification = new SpeakerIdentificationOptions { @@ -76,9 +109,8 @@ public sealed class PyannoteSpeakerIdentityMatchValidatorTests Enabled = enabled, MinimumSingleSpeakerCoverage = 0.90, MinimumMatchingKnownSnippetRatio = 1, - Diarization = new PyannoteDiarizationOptions + Diarization = new PyannoteRuntimeOptions { - Enabled = true, BuildImage = false, DockerCommand = "docker", Image = "meeting-assistant-pyannote:local", @@ -94,7 +126,20 @@ public sealed class PyannoteSpeakerIdentityMatchValidatorTests } } } - }), + }); + } + + private static PyannoteSpeakerIdentityMatchValidator CreateValidator( + CapturingCommandRunner commandRunner, + MeetingAssistantOptions configuredOptions) + { + var finalizer = new PyannoteTranscriptFinalizer( + commandRunner, + Options.Create(new MeetingAssistantOptions()), + NullLogger.Instance); + return new PyannoteSpeakerIdentityMatchValidator( + finalizer, + Options.Create(configuredOptions), NullLogger.Instance); } diff --git a/MeetingAssistant.Tests/PyannoteTranscriptFinalizerTests.cs b/MeetingAssistant.Tests/PyannoteTranscriptFinalizerTests.cs index d6d33de..91698e4 100644 --- a/MeetingAssistant.Tests/PyannoteTranscriptFinalizerTests.cs +++ b/MeetingAssistant.Tests/PyannoteTranscriptFinalizerTests.cs @@ -156,9 +156,8 @@ public sealed class PyannoteTranscriptFinalizerTests commandRunner, Options.Create(new MeetingAssistantOptions()), NullLogger.Instance); - var explicitDiarization = new PyannoteDiarizationOptions + var explicitDiarization = new PyannoteRuntimeOptions { - Enabled = true, DockerCommand = "docker", BaseImage = "python:3.11-slim", Image = "meeting-assistant-pyannote-azure:local", @@ -169,7 +168,7 @@ public sealed class PyannoteTranscriptFinalizerTests CommandTimeout = TimeSpan.FromMinutes(1) }; - await finalizer.FinalizeAsync( + await finalizer.FinalizeEnabledAsync( audioPath, [new TranscriptionSegment(TimeSpan.Zero, TimeSpan.FromSeconds(1), "Unknown", "hello")], explicitDiarization, @@ -186,9 +185,8 @@ public sealed class PyannoteTranscriptFinalizerTests { var commandRunner = new CapturingCommandRunner(""); var finalizer = CreateFinalizer(commandRunner, token: "hf_test"); - var diarization = new PyannoteDiarizationOptions + var diarization = new PyannoteRuntimeOptions { - Enabled = true, DockerCommand = "docker", BaseImage = "python:3.11-slim", Image = "meeting-assistant-pyannote-warmup:local", @@ -218,9 +216,8 @@ public sealed class PyannoteTranscriptFinalizerTests var commandRunner = new CapturingCommandRunner(""); var finalizer = CreateFinalizer(commandRunner, token: null); - await finalizer.WarmUpAsync(new PyannoteDiarizationOptions + await finalizer.WarmUpAsync(new PyannoteRuntimeOptions { - Enabled = true, Token = "", TokenEnv = "", ModelsFolder = Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N"), "models") diff --git a/MeetingAssistant.Tests/RecordingCoordinatorTests.cs b/MeetingAssistant.Tests/RecordingCoordinatorTests.cs index 902af5c..52a8e1d 100644 --- a/MeetingAssistant.Tests/RecordingCoordinatorTests.cs +++ b/MeetingAssistant.Tests/RecordingCoordinatorTests.cs @@ -9,6 +9,7 @@ using MeetingAssistant.Workflow; using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging.Abstractions; using Microsoft.Extensions.Options; +using System.Collections.Concurrent; using System.Threading.Channels; namespace MeetingAssistant.Tests; @@ -116,6 +117,52 @@ public sealed class RecordingCoordinatorTests Assert.False(stopped.IsRecording); } + [Fact] + public async Task PauseRoutesSilenceAndUnpauseResumesCapturedAudioThroughSameRun() + { + var audioSource = new ControlledAudioSource(); + var provider = new CapturingAudioStreamingTranscriptionProvider(); + var audioArchive = new InMemoryRecordedAudioStore(); + var pipelineFactory = new TestSpeechRecognitionPipelineFactory(provider); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + pipelineFactory, + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + audioArchive, + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions()), + NullLogger.Instance); + + await coordinator.StartAsync(CancellationToken.None); + var paused = await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + await audioSource.WriteAsync(new AudioChunk([1, 2, 3, 4], 16000, 1), CancellationToken.None); + await WaitUntilAsync(() => provider.Chunks.Count >= 1); + + Assert.True(paused.IsRecording); + Assert.True(paused.IsPaused); + Assert.Equal([0, 0, 0, 0], provider.Chunks.ElementAt(0).Pcm); + Assert.Equal([0, 0, 0, 0], audioArchive.AppendedChunks[0].Pcm); + + var unpaused = await coordinator.SetTranscriptionPausedAsync(false, CancellationToken.None); + await audioSource.WriteAsync(new AudioChunk([5, 6], 16000, 1), CancellationToken.None); + await WaitUntilAsync(() => provider.Chunks.Count >= 2); + + Assert.True(unpaused.IsRecording); + Assert.False(unpaused.IsPaused); + Assert.Equal([5, 6], provider.Chunks.ElementAt(1).Pcm); + Assert.Equal([5, 6], audioArchive.AppendedChunks[1].Pcm); + Assert.Equal(1, pipelineFactory.CreateCount); + + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + var stopped = await coordinator.StopAsync(CancellationToken.None); + + Assert.False(stopped.IsRecording); + Assert.Equal(1, pipelineFactory.CreateCount); + } + [Fact] public async Task TranscriptLineWorkflowRuleTransformsLiveTranscriptAfterDurableAppend() { @@ -596,6 +643,50 @@ public sealed class RecordingCoordinatorTests noteStore.SavedNote?.Frontmatter.EndTime); } + [Fact] + public async Task InactivityPromptCanPauseItsOriginatingActiveMeeting() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T10:30:00+02:00")); + var promptService = new CapturingMeetingInactivityPromptService(MeetingInactivityPromptResponse.Pause); + var coordinator = new MeetingRecordingCoordinator( + new ControlledAudioSource(), + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + await WaitUntilAsync(() => coordinator.CurrentStatus.IsPaused); + await WaitUntilAsync(() => promptService.DismissAllCount >= 1); + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.True(promptService.DismissAllCount >= 1); + + await coordinator.StopAsync(CancellationToken.None); + } + [Fact] public async Task InactivitySafeguardAutoStopsNormallyAndUsesMeetingStartWhenNoTranscriptArrives() { @@ -749,6 +840,645 @@ public sealed class RecordingCoordinatorTests await coordinator.StopAsync(CancellationToken.None); } + [Fact] + public async Task NewTranscriptTextDismissesOutstandingInactivityPrompts() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:30:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var audioSource = new ControlledAudioSource(); + var transcriptStore = new InMemoryTranscriptStore(); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + transcriptStore, + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + + await audioSource.WriteAsync(new AudioChunk([1, 0], 16000, 1), CancellationToken.None); + await transcriptStore.WaitForTextAsync("chunk:2"); + + await WaitUntilAsync(() => promptService.DismissAllCount == 1); + Assert.Single(promptService.Requests); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task BlankTranscriptSegmentDoesNotDismissOutstandingInactivityPrompts() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:40:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var audioSource = new ControlledAudioSource(); + var transcriptStore = new InMemoryTranscriptStore(); + var provider = new SequencedStreamingTranscriptionProvider( + [ + new TranscriptionSegment(TimeSpan.Zero, TimeSpan.Zero, "Unknown", " "), + new TranscriptionSegment( + TimeSpan.Zero, + TimeSpan.Zero, + "System", + "", + TranscriptionSegmentKind.Marker, + MarkerId: "test-marker") + ]); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(provider), + transcriptStore, + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + + await audioSource.WriteAsync(new AudioChunk([1, 0], 16000, 1), CancellationToken.None); + await audioSource.WriteAsync(new AudioChunk([2, 0], 16000, 1), CancellationToken.None); + await WaitUntilAsync(() => transcriptStore.Segments.Count == 2); + + Assert.Equal(0, promptService.DismissAllCount); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task LateTranscriptFromStoppedRunDoesNotDismissCurrentRunInactivityPrompt() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:42:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var delayedProvider = new DelayedSegmentOnAudioCompletionProvider( + new TranscriptionSegment(TimeSpan.Zero, TimeSpan.FromSeconds(1), "Guest-01", "late old-run text")); + var coordinator = new MeetingRecordingCoordinator( + new ControlledAudioSource(), + new SequencedSpeechRecognitionPipelineFactory( + delayedProvider, + new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + var oldStopTask = coordinator.StopAsync(CancellationToken.None); + await delayedProvider.WaitForAudioCompletionAsync(); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount >= 2); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + Assert.Equal(0, promptService.DismissAllCount); + + delayedProvider.ReleaseSegment(); + await oldStopTask; + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.Equal(0, promptService.DismissAllCount); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task NewTranscriptTextInvalidatesAnAlreadyDequeuedInactivityAction() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:45:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var audioSource = new ControlledAudioSource(); + var transcriptStore = new InMemoryTranscriptStore(); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + transcriptStore, + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + + await audioSource.WriteAsync(new AudioChunk([1, 0], 16000, 1), CancellationToken.None); + await transcriptStore.WaitForTextAsync("chunk:2"); + await WaitUntilAsync(() => promptService.DismissAllCount == 1); + await promptService.RespondAsync(MeetingInactivityPromptResponse.Stop); + + Assert.True(coordinator.CurrentStatus.IsRecording); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task InactivityActionRemainsValidUntilNewTranscriptIsDurablyWritten() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:47:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var audioSource = new ControlledAudioSource(); + var transcriptStore = new InMemoryTranscriptStore(); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + transcriptStore, + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + + transcriptStore.BlockNextAppend(); + await audioSource.WriteAsync(new AudioChunk([1, 0], 16000, 1), CancellationToken.None); + await transcriptStore.WaitForBlockedAppendAsync(); + var responseTask = promptService.RespondAsync(MeetingInactivityPromptResponse.Stop); + await WaitUntilAsync(() => !coordinator.CurrentStatus.IsRecording || responseTask.IsCompleted); + + try + { + Assert.False(coordinator.CurrentStatus.IsRecording); + } + finally + { + transcriptStore.ReleaseBlockedAppend(); + await responseTask; + if (coordinator.CurrentStatus.IsRecording) + { + await coordinator.StopAsync(CancellationToken.None); + } + } + } + + [Fact] + public async Task PausingTranscriptionInvalidatesAnAlreadyDequeuedInactivityAction() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:50:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var coordinator = new MeetingRecordingCoordinator( + new ControlledAudioSource(), + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + await promptService.RespondAsync(MeetingInactivityPromptResponse.Stop); + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.True(coordinator.CurrentStatus.IsPaused); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task TranscriptWrittenWhilePauseActionWaitsForCoordinatorInvalidatesThatAction() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:52:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var audioSource = new ControlledAudioSource(); + var transcriptStore = new InMemoryTranscriptStore(); + var noteStore = new InMemoryMeetingNoteStore(); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + transcriptStore, + noteStore, + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForPromptAsync(); + + noteStore.BlockNextRead(); + promptService.BlockNextDismiss(); + var metadataTask = coordinator.AttachMetadataToCurrentMeetingAsync( + new MeetingMetadata("Updated meeting", [], ""), + CancellationToken.None); + await noteStore.WaitForBlockedReadAsync(); + var responseTask = promptService.RespondAsync(MeetingInactivityPromptResponse.Pause); + + await audioSource.WriteAsync(new AudioChunk([1, 0], 16000, 1), CancellationToken.None); + await transcriptStore.WaitForTextAsync("chunk:2"); + await Task.Delay(25); + noteStore.ReleaseBlockedRead(); + await promptService.WaitForBlockedDismissAsync(); + try + { + Assert.Same(responseTask, await Task.WhenAny(responseTask, Task.Delay(TimeSpan.FromSeconds(1)))); + } + finally + { + promptService.ReleaseBlockedDismiss(); + } + + await Task.WhenAll(metadataTask, responseTask); + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.False(coordinator.CurrentStatus.IsPaused); + Assert.Equal(1, promptService.DismissAllCount); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task PausingWhileAutoStopDecisionIsInFlightKeepsTheMeetingActive() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:55:00+02:00")); + var coordinator = new MeetingRecordingCoordinator( + new ControlledAudioSource(), + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.Zero, + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromSeconds(5), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.BlockNextNowRead(); + clock.Advance(TimeSpan.FromSeconds(5)); + await clock.WaitForBlockedNowReadAsync(); + + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + clock.ReleaseBlockedNowRead(); + await WaitUntilAsync(() => !coordinator.CurrentStatus.IsRecording || clock.PendingDelayCount > 0); + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.True(coordinator.CurrentStatus.IsPaused); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task PausingWhileAnInactivityPromptIsBeingShownDismissesTheLatePrompt() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T12:57:00+02:00")); + var promptService = new BlockingMeetingInactivityPromptService(); + var coordinator = new MeetingRecordingCoordinator( + new ControlledAudioSource(), + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromMinutes(30), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + clock.Advance(TimeSpan.FromSeconds(2)); + await promptService.WaitForShowStartedAsync(); + + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + promptService.ReleaseShow(); + await promptService.WaitForShowCompletedAsync(); + await promptService.WaitForLateDismissAsync(); + + Assert.False(promptService.IsVisible); + Assert.True(coordinator.CurrentStatus.IsPaused); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task PausedTranscriptionSuspendsInactivitySafeguardUntilUnpaused() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T13:00:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var audioSource = new ControlledAudioSource(); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromSeconds(5), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + + clock.Advance(TimeSpan.FromSeconds(10)); + await WaitUntilAsync(() => !coordinator.CurrentStatus.IsRecording || clock.PendingDelayCount > 0); + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.True(coordinator.CurrentStatus.IsPaused); + Assert.Empty(promptService.Requests); + + await coordinator.SetTranscriptionPausedAsync(false, CancellationToken.None); + clock.Advance(TimeSpan.FromSeconds(1)); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + Assert.Empty(promptService.Requests); + + clock.Advance(TimeSpan.FromSeconds(1)); + await promptService.WaitForPromptAsync(); + + Assert.Single(promptService.Requests); + Assert.True(coordinator.CurrentStatus.IsRecording); + + await coordinator.StopAsync(CancellationToken.None); + } + + [Fact] + public async Task PausedTranscriptionAutoStopsNormallyAfterFourHoursWithoutShowingPrompt() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T14:00:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var coordinator = new MeetingRecordingCoordinator( + new ControlledAudioSource(), + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + Enabled = false, + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromSeconds(5), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + + clock.Advance(TimeSpan.FromHours(4) - TimeSpan.FromMinutes(1)); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.Empty(promptService.Requests); + + clock.Advance(TimeSpan.FromMinutes(1)); + await WaitUntilAsync(() => !coordinator.CurrentStatus.IsRecording || clock.PendingDelayCount > 0); + + Assert.False(coordinator.CurrentStatus.IsRecording); + Assert.Empty(promptService.Requests); + } + + [Fact] + public async Task UnpauseAtMaximumPauseDeadlineKeepsMeetingActiveAndResetsPauseTimer() + { + var clock = new ManualMeetingInactivityClock(DateTimeOffset.Parse("2026-06-02T18:00:00+02:00")); + var promptService = new IgnoringMeetingInactivityPromptService(); + var coordinator = new MeetingRecordingCoordinator( + new ControlledAudioSource(), + new TestSpeechRecognitionPipelineFactory(new EchoStreamingTranscriptionProvider()), + new InMemoryTranscriptStore(), + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + Recording = + { + InactivitySafeguard = + { + FirstPromptAfter = TimeSpan.FromSeconds(2), + ReminderPromptAfter = [], + AutoStopAfter = TimeSpan.FromSeconds(3), + MaximumPauseDuration = TimeSpan.FromSeconds(5), + CheckInterval = TimeSpan.FromSeconds(1) + } + } + }), + NullLogger.Instance, + inactivityPromptService: promptService, + inactivityClock: clock); + + await coordinator.StartAsync(CancellationToken.None); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + + clock.BlockNextNowRead(); + clock.Advance(TimeSpan.FromSeconds(5)); + await clock.WaitForBlockedNowReadAsync(); + await coordinator.SetTranscriptionPausedAsync(false, CancellationToken.None); + clock.ReleaseBlockedNowRead(); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.False(coordinator.CurrentStatus.IsPaused); + + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + clock.Advance(TimeSpan.FromSeconds(4)); + await WaitUntilAsync(() => clock.PendingDelayCount > 0); + + Assert.True(coordinator.CurrentStatus.IsRecording); + Assert.Empty(promptService.Requests); + + clock.Advance(TimeSpan.FromSeconds(1)); + await WaitUntilAsync(() => !coordinator.CurrentStatus.IsRecording || clock.PendingDelayCount > 0); + + Assert.False(coordinator.CurrentStatus.IsRecording); + Assert.Empty(promptService.Requests); + } + [Fact] public async Task StartUsesCurrentOutlookMeetingMetadataWhenAvailable() { @@ -1695,12 +2425,17 @@ public sealed class RecordingCoordinatorTests await coordinator.StopAsync(CancellationToken.None); } - [Fact] - public async Task ToggleToDifferentLaunchProfileBuffersAudioCapturedWhilePreviousPipelineDrains() + [Theory] + [InlineData("default", "english")] + [InlineData("english", "default")] + public async Task ProfileSwitchDrainsFinalTranscriptAndResumesBufferedAudioWithoutBlockingNotifications( + string initialProfile, + string targetProfile) { var audioSource = new ControlledAudioSource(); var pipelineFactory = new BlockingProfileSwitchSpeechRecognitionPipelineFactory(); var transcriptStore = new InMemoryTranscriptStore(); + var promptService = new IgnoringMeetingInactivityPromptService(); var launchProfiles = CreateLaunchProfiles( Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N"), "default"), Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N"), "english")); @@ -1715,13 +2450,16 @@ public sealed class RecordingCoordinatorTests new CapturingMeetingSummaryPipeline(), Options.Create(launchProfiles.GetRequiredProfile(null).Options), NullLogger.Instance, + inactivityPromptService: promptService, launchProfiles: launchProfiles); - await coordinator.StartAsync(CancellationToken.None); + await coordinator.StartAsync(initialProfile, CancellationToken.None); await audioSource.WriteAsync(new AudioChunk([1, 0], 16000, 1), CancellationToken.None); - await transcriptStore.WaitForTextAsync("default chunk:2"); + await transcriptStore.WaitForTextAsync($"{initialProfile} chunk:2"); + await WaitUntilAsync(() => promptService.DismissAllCount == 1); - var switchTask = coordinator.ToggleAsync("english", CancellationToken.None); + using var switchCancellation = new CancellationTokenSource(TimeSpan.FromSeconds(10)); + var switchTask = coordinator.ToggleAsync(targetProfile, switchCancellation.Token); await pipelineFactory.WaitUntilFirstPipelineDrainIsBlockedAsync(); await audioSource.WriteAsync(new AudioChunk([1, 0, 2, 0, 3, 0], 16000, 1), CancellationToken.None); @@ -1729,10 +2467,21 @@ public sealed class RecordingCoordinatorTests pipelineFactory.ReleaseFirstPipelineDrain(); await switchTask.WaitAsync(TimeSpan.FromSeconds(5)); - await transcriptStore.WaitForTextAsync("english chunk:6"); + await transcriptStore.WaitForTextAsync($"{targetProfile} chunk:6"); + await WaitUntilAsync(() => promptService.DismissAllCount == 3); - Assert.Equal([null, "english"], pipelineFactory.ProfileNames); - await coordinator.StopAsync(CancellationToken.None); + Assert.Equal(targetProfile, coordinator.CurrentStatus.LaunchProfile); + Assert.Equal([initialProfile, targetProfile], pipelineFactory.ProfileNames); + var segments = transcriptStore.Segments.ToList(); + var finalResultIndex = segments.FindIndex(segment => segment.Text == "Final result emitted while switching profiles."); + var markerIndex = segments.FindIndex(segment => segment.Text.Contains($"Transcription profile changed to {targetProfile}", StringComparison.Ordinal)); + Assert.True(finalResultIndex >= 0 && finalResultIndex < markerIndex); + + await coordinator.ToggleAsync(initialProfile, CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(5)); + await audioSource.WriteAsync(new AudioChunk([1, 0, 2, 0], 16000, 1), CancellationToken.None); + await transcriptStore.WaitForTextAsync($"{initialProfile} chunk:4"); + await coordinator.StopAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(5)); + Assert.False(coordinator.CurrentStatus.IsRecording); } [Fact] @@ -2087,9 +2836,10 @@ public sealed class RecordingCoordinatorTests var transcriptStore = new InMemoryTranscriptStore(); var speakerIdentification = new BlockingSpeakerIdentificationService("Guest03", "Chris"); var provider = new OrderedChunkProvider(); + var pipelineFactory = new TestSpeechRecognitionPipelineFactory(provider); var coordinator = new MeetingRecordingCoordinator( audioSource, - new TestSpeechRecognitionPipelineFactory(provider), + pipelineFactory, transcriptStore, new InMemoryMeetingNoteStore(), new CapturingMeetingNoteOpener(), @@ -2115,6 +2865,8 @@ public sealed class RecordingCoordinatorTests await WaitUntilAsync(() => transcriptStore.ReplacedSegments.Any(segment => segment.Text?.Contains("first", StringComparison.Ordinal) == true)); Assert.Equal("Chris", transcriptStore.ReplacedSegments.Single().Speaker); await Task.Delay(50); + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + await coordinator.SetTranscriptionPausedAsync(false, CancellationToken.None); await audioSource.WriteAsync(new AudioChunk(Samples(8, 9, 10, 11, 12, 13, 14, 15), 4, 1), CancellationToken.None); await transcriptStore.WaitForTextAsync("second"); @@ -2127,6 +2879,7 @@ public sealed class RecordingCoordinatorTests Assert.Contains( transcriptStore.ReplacedSegments, segment => segment.Text?.Contains("first", StringComparison.Ordinal) == true && segment.Speaker == "Chris"); + Assert.Equal(1, pipelineFactory.CreateCount); } [Fact] @@ -2204,6 +2957,65 @@ public sealed class RecordingCoordinatorTests Assert.Equal(["Chris"], speakerIdentification.Requests.Last().MeetingNote.Frontmatter.Attendees); } + [Fact] + public async Task LiveResemblyzerMatchingRetriesWhenSameSpeakerCollectsRequiredSamples() + { + var audioSource = new ControlledAudioSource(); + var transcriptStore = new InMemoryTranscriptStore(); + var speakerIdentification = new CountingSpeakerIdentificationService(); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(new OrderedChunkProvider()), + transcriptStore, + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + SpeakerIdentification = new SpeakerIdentificationOptions + { + InitialDelay = TimeSpan.Zero, + Interval = TimeSpan.FromMilliseconds(20), + MinimumSampleSpeechDuration = TimeSpan.Zero, + Resemblyzer = new ResemblyzerSpeakerRecognitionOptions + { + Enabled = true, + RequiredVectorsPerSpeaker = 5 + } + } + }), + NullLogger.Instance, + speakerIdentificationService: speakerIdentification, + speakerSampleCollectionPolicy: SpeakerSampleCollectionPolicy.IndependentVectors(1000)); + + await coordinator.StartAsync(CancellationToken.None); + try + { + await audioSource.WriteAsync(new AudioChunk(Samples(0, 1, 2, 3, 4, 5, 6, 7), 4, 1), CancellationToken.None); + await WaitUntilAsync(() => speakerIdentification.Requests.Count == 1); + Assert.Single(speakerIdentification.Requests.Single().Samples!); + + // The provider emits overlapping two-second segments, so every second + // segment supplies another independent recognition sample. + for (var index = 1; index < 9; index++) + { + await audioSource.WriteAsync(new AudioChunk(Samples(0, 1, 2, 3, 4, 5, 6, 7), 4, 1), CancellationToken.None); + } + + await WaitUntilAsync(() => speakerIdentification.Requests.Last().Samples?.Count == 5); + var attempts = speakerIdentification.Requests.Count; + await Task.Delay(100); + Assert.Equal(attempts, speakerIdentification.Requests.Count); + Assert.All(speakerIdentification.Requests.Last().Samples!, sample => Assert.Equal("Guest03", sample.Speaker)); + } + finally + { + await coordinator.StopAsync(CancellationToken.None); + } + } + [Fact] public async Task LiveSpeakerIdentificationRetriesWhenNewUnmappedSpeakerAppears() { @@ -2967,6 +3779,9 @@ public sealed class RecordingCoordinatorTests private readonly object gate = new(); private readonly List segments = []; private readonly string transcriptPath; + private TaskCompletionSource? blockedAppendStarted; + private TaskCompletionSource? releaseBlockedAppend; + private bool blockNextAppend; public InMemoryTranscriptStore(string transcriptPath = "memory-transcript.md") { @@ -2990,8 +3805,29 @@ public sealed class RecordingCoordinatorTests return CreateSessionAsync(cancellationToken); } - public Task AppendLineAsync(TranscriptSession session, string line, CancellationToken cancellationToken) + public async Task AppendLineAsync( + TranscriptSession session, + string line, + CancellationToken cancellationToken) { + TaskCompletionSource? appendStarted = null; + TaskCompletionSource? appendRelease = null; + lock (gate) + { + if (blockNextAppend) + { + blockNextAppend = false; + appendStarted = blockedAppendStarted; + appendRelease = releaseBlockedAppend; + } + } + + if (appendRelease is not null) + { + appendStarted!.TrySetResult(); + await appendRelease.Task.WaitAsync(cancellationToken); + } + int index; lock (gate) { @@ -2999,7 +3835,33 @@ public sealed class RecordingCoordinatorTests segments.Add(ParseTranscriptLine(line)); } - return Task.FromResult(new TranscriptLineReference(session.TranscriptPath, index, line)); + return new TranscriptLineReference(session.TranscriptPath, index, line); + } + + public void BlockNextAppend() + { + lock (gate) + { + blockNextAppend = true; + blockedAppendStarted = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + releaseBlockedAppend = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + } + } + + public Task WaitForBlockedAppendAsync() + { + lock (gate) + { + return blockedAppendStarted!.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + } + + public void ReleaseBlockedAppend() + { + lock (gate) + { + releaseBlockedAppend?.TrySetResult(); + } } public Task ReplaceLineAsync( @@ -3150,6 +4012,9 @@ public sealed class RecordingCoordinatorTests private sealed class InMemoryMeetingNoteStore : IMeetingNoteStore { private readonly string notePath; + private TaskCompletionSource? blockedReadStarted; + private TaskCompletionSource? releaseBlockedRead; + private bool blockNextRead; public InMemoryMeetingNoteStore(string notePath = "memory-meeting.md") { @@ -3164,9 +4029,33 @@ public sealed class RecordingCoordinatorTests return Task.FromResult(SavedNote); } - public Task ReadAsync(string path, CancellationToken cancellationToken) + public async Task ReadAsync(string path, CancellationToken cancellationToken) { - return Task.FromResult(SavedNote ?? throw new FileNotFoundException(path)); + if (blockNextRead) + { + blockNextRead = false; + blockedReadStarted!.TrySetResult(); + await releaseBlockedRead!.Task.WaitAsync(cancellationToken); + } + + return SavedNote ?? throw new FileNotFoundException(path); + } + + public void BlockNextRead() + { + blockNextRead = true; + blockedReadStarted = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + releaseBlockedRead = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + } + + public Task WaitForBlockedReadAsync() + { + return blockedReadStarted!.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + + public void ReleaseBlockedRead() + { + releaseBlockedRead?.TrySetResult(); } public void UpdateAttendees(IEnumerable attendees) @@ -4057,6 +4946,8 @@ public sealed class RecordingCoordinatorTests public List AppendedChunkSizes { get; } = []; + public List AppendedChunks { get; } = []; + public bool Completed { get; private set; } public bool Deleted { get; private set; } @@ -4094,6 +4985,7 @@ public sealed class RecordingCoordinatorTests public Task AppendAsync(AudioChunk chunk, CancellationToken cancellationToken) { store.AppendedChunkSizes.Add(chunk.Pcm.Length); + store.AppendedChunks.Add(chunk); store.AppendObserved.TrySetResult(); return Task.CompletedTask; } @@ -4152,8 +5044,11 @@ public sealed class RecordingCoordinatorTests this.finalize = finalize ?? ((_, _, _, _) => Task.FromResult>([])); } + public int CreateCount { get; private set; } + public ISpeechRecognitionPipeline Create() { + CreateCount++; return new TestSpeechRecognitionPipeline(provider, finalize); } } @@ -4303,6 +5198,11 @@ public sealed class RecordingCoordinatorTests drainBlocked.TrySetResult(); await releaseDrain.Task.WaitAsync(cancellationToken); + yield return new TranscriptionSegment( + TimeSpan.Zero, + TimeSpan.FromSeconds(3), + "Unknown", + "Final result emitted while switching profiles."); } } @@ -4744,13 +5644,44 @@ public sealed class RecordingCoordinatorTests { private readonly object gate = new(); private readonly List delays = []; + private readonly TaskCompletionSource blockedNowReadObserved = + new(TaskCreationOptions.RunContinuationsAsynchronously); + private readonly TaskCompletionSource releaseBlockedNowRead = + new(TaskCreationOptions.RunContinuationsAsynchronously); + private DateTimeOffset now; + private bool blockNextNowRead; public ManualMeetingInactivityClock(DateTimeOffset now) { - Now = now; + this.now = now; } - public DateTimeOffset Now { get; private set; } + public DateTimeOffset Now + { + get + { + var shouldBlock = false; + lock (gate) + { + if (blockNextNowRead) + { + blockNextNowRead = false; + shouldBlock = true; + } + } + + if (shouldBlock) + { + blockedNowReadObserved.TrySetResult(); + releaseBlockedNowRead.Task.GetAwaiter().GetResult(); + } + + lock (gate) + { + return now; + } + } + } public int PendingDelayCount { @@ -4798,8 +5729,8 @@ public sealed class RecordingCoordinatorTests List due; lock (gate) { - Now += duration; - due = delays.Where(delay => delay.DueAt <= Now).ToList(); + now += duration; + due = delays.Where(delay => delay.DueAt <= now).ToList(); foreach (var delay in due) { delays.Remove(delay); @@ -4813,6 +5744,24 @@ public sealed class RecordingCoordinatorTests } } + public void BlockNextNowRead() + { + lock (gate) + { + blockNextNowRead = true; + } + } + + public Task WaitForBlockedNowReadAsync() + { + return blockedNowReadObserved.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + + public void ReleaseBlockedNowRead() + { + releaseBlockedNowRead.TrySetResult(); + } + private sealed class ScheduledDelay { public ScheduledDelay(DateTimeOffset dueAt, TaskCompletionSource completion) @@ -4842,6 +5791,8 @@ public sealed class RecordingCoordinatorTests public List Requests { get; } = []; + public int DismissAllCount { get; private set; } + public async Task ShowStopPromptAsync( MeetingInactivityPromptRequest request, Func handleResponseAsync, @@ -4859,21 +5810,50 @@ public sealed class RecordingCoordinatorTests { return promptObserved.Task.WaitAsync(TimeSpan.FromSeconds(5)); } + + public Task DismissAllAsync(CancellationToken cancellationToken) + { + DismissAllCount++; + return Task.CompletedTask; + } + } + + private sealed class SequencedSpeechRecognitionPipelineFactory : ISpeechRecognitionPipelineFactory + { + private readonly Queue providers; + + public SequencedSpeechRecognitionPipelineFactory(params IStreamingTranscriptionProvider[] providers) + { + this.providers = new Queue(providers); + } + + public ISpeechRecognitionPipeline Create() + { + return new TestSpeechRecognitionPipeline( + providers.Dequeue(), + (_, _, _, _) => Task.FromResult>([])); + } } private sealed class IgnoringMeetingInactivityPromptService : IMeetingInactivityPromptService { private readonly TaskCompletionSource promptObserved = new(TaskCreationOptions.RunContinuationsAsynchronously); + private TaskCompletionSource? blockedDismissObserved; + private TaskCompletionSource? releaseBlockedDismiss; + private Func? handleResponseAsync; public List Requests { get; } = []; + public int DismissAllCount { get; private set; } + public Task ShowStopPromptAsync( MeetingInactivityPromptRequest request, Func handleResponseAsync, CancellationToken cancellationToken) { Requests.Add(request); + this.handleResponseAsync = handleResponseAsync; promptObserved.TrySetResult(); return Task.CompletedTask; } @@ -4882,6 +5862,98 @@ public sealed class RecordingCoordinatorTests { return promptObserved.Task.WaitAsync(TimeSpan.FromSeconds(5)); } + + public async Task DismissAllAsync(CancellationToken cancellationToken) + { + DismissAllCount++; + if (blockedDismissObserved is not null && releaseBlockedDismiss is not null) + { + blockedDismissObserved.TrySetResult(); + await releaseBlockedDismiss.Task.WaitAsync(cancellationToken); + blockedDismissObserved = null; + releaseBlockedDismiss = null; + } + } + + public void BlockNextDismiss() + { + blockedDismissObserved = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + releaseBlockedDismiss = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + } + + public Task WaitForBlockedDismissAsync() + { + return blockedDismissObserved!.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + + public void ReleaseBlockedDismiss() + { + releaseBlockedDismiss!.TrySetResult(); + } + + public Task RespondAsync(MeetingInactivityPromptResponse response) + { + return handleResponseAsync is null + ? Task.CompletedTask + : handleResponseAsync(response, CancellationToken.None); + } + } + + private sealed class BlockingMeetingInactivityPromptService : IMeetingInactivityPromptService + { + private readonly TaskCompletionSource showStarted = + new(TaskCreationOptions.RunContinuationsAsynchronously); + private readonly TaskCompletionSource releaseShow = + new(TaskCreationOptions.RunContinuationsAsynchronously); + private readonly TaskCompletionSource showCompleted = + new(TaskCreationOptions.RunContinuationsAsynchronously); + private readonly TaskCompletionSource lateDismissed = + new(TaskCreationOptions.RunContinuationsAsynchronously); + private int dismissCount; + + public bool IsVisible { get; private set; } + + public async Task ShowStopPromptAsync( + MeetingInactivityPromptRequest request, + Func handleResponseAsync, + CancellationToken cancellationToken) + { + showStarted.TrySetResult(); + await releaseShow.Task.WaitAsync(cancellationToken); + IsVisible = true; + showCompleted.TrySetResult(); + } + + public Task DismissAllAsync(CancellationToken cancellationToken) + { + IsVisible = false; + if (Interlocked.Increment(ref dismissCount) >= 2) + { + lateDismissed.TrySetResult(); + } + + return Task.CompletedTask; + } + + public Task WaitForShowStartedAsync() + { + return showStarted.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + + public Task WaitForShowCompletedAsync() + { + return showCompleted.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + + public Task WaitForLateDismissAsync() + { + return lateDismissed.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + + public void ReleaseShow() + { + releaseShow.TrySetResult(); + } } private sealed class CapturedChunkThenCancelAudioSource : IMeetingAudioSource @@ -4937,6 +6009,31 @@ public sealed class RecordingCoordinatorTests } } + private sealed class SequencedStreamingTranscriptionProvider : IStreamingTranscriptionProvider + { + private readonly IReadOnlyList segments; + + public SequencedStreamingTranscriptionProvider(IReadOnlyList segments) + { + this.segments = segments; + } + + public async IAsyncEnumerable TranscribeAsync( + IAsyncEnumerable audio, + SpeechRecognitionPipelineOptions options, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken cancellationToken) + { + var index = 0; + await foreach (var _ in audio.WithCancellation(cancellationToken)) + { + if (index < segments.Count) + { + yield return segments[index++]; + } + } + } + } + private sealed class EchoStreamingTranscriptionProvider : IStreamingTranscriptionProvider { public bool FirstChunkWasObservedBeforeSourceCompleted { get; private set; } @@ -4954,6 +6051,62 @@ public sealed class RecordingCoordinatorTests } } + private sealed class CapturingAudioStreamingTranscriptionProvider : IStreamingTranscriptionProvider + { + public ConcurrentQueue Chunks { get; } = new(); + + public async IAsyncEnumerable TranscribeAsync( + IAsyncEnumerable audio, + SpeechRecognitionPipelineOptions options, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken cancellationToken) + { + await foreach (var chunk in audio.WithCancellation(cancellationToken)) + { + Chunks.Enqueue(chunk); + } + + yield break; + } + } + + private sealed class DelayedSegmentOnAudioCompletionProvider : IStreamingTranscriptionProvider + { + private readonly TranscriptionSegment segment; + private readonly TaskCompletionSource audioCompleted = + new(TaskCreationOptions.RunContinuationsAsynchronously); + private readonly TaskCompletionSource releaseSegment = + new(TaskCreationOptions.RunContinuationsAsynchronously); + + public DelayedSegmentOnAudioCompletionProvider(TranscriptionSegment segment) + { + this.segment = segment; + } + + public async IAsyncEnumerable TranscribeAsync( + IAsyncEnumerable audio, + SpeechRecognitionPipelineOptions options, + [System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken cancellationToken) + { + await foreach (var _ in audio.WithCancellation(cancellationToken)) + { + } + + audioCompleted.TrySetResult(); + await releaseSegment.Task.WaitAsync(cancellationToken); + yield return segment; + } + + public Task WaitForAudioCompletionAsync() + { + return audioCompleted.Task.WaitAsync(TimeSpan.FromSeconds(5)); + } + + public void ReleaseSegment() + { + releaseSegment.TrySetResult(); + } + } + private sealed class FinalSegmentOnAudioCompletionProvider : IStreamingTranscriptionProvider { public async IAsyncEnumerable TranscribeAsync( diff --git a/MeetingAssistant.Tests/ResemblyzerSpeakerIdentificationServiceTests.cs b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentificationServiceTests.cs new file mode 100644 index 0000000..c81bba7 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentificationServiceTests.cs @@ -0,0 +1,596 @@ +using MeetingAssistant; +using MeetingAssistant.MeetingNotes; +using MeetingAssistant.Speakers; +using MeetingAssistant.Transcription; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerSpeakerIdentificationServiceTests +{ + [Fact] + public async Task LiveMatchRelabelsSpeakerAndStoresFiveVectorsWithoutWavSnippets() + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync("Chris", [UnitVector(0), UnitVector(0, 20, 0.02f)]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + var result = await service.IdentifyKnownSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 5), + CancellationToken.None); + + Assert.Equal("Chris", result.Segments.Single().Speaker); + Assert.Equal("Chris", result.SpeakerMappings["Guest03"]); + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal(7, saved.VoiceVectors.Count); + Assert.Empty(saved.Snippets); + Assert.Single(saved.References); + Assert.Single(fixture.Encoder.Requests); + Assert.Equal(5, fixture.Encoder.Requests[0].Count); + } + + [Fact] + public async Task SummaryOverrideCreatesNamedIdentityFromThreeAvailableVectors() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 3) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + var request = fixture.CreateRequest("Guest-01", sampleCount: 3); + + await service.ApplySpeakerOverrideAsync( + request, + "Guest-01", + "Sabrina", + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal("Sabrina", saved.CanonicalName); + Assert.Equal(3, saved.VoiceVectors.Count); + Assert.Empty(saved.Snippets); + Assert.Single(saved.References); + Assert.Equal(3, fixture.Encoder.Requests.Single().Count); + } + + [Fact] + public async Task FinalProcessingLearnsUnmatchedSpeakerFromFiveVectorsAndAttendees() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 5, attendees: ["John", "Mike"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Null(saved.CanonicalName); + Assert.Equal(["John", "Mike"], saved.CandidateNames.Select(candidate => candidate.Name).Order()); + Assert.Equal(5, saved.VoiceVectors.Count); + Assert.Empty(saved.Snippets); + Assert.Single(saved.References); + } + + [Fact] + public async Task FiveVectorThresholdDoesNotCapQualifyingCurrentRunEvidence() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 8) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 8, attendees: ["John", "Mike"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal(8, saved.VoiceVectors.Count); + Assert.Equal(8, fixture.Encoder.Requests.Single().Count); + } + + [Fact] + public async Task FinalProcessingPrunesForeignVectorsFromNewIdentityAtMinimum() + { + await using var fixture = await Fixture.CreateAsync(options => + { + options.OutlierPruningMinimumVectors = 20; + options.OutlierPruningNeighborSimilarity = 0.90; + options.OutlierPruningMinimumNeighbors = 3; + options.OutlierPruningMinimumClusterRatio = 0.60; + }); + fixture.Encoder.Vectors = Enumerable.Range(0, 16) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 20, attendees: ["John", "Mike"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal(16, saved.VoiceVectors.Count); + Assert.All( + saved.VoiceVectors, + stored => Assert.True(SpeakerVoiceVectors.Decode(stored)[0] > 0.9f)); + } + + [Theory] + [InlineData(false)] + [InlineData(true)] + public async Task FinalProcessingStoresVectorsCollectedAfterLiveMatch(bool transcriptAlreadyRelabeled) + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync( + "Chris", + [UnitVector(0), UnitVector(0, 20, 0.02f)]); + var currentRunVectors = Enumerable.Range(0, 8) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + fixture.Encoder.Vectors = currentRunVectors.Take(5).ToList(); + var service = fixture.CreateService(); + + var liveResult = await service.IdentifyKnownSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 5), + CancellationToken.None); + fixture.Encoder.Vectors = currentRunVectors; + var finalRequest = fixture.CreateRequest("Guest03", sampleCount: 8) with + { + KnownSpeakerMappings = liveResult.SpeakerMappings + }; + if (transcriptAlreadyRelabeled) + { + finalRequest = finalRequest with { Segments = liveResult.Segments }; + } + + await service.ProcessFinishedTranscriptAsync(finalRequest, CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal(10, saved.VoiceVectors.Count); + Assert.Equal([5, 8], fixture.Encoder.Requests.Select(request => request.Count)); + } + + [Fact] + public async Task RepeatedMatchDeduplicatesVectorsAndHonorsConfiguredLimit() + { + await using var fixture = await Fixture.CreateAsync(options => options.MaxVectorsPerIdentity = 6); + var identity = await fixture.AddIdentityAsync("Chris", [UnitVector(0), UnitVector(0, 20, 0.02f)]); + fixture.Encoder.Vectors = + [ + UnitVector(0), + UnitVector(0, 2, 0.01f), + UnitVector(0, 3, 0.02f), + UnitVector(0, 4, 0.03f), + UnitVector(0, 5, 0.04f) + ]; + var service = fixture.CreateService(); + var request = fixture.CreateRequest("Guest03", sampleCount: 5); + + await service.IdentifyKnownSpeakersAsync(request, CancellationToken.None); + await service.IdentifyKnownSpeakersAsync(request, CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal(6, saved.VoiceVectors.Count); + Assert.Single(saved.References); + } + + [Fact] + public async Task FinishedMatchingExtractsFiveNonOverlappingSamplesWhenLiveSamplesAreMissing() + { + await using var fixture = await Fixture.CreateAsync(); + await fixture.AddIdentityAsync("Chris", [UnitVector(0), UnitVector(0, 20, 0.02f)]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + var result = await service.IdentifyFinishedSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 0, segmentCount: 5), + CancellationToken.None); + + Assert.Equal("Chris", result.Segments[0].Speaker); + Assert.Equal(5, fixture.SnippetExtractor.Requests.Count); + Assert.All( + fixture.SnippetExtractor.Requests.Zip(fixture.SnippetExtractor.Requests.Skip(1)), + pair => Assert.True(pair.First[^1].End <= pair.Second[0].Start)); + Assert.Equal(5, fixture.Encoder.Requests.Single().Count); + } + + [Fact] + public async Task FinalMatchPromotesTheRemainingCandidateAndAuditsTheTranscript() + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync( + null, + [UnitVector(0), UnitVector(0, 20, 0.02f)], + candidates: ["John", "Mike"]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + var result = await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest03", sampleCount: 5, attendees: ["Jane", "John", "Chris"]), + CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal("John", saved.CanonicalName); + Assert.Equal(["John"], saved.CandidateNames.Select(candidate => candidate.Name)); + Assert.Equal("John", result.Segments.Single().Speaker); + Assert.All(saved.References, reference => + Assert.Contains("Guest03 was identified as John", File.ReadAllText(reference.TranscriptPath))); + } + + [Fact] + public async Task FinalMatchResetsCandidatesWhenAttendeesDoNotIntersect() + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync( + null, + [UnitVector(0), UnitVector(0, 20, 0.02f)], + candidates: ["John", "Mike"]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest03", sampleCount: 5, attendees: ["Jane", "Chris"]), + CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Null(saved.CanonicalName); + Assert.Equal(["Chris", "Jane"], saved.CandidateNames.Select(candidate => candidate.Name).Order()); + } + + [Fact] + public async Task FinalUnmatchedSpeakerWithOneCandidateIsAuditedWhenLearned() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 5, attendees: ["Manuel"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal("Manuel", saved.CanonicalName); + Assert.Contains("Guest-01 was identified as Manuel", File.ReadAllText(saved.References.Single().TranscriptPath)); + } + + [Fact] + public async Task CandidateLimitCountsOnlyIdentitiesWithCompatibleVectors() + { + await using var fixture = await Fixture.CreateAsync( + configureSpeaker: options => options.MaxMatchCandidates = 1); + await fixture.AddIdentityAsync("Legacy WAV identity", []); + await fixture.AddIdentityAsync("Chris", [UnitVector(0)]); + fixture.Encoder.Vectors = Enumerable.Repeat(UnitVector(0), 5).ToList(); + var service = fixture.CreateService(); + + var result = await service.IdentifyKnownSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 5, attendees: []), + CancellationToken.None); + + Assert.Equal("Chris", result.Segments.Single().Speaker); + } + + [Fact] + public async Task SummaryOverrideMergePreservesAllEvidenceFromCurrentRunCandidate() + { + await using var fixture = await Fixture.CreateAsync(); + var target = await fixture.AddIdentityAsync("Sabrina", [UnitVector(0)]); + var request = fixture.CreateRequest("Guest-01", sampleCount: 3, attendees: ["Sabrina"]); + await fixture.AddCurrentRunCandidateAsync(request, "Sabrina"); + fixture.Encoder.Vectors = Enumerable.Range(0, 3) + .Select(index => UnitVector(0, index + 2, 0.02f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ApplySpeakerOverrideAsync( + request, + "Guest-01", + "Sabrina", + CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(target.Id); + Assert.Single(saved.Snippets); + Assert.Equal(5, saved.VoiceVectors.Count); + Assert.Contains(saved.VoiceVectors, vector => vector.ModelId == "older-model"); + await using var context = new TestDbContextFactory(fixture.DatabasePath).CreateDbContext(); + Assert.Single(await context.SpeakerIdentities.ToListAsync()); + } + + private static float[] UnitVector( + int primaryDimension, + int? secondaryDimension = null, + float secondaryValue = 0) + { + var vector = new float[256]; + vector[primaryDimension] = 1; + if (secondaryDimension is { } dimension) + { + vector[dimension] = secondaryValue; + } + + return vector; + } + + private static byte[] ToBytes(float[] vector) + { + var bytes = new byte[vector.Length * sizeof(float)]; + Buffer.BlockCopy(vector, 0, bytes, 0, bytes.Length); + return bytes; + } + + private sealed class Fixture : IAsyncDisposable + { + private readonly string directory; + private readonly string databasePath; + private readonly SpeakerIdentificationOptions speakerOptions; + + private Fixture(string directory, string databasePath, SpeakerIdentificationOptions speakerOptions) + { + this.directory = directory; + this.databasePath = databasePath; + this.speakerOptions = speakerOptions; + } + + public FakeEncoder Encoder { get; } = new(); + + public FakeSnippetExtractor SnippetExtractor { get; } = new(); + + public string DatabasePath => databasePath; + + public static async Task CreateAsync( + Action? configure = null, + Action? configureSpeaker = null) + { + var directory = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + var databasePath = Path.Combine(directory, "speaker-identities.db"); + var speakerOptions = new SpeakerIdentificationOptions + { + DatabasePath = databasePath, + MatchBatchSize = 6, + MaxMatchCandidates = 100, + MatchIdentityActiveAge = TimeSpan.FromDays(365), + MinimumSampleSpeechDuration = TimeSpan.Zero, + Resemblyzer = new ResemblyzerSpeakerRecognitionOptions + { + Enabled = true, + RequiredVectorsPerSpeaker = 5, + MaxVectorsPerIdentity = 1000, + MinimumClusterCohesion = 0.75, + MinimumIdentitySimilarity = 0.75, + MinimumSimilarityMargin = 0.05, + ModelId = "resemblyzer-0.1.4-pretrained" + } + }; + configure?.Invoke(speakerOptions.Resemblyzer); + configureSpeaker?.Invoke(speakerOptions); + await using var context = new SpeakerIdentityDbContext( + new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath};Pooling=False") + .Options); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + return new Fixture(directory, databasePath, speakerOptions); + } + + public ResemblyzerSpeakerIdentificationService CreateService() + { + var appOptions = new MeetingAssistantOptions { SpeakerIdentification = speakerOptions }; + return new ResemblyzerSpeakerIdentificationService( + new TestDbContextFactory(databasePath), + SnippetExtractor, + Encoder, + new ResemblyzerVoiceClusterMatcher( + speakerOptions.Resemblyzer, + NullLogger.Instance), + new ResemblyzerVoiceVectorOutlierPruner( + speakerOptions.Resemblyzer, + NullLogger.Instance), + Options.Create(appOptions), + NullLogger.Instance); + } + + public SpeakerIdentificationRequest CreateRequest( + string speaker, + int sampleCount, + IReadOnlyList? attendees = null, + int segmentCount = 1) + { + var transcriptPath = Path.Combine(directory, "transcript.md"); + File.WriteAllText(transcriptPath, "Transcript"); + var segments = Enumerable.Range(0, segmentCount) + .Select(index => new TranscriptionSegment( + TimeSpan.FromSeconds(index * 30), + TimeSpan.FromSeconds((index + 1) * 30), + speaker, + "enough useful words for a speaker sample")) + .ToList(); + var segment = segments[0]; + return new SpeakerIdentificationRequest( + Path.Combine(directory, "meeting.wav"), + new MeetingNote( + Path.Combine(directory, "meeting.md"), + new MeetingNoteFrontmatter + { + Title = "Test", + Attendees = attendees?.ToList() ?? ["Chris"], + Transcript = transcriptPath, + AssistantContext = Path.Combine(directory, "context.md"), + Summary = Path.Combine(directory, "summary.md") + }, + ""), + segments, + Enumerable.Range(0, sampleCount) + .Select(index => new SpeakerAudioSample(speaker, segment, [(byte)(index + 1)], 100 - index)) + .ToList()); + } + + public async Task AddIdentityAsync( + string? name, + IReadOnlyList vectors, + IReadOnlyList? candidates = null, + IReadOnlyList? aliases = null) + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + var now = DateTimeOffset.UtcNow; + var identity = new SpeakerIdentity + { + CanonicalName = name, + CreatedAt = now, + UpdatedAt = now, + CandidateNames = candidates?.Select(candidate => new SpeakerCandidateName { Name = candidate }).ToList() ?? [], + Aliases = aliases?.Select(alias => new SpeakerAlias { Name = alias }).ToList() ?? [], + VoiceVectors = vectors.Select((vector, index) => new SpeakerVoiceVector + { + ModelId = speakerOptions.Resemblyzer.ModelId, + Dimensions = 256, + VectorBytes = ToBytes(vector), + Fingerprint = $"known-{index}", + CreatedAt = now.AddMinutes(index) + }).ToList() + }; + context.SpeakerIdentities.Add(identity); + await context.SaveChangesAsync(); + return identity; + } + + public async Task AddCurrentRunCandidateAsync( + SpeakerIdentificationRequest request, + string candidateName) + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + var now = DateTimeOffset.UtcNow; + context.SpeakerIdentities.Add(new SpeakerIdentity + { + CreatedAt = now, + UpdatedAt = now, + CandidateNames = [new SpeakerCandidateName { Name = candidateName }], + Snippets = [new SpeakerSnippet { WavBytes = [9, 8, 7], CreatedAt = now }], + VoiceVectors = + [ + new SpeakerVoiceVector + { + ModelId = "older-model", + Dimensions = 256, + VectorBytes = ToBytes(UnitVector(1)), + Fingerprint = "older-model-vector", + CreatedAt = now + } + ], + References = + [ + SpeakerIdentityReferences.Create( + request.MeetingNote.Path, + request.MeetingNote.Frontmatter.Transcript, + now) + ] + }); + await context.SaveChangesAsync(); + } + + public async Task LoadIdentityAsync(int id) + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + return await context.SpeakerIdentities + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.Aliases) + .Include(identity => identity.References) + .SingleAsync(identity => identity.Id == id); + } + + public async Task LoadOnlyIdentityAsync() + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + return await context.SpeakerIdentities + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.References) + .SingleAsync(); + } + + public ValueTask DisposeAsync() + { + if (Directory.Exists(directory)) + { + Directory.Delete(directory, recursive: true); + } + + return ValueTask.CompletedTask; + } + } + + private sealed class FakeEncoder : IResemblyzerVoiceEncoder + { + public IReadOnlyList Vectors { get; set; } = []; + + public List> Requests { get; } = []; + + public Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + Requests.Add(wavSamples.Select(sample => sample.ToArray()).ToList()); + return Task.FromResult>(Vectors.Take(wavSamples.Count).ToList()); + } + + public Task WarmUpAsync(CancellationToken cancellationToken) + { + return Task.CompletedTask; + } + } + + private sealed class FakeSnippetExtractor : ISpeakerSnippetExtractor + { + public List> Requests { get; } = []; + + public Task ExtractSnippetAsync( + string audioPath, + IReadOnlyList speakerSegments, + CancellationToken cancellationToken) + { + Requests.Add(speakerSegments.ToList()); + return Task.FromResult([checked((byte)Requests.Count)]); + } + } + + private sealed class TestDbContextFactory : IDbContextFactory + { + private readonly string databasePath; + + public TestDbContextFactory(string databasePath) + { + this.databasePath = databasePath; + } + + public SpeakerIdentityDbContext CreateDbContext() + { + return new SpeakerIdentityDbContext( + new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath};Pooling=False") + .Options); + } + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerSpeakerIdentityMergeServiceTests.cs b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentityMergeServiceTests.cs new file mode 100644 index 0000000..5711692 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentityMergeServiceTests.cs @@ -0,0 +1,174 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerSpeakerIdentityMergeServiceTests +{ + [Fact] + public async Task MergeRecentIdentitiesRequiresTwoDisjointMatchingVectorClusters() + { + var directory = Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + try + { + var databasePath = Path.Combine(directory, "identities.db"); + var factory = new TestDbContextFactory(databasePath); + await using (var context = factory.CreateDbContext()) + { + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + context.SpeakerIdentities.Add(CreateIdentity( + "Chris", + DateTimeOffset.UtcNow.AddMonths(-2), + Enumerable.Range(0, 10) + .Select(index => UnitVector(0, index + 20, 0.02f)) + .ToList(), + directory)); + context.SpeakerIdentities.Add(CreateIdentity( + "Chris duplicate", + DateTimeOffset.UtcNow.AddDays(-1), + Enumerable.Range(0, 10) + .Select(index => UnitVector(0, index + 2, 0.02f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 40, 0.02f))) + .ToList(), + directory)); + context.SpeakerIdentities.Add(CreateIdentity( + "Decoy", + DateTimeOffset.UtcNow.AddMonths(-1), + [UnitVector(0)], + directory)); + await context.SaveChangesAsync(); + } + + var resemblyzerOptions = new ResemblyzerSpeakerRecognitionOptions + { + Enabled = true, + RequiredVectorsPerSpeaker = 5, + MaxVectorsPerIdentity = 1000, + MinimumClusterCohesion = 0.75, + MinimumIdentitySimilarity = 0.75, + MinimumSimilarityMargin = 0.05, + ModelId = "resemblyzer-0.1.4-pretrained" + }; + var service = new ResemblyzerSpeakerIdentityMergeService( + factory, + new ResemblyzerVoiceClusterMatcher( + resemblyzerOptions, + NullLogger.Instance), + new ResemblyzerVoiceVectorOutlierPruner( + resemblyzerOptions, + NullLogger.Instance), + Options.Create(new MeetingAssistantOptions + { + SpeakerIdentification = new SpeakerIdentificationOptions + { + MergeRecentIdentityAge = TimeSpan.FromDays(14), + MaxMatchCandidates = 1, + MaxSnippetsPerSpeaker = 3, + Resemblyzer = resemblyzerOptions + } + }), + NullLogger.Instance); + + var result = await service.MergeRecentIdentitiesAsync(TimeSpan.FromDays(14), CancellationToken.None); + + Assert.Equal(2, result.MatchAttempts); + Assert.Equal(1, result.MergedPairs); + await using var verification = factory.CreateDbContext(); + var saved = await verification.SpeakerIdentities + .Include(identity => identity.Aliases) + .Include(identity => identity.VoiceVectors) + .SingleAsync(identity => identity.CanonicalName == "Chris"); + Assert.Equal(2, await verification.SpeakerIdentities.CountAsync()); + Assert.Equal("Chris", saved.CanonicalName); + Assert.Contains(saved.Aliases, alias => alias.Name == "Chris duplicate"); + Assert.Equal(20, saved.VoiceVectors.Count); + Assert.All(saved.VoiceVectors, vector => Assert.True(ToVector(vector.VectorBytes)[0] > 0.9f)); + } + finally + { + Directory.Delete(directory, recursive: true); + } + } + + private static SpeakerIdentity CreateIdentity( + string name, + DateTimeOffset createdAt, + IReadOnlyList vectors, + string directory) + { + var transcriptPath = Path.Combine(directory, $"{Guid.NewGuid():N}.md"); + File.WriteAllText(transcriptPath, "Transcript"); + return new SpeakerIdentity + { + CanonicalName = name, + CreatedAt = createdAt, + UpdatedAt = createdAt, + VoiceVectors = vectors.Select((vector, index) => new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = ToBytes(vector), + Fingerprint = $"{name}-{index}", + CreatedAt = createdAt.AddMinutes(index) + }).ToList(), + References = + [ + new SpeakerIdentityReference + { + MeetingNotePath = Path.Combine(directory, $"{Guid.NewGuid():N}.md"), + TranscriptPath = transcriptPath, + CreatedAt = createdAt + } + ] + }; + } + + private static float[] UnitVector(int primary, int? secondary = null, float secondaryValue = 0) + { + var vector = new float[256]; + vector[primary] = 1; + if (secondary is { } index) + { + vector[index] = secondaryValue; + } + + return vector; + } + + private static byte[] ToBytes(float[] vector) + { + var bytes = new byte[vector.Length * sizeof(float)]; + Buffer.BlockCopy(vector, 0, bytes, 0, bytes.Length); + return bytes; + } + + private static float[] ToVector(byte[] bytes) + { + var vector = new float[bytes.Length / sizeof(float)]; + Buffer.BlockCopy(bytes, 0, vector, 0, bytes.Length); + return vector; + } + + private sealed class TestDbContextFactory : IDbContextFactory + { + private readonly string databasePath; + + public TestDbContextFactory(string databasePath) + { + this.databasePath = databasePath; + } + + public SpeakerIdentityDbContext CreateDbContext() + { + return new SpeakerIdentityDbContext( + new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath};Pooling=False") + .Options); + } + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerVoiceClusterMatcherTests.cs b/MeetingAssistant.Tests/ResemblyzerVoiceClusterMatcherTests.cs new file mode 100644 index 0000000..19f93f3 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerVoiceClusterMatcherTests.cs @@ -0,0 +1,122 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerVoiceClusterMatcherTests +{ + [Fact] + public void CoherentSimilarClusterSelectsUnambiguousIdentity() + { + var matcher = CreateMatcher(); + var query = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, secondaryDimension: index + 2, secondaryValue: 0.05f)) + .ToList(); + var candidates = new[] + { + new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0), UnitVector(0, 10, 0.02f)]), + new ResemblyzerVoiceVectorCandidate(77, [UnitVector(1), UnitVector(1, 11, 0.02f)]) + }; + + var result = matcher.Match(query, candidates); + + Assert.True(result.Accepted); + Assert.Equal(42, result.IdentityId); + Assert.True(result.Cohesion >= 0.99); + Assert.True(result.BestSimilarity >= 0.99); + Assert.True(result.RunnerUpSimilarity < 0.1); + } + + [Fact] + public void FourVectorsDoNotTriggerAutomaticMatching() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 4).ToList(), + [new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)])]); + + Assert.False(result.Accepted); + Assert.Contains("4/5", result.Reason); + } + + [Fact] + public void IncoherentClusterIsRejectedBeforeIdentityScoring() + { + var result = CreateMatcher().Match( + Enumerable.Range(0, 5).Select(index => UnitVector(index)).ToList(), + [new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)])]); + + Assert.False(result.Accepted); + Assert.Contains("cohesion", result.Reason); + Assert.Null(result.BestSimilarity); + } + + [Fact] + public void SimilarityBelowThresholdIsRejected() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 5).ToList(), + [new ResemblyzerVoiceVectorCandidate(42, [UnitVector(1)])]); + + Assert.False(result.Accepted); + Assert.Contains("best similarity", result.Reason); + Assert.Equal(0, result.BestSimilarity); + } + + [Fact] + public void SimilarCandidatesWithinMarginAreRejected() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 5).ToList(), + [ + new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)]), + new ResemblyzerVoiceVectorCandidate(77, [UnitVector(0, 1, 0.01f)]) + ]); + + Assert.False(result.Accepted); + Assert.Contains("margin", result.Reason); + Assert.NotNull(result.RunnerUpSimilarity); + } + + [Fact] + public void InvalidStoredCandidateDoesNotAbortScoringOtherIdentities() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 5).ToList(), + [ + new ResemblyzerVoiceVectorCandidate(13, [new float[256]]), + new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)]) + ]); + + Assert.True(result.Accepted); + Assert.Equal(42, result.IdentityId); + } + + private static ResemblyzerVoiceClusterMatcher CreateMatcher() + { + return new ResemblyzerVoiceClusterMatcher( + new ResemblyzerSpeakerRecognitionOptions + { + RequiredVectorsPerSpeaker = 5, + MinimumClusterCohesion = 0.75, + MinimumIdentitySimilarity = 0.75, + MinimumSimilarityMargin = 0.05 + }, + NullLogger.Instance); + } + + private static float[] UnitVector( + int primaryDimension, + int? secondaryDimension = null, + float secondaryValue = 0) + { + var vector = new float[256]; + vector[primaryDimension] = 1; + if (secondaryDimension is { } dimension) + { + vector[dimension] = secondaryValue; + } + + return vector; + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerVoiceVectorOutlierPrunerTests.cs b/MeetingAssistant.Tests/ResemblyzerVoiceVectorOutlierPrunerTests.cs new file mode 100644 index 0000000..050159f --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerVoiceVectorOutlierPrunerTests.cs @@ -0,0 +1,141 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerVoiceVectorOutlierPrunerTests +{ + [Fact] + public void DominantDensityClusterRemovesForeignSpeakerVectorsAtThreshold() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 16) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(4, result); + Assert.Equal(16, identity.VoiceVectors.Count); + Assert.All( + identity.VoiceVectors, + stored => Assert.True(SpeakerVoiceVectors.Decode(stored)[0] > 0.9f)); + } + + [Fact] + public void AmbiguousDenseClustersPreserveAllVectors() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 10) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 10) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(0, result); + Assert.Equal(20, identity.VoiceVectors.Count); + } + + [Fact] + public void BelowMinimumVectorCountPreservesAllVectorsWithoutEvaluation() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 15) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(0, result); + Assert.Equal(19, identity.VoiceVectors.Count); + } + + [Fact] + public void PruningPreservesOtherModelsAndMalformedStoredRows() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 16) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + SpeakerVoiceVectors.AddDistinct( + identity, + [UnitVector(2, 60, 0.04f)], + "older-model", + 1000, + DateTimeOffset.UtcNow); + identity.VoiceVectors.Add(new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = [1], + Fingerprint = "malformed", + CreatedAt = DateTimeOffset.UtcNow + }); + identity.VoiceVectors.Add(new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = "zero-magnitude", + CreatedAt = DateTimeOffset.UtcNow + }); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(4, result); + Assert.Equal(19, identity.VoiceVectors.Count); + Assert.Contains(identity.VoiceVectors, vector => vector.ModelId == "older-model"); + Assert.Contains(identity.VoiceVectors, vector => vector.Fingerprint == "malformed"); + Assert.Contains(identity.VoiceVectors, vector => vector.Fingerprint == "zero-magnitude"); + } + + private static ResemblyzerVoiceVectorOutlierPruner CreatePruner() + { + return new ResemblyzerVoiceVectorOutlierPruner( + new ResemblyzerSpeakerRecognitionOptions + { + ModelId = "resemblyzer-0.1.4-pretrained", + OutlierPruningMinimumVectors = 20, + OutlierPruningNeighborSimilarity = 0.90, + OutlierPruningMinimumNeighbors = 3, + OutlierPruningMinimumClusterRatio = 0.60 + }, + NullLogger.Instance); + } + + private static SpeakerIdentity IdentityWithVectors(IReadOnlyList vectors) + { + var identity = new SpeakerIdentity(); + SpeakerVoiceVectors.AddDistinct( + identity, + vectors, + "resemblyzer-0.1.4-pretrained", + 1000, + DateTimeOffset.UtcNow); + return identity; + } + + private static float[] UnitVector( + int primaryDimension, + int secondaryDimension, + float secondaryValue) + { + var vector = new float[256]; + vector[primaryDimension] = 1; + vector[secondaryDimension] = secondaryValue; + return vector; + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerWarmupHostedServiceTests.cs b/MeetingAssistant.Tests/ResemblyzerWarmupHostedServiceTests.cs new file mode 100644 index 0000000..c61ed78 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerWarmupHostedServiceTests.cs @@ -0,0 +1,78 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerWarmupHostedServiceTests +{ + [Fact] + public async Task EnabledWarmupStartsWithoutBlockingApplicationStartup() + { + var encoder = new BlockingEncoder(); + var options = new MeetingAssistantOptions(); + options.SpeakerIdentification.Resemblyzer.Enabled = true; + var service = new ResemblyzerWarmupHostedService( + encoder, + Options.Create(options), + NullLogger.Instance); + + await service.StartAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(1)); + await encoder.WaitForWarmupAsync(); + + await service.StopAsync(CancellationToken.None); + + Assert.True(encoder.CancellationObserved); + } + + [Fact] + public async Task DisabledWarmupDoesNotInvokeEncoder() + { + var encoder = new BlockingEncoder(); + var service = new ResemblyzerWarmupHostedService( + encoder, + Options.Create(new MeetingAssistantOptions()), + NullLogger.Instance); + + await service.StartAsync(CancellationToken.None); + await service.StopAsync(CancellationToken.None); + + Assert.False(encoder.WarmupStarted); + } + + private sealed class BlockingEncoder : IResemblyzerVoiceEncoder + { + private readonly TaskCompletionSource started = new(TaskCreationOptions.RunContinuationsAsynchronously); + + public bool WarmupStarted { get; private set; } + + public bool CancellationObserved { get; private set; } + + public Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + return Task.FromResult>([]); + } + + public async Task WarmUpAsync(CancellationToken cancellationToken) + { + WarmupStarted = true; + started.TrySetResult(); + try + { + await Task.Delay(Timeout.InfiniteTimeSpan, cancellationToken); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + CancellationObserved = true; + } + } + + public Task WaitForWarmupAsync() + { + return started.Task.WaitAsync(TimeSpan.FromSeconds(1)); + } + } +} diff --git a/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs b/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs index 2253534..230ebb1 100644 --- a/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs +++ b/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs @@ -8,13 +8,22 @@ namespace MeetingAssistant.Tests; public sealed class SpeakerAudioSampleCollectorTests { [Fact] - public void CollectorWaitsForConfiguredUninterruptedSpeechDuration() + public void DefaultSampleDurationsAreTenAndSixtySeconds() + { + var options = new SpeakerIdentificationOptions(); + + Assert.Equal(TimeSpan.FromSeconds(10), options.MinimumSampleSpeechDuration); + Assert.Equal(TimeSpan.FromSeconds(60), options.MaximumSampleDuration); + } + + [Fact] + public void CollectorWaitsForConfiguredSpeakerAudioDuration() { var collector = new SpeakerAudioSampleCollector( TimeSpan.FromMinutes(2), maxSamplesPerSpeaker: 3, - minimumUninterruptedSpeechDuration: TimeSpan.FromSeconds(30), - maximumSegmentGap: TimeSpan.FromSeconds(1)); + minimumSampleSpeechDuration: TimeSpan.FromSeconds(30), + maximumSampleDuration: TimeSpan.FromSeconds(60)); collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(35))); collector.TryAdd(Segment(0, 12, "Guest01", "one two three four five.")); @@ -37,8 +46,8 @@ public sealed class SpeakerAudioSampleCollectorTests var collector = new SpeakerAudioSampleCollector( TimeSpan.FromMinutes(2), maxSamplesPerSpeaker: 3, - minimumUninterruptedSpeechDuration: TimeSpan.FromSeconds(30), - maximumSegmentGap: TimeSpan.FromSeconds(1)); + minimumSampleSpeechDuration: TimeSpan.FromSeconds(30), + maximumSampleDuration: TimeSpan.FromSeconds(60)); collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(50))); collector.TryAdd(Segment(0, 20, "Guest01", "one two three four five.")); @@ -48,6 +57,61 @@ public sealed class SpeakerAudioSampleCollectorTests Assert.DoesNotContain(collector.Snapshot(), sample => sample.Speaker == "Guest01"); } + [Fact] + public void CollectorCombinesConsecutiveSameSpeakerSegmentsAcrossProviderPauses() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 3, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(10), + maximumSampleDuration: TimeSpan.FromSeconds(60)); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(20))); + + collector.TryAdd(Segment(0, 4, "Guest01", "one two three four.")); + collector.TryAdd(Segment(7, 13, "Guest01", "five six seven eight.")); + + var sample = Assert.Single(collector.Snapshot()); + Assert.Equal(TimeSpan.Zero, sample.Segment.Start); + Assert.Equal(TimeSpan.FromSeconds(13), sample.Segment.End); + } + + [Fact] + public void CollectorDoesNotCountProviderPausesAsSpeakerAudio() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 3, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(10), + maximumSampleDuration: TimeSpan.FromSeconds(60)); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(30))); + + collector.TryAdd(Segment(0, 4, "Guest01", "one two three four.")); + collector.TryAdd(Segment(20, 25, "Guest01", "five six seven eight.")); + + Assert.Empty(collector.Snapshot()); + } + + [Fact] + public void CollectorCapsRecognitionWavAtConfiguredMaximumDuration() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 3, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(10), + maximumSampleDuration: TimeSpan.FromSeconds(60)); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(75))); + + var sample = collector.TryAdd(Segment( + 0, + 70, + "Guest01", + "one two three four five six seven eight nine ten.")); + + Assert.NotNull(sample); + Assert.Equal(TimeSpan.FromSeconds(60), sample.Segment.End); + Assert.True(ReadDuration(sample.WavBytes) <= TimeSpan.FromSeconds(60)); + } + [Fact] public void CollectorLogsWhenSampleIsDiscardedBecauseSpeechIsTooShort() { @@ -55,8 +119,8 @@ public sealed class SpeakerAudioSampleCollectorTests var collector = new SpeakerAudioSampleCollector( TimeSpan.FromMinutes(2), maxSamplesPerSpeaker: 3, - minimumUninterruptedSpeechDuration: TimeSpan.FromSeconds(30), - maximumSegmentGap: TimeSpan.FromSeconds(1), + minimumSampleSpeechDuration: TimeSpan.FromSeconds(30), + maximumSampleDuration: TimeSpan.FromSeconds(60), logger: logger); collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(35))); @@ -69,6 +133,47 @@ public sealed class SpeakerAudioSampleCollectorTests Assert.Contains("minimum duration", message); } + [Fact] + public void CollectorCanResetThePendingSpanAfterEachAcceptedSample() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 5, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(2), + maximumSampleDuration: TimeSpan.FromSeconds(60), + requireNonOverlappingSamples: true); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(6))); + + collector.TryAdd(Segment(0, 2, "Guest01", "one two three.")); + collector.TryAdd(Segment(2, 4, "Guest01", "four five six.")); + collector.TryAdd(Segment(4, 6, "Guest01", "seven eight nine.")); + + var samples = collector.Snapshot().OrderBy(sample => sample.Segment.Start).ToList(); + + Assert.Collection( + samples, + sample => Assert.Equal((TimeSpan.Zero, TimeSpan.FromSeconds(2)), (sample.Segment.Start, sample.Segment.End)), + sample => Assert.Equal((TimeSpan.FromSeconds(2), TimeSpan.FromSeconds(4)), (sample.Segment.Start, sample.Segment.End)), + sample => Assert.Equal((TimeSpan.FromSeconds(4), TimeSpan.FromSeconds(6)), (sample.Segment.Start, sample.Segment.End))); + } + + [Fact] + public void ResettingCollectorRejectsASegmentThatOverlapsAnAcceptedSample() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 5, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(2), + maximumSampleDuration: TimeSpan.FromSeconds(60), + requireNonOverlappingSamples: true); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(4))); + + collector.TryAdd(Segment(0, 2, "Guest01", "one two three.")); + collector.TryAdd(Segment(1, 3, "Guest01", "four five six.")); + + Assert.Single(collector.Snapshot()); + } + private static TranscriptionSegment Segment( double start, double end, diff --git a/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs b/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs index 32e17d4..d0d62cf 100644 --- a/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs +++ b/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs @@ -107,6 +107,9 @@ public sealed class SpeakerIdentityMergeServiceTests return new SpeakerIdentityMergeService( new TestSpeakerIdentityDbContextFactory(dbPath), Matcher, + new ResemblyzerVoiceVectorOutlierPruner( + new ResemblyzerSpeakerRecognitionOptions(), + NullLogger.Instance), Options.Create(new MeetingAssistantOptions { SpeakerIdentification = new SpeakerIdentificationOptions diff --git a/MeetingAssistant.Tests/SpeakerIdentityMergerTests.cs b/MeetingAssistant.Tests/SpeakerIdentityMergerTests.cs new file mode 100644 index 0000000..daf3fba --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerIdentityMergerTests.cs @@ -0,0 +1,47 @@ +using MeetingAssistant.Speakers; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerIdentityMergerTests +{ + [Fact] + public void MergeIntoRetainsNewestDistinctVoiceVectorsUpToConfiguredLimit() + { + var now = DateTimeOffset.UtcNow; + var target = new SpeakerIdentity + { + CanonicalName = "Chris", + VoiceVectors = + [ + Vector("oldest", 1, now.AddMinutes(-4)), + Vector("duplicate", 2, now.AddMinutes(-3)) + ] + }; + var source = new SpeakerIdentity + { + VoiceVectors = + [ + Vector("duplicate", 2, now.AddMinutes(-2)), + Vector("newer", 3, now.AddMinutes(-1)), + Vector("newest", 4, now) + ] + }; + + SpeakerIdentityMerger.MergeInto(target, source, maxSnippets: 3, maxVoiceVectors: 3); + + Assert.Equal(["duplicate", "newer", "newest"], target.VoiceVectors.Select(vector => vector.Fingerprint).Order()); + Assert.All(target.VoiceVectors, vector => Assert.Same(target, vector.SpeakerIdentity)); + } + + private static SpeakerVoiceVector Vector(string fingerprint, byte value, DateTimeOffset createdAt) + { + return new SpeakerVoiceVector + { + ModelId = "model", + Dimensions = 256, + VectorBytes = Enumerable.Repeat(value, 256 * sizeof(float)).ToArray(), + Fingerprint = fingerprint, + CreatedAt = createdAt + }; + } +} diff --git a/MeetingAssistant.Tests/SpeakerRecognitionFeatureSelectionTests.cs b/MeetingAssistant.Tests/SpeakerRecognitionFeatureSelectionTests.cs new file mode 100644 index 0000000..a62005c --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerRecognitionFeatureSelectionTests.cs @@ -0,0 +1,80 @@ +using MeetingAssistant.Speakers; +using Microsoft.Data.Sqlite; +using Microsoft.AspNetCore.Hosting; +using Microsoft.AspNetCore.Mvc.Testing; +using Microsoft.AspNetCore.TestHost; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.DependencyInjection.Extensions; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerRecognitionFeatureSelectionTests +{ + [Theory] + [InlineData(false, typeof(SpeakerIdentityService), typeof(SpeakerIdentityMergeService), 1, false)] + [InlineData(true, typeof(ResemblyzerSpeakerIdentificationService), typeof(ResemblyzerSpeakerIdentityMergeService), 1000, true)] + public async Task FeatureFlagSelectsExactlyOneIdentificationBackend( + bool enabled, + Type expectedIdentificationType, + Type expectedMergeType, + int expectedMinimumSamples, + bool expectedNonOverlappingSamples) + { + var directory = Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + try + { + await using var factory = new WebApplicationFactory() + .WithWebHostBuilder(builder => + { + builder.ConfigureAppConfiguration((_, configuration) => + { + configuration.AddInMemoryCollection(new Dictionary + { + ["MeetingAssistant:FunAsr:Backend:Enabled"] = "false", + ["MeetingAssistant:SpeakerIdentification:DatabasePath"] = Path.Combine(directory, "identities.db"), + ["MeetingAssistant:SpeakerIdentification:PyannoteValidation:Enabled"] = "false", + ["MeetingAssistant:SpeakerIdentification:Resemblyzer:Enabled"] = enabled.ToString() + }); + }); + builder.ConfigureTestServices(services => + { + services.RemoveAll(); + services.AddSingleton(); + }); + }); + + var service = factory.Services.GetRequiredService(); + + Assert.IsType(expectedIdentificationType, service); + Assert.Single(factory.Services.GetServices()); + var mergeService = factory.Services.GetRequiredService(); + Assert.IsType(expectedMergeType, mergeService); + Assert.Single(factory.Services.GetServices()); + var policy = factory.Services.GetRequiredService(); + Assert.Equal(expectedMinimumSamples, policy.MinimumRetainedSamples); + Assert.Equal(expectedNonOverlappingSamples, policy.RequireNonOverlappingSamples); + } + finally + { + SqliteConnection.ClearAllPools(); + Directory.Delete(directory, recursive: true); + } + } + + private sealed class NoopEncoder : IResemblyzerVoiceEncoder + { + public Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + return Task.FromResult>([]); + } + + public Task WarmUpAsync(CancellationToken cancellationToken) + { + return Task.CompletedTask; + } + } +} diff --git a/MeetingAssistant.Tests/SpeakerSampleSpanSelectorTests.cs b/MeetingAssistant.Tests/SpeakerSampleSpanSelectorTests.cs new file mode 100644 index 0000000..0d6d88a --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerSampleSpanSelectorTests.cs @@ -0,0 +1,142 @@ +using MeetingAssistant.Speakers; +using MeetingAssistant.Transcription; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerSampleSpanSelectorTests +{ + [Fact] + public void SelectBestSameSpeakerSpanCombinesLinesAndCapsTheClip() + { + var span = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpan( + [ + Segment(0, 4), + Segment(7, 13), + Segment(20, 70) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60)); + + Assert.Equal(3, span.Count); + Assert.Equal(TimeSpan.Zero, span[0].Start); + Assert.Equal(TimeSpan.FromSeconds(60), span[^1].End); + Assert.Equal(TimeSpan.FromSeconds(60), SpeakerSampleSpanSelector.SpanDuration(span)); + } + + [Fact] + public void SelectBestSameSpeakerSpanDoesNotCountProviderPausesAsSpeakerAudio() + { + var span = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpan( + [ + Segment(0, 4), + Segment(20, 25) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60)); + + Assert.Empty(span); + } + + [Fact] + public void SelectBestSameSpeakerSpansReturnsDistinctMinimumLengthSamples() + { + var segments = Enumerable.Range(0, 6) + .Select(index => new TranscriptionSegment( + TimeSpan.FromSeconds(index * 10), + TimeSpan.FromSeconds((index + 1) * 10), + "Guest01", + $"segment {index} has enough words")) + .ToList(); + + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + segments, + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + Assert.Equal(5, spans.Count); + Assert.All(spans, span => Assert.True(SpeakerSampleSpanSelector.SpanDuration(span) >= TimeSpan.FromSeconds(10))); + Assert.All( + spans.Zip(spans.Skip(1)), + pair => Assert.True(pair.First[^1].End <= pair.Second[0].Start)); + } + + [Fact] + public void SelectBestSameSpeakerSpansSkipsSegmentsThatOverlapACompletedSample() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [ + Segment(0, 10), + Segment(9, 19), + Segment(20, 30) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + Assert.Equal(2, spans.Count); + Assert.Equal(TimeSpan.Zero, spans[0][0].Start); + Assert.Equal(TimeSpan.FromSeconds(20), spans[1][0].Start); + } + + [Fact] + public void SelectBestSameSpeakerSpansCombinesConsecutiveLinesAcrossProviderPauses() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [ + Segment(0, 4), + Segment(7, 13) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + var span = Assert.Single(spans); + Assert.Equal(TimeSpan.Zero, span[0].Start); + Assert.Equal(TimeSpan.FromSeconds(13), span[^1].End); + } + + [Fact] + public void SelectBestSameSpeakerSpansDoesNotCountProviderPausesAsSpeakerAudio() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [ + Segment(0, 4), + Segment(20, 25) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + Assert.Empty(spans); + } + + [Fact] + public void SelectBestSameSpeakerSpansCapsFinalizedSamplesAtConfiguredMaximumDuration() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [Segment(0, 70)], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + var span = Assert.Single(spans); + Assert.Equal(TimeSpan.FromSeconds(60), SpeakerSampleSpanSelector.SpanDuration(span)); + } + + private static TranscriptionSegment Segment(double start, double end) + { + return new TranscriptionSegment( + TimeSpan.FromSeconds(start), + TimeSpan.FromSeconds(end), + "Guest01", + "enough words for this segment"); + } +} diff --git a/MeetingAssistant.Tests/SpeakerVoiceVectorPersistenceTests.cs b/MeetingAssistant.Tests/SpeakerVoiceVectorPersistenceTests.cs new file mode 100644 index 0000000..21464cc --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerVoiceVectorPersistenceTests.cs @@ -0,0 +1,106 @@ +using MeetingAssistant.Speakers; +using Microsoft.EntityFrameworkCore; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerVoiceVectorPersistenceTests +{ + [Fact] + public async Task IdentityStoresVersionedVoiceVectorAndDeletesItWithIdentity() + { + var databasePath = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "speaker-identities.db"); + Directory.CreateDirectory(Path.GetDirectoryName(databasePath)!); + var options = new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath}") + .Options; + + await using (var context = new SpeakerIdentityDbContext(options)) + { + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + context.SpeakerIdentities.Add(new SpeakerIdentity + { + CanonicalName = "Chris", + CreatedAt = DateTimeOffset.UtcNow, + UpdatedAt = DateTimeOffset.UtcNow, + VoiceVectors = + [ + new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = "vector-a", + CreatedAt = DateTimeOffset.UtcNow + } + ] + }); + await context.SaveChangesAsync(); + } + + await using (var context = new SpeakerIdentityDbContext(options)) + { + var identity = await context.SpeakerIdentities + .Include(candidate => candidate.VoiceVectors) + .SingleAsync(); + var vector = Assert.Single(identity.VoiceVectors); + Assert.Equal("resemblyzer-0.1.4-pretrained", vector.ModelId); + Assert.Equal(256, vector.Dimensions); + Assert.Equal(256 * sizeof(float), vector.VectorBytes.Length); + + context.SpeakerIdentities.Remove(identity); + await context.SaveChangesAsync(); + } + + await using (var context = new SpeakerIdentityDbContext(options)) + { + Assert.Empty(await context.SpeakerVoiceVectors.ToListAsync()); + } + } + + [Fact] + public async Task IdentityRejectsDuplicateVoiceVectorFingerprint() + { + var databasePath = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "speaker-identities.db"); + Directory.CreateDirectory(Path.GetDirectoryName(databasePath)!); + var options = new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath}") + .Options; + await using var context = new SpeakerIdentityDbContext(options); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + var identity = new SpeakerIdentity + { + CanonicalName = "Chris", + CreatedAt = DateTimeOffset.UtcNow, + UpdatedAt = DateTimeOffset.UtcNow + }; + context.SpeakerIdentities.Add(identity); + await context.SaveChangesAsync(); + + context.SpeakerVoiceVectors.AddRange( + CreateVector(identity.Id, "same-vector"), + CreateVector(identity.Id, "same-vector")); + + await Assert.ThrowsAsync(() => context.SaveChangesAsync()); + } + + private static SpeakerVoiceVector CreateVector(int identityId, string fingerprint) + { + return new SpeakerVoiceVector + { + SpeakerIdentityId = identityId, + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = fingerprint, + CreatedAt = DateTimeOffset.UtcNow + }; + } +} diff --git a/MeetingAssistant.Tests/TaskbarIconTests.cs b/MeetingAssistant.Tests/TaskbarIconTests.cs index b1cdbb6..ec2eea3 100644 --- a/MeetingAssistant.Tests/TaskbarIconTests.cs +++ b/MeetingAssistant.Tests/TaskbarIconTests.cs @@ -74,7 +74,8 @@ public sealed class TaskbarIconTests menu, ("Open agent", MeetingTaskbarAction.EditRules, false), ("Finish meeting", MeetingTaskbarAction.StopRecording, true), - ("Microphone", MeetingTaskbarAction.OpenSubmenu, true), + ("Pause transcription", MeetingTaskbarAction.PauseTranscription, true), + ("Microphone", MeetingTaskbarAction.OpenSubmenu, false), ("Cancel meeting recording and discard", MeetingTaskbarAction.AbortRecording, false), ("Switch to english\tCtrl+Alt+L", MeetingTaskbarAction.SwitchProfile, false), ("Exit", MeetingTaskbarAction.Exit, true)); @@ -94,10 +95,32 @@ public sealed class TaskbarIconTests menu, ("Open agent", MeetingTaskbarAction.EditRules, false), ("Finish meeting", MeetingTaskbarAction.StopRecording, true), - ("Cancel meeting recording and discard", MeetingTaskbarAction.AbortRecording, true), + ("Pause transcription", MeetingTaskbarAction.PauseTranscription, true), + ("Cancel meeting recording and discard", MeetingTaskbarAction.AbortRecording, false), ("Exit", MeetingTaskbarAction.Exit, true)); } + [Fact] + public void TranscriptionPauseActionTracksActivePauseStateWithoutHidingFinish() + { + var runningMenu = MeetingTaskbarMenuBuilder.Build( + Status(isRecording: true, state: RecordingProcessState.Recording, profile: "default"), + [Profile("default")]); + var pausedMenu = MeetingTaskbarMenuBuilder.Build( + Status(isRecording: true, state: RecordingProcessState.Recording, profile: "default", isPaused: true), + [Profile("default")]); + + Assert.Contains(runningMenu.Items, item => + item.Action == MeetingTaskbarAction.PauseTranscription && + item.Text == "Pause transcription"); + Assert.Contains(pausedMenu.Items, item => + item.Action == MeetingTaskbarAction.UnpauseTranscription && + item.Text == "Unpause transcription"); + Assert.Contains(pausedMenu.Items, item => + item.Action == MeetingTaskbarAction.StopRecording && + item.Text == "Finish meeting"); + } + [Fact] public void ProcessingMenuShowsSummarizingButAllowsStartingNewRecordings() { @@ -200,7 +223,8 @@ public sealed class TaskbarIconTests private static RecordingStatus Status( bool isRecording = false, RecordingProcessState state = RecordingProcessState.Idle, - string? profile = null) + string? profile = null, + bool isPaused = false) { return new RecordingStatus( isRecording, @@ -209,7 +233,8 @@ public sealed class TaskbarIconTests state == RecordingProcessState.Idle ? null : "context.md", state == RecordingProcessState.Idle ? null : "summary.md", state, - profile); + profile, + isPaused); } [SupportedOSPlatform("windows")] diff --git a/MeetingAssistant.Tests/VenvResemblyzerVoiceEncoderTests.cs b/MeetingAssistant.Tests/VenvResemblyzerVoiceEncoderTests.cs new file mode 100644 index 0000000..ecdac99 --- /dev/null +++ b/MeetingAssistant.Tests/VenvResemblyzerVoiceEncoderTests.cs @@ -0,0 +1,249 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using MeetingAssistant.Transcription; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; +using System.Text.Json; + +namespace MeetingAssistant.Tests; + +public sealed class VenvResemblyzerVoiceEncoderTests +{ + [Fact] + public async Task WarmUpAsyncProvisionsVersionedCpuOnlyEnvironmentWithoutDocker() + { + using var fixture = new EncoderFixture(); + + await fixture.Encoder.WarmUpAsync(CancellationToken.None); + + Assert.DoesNotContain( + fixture.Runner.Commands, + command => command.FileName.Contains("docker", StringComparison.OrdinalIgnoreCase)); + Assert.Contains( + fixture.Runner.Commands, + command => command.FileName == "python-test" + && command.Arguments.Take(2).SequenceEqual(["-m", "venv"])); + Assert.Contains( + fixture.Runner.Commands, + command => command.Arguments.Contains("torch==2.14.0+cpu") + && command.Arguments.Contains("https://download.pytorch.org/whl/cpu")); + Assert.Contains( + fixture.Runner.Commands, + command => command.Arguments.Contains("webrtcvad-wheels==2.0.14")); + Assert.Contains( + fixture.Runner.Commands, + command => command.Arguments.Contains("Resemblyzer==0.1.4") + && command.Arguments.Contains("--no-deps")); + } + + [Fact] + public async Task WarmUpAsyncDoesNotMarkFailedEnvironmentReady() + { + using var fixture = new EncoderFixture( + warmupExitCode: 21, + warmupError: "incompatible environment"); + + await Assert.ThrowsAsync( + () => fixture.Encoder.WarmUpAsync(CancellationToken.None)); + + Assert.Empty(Directory.EnumerateFiles( + fixture.RuntimeFolder, + ".ready", + SearchOption.AllDirectories)); + } + + [Fact] + public async Task EncodeAsyncRunsBatchWithManagedEnvironmentPython() + { + var first = new float[256]; + first[0] = 2; + var second = new float[256]; + second[1] = 3; + using var fixture = new EncoderFixture(VectorOutput([first, second])); + + var vectors = await fixture.Encoder.EncodeAsync([[1, 2, 3], [4, 5, 6]], CancellationToken.None); + + Assert.Equal(2, vectors.Count); + Assert.Equal(1f, vectors[0][0], 5); + Assert.Equal(1f, vectors[1][1], 5); + var encodingCommand = Assert.Single( + fixture.Runner.Commands, + command => command.Arguments.Any( + argument => argument.EndsWith("encode.py", StringComparison.Ordinal))); + Assert.EndsWith( + Path.Combine("Scripts", "python.exe"), + encodingCommand.FileName, + StringComparison.OrdinalIgnoreCase); + Assert.Equal(2, encodingCommand.Arguments.Count); + Assert.True(Path.IsPathFullyQualified(encodingCommand.Arguments[1])); + } + + [Fact] + public async Task EncodeAsyncRejectsWrongVectorDimension() + { + var vector = new float[255]; + vector[0] = 1; + using var fixture = new EncoderFixture(VectorOutput([vector])); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("255 dimensions", error.Message); + } + + [Fact] + public async Task EncodeAsyncRejectsMalformedVectorJson() + { + using var fixture = new EncoderFixture( + "__MEETING_ASSISTANT_RESEMBLYZER_JSON_START__\n[not-json]\n__MEETING_ASSISTANT_RESEMBLYZER_JSON_END__"); + + await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + } + + [Fact] + public async Task EncodeAsyncRejectsNonFiniteVector() + { + var values = string.Join(',', new[] { "NaN" }.Concat(Enumerable.Repeat("0", 255))); + using var fixture = new EncoderFixture( + $"__MEETING_ASSISTANT_RESEMBLYZER_JSON_START__\n[[{values}]]\n__MEETING_ASSISTANT_RESEMBLYZER_JSON_END__"); + + await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + } + + [Fact] + public async Task EncodeAsyncRejectsWrongResultCount() + { + var vector = new float[256]; + vector[0] = 1; + using var fixture = new EncoderFixture(VectorOutput([vector, vector])); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("2 vectors for 1 WAV samples", error.Message); + } + + [Fact] + public async Task EncodeAsyncRejectsZeroMagnitudeVector() + { + using var fixture = new EncoderFixture(VectorOutput([new float[256]])); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("zero magnitude", error.Message); + } + + [Fact] + public async Task EncodeAsyncReportsFailedLocalCommand() + { + using var fixture = new EncoderFixture("", 17, "runtime failed"); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("exit code 17", error.Message); + Assert.Contains("runtime failed", error.Message); + } + + private static string VectorOutput(IReadOnlyList vectors) + { + return $""" + __MEETING_ASSISTANT_RESEMBLYZER_JSON_START__ + {JsonSerializer.Serialize(vectors)} + __MEETING_ASSISTANT_RESEMBLYZER_JSON_END__ + """; + } + + private sealed class EncoderFixture : IDisposable + { + public EncoderFixture( + string encodingOutput = "", + int encodingExitCode = 0, + string encodingError = "", + int warmupExitCode = 0, + string warmupError = "") + { + RuntimeFolder = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "resemblyzer"); + Runner = new CapturingCommandRunner( + encodingOutput, + encodingExitCode, + encodingError, + warmupExitCode, + warmupError); + var options = new MeetingAssistantOptions(); + options.SpeakerIdentification.Resemblyzer.PythonCommand = "python-test"; + options.SpeakerIdentification.Resemblyzer.RuntimeFolder = RuntimeFolder; + Encoder = new VenvResemblyzerVoiceEncoder( + Runner, + Options.Create(options), + NullLogger.Instance); + } + + public CapturingCommandRunner Runner { get; } + + public VenvResemblyzerVoiceEncoder Encoder { get; } + + public string RuntimeFolder { get; } + + public void Dispose() + { + if (Directory.Exists(RuntimeFolder)) + { + Directory.Delete(RuntimeFolder, recursive: true); + } + } + } + + private sealed class CapturingCommandRunner : ICommandRunner + { + private readonly string encodingOutput; + private readonly int encodingExitCode; + private readonly string encodingError; + private readonly int warmupExitCode; + private readonly string warmupError; + + public CapturingCommandRunner( + string encodingOutput = "", + int encodingExitCode = 0, + string encodingError = "", + int warmupExitCode = 0, + string warmupError = "") + { + this.encodingOutput = encodingOutput; + this.encodingExitCode = encodingExitCode; + this.encodingError = encodingError; + this.warmupExitCode = warmupExitCode; + this.warmupError = warmupError; + } + + public List Commands { get; } = []; + + public Task RunAsync( + string fileName, + IReadOnlyList arguments, + CancellationToken cancellationToken, + IReadOnlyDictionary? environment = null) + { + Commands.Add(new CapturedCommand(fileName, arguments.ToList())); + var isEncoding = arguments.Any(argument => argument.EndsWith("encode.py", StringComparison.Ordinal)); + if (isEncoding) + { + return Task.FromResult(new CommandResult(encodingExitCode, encodingOutput, encodingError)); + } + + var isWarmup = arguments.Any(argument => argument.Contains("VoiceEncoder", StringComparison.Ordinal)); + return Task.FromResult(isWarmup + ? new CommandResult(warmupExitCode, string.Empty, warmupError) + : new CommandResult(0, string.Empty, string.Empty)); + } + } + + private sealed record CapturedCommand(string FileName, IReadOnlyList Arguments); +} diff --git a/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs b/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs index 170bf22..032c57f 100644 --- a/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs +++ b/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs @@ -864,6 +864,32 @@ public sealed class WorkflowRulesEditorTests Assert.Empty(await fixture.Context.SpeakerIdentities.ToListAsync()); } + [Fact] + public async Task RulesEditorToolsExposeVoiceVectorCountsSeparatelyFromWavSamples() + { + await using var fixture = await WorkflowRulesEditorIdentityFixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync("Sabrina", sample: [1, 2, 3]); + fixture.Context.SpeakerVoiceVectors.Add(new SpeakerVoiceVector + { + SpeakerIdentityId = identity.Id, + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = "vector-1", + CreatedAt = DateTimeOffset.UtcNow + }); + await fixture.Context.SaveChangesAsync(); + var tools = fixture.CreateTools(); + + var searchResult = await tools.SearchIdentities("Sabrina"); + var readResult = await tools.ReadIdentity(identity.Id); + + Assert.Contains("\"sampleCount\": 1", searchResult); + Assert.Contains("\"voiceVectorCount\": 1", searchResult); + Assert.Contains("\"voiceVectorCount\": 1", readResult); + Assert.DoesNotContain("vector-1", readResult); + } + [Fact] public async Task RulesEditorToolsRefusesSamplelessSpeakerIdentityCreation() { @@ -872,7 +898,7 @@ public sealed class WorkflowRulesEditorTests var result = await tools.CreateIdentity("Sabrina", ["Sabi"], ["Guest-01"]); - Assert.StartsWith("Refused: speaker identities require at least one audio sample.", result); + Assert.StartsWith("Refused: speaker identities require audio evidence", result); Assert.Empty(await fixture.Context.SpeakerIdentities.ToListAsync()); } diff --git a/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs b/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs index 45f3675..692b6c8 100644 --- a/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs +++ b/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs @@ -1,4 +1,5 @@ using Microsoft.Extensions.Configuration; +using MeetingAssistant.Speakers; namespace MeetingAssistant.LaunchProfiles; @@ -48,6 +49,7 @@ public sealed class ConfigurationLaunchProfileOptionsProvider : ILaunchProfileOp var options = BindDefaultOptions(); if (profileName.Equals(DefaultProfileName, StringComparison.OrdinalIgnoreCase)) { + ValidateSpeakerSampleDurations(options, DefaultProfileName); return new LaunchProfile(DefaultProfileName, options); } @@ -59,6 +61,7 @@ public sealed class ConfigurationLaunchProfileOptionsProvider : ILaunchProfileOp profileSection.Bind(options); ApplyArrayOverrides(profileSection, options); + ValidateSpeakerSampleDurations(options, profileName); return new LaunchProfile(profileName, options); } @@ -149,6 +152,18 @@ public sealed class ConfigurationLaunchProfileOptionsProvider : ILaunchProfileOp : name.Trim(); } + private static void ValidateSpeakerSampleDurations( + MeetingAssistantOptions options, + string profileName) + { + SpeakerSampleDurationConfiguration.ValidateOrThrow( + options.SpeakerIdentification, + $"Launch profile '{profileName}' has invalid speaker sample durations"); + ResemblyzerSpeakerRecognitionConfiguration.ValidateOrThrow( + options.SpeakerIdentification.Resemblyzer, + $"Launch profile '{profileName}' has invalid Resemblyzer speaker-recognition settings"); + } + private static void ApplyArrayOverrides( IConfigurationSection profileSection, MeetingAssistantOptions options) diff --git a/MeetingAssistant/MeetingAssistant.csproj b/MeetingAssistant/MeetingAssistant.csproj index 2df711b..ae0fb77 100644 --- a/MeetingAssistant/MeetingAssistant.csproj +++ b/MeetingAssistant/MeetingAssistant.csproj @@ -5,7 +5,7 @@ enable enable true - 1.51.1 + 1.51.2 @@ -36,16 +36,16 @@ - + - - - + + + - + diff --git a/MeetingAssistant/MeetingAssistantOptions.cs b/MeetingAssistant/MeetingAssistantOptions.cs index 4ec402e..25da41b 100644 --- a/MeetingAssistant/MeetingAssistantOptions.cs +++ b/MeetingAssistant/MeetingAssistantOptions.cs @@ -165,6 +165,8 @@ public sealed class RecordingInactivitySafeguardOptions public TimeSpan AutoStopAfter { get; set; } = TimeSpan.FromMinutes(30); + public TimeSpan MaximumPauseDuration { get; set; } = TimeSpan.FromHours(4); + public TimeSpan InferredEndPadding { get; set; } = TimeSpan.FromMinutes(1); public TimeSpan CheckInterval { get; set; } = TimeSpan.FromSeconds(15); @@ -181,10 +183,8 @@ public sealed class WhisperLocalOptions public PyannoteDiarizationOptions Diarization { get; set; } = new(); } -public sealed class PyannoteDiarizationOptions +public class PyannoteRuntimeOptions { - public bool Enabled { get; set; } = true; - public string DockerCommand { get; set; } = "docker"; public string BaseImage { get; set; } = "python:3.11-slim"; @@ -208,6 +208,11 @@ public sealed class PyannoteDiarizationOptions public TimeSpan CommandTimeout { get; set; } = TimeSpan.FromMinutes(30); } +public sealed class PyannoteDiarizationOptions : PyannoteRuntimeOptions +{ + public bool Enabled { get; set; } = true; +} + public sealed class AzureSpeechOptions { public string Endpoint { get; set; } = ""; @@ -340,9 +345,9 @@ public sealed class SpeakerIdentificationOptions public int MaxSnippetsPerSpeaker { get; set; } = 3; - public TimeSpan MinimumSampleSpeechDuration { get; set; } = TimeSpan.FromSeconds(30); + public TimeSpan MinimumSampleSpeechDuration { get; set; } = TimeSpan.FromSeconds(10); - public TimeSpan MaximumSampleSegmentGap { get; set; } = TimeSpan.FromSeconds(1); + public TimeSpan MaximumSampleDuration { get; set; } = TimeSpan.FromSeconds(60); public double SilenceBetweenSnippetsSeconds { get; set; } = 1; @@ -355,6 +360,47 @@ public sealed class SpeakerIdentificationOptions public AzureSpeechOptions AzureSpeech { get; set; } = new(); public SpeakerIdentityPyannoteValidationOptions PyannoteValidation { get; set; } = new(); + + public ResemblyzerSpeakerRecognitionOptions Resemblyzer { get; set; } = new(); +} + +public sealed class ResemblyzerSpeakerRecognitionOptions +{ + public bool Enabled { get; set; } + + public int RequiredVectorsPerSpeaker { get; set; } = 5; + + public int MaxVectorsPerIdentity { get; set; } = 1000; + + public int OutlierPruningMinimumVectors { get; set; } = 20; + + public double OutlierPruningNeighborSimilarity { get; set; } = 0.75; + + public int OutlierPruningMinimumNeighbors { get; set; } = 3; + + public double OutlierPruningMinimumClusterRatio { get; set; } = 0.60; + + public double MinimumClusterCohesion { get; set; } = 0.75; + + public double MinimumIdentitySimilarity { get; set; } = 0.75; + + public double MinimumSimilarityMargin { get; set; } = 0.05; + + public string ModelId { get; set; } = "resemblyzer-0.1.4-pretrained"; + + public string PackageVersion { get; set; } = "0.1.4"; + + public string PythonCommand { get; set; } = "python"; + + public string TorchVersion { get; set; } = "2.14.0+cpu"; + + public string TorchIndexUrl { get; set; } = "https://download.pytorch.org/whl/cpu"; + + public string WebRtcVadVersion { get; set; } = "2.0.14"; + + public string RuntimeFolder { get; set; } = @"%LOCALAPPDATA%\MeetingAssistant\Resemblyzer"; + + public TimeSpan CommandTimeout { get; set; } = TimeSpan.FromMinutes(15); } public sealed class SpeakerIdentityPyannoteValidationOptions @@ -365,9 +411,8 @@ public sealed class SpeakerIdentityPyannoteValidationOptions public double MinimumMatchingKnownSnippetRatio { get; set; } = 0.50; - public PyannoteDiarizationOptions Diarization { get; set; } = new() + public PyannoteRuntimeOptions Diarization { get; set; } = new() { - Enabled = true, CommandTimeout = TimeSpan.FromHours(1), AlignmentMode = PyannoteAlignmentMode.PyannoteTurns }; diff --git a/MeetingAssistant/Program.cs b/MeetingAssistant/Program.cs index 2ee7661..d3a7b73 100644 --- a/MeetingAssistant/Program.cs +++ b/MeetingAssistant/Program.cs @@ -17,7 +17,10 @@ using Microsoft.Extensions.Options; var builder = WebApplication.CreateBuilder(args); builder.Logging.AddProvider(new MeetingAssistantFileLoggerProvider()); -builder.Services.Configure(builder.Configuration.GetSection("MeetingAssistant")); +builder.Services.AddSingleton, MeetingAssistantSpeakerSampleOptionsValidator>(); +builder.Services.AddOptions() + .Bind(builder.Configuration.GetSection("MeetingAssistant")) + .ValidateOnStart(); builder.Services.AddSingleton(HostOperatingSystem.DetectCurrent()); builder.Services.AddSingleton(); #if WINDOWS @@ -88,8 +91,35 @@ builder.Services.AddSingleton(); builder.Services.AddSingleton(); builder.Services.AddSingleton(); -builder.Services.AddSingleton(); -builder.Services.AddSingleton(); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(services => new ResemblyzerVoiceClusterMatcher( + services.GetRequiredService>().Value.SpeakerIdentification.Resemblyzer, + services.GetRequiredService>())); +builder.Services.AddSingleton(services => new ResemblyzerVoiceVectorOutlierPruner( + services.GetRequiredService>().Value.SpeakerIdentification.Resemblyzer, + services.GetRequiredService>())); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(services => +{ + var resemblyzer = services.GetRequiredService>() + .Value.SpeakerIdentification.Resemblyzer; + return resemblyzer.Enabled + ? SpeakerSampleCollectionPolicy.IndependentVectors(resemblyzer.MaxVectorsPerIdentity) + : SpeakerSampleCollectionPolicy.ExistingBackend; +}); +builder.Services.AddSingleton(services => + services.GetRequiredService>() + .Value.SpeakerIdentification.Resemblyzer.Enabled + ? services.GetRequiredService() + : services.GetRequiredService()); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(services => + services.GetRequiredService>() + .Value.SpeakerIdentification.Resemblyzer.Enabled + ? services.GetRequiredService() + : services.GetRequiredService()); builder.Services.AddSingleton(); builder.Services.AddSingleton(); builder.Services.AddSingleton(); @@ -99,7 +129,11 @@ builder.Services.AddSingleton(); builder.Services.AddSingleton(); builder.Services.AddSingleton(); +#if WINDOWS builder.Services.AddSingleton(); +#else +builder.Services.AddSingleton(); +#endif builder.Services.AddSingleton(); builder.Services.AddTransient(); #if WINDOWS @@ -132,6 +166,7 @@ builder.Services.AddHostedService(); builder.Services.AddHostedService(); builder.Services.AddHostedService(); builder.Services.AddHostedService(); +builder.Services.AddHostedService(); #if WINDOWS builder.Services.AddHostedService(); builder.Services.AddHostedService(); diff --git a/MeetingAssistant/Recording/IMeetingInactivityPromptService.cs b/MeetingAssistant/Recording/IMeetingInactivityPromptService.cs index 58fa9d6..f2ddd67 100644 --- a/MeetingAssistant/Recording/IMeetingInactivityPromptService.cs +++ b/MeetingAssistant/Recording/IMeetingInactivityPromptService.cs @@ -6,6 +6,8 @@ public interface IMeetingInactivityPromptService MeetingInactivityPromptRequest request, Func handleResponseAsync, CancellationToken cancellationToken); + + Task DismissAllAsync(CancellationToken cancellationToken); } public sealed record MeetingInactivityPromptRequest( @@ -18,9 +20,32 @@ public enum MeetingInactivityPromptResponse { Dismissed, Continue, + Pause, Stop } +internal sealed record MeetingInactivityPromptAction( + string Content, + string ResponseArgument, + MeetingInactivityPromptResponse Response); + +internal static class MeetingInactivityPromptActions +{ + public static IReadOnlyList All { get; } = + [ + new("Yes", "stop", MeetingInactivityPromptResponse.Stop), + new("No", "continue", MeetingInactivityPromptResponse.Continue), + new("Pause transcription", "pause", MeetingInactivityPromptResponse.Pause) + ]; + + public static MeetingInactivityPromptResponse ParseResponse(string? value) + { + return All.FirstOrDefault(action => + action.ResponseArgument.Equals(value, StringComparison.OrdinalIgnoreCase)) + ?.Response ?? MeetingInactivityPromptResponse.Continue; + } +} + public sealed class NoopMeetingInactivityPromptService : IMeetingInactivityPromptService { public Task ShowStopPromptAsync( @@ -30,4 +55,9 @@ public sealed class NoopMeetingInactivityPromptService : IMeetingInactivityPromp { return Task.CompletedTask; } + + public Task DismissAllAsync(CancellationToken cancellationToken) + { + return Task.CompletedTask; + } } diff --git a/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs b/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs index 7fc9d0b..788717b 100644 --- a/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs +++ b/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs @@ -34,6 +34,7 @@ public sealed class MeetingRecordingCoordinator private readonly IMeetingInactivityClock inactivityClock; private readonly IOfflineTranscriptionBacklog offlineTranscriptionBacklog; private readonly MeetingAssistantOptions options; + private readonly SpeakerSampleCollectionPolicy speakerSampleCollectionPolicy; private readonly ILogger logger; private readonly SemaphoreSlim gate = new(1, 1); private RecordingRun? currentRun; @@ -63,7 +64,8 @@ public sealed class MeetingRecordingCoordinator IMeetingRunArtifactCleaner? artifactCleaner = null, IMeetingInactivityPromptService? inactivityPromptService = null, IMeetingInactivityClock? inactivityClock = null, - IOfflineTranscriptionBacklog? offlineTranscriptionBacklog = null) + IOfflineTranscriptionBacklog? offlineTranscriptionBacklog = null, + SpeakerSampleCollectionPolicy? speakerSampleCollectionPolicy = null) { this.audioSource = audioSource; this.speechRecognitionPipelineFactory = speechRecognitionPipelineFactory; @@ -86,6 +88,8 @@ public sealed class MeetingRecordingCoordinator this.inactivityClock = inactivityClock ?? new SystemMeetingInactivityClock(); this.offlineTranscriptionBacklog = offlineTranscriptionBacklog ?? NoopOfflineTranscriptionBacklog.Instance; this.options = options.Value; + this.speakerSampleCollectionPolicy = speakerSampleCollectionPolicy + ?? SpeakerSampleCollectionPolicy.ExistingBackend; this.logger = logger; } @@ -96,7 +100,8 @@ public sealed class MeetingRecordingCoordinator currentArtifacts?.AssistantContextPath, currentArtifacts?.SummaryPath, GetProcessState(currentRun), - currentRun?.LaunchProfileName); + currentRun?.LaunchProfileName, + IsRecording && currentRun?.IsTranscriptionPaused == true); public MeetingSessionArtifacts? CurrentArtifacts => currentArtifacts; @@ -289,10 +294,12 @@ public sealed class MeetingRecordingCoordinator startedAt, launchProfile.Name, runOptions.SpeakerIdentification.LiveSampleBufferDuration, - runOptions.SpeakerIdentification.MaxSnippetsPerSpeaker, + speakerSampleCollectionPolicy.ResolveRetainedSampleLimit( + runOptions.SpeakerIdentification.MaxSnippetsPerSpeaker), + speakerSampleCollectionPolicy.RequireNonOverlappingSamples, logger); run.Task = Task.Run(() => RecordAsync(run), CancellationToken.None); - if (runOptions.Recording.InactivitySafeguard.Enabled) + if (ShouldRunInactivitySafeguard(runOptions.Recording.InactivitySafeguard)) { _ = Task.Run( () => RunInactivitySafeguardAsync(run), @@ -334,11 +341,14 @@ public sealed class MeetingRecordingCoordinator public async Task StopAsync(CancellationToken cancellationToken) { - return await StopAsync(null, cancellationToken); + return await StopAsync(null, null, null, null, cancellationToken); } private async Task StopAsync( DateTimeOffset? inferredEndTime, + RecordingRun? expectedRun, + long? expectedActivityVersion, + DateTimeOffset? expectedPauseStartedAt, CancellationToken cancellationToken) { RecordingRun run; @@ -347,13 +357,30 @@ public sealed class MeetingRecordingCoordinator await gate.WaitAsync(cancellationToken); try { - if (currentRun is null || currentRun.IsCaptureStopping) + if (currentRun is null || + currentRun.IsCaptureStopping || + expectedRun is not null && !ReferenceEquals(currentRun, expectedRun)) { return CurrentStatus; } run = currentRun; - if (inferredEndTime is not null) + var inactivityStopClaimed = expectedActivityVersion is not null; + if (expectedActivityVersion is not null && + (inferredEndTime is null || + !run.TryClaimInactivityStop(expectedActivityVersion.Value, inferredEndTime.Value))) + { + return CurrentStatus; + } + + var maximumPauseStopClaimed = expectedPauseStartedAt is not null; + if (expectedPauseStartedAt is not null && + !run.TryClaimMaximumPauseStop(expectedPauseStartedAt.Value)) + { + return CurrentStatus; + } + + if (!inactivityStopClaimed && !maximumPauseStopClaimed && inferredEndTime is not null) { run.SetInferredEndTime(inferredEndTime.Value); } @@ -363,7 +390,7 @@ public sealed class MeetingRecordingCoordinator { run.Abort(); } - else + else if (!inactivityStopClaimed && !maximumPauseStopClaimed) { run.StopCapture(); } @@ -409,6 +436,65 @@ public sealed class MeetingRecordingCoordinator return CurrentStatus; } + public async Task SetTranscriptionPausedAsync( + bool isPaused, + CancellationToken cancellationToken) + { + return await SetTranscriptionPausedAsync(isPaused, null, null, cancellationToken); + } + + private async Task SetTranscriptionPausedAsync( + bool isPaused, + RecordingRun? expectedRun, + long? expectedActivityVersion, + CancellationToken cancellationToken) + { + await gate.WaitAsync(cancellationToken); + try + { + var run = currentRun; + if (run is null || + run.IsCaptureStopping || + expectedRun is not null && !ReferenceEquals(run, expectedRun)) + { + return CurrentStatus; + } + + var changed = false; + await run.PipelineGate.WaitAsync(cancellationToken); + try + { + var changedAt = inactivityClock.Now; + changed = run.TrySetTranscriptionPaused( + isPaused, + changedAt, + expectedActivityVersion); + } + finally + { + run.PipelineGate.Release(); + } + + if (changed && isPaused) + { + await DismissInactivityPromptsAsync(cancellationToken); + } + + if (changed) + { + logger.LogInformation( + "Meeting transcription {PauseState}", + isPaused ? "paused" : "unpaused"); + } + + return CurrentStatus; + } + finally + { + gate.Release(); + } + } + public async Task AbortAsync(CancellationToken cancellationToken) { RecordingRun run; @@ -679,8 +765,7 @@ public sealed class MeetingRecordingCoordinator chunk.Channels); } - await run.RecordedAudio.AppendAsync(chunk, run.CaptureCancellation); - await run.WriteAudioAsync(chunk, run.CaptureCancellation); + await run.RouteCapturedAudioAsync(chunk, run.CaptureCancellation); } logger.LogInformation( @@ -719,7 +804,6 @@ public sealed class MeetingRecordingCoordinator run.ResetSpeakerIdentification(); } - run.RecordTranscriptActivity(segment, inactivityClock.Now); var sample = run.TryAddSpeakerSample(segment); if (sample is not null) { @@ -736,6 +820,56 @@ public sealed class MeetingRecordingCoordinator var relabeledSegment = run.Relabel(segment); run.AddLiveSegment(relabeledSegment); await AppendTranscriptSegmentAsync(run, relabeledSegment, cancellationToken); + run.RecordTranscriptActivity(segment, inactivityClock.Now); + if (!string.IsNullOrWhiteSpace(segment.Text)) + { + // Switching profiles holds the coordinator gate while draining this reader. + // Notification cleanup must not make transcript draining wait for that gate. + _ = DismissInactivityPromptsForRunAsync(run, run.CaptureCancellation); + } + } + } + + private async Task DismissInactivityPromptsForRunAsync( + RecordingRun run, + CancellationToken cancellationToken) + { + try + { + await gate.WaitAsync(cancellationToken); + try + { + if (!ReferenceEquals(currentRun, run) || run.IsCaptureStopping) + { + return; + } + + await DismissInactivityPromptsAsync(cancellationToken); + } + finally + { + gate.Release(); + } + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + // Pending cleanup is no longer needed after this run stops capturing. + } + } + + private async Task DismissInactivityPromptsAsync(CancellationToken cancellationToken) + { + try + { + await inactivityPromptService.DismissAllAsync(cancellationToken); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + throw; + } + catch (Exception exception) + { + logger.LogWarning(exception, "Failed to dismiss transcript inactivity prompts"); } } @@ -863,7 +997,10 @@ public sealed class MeetingRecordingCoordinator private async Task RunInactivitySafeguardAsync(RecordingRun run) { var safeguardOptions = run.Options.Recording.InactivitySafeguard; - var promptThresholds = GetInactivityPromptThresholds(safeguardOptions); + var promptThresholds = safeguardOptions.Enabled + ? GetInactivityPromptThresholds(safeguardOptions) + : []; + var checkInterval = GetInactivitySafeguardCheckInterval(safeguardOptions); var promptedThresholds = new HashSet(); var lastActivityVersion = run.GetTranscriptActivitySnapshot().ActivityVersion; @@ -871,7 +1008,7 @@ public sealed class MeetingRecordingCoordinator "Recording inactivity safeguard started with prompt thresholds {PromptThresholds}, auto-stop {AutoStopAfter}, check interval {CheckInterval}", string.Join(", ", promptThresholds), safeguardOptions.AutoStopAfter, - safeguardOptions.CheckInterval); + checkInterval); try { @@ -887,9 +1024,38 @@ public sealed class MeetingRecordingCoordinator snapshot.LastTranscriptActivityAt); } + var pauseStartedAt = snapshot.TranscriptionPausedAt; + if (pauseStartedAt is not null) + { + var maximumPauseDuration = safeguardOptions.MaximumPauseDuration; + if (maximumPauseDuration > TimeSpan.Zero && + inactivityClock.Now - pauseStartedAt.Value >= maximumPauseDuration) + { + logger.LogWarning( + "Recording inactivity safeguard stopping meeting after transcription remained paused for {PauseDuration}", + maximumPauseDuration); + var stopStatus = await StopAsync( + null, + run, + null, + pauseStartedAt, + CancellationToken.None); + if (!stopStatus.IsRecording) + { + return; + } + } + + await inactivityClock.DelayAsync( + checkInterval, + run.CaptureCancellation); + continue; + } + var now = inactivityClock.Now; var inactivityDuration = now - snapshot.LastTranscriptActivityAt; - if (safeguardOptions.AutoStopAfter > TimeSpan.Zero && + if (safeguardOptions.Enabled && + safeguardOptions.AutoStopAfter > TimeSpan.Zero && inactivityDuration >= safeguardOptions.AutoStopAfter) { var inferredEndTime = run.GetInactivitySafeguardEndTime(safeguardOptions.InferredEndPadding); @@ -897,8 +1063,16 @@ public sealed class MeetingRecordingCoordinator "Recording inactivity safeguard auto-stopping meeting after {InactivityDuration} without transcript text; inferred end time {InferredEndTime}", inactivityDuration, inferredEndTime); - await StopAsync(inferredEndTime, CancellationToken.None); - return; + var stopStatus = await StopAsync( + inferredEndTime, + run, + snapshot.ActivityVersion, + null, + CancellationToken.None); + if (!stopStatus.IsRecording) + { + return; + } } foreach (var threshold in promptThresholds) @@ -915,6 +1089,7 @@ public sealed class MeetingRecordingCoordinator inactivityDuration, threshold, inferredEndTime); + var promptActivityVersion = snapshot.ActivityVersion; await inactivityPromptService.ShowStopPromptAsync( new MeetingInactivityPromptRequest( inactivityDuration, @@ -930,16 +1105,36 @@ public sealed class MeetingRecordingCoordinator if (response == MeetingInactivityPromptResponse.Stop) { - await StopAsync(inferredEndTime, CancellationToken.None); + await StopAsync( + inferredEndTime, + run, + promptActivityVersion, + null, + CancellationToken.None); + } + else if (response == MeetingInactivityPromptResponse.Pause) + { + await SetTranscriptionPausedAsync( + true, + run, + promptActivityVersion, + CancellationToken.None); } }, run.CaptureCancellation); + var currentActivity = run.GetTranscriptActivitySnapshot(); + if (currentActivity.IsTranscriptionPaused || + currentActivity.ActivityVersion != promptActivityVersion) + { + await DismissInactivityPromptsAsync(run.CaptureCancellation); + } + break; } await inactivityClock.DelayAsync( - GetInactivitySafeguardCheckInterval(safeguardOptions), + checkInterval, run.CaptureCancellation); } } @@ -1007,11 +1202,11 @@ public sealed class MeetingRecordingCoordinator try { var meetingNote = await meetingNoteStore.ReadAsync(run.MeetingNotePath, cancellationToken); - var checkpoint = run.CreateLiveIdentificationCheckpoint(meetingNote); + var checkpoint = run.CreateLiveIdentificationCheckpoint(meetingNote, samples); if (checkpoint is null) { logger.LogInformation( - "Skipping live speaker identity matching because no new unmapped sample speakers or attendee changes were found"); + "Skipping live speaker identity matching because no new eligible speaker evidence or attendee changes were found"); return; } @@ -1839,6 +2034,11 @@ public sealed class MeetingRecordingCoordinator .ToList(); } + private static bool ShouldRunInactivitySafeguard(RecordingInactivitySafeguardOptions options) + { + return options.Enabled || options.MaximumPauseDuration > TimeSpan.Zero; + } + private static TimeSpan GetInactivitySafeguardCheckInterval( RecordingInactivitySafeguardOptions options) { @@ -1895,6 +2095,8 @@ public sealed class MeetingRecordingCoordinator private TimeSpan? lastTranscriptSegmentEnd; private long transcriptActivityVersion; private DateTimeOffset? inferredEndTime; + private bool isTranscriptionPaused; + private DateTimeOffset? transcriptionPausedAt; public RecordingRun( CancellationTokenSource captureCancellation, @@ -1910,6 +2112,7 @@ public sealed class MeetingRecordingCoordinator string launchProfileName, TimeSpan liveSampleBufferDuration, int maxSpeakerSamples, + bool requireNonOverlappingSpeakerSamples, ILogger logger) { CaptureCancellationSource = captureCancellation; @@ -1930,8 +2133,9 @@ public sealed class MeetingRecordingCoordinator liveSampleBufferDuration, maxSpeakerSamples, options.SpeakerIdentification.MinimumSampleSpeechDuration, - options.SpeakerIdentification.MaximumSampleSegmentGap, - logger); + options.SpeakerIdentification.MaximumSampleDuration, + logger, + requireNonOverlappingSpeakerSamples); } public CancellationTokenSource CaptureCancellationSource { get; } @@ -1978,6 +2182,17 @@ public sealed class MeetingRecordingCoordinator public bool HasAttachedPromptMetadata { get; private set; } + public bool IsTranscriptionPaused + { + get + { + lock (transcriptActivityGate) + { + return isTranscriptionPaused; + } + } + } + public AssistantContextState ContextState { get; private set; } = AssistantContextState.CollectingMetadata; public DateTimeOffset? InferredEndTime @@ -2001,6 +2216,37 @@ public sealed class MeetingRecordingCoordinator HasAttachedPromptMetadata = true; } + public bool TrySetTranscriptionPaused( + bool isPaused, + DateTimeOffset changedAt, + long? expectedActivityVersion) + { + lock (transcriptActivityGate) + { + if (expectedActivityVersion is not null && + transcriptActivityVersion != expectedActivityVersion.Value) + { + return false; + } + + if (isTranscriptionPaused == isPaused) + { + return false; + } + + isTranscriptionPaused = isPaused; + transcriptionPausedAt = isPaused ? changedAt : null; + + if (!isPaused) + { + lastTranscriptActivityAt = changedAt; + } + + transcriptActivityVersion++; + return true; + } + } + public void Abort() { IsAborted = true; @@ -2029,13 +2275,17 @@ public sealed class MeetingRecordingCoordinator return string.Equals(LaunchProfileName, launchProfileName, StringComparison.OrdinalIgnoreCase); } - public async ValueTask WriteAudioAsync(AudioChunk chunk, CancellationToken cancellationToken) + public async ValueTask RouteCapturedAudioAsync(AudioChunk chunk, CancellationToken cancellationToken) { await PipelineGate.WaitAsync(cancellationToken); try { - AppendAudio(chunk); - await Pipeline.WriteAsync(chunk, cancellationToken); + var routedChunk = IsTranscriptionPaused + ? chunk with { Pcm = new byte[chunk.Pcm.Length] } + : chunk; + await RecordedAudio.AppendAsync(routedChunk, cancellationToken); + AppendAudio(routedChunk); + await Pipeline.WriteAsync(routedChunk, cancellationToken); } finally { @@ -2219,7 +2469,9 @@ public sealed class MeetingRecordingCoordinator return new TranscriptActivitySnapshot( lastTranscriptActivityAt, lastTranscriptSegmentEnd, - transcriptActivityVersion); + transcriptActivityVersion, + isTranscriptionPaused, + transcriptionPausedAt); } } @@ -2239,6 +2491,35 @@ public sealed class MeetingRecordingCoordinator } } + public bool TryClaimInactivityStop(long expectedActivityVersion, DateTimeOffset endTime) + { + lock (transcriptActivityGate) + { + if (isTranscriptionPaused || transcriptActivityVersion != expectedActivityVersion) + { + return false; + } + + inferredEndTime = endTime; + StopCapture(); + return true; + } + } + + public bool TryClaimMaximumPauseStop(DateTimeOffset expectedPauseStartedAt) + { + lock (transcriptActivityGate) + { + if (!isTranscriptionPaused || transcriptionPausedAt != expectedPauseStartedAt) + { + return false; + } + + StopCapture(); + return true; + } + } + public void AppendAudio(AudioChunk chunk) { speakerSampleCollector.AppendAudio(chunk); @@ -2262,9 +2543,11 @@ public sealed class MeetingRecordingCoordinator return speakerSampleCollector.Snapshot(); } - public LiveIdentificationCheckpoint? CreateLiveIdentificationCheckpoint(MeetingNote meetingNote) + public LiveIdentificationCheckpoint? CreateLiveIdentificationCheckpoint( + MeetingNote meetingNote, + IReadOnlyList samples) { - var speakers = GetUnmappedSampleSpeakers(); + var speakers = GetUnmappedSampleSpeakers(samples); if (speakers.Count == 0) { return null; @@ -2272,7 +2555,11 @@ public sealed class MeetingRecordingCoordinator var checkpoint = new LiveIdentificationCheckpoint( speakers, - BuildAttendeeSignature(meetingNote)); + BuildAttendeeSignature(meetingNote), + Options.SpeakerIdentification.Resemblyzer.Enabled + ? speakers.Select(speaker => samples.Count(sample => + string.Equals(sample.Speaker, speaker, StringComparison.OrdinalIgnoreCase))).ToArray() + : []); lock (liveIdentificationGate) { return lastLiveIdentificationCheckpoint?.Matches(checkpoint) == true @@ -2330,9 +2617,9 @@ public sealed class MeetingRecordingCoordinator StringComparer.OrdinalIgnoreCase); } - private IReadOnlyList GetUnmappedSampleSpeakers() + private IReadOnlyList GetUnmappedSampleSpeakers(IReadOnlyList samples) { - return GetSpeakerSamplesSnapshot() + return samples .Select(sample => sample.Speaker) .Where(speaker => !string.IsNullOrWhiteSpace(speaker)) .Where(speaker => !string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase)) @@ -2368,7 +2655,9 @@ public sealed class MeetingRecordingCoordinator public sealed record TranscriptActivitySnapshot( DateTimeOffset LastTranscriptActivityAt, TimeSpan? LastTranscriptSegmentEnd, - long ActivityVersion); + long ActivityVersion, + bool IsTranscriptionPaused, + DateTimeOffset? TranscriptionPausedAt); private static int StateRank(AssistantContextState state) { @@ -2386,11 +2675,13 @@ public sealed class MeetingRecordingCoordinator public sealed record LiveIdentificationCheckpoint( IReadOnlyList Speakers, - string AttendeeSignature) + string AttendeeSignature, + IReadOnlyList SampleCounts) { public bool Matches(LiveIdentificationCheckpoint other) { return string.Equals(AttendeeSignature, other.AttendeeSignature, StringComparison.Ordinal) && + SampleCounts.SequenceEqual(other.SampleCounts) && Speakers.Count == other.Speakers.Count && Speakers.SequenceEqual(other.Speakers, StringComparer.OrdinalIgnoreCase); } @@ -2423,7 +2714,8 @@ public sealed record RecordingStatus( string? AssistantContextPath, string? SummaryPath, RecordingProcessState State = RecordingProcessState.Idle, - string? LaunchProfile = null); + string? LaunchProfile = null, + bool IsPaused = false); public enum RecordingProcessState { diff --git a/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs b/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs index dccb9a3..6f1226c 100644 --- a/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs +++ b/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs @@ -9,37 +9,31 @@ internal sealed class SpeakerAudioSampleCollector private readonly object gate = new(); private readonly RollingAudioBuffer audioBuffer; private readonly Dictionary> samplesBySpeaker = new(StringComparer.OrdinalIgnoreCase); + private readonly Dictionary lastAcceptedEndBySpeaker = new(StringComparer.OrdinalIgnoreCase); private readonly int maxSamplesPerSpeaker; - private readonly TimeSpan minimumUninterruptedSpeechDuration; - private readonly TimeSpan maximumSegmentGap; + private readonly TimeSpan minimumSampleSpeechDuration; + private readonly TimeSpan maximumSampleDuration; + private readonly bool requireNonOverlappingSamples; private readonly ILogger? logger; - private PendingSpeakerSpan? pendingSpan; - - public SpeakerAudioSampleCollector(TimeSpan bufferDuration, int maxSamplesPerSpeaker) - : this( - bufferDuration, - maxSamplesPerSpeaker, - TimeSpan.FromSeconds(30), - TimeSpan.FromSeconds(1), - logger: null) - { - } + private SpeakerSampleSpan? pendingSpan; public SpeakerAudioSampleCollector( TimeSpan bufferDuration, int maxSamplesPerSpeaker, - TimeSpan minimumUninterruptedSpeechDuration, - TimeSpan maximumSegmentGap, - ILogger? logger = null) + TimeSpan minimumSampleSpeechDuration, + TimeSpan maximumSampleDuration, + ILogger? logger = null, + bool requireNonOverlappingSamples = false) { + SpeakerSampleDurationConfiguration.ValidateOrThrow( + minimumSampleSpeechDuration, + maximumSampleDuration, + "Speaker sample collector configuration is invalid"); audioBuffer = new RollingAudioBuffer(bufferDuration); this.maxSamplesPerSpeaker = Math.Max(1, maxSamplesPerSpeaker); - this.minimumUninterruptedSpeechDuration = minimumUninterruptedSpeechDuration > TimeSpan.Zero - ? minimumUninterruptedSpeechDuration - : TimeSpan.Zero; - this.maximumSegmentGap = maximumSegmentGap >= TimeSpan.Zero - ? maximumSegmentGap - : TimeSpan.Zero; + this.minimumSampleSpeechDuration = minimumSampleSpeechDuration; + this.maximumSampleDuration = maximumSampleDuration; + this.requireNonOverlappingSamples = requireNonOverlappingSamples; this.logger = logger; } @@ -53,6 +47,7 @@ internal sealed class SpeakerAudioSampleCollector lock (gate) { samplesBySpeaker.Clear(); + lastAcceptedEndBySpeaker.Clear(); pendingSpan = null; audioBuffer.Reset(); } @@ -68,34 +63,54 @@ internal sealed class SpeakerAudioSampleCollector return null; } - TranscriptionSegment sampleSegment; - PendingSpanReset? reset; + SpeakerSampleSpan? sampleSpan; + SpeakerSampleSpan? previousSpan; lock (gate) { - (sampleSegment, reset) = ExtendPendingSpan(segment); + if (requireNonOverlappingSamples && + lastAcceptedEndBySpeaker.TryGetValue(segment.Speaker, out var lastAcceptedEnd) && + segment.Start < lastAcceptedEnd) + { + logger?.LogInformation( + "Discarding speaker identity sample for {Speaker} because it overlaps an accepted sample ending at {AcceptedSampleEnd}", + segment.Speaker, + lastAcceptedEnd); + return null; + } + + (sampleSpan, previousSpan) = ExtendPendingSpan(segment); } - if (reset is not null) + if (sampleSpan is null) { logger?.LogInformation( - "Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}, gap {Gap}, maximum gap {MaximumGap}", - reset.PreviousSpeaker, - segment.Speaker, - reset.PreviousEnd, - segment.Start, - reset.Gap, - maximumSegmentGap); + "Discarding speaker identity sample for {Speaker} because the segment duration is not positive", + segment.Speaker); + return null; } - var score = Score(sampleSegment, minimumUninterruptedSpeechDuration); + var sampleSegment = sampleSpan.ToSegment(); + + if (previousSpan is not null) + { + logger?.LogInformation( + "Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}", + previousSpan.Speaker, + segment.Speaker, + previousSpan.End, + segment.Start); + } + + var score = Score(sampleSegment, sampleSpan.SpeechDuration, minimumSampleSpeechDuration); if (!score.Accepted) { logger?.LogInformation( - "Discarding speaker identity sample for {Speaker} because {Reason}: duration {Duration}, minimum duration {MinimumDuration}, word count {WordCount}", + "Discarding speaker identity sample for {Speaker} because {Reason}: speaker audio {SpeakerAudioDuration}, clip duration {ClipDuration}, minimum duration {MinimumDuration}, word count {WordCount}", sampleSegment.Speaker, score.Reason, + sampleSpan.SpeechDuration, sampleSegment.End - sampleSegment.Start, - minimumUninterruptedSpeechDuration, + minimumSampleSpeechDuration, score.WordCount); return null; } @@ -115,6 +130,13 @@ internal sealed class SpeakerAudioSampleCollector var sample = new SpeakerAudioSample(sampleSegment.Speaker, sampleSegment, wavBytes, score.Value); lock (gate) { + if (requireNonOverlappingSamples && + ReferenceEquals(pendingSpan, sampleSpan)) + { + pendingSpan = null; + lastAcceptedEndBySpeaker[sampleSegment.Speaker] = sampleSegment.End; + } + if (!samplesBySpeaker.TryGetValue(sampleSegment.Speaker, out var samples)) { samples = []; @@ -163,35 +185,26 @@ internal sealed class SpeakerAudioSampleCollector !string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase); } - private (TranscriptionSegment Segment, PendingSpanReset? Reset) ExtendPendingSpan(TranscriptionSegment segment) + private (SpeakerSampleSpan? Span, SpeakerSampleSpan? PreviousSpan) ExtendPendingSpan(TranscriptionSegment segment) { - if (pendingSpan is null || - !SpeakerSampleSpanSelector.CanExtend(pendingSpan.Speaker, pendingSpan.End, segment, maximumSegmentGap)) + if (pendingSpan is not null && pendingSpan.TryExtend(segment, out var extended)) { - var reset = pendingSpan is null - ? null - : new PendingSpanReset( - pendingSpan.Speaker, - pendingSpan.End, - segment.Start - pendingSpan.End); - pendingSpan = new PendingSpeakerSpan( - segment.Speaker, - segment.Start, - segment.End, - [segment.Text]); - return (pendingSpan.ToSegment(), reset); + pendingSpan = extended; + return (pendingSpan, null); } - pendingSpan = pendingSpan.Extend(segment); - return (pendingSpan.ToSegment(), null); + var previousSpan = pendingSpan; + pendingSpan = SpeakerSampleSpan.Create(segment, maximumSampleDuration); + return (pendingSpan, previousSpan); } private static SampleScore Score( TranscriptionSegment segment, - TimeSpan minimumUninterruptedSpeechDuration) + TimeSpan speechDuration, + TimeSpan minimumSampleSpeechDuration) { - var durationSeconds = (segment.End - segment.Start).TotalSeconds; - if (durationSeconds < minimumUninterruptedSpeechDuration.TotalSeconds) + var durationSeconds = speechDuration.TotalSeconds; + if (durationSeconds < minimumSampleSpeechDuration.TotalSeconds) { return new SampleScore(false, 0, "speech duration is below the configured minimum", WordCount(segment.Text)); } @@ -202,7 +215,7 @@ internal sealed class SpeakerAudioSampleCollector return new SampleScore(false, 0, "word count is below the minimum useful sample length", words); } - var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumUninterruptedSpeechDuration.TotalSeconds), 2); + var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumSampleSpeechDuration.TotalSeconds), 2); var wordScore = Math.Min(words / 60.0, 1); var sentenceBonus = segment.Text.TrimEnd().EndsWith('.') || segment.Text.TrimEnd().EndsWith('?') || @@ -221,30 +234,4 @@ internal sealed class SpeakerAudioSampleCollector private sealed record SampleScore(bool Accepted, double Value, string? Reason, int WordCount); - private sealed record PendingSpanReset(string PreviousSpeaker, TimeSpan PreviousEnd, TimeSpan Gap); - - private sealed record PendingSpeakerSpan( - string Speaker, - TimeSpan Start, - TimeSpan End, - IReadOnlyList TextParts) - { - public PendingSpeakerSpan Extend(TranscriptionSegment segment) - { - return this with - { - End = segment.End > End ? segment.End : End, - TextParts = TextParts.Append(segment.Text).ToList() - }; - } - - public TranscriptionSegment ToSegment() - { - return new TranscriptionSegment( - Start, - End, - Speaker, - string.Join(' ', TextParts.Where(part => !string.IsNullOrWhiteSpace(part)))); - } - } } diff --git a/MeetingAssistant/Recording/WindowsMeetingInactivityPromptService.Windows.cs b/MeetingAssistant/Recording/WindowsMeetingInactivityPromptService.Windows.cs index 3cef4c6..79a2904 100644 --- a/MeetingAssistant/Recording/WindowsMeetingInactivityPromptService.Windows.cs +++ b/MeetingAssistant/Recording/WindowsMeetingInactivityPromptService.Windows.cs @@ -11,7 +11,7 @@ public sealed class WindowsMeetingInactivityPromptService : IMeetingInactivityPr private const string NotificationGroup = "meeting-inactivity"; private readonly ConcurrentDictionary pendingPrompts = new(StringComparer.OrdinalIgnoreCase); private readonly ILogger logger; - private readonly object registrationGate = new(); + private readonly object notificationGate = new(); private bool registered; public WindowsMeetingInactivityPromptService(ILogger logger) @@ -37,24 +37,27 @@ public sealed class WindowsMeetingInactivityPromptService : IMeetingInactivityPr try { - EnsureRegistered(); - var promptId = Guid.NewGuid().ToString("N"); - pendingPrompts[promptId] = new PendingPrompt(handleResponseAsync); - logger.LogInformation( - "Registered native Windows inactivity notification {PromptId} response callback at threshold {Threshold}", - promptId, - request.Threshold); - var notification = BuildNotification(promptId, request); - notification.Show(toast => + lock (notificationGate) { - toast.Group = NotificationGroup; - toast.Tag = promptId; - toast.ExpirationTime = MeetingToastExpirationPolicy.StopReminderExpiration(DateTimeOffset.Now); - }); - logger.LogInformation( - "Displayed native Windows inactivity notification {PromptId} at threshold {Threshold}", - promptId, - request.Threshold); + EnsureRegistered(); + var promptId = Guid.NewGuid().ToString("N"); + pendingPrompts[promptId] = new PendingPrompt(handleResponseAsync); + logger.LogInformation( + "Registered native Windows inactivity notification {PromptId} response callback at threshold {Threshold}", + promptId, + request.Threshold); + var notification = BuildNotification(promptId, request); + notification.Show(toast => + { + toast.Group = NotificationGroup; + toast.Tag = promptId; + toast.ExpirationTime = MeetingToastExpirationPolicy.StopReminderExpiration(DateTimeOffset.Now); + }); + logger.LogInformation( + "Displayed native Windows inactivity notification {PromptId} at threshold {Threshold}", + promptId, + request.Threshold); + } } catch (Exception exception) { @@ -76,6 +79,36 @@ public sealed class WindowsMeetingInactivityPromptService : IMeetingInactivityPr return Task.CompletedTask; } + public Task DismissAllAsync(CancellationToken cancellationToken) + { + cancellationToken.ThrowIfCancellationRequested(); + lock (notificationGate) + { + if (pendingPrompts.IsEmpty) + { + return Task.CompletedTask; + } + + pendingPrompts.Clear(); + if (!OperatingSystem.IsWindowsVersionAtLeast(10, 0, 17763)) + { + return Task.CompletedTask; + } + + try + { + ToastNotificationManagerCompat.History.RemoveGroup(NotificationGroup); + logger.LogInformation("Dismissed all native Windows inactivity notifications"); + } + catch (Exception exception) + { + logger.LogWarning(exception, "Failed to dismiss native Windows inactivity notifications"); + } + + return Task.CompletedTask; + } + } + public Task StopAsync(CancellationToken cancellationToken) { return Task.CompletedTask; @@ -83,29 +116,28 @@ public sealed class WindowsMeetingInactivityPromptService : IMeetingInactivityPr public void Dispose() { - if (!registered) + lock (notificationGate) { - return; - } + if (!registered) + { + return; + } - try - { - ToastNotificationManagerCompat.OnActivated -= OnNotificationInvoked; - } - catch (Exception exception) - { - logger.LogWarning(exception, "Failed to detach native Windows toast notification activation handler"); + try + { + ToastNotificationManagerCompat.OnActivated -= OnNotificationInvoked; + registered = false; + } + catch (Exception exception) + { + logger.LogWarning(exception, "Failed to detach native Windows toast notification activation handler"); + } } } private void EnsureRegistered() { - if (registered) - { - return; - } - - lock (registrationGate) + lock (notificationGate) { if (registered) { @@ -122,29 +154,25 @@ public sealed class WindowsMeetingInactivityPromptService : IMeetingInactivityPr string promptId, MeetingInactivityPromptRequest request) { - var yesButton = new ToastButton() - .SetContent("Yes") - .AddArgument("source", NotificationSource) - .AddArgument("promptId", promptId) - .AddArgument("response", "stop") - .SetBackgroundActivation(); - - var noButton = new ToastButton() - .SetContent("No") - .AddArgument("source", NotificationSource) - .AddArgument("promptId", promptId) - .AddArgument("response", "continue") - .SetBackgroundActivation(); - - return new ToastContentBuilder() + var builder = new ToastContentBuilder() .AddArgument("source", NotificationSource) .AddArgument("promptId", promptId) .SetToastScenario(ToastScenario.Reminder) .SetToastDuration(ToastDuration.Long) .AddText("Stop meeting?") - .AddText($"No transcript text has arrived for {FormatDuration(request.InactivityDuration)}.") - .AddButton(yesButton) - .AddButton(noButton); + .AddText($"No transcript text has arrived for {FormatDuration(request.InactivityDuration)}."); + + foreach (var action in MeetingInactivityPromptActions.All) + { + builder.AddButton(new ToastButton() + .SetContent(action.Content) + .AddArgument("source", NotificationSource) + .AddArgument("promptId", promptId) + .AddArgument("response", action.ResponseArgument) + .SetBackgroundActivation()); + } + + return builder; } private void OnNotificationInvoked(ToastNotificationActivatedEventArgsCompat args) @@ -177,10 +205,9 @@ public sealed class WindowsMeetingInactivityPromptService : IMeetingInactivityPr return; } - var response = TryGetArgument(arguments, "response", out var responseValue) && - string.Equals(responseValue, "stop", StringComparison.OrdinalIgnoreCase) - ? MeetingInactivityPromptResponse.Stop - : MeetingInactivityPromptResponse.Continue; + var response = TryGetArgument(arguments, "response", out var responseValue) + ? MeetingInactivityPromptActions.ParseResponse(responseValue) + : MeetingInactivityPromptResponse.Continue; _ = Task.Run(async () => { try diff --git a/MeetingAssistant/Speakers/PyannoteSpeakerIdentityMatchValidator.cs b/MeetingAssistant/Speakers/PyannoteSpeakerIdentityMatchValidator.cs index 46ab137..62228a6 100644 --- a/MeetingAssistant/Speakers/PyannoteSpeakerIdentityMatchValidator.cs +++ b/MeetingAssistant/Speakers/PyannoteSpeakerIdentityMatchValidator.cs @@ -190,7 +190,7 @@ public sealed class PyannoteSpeakerIdentityMatchValidator : ISpeakerIdentityMatc return []; } - return await finalizer.FinalizeAsync( + return await finalizer.FinalizeEnabledAsync( wavPath, [new TranscriptionSegment(TimeSpan.Zero, duration, "Unknown", "speaker identity validation sample")], options.Diarization, diff --git a/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentificationService.cs b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentificationService.cs new file mode 100644 index 0000000..cafebc0 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentificationService.cs @@ -0,0 +1,631 @@ +using MeetingAssistant.MeetingNotes; +using MeetingAssistant.Transcription; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed class ResemblyzerSpeakerIdentificationService : ISpeakerIdentificationService +{ + private readonly IDbContextFactory dbContextFactory; + private readonly ISpeakerSnippetExtractor snippetExtractor; + private readonly IResemblyzerVoiceEncoder encoder; + private readonly ResemblyzerVoiceClusterMatcher clusterMatcher; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; + private readonly SpeakerIdentificationOptions options; + private readonly ResemblyzerSpeakerRecognitionOptions resemblyzerOptions; + private readonly ILogger logger; + + public ResemblyzerSpeakerIdentificationService( + IDbContextFactory dbContextFactory, + ISpeakerSnippetExtractor snippetExtractor, + IResemblyzerVoiceEncoder encoder, + ResemblyzerVoiceClusterMatcher clusterMatcher, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, + IOptions options, + ILogger logger) + { + this.dbContextFactory = dbContextFactory; + this.snippetExtractor = snippetExtractor; + this.encoder = encoder; + this.clusterMatcher = clusterMatcher; + this.outlierPruner = outlierPruner; + this.options = options.Value.SpeakerIdentification; + resemblyzerOptions = this.options.Resemblyzer; + this.logger = logger; + } + + public Task IdentifyKnownSpeakersAsync( + SpeakerIdentificationRequest request, + CancellationToken cancellationToken) + { + return ProcessTranscriptAsync(request, final: false, allowAudioFallback: false, cancellationToken); + } + + public Task IdentifyFinishedSpeakersAsync( + SpeakerIdentificationRequest request, + CancellationToken cancellationToken) + { + return ProcessTranscriptAsync(request, final: false, allowAudioFallback: true, cancellationToken); + } + + public Task ProcessFinishedTranscriptAsync( + SpeakerIdentificationRequest request, + CancellationToken cancellationToken) + { + return ProcessTranscriptAsync(request, final: true, allowAudioFallback: true, cancellationToken); + } + + public async Task ApplySpeakerOverrideAsync( + SpeakerIdentificationRequest request, + string sourceSpeaker, + string targetSpeaker, + CancellationToken cancellationToken) + { + if (!options.Enabled || + string.IsNullOrWhiteSpace(sourceSpeaker) || + string.IsNullOrWhiteSpace(targetSpeaker) || + string.Equals(sourceSpeaker, targetSpeaker, StringComparison.OrdinalIgnoreCase)) + { + return; + } + + var sourceLabel = sourceSpeaker.Trim(); + var targetName = targetSpeaker.Trim(); + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var identities = await LoadIdentities(context).ToListAsync(cancellationToken); + var target = identities + .Where(identity => SpeakerIdentityNaming.GetAcceptedNames(identity).Contains(targetName)) + .OrderBy(identity => string.Equals(identity.CanonicalName, targetName, StringComparison.OrdinalIgnoreCase) ? 0 : 1) + .ThenBy(identity => identity.Id) + .FirstOrDefault(); + var reference = CreateReference(request.MeetingNote, DateTimeOffset.UtcNow); + var sourceCandidate = identities + .Where(identity => string.IsNullOrWhiteSpace(identity.CanonicalName)) + .Where(identity => identity.References.Any(existing => SpeakerIdentityReferences.IsSame(existing, reference))) + .Where(identity => SpeakerIdentityNaming.GetAcceptedNames(identity).Contains(targetName)) + .OrderBy(identity => identity.Id) + .FirstOrDefault(); + var vectors = await ResolveAvailableVectorsAsync(request, sourceLabel, cancellationToken); + if (target is null && sourceCandidate is null && vectors.Count == 0) + { + logger.LogWarning( + "Skipping Resemblyzer speaker override from {SourceSpeaker} to {TargetSpeaker} because no source evidence was available", + sourceLabel, + targetName); + return; + } + + var now = DateTimeOffset.UtcNow; + if (target is null) + { + target = sourceCandidate ?? new SpeakerIdentity + { + CreatedAt = now + }; + if (target.Id == 0) + { + context.SpeakerIdentities.Add(target); + } + } + else if (sourceCandidate is not null && sourceCandidate.Id != target.Id) + { + SpeakerIdentityMerger.MergeIntoAndPrune( + target, + sourceCandidate, + options.MaxSnippetsPerSpeaker, + resemblyzerOptions.MaxVectorsPerIdentity, + outlierPruner); + context.SpeakerIdentities.Remove(sourceCandidate); + } + + target.CanonicalName = targetName; + SpeakerIdentityNaming.SetCandidates(target, [targetName]); + SpeakerIdentityReferences.AddIfMissing(target, reference, now); + outlierPruner.AddAndPrune( + target, + vectors, + now); + target.UpdatedAt = now; + await context.SaveChangesAsync(cancellationToken); + await SpeakerIdentityTranscriptAudit.AppendIdentifiedAsync( + target.References, + sourceLabel, + targetName, + cancellationToken); + } + + public async Task DeleteSpeakerIdentityAsync( + string identity, + CancellationToken cancellationToken) + { + if (!options.Enabled || string.IsNullOrWhiteSpace(identity)) + { + return; + } + + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var identities = await LoadIdentities(context).ToListAsync(cancellationToken); + var target = identities.FirstOrDefault(candidate => + SpeakerIdentityNaming.GetAcceptedNames(candidate).Contains(identity.Trim())); + if (target is null) + { + return; + } + + context.SpeakerIdentities.Remove(target); + await context.SaveChangesAsync(cancellationToken); + } + + private async Task ProcessTranscriptAsync( + SpeakerIdentificationRequest request, + bool final, + bool allowAudioFallback, + CancellationToken cancellationToken) + { + if (!options.Enabled || request.Segments.Count == 0) + { + return EmptyResult(request.Segments); + } + + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var attendees = SpeakerIdentityNaming.NormalizeAttendees(request.MeetingNote.Frontmatter.Attendees); + var knownMappings = request.KnownSpeakerMappings ?? + new Dictionary(StringComparer.OrdinalIgnoreCase); + var knownLabels = knownMappings.Keys.ToHashSet(StringComparer.OrdinalIgnoreCase); + if (final && knownMappings.Count > 0) + { + await PersistMappedSpeakerEvidenceAsync( + context, + request, + knownMappings, + cancellationToken); + } + + var identifiedNames = knownMappings.Values + .Where(name => !string.IsNullOrWhiteSpace(name)) + .Select(name => name.Trim()) + .ToHashSet(StringComparer.OrdinalIgnoreCase); + foreach (var segmentSpeaker in request.Segments + .Select(segment => segment.Speaker) + .Where(speaker => !string.IsNullOrWhiteSpace(speaker) && !SpeakerIdentityNaming.IsDiarizedSpeakerLabel(speaker))) + { + identifiedNames.Add(segmentSpeaker.Trim()); + } + + var mappings = new Dictionary(StringComparer.OrdinalIgnoreCase); + var attendeeMatches = new List(); + var pendingAudits = new List(); + var matchedAcceptedNames = identifiedNames.ToHashSet(StringComparer.OrdinalIgnoreCase); + var unmatchedSpeakers = new List<(string Speaker, IReadOnlyList Vectors)>(); + foreach (var speaker in request.Segments + .Select(segment => segment.Speaker) + .Where(speaker => !string.IsNullOrWhiteSpace(speaker)) + .Distinct(StringComparer.OrdinalIgnoreCase)) + { + if (knownLabels.Contains(speaker) || identifiedNames.Contains(speaker) || !SpeakerIdentityNaming.IsDiarizedSpeakerLabel(speaker)) + { + continue; + } + + var vectors = await ResolveAutomaticVectorsAsync( + request, + speaker, + allowAudioFallback, + cancellationToken); + if (vectors.Count < resemblyzerOptions.RequiredVectorsPerSpeaker) + { + logger.LogInformation( + "Resemblyzer matching waits for more vectors for {Speaker}: {VectorCount}/{RequiredVectorCount}", + speaker, + vectors.Count, + resemblyzerOptions.RequiredVectorsPerSpeaker); + continue; + } + + var (identity, decision) = await FindMatchAsync( + context, + attendees, + identifiedNames, + vectors, + cancellationToken); + if (identity is null) + { + if (final && decision.Cohesion >= resemblyzerOptions.MinimumClusterCohesion) + { + unmatchedSpeakers.Add((speaker, vectors)); + } + + continue; + } + + var now = DateTimeOffset.UtcNow; + var previousCanonicalName = identity.CanonicalName; + var previousReferenceCount = identity.References.Count; + outlierPruner.AddAndPrune( + identity, + vectors, + now); + SpeakerIdentityReferences.AddIfMissing( + identity, + CreateReference(request.MeetingNote, now), + now); + if (identity.References.Count != previousReferenceCount) + { + identity.UpdatedAt = now; + } + + if (final) + { + UpdateMatchedIdentity(identity, attendees); + if (string.IsNullOrWhiteSpace(previousCanonicalName) && + !string.IsNullOrWhiteSpace(identity.CanonicalName)) + { + pendingAudits.Add(new PendingIdentificationAudit( + identity, + speaker, + identity.CanonicalName)); + } + + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) + { + matchedAcceptedNames.Add(acceptedName); + } + } + + var displayName = identity.GetDisplayName(); + if (!string.IsNullOrWhiteSpace(displayName)) + { + mappings[speaker] = displayName; + identifiedNames.Add(displayName); + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) + { + identifiedNames.Add(acceptedName); + } + attendeeMatches.Add(new SpeakerIdentityAttendeeMatch( + displayName, + SpeakerIdentityNaming.GetAcceptedNames(identity).ToList())); + } + } + + if (final) + { + LearnUnmatchedSpeakers( + context, + request.MeetingNote, + attendees, + matchedAcceptedNames, + unmatchedSpeakers, + pendingAudits); + } + + await context.SaveChangesAsync(cancellationToken); + await AppendAuditsAsync(pendingAudits, cancellationToken); + var relabeled = request.Segments + .Select(segment => mappings.TryGetValue(segment.Speaker, out var name) + ? segment with { Speaker = name } + : segment) + .ToList(); + return new SpeakerIdentificationResult(relabeled, mappings, attendeeMatches); + } + + private async Task<(SpeakerIdentity? Identity, ResemblyzerVoiceClusterMatchResult Decision)> FindMatchAsync( + SpeakerIdentityDbContext context, + IReadOnlyList attendees, + IReadOnlySet identifiedNames, + IReadOnlyList queryVectors, + CancellationToken cancellationToken) + { + var activeCutoff = DateTimeOffset.UtcNow - options.MatchIdentityActiveAge; + var identities = await LoadIdentities(context) + .OrderByDescending(identity => identity.References.Count) + .ThenBy(identity => identity.Id) + .ToListAsync(cancellationToken); + var candidates = identities + .Select(identity => new + { + Identity = identity, + IsAttendee = SpeakerIdentityNaming.MatchesAttendees(identity, attendees), + IsActive = identity.UpdatedAt >= activeCutoff + }) + .Where(candidate => candidate.IsAttendee || candidate.IsActive) + .Where(candidate => !SpeakerIdentityNaming.MatchesAnyAcceptedName(candidate.Identity, identifiedNames)) + .Where(candidate => candidate.Identity.VoiceVectors.Any(vector => + string.Equals(vector.ModelId, resemblyzerOptions.ModelId, StringComparison.Ordinal))) + .OrderByDescending(candidate => candidate.IsAttendee) + .ThenByDescending(candidate => candidate.Identity.ReferenceCount) + .ThenBy(candidate => candidate.Identity.Id) + .Take(Math.Max(1, options.MaxMatchCandidates)) + .Select(candidate => new ResemblyzerVoiceVectorCandidate( + candidate.Identity.Id, + SpeakerVoiceVectors.DecodeCompatible( + candidate.Identity, + resemblyzerOptions.ModelId, + logger))) + .Where(candidate => candidate.Vectors.Count > 0) + .ToList(); + var match = clusterMatcher.Match(queryVectors, candidates); + return ( + match.IdentityId is { } identityId + ? identities.Single(identity => identity.Id == identityId) + : null, + match); + } + + private void LearnUnmatchedSpeakers( + SpeakerIdentityDbContext context, + MeetingNote meetingNote, + IReadOnlyList attendees, + IReadOnlySet matchedAcceptedNames, + IReadOnlyList<(string Speaker, IReadOnlyList Vectors)> unmatchedSpeakers, + ICollection pendingAudits) + { + var candidates = attendees + .Except(matchedAcceptedNames, StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .ToList(); + if (candidates.Count == 0) + { + return; + } + + foreach (var (speaker, vectors) in unmatchedSpeakers) + { + var now = DateTimeOffset.UtcNow; + var identity = new SpeakerIdentity + { + CanonicalName = candidates.Count == 1 ? candidates[0] : null, + CreatedAt = now, + UpdatedAt = now, + CandidateNames = candidates + .Select(name => new SpeakerCandidateName { Name = name }) + .ToList(), + References = [CreateReference(meetingNote, now)] + }; + outlierPruner.AddAndPrune( + identity, + vectors, + now); + context.SpeakerIdentities.Add(identity); + logger.LogInformation( + "Created Resemblyzer identity candidate for {Speaker} with {VectorCount} vector(s) and candidates {Candidates}", + speaker, + identity.VoiceVectors.Count, + string.Join(", ", candidates)); + if (!string.IsNullOrWhiteSpace(identity.CanonicalName)) + { + pendingAudits.Add(new PendingIdentificationAudit( + identity, + speaker, + identity.CanonicalName)); + } + } + } + + private async Task AppendAuditsAsync( + IEnumerable pendingAudits, + CancellationToken cancellationToken) + { + foreach (var audit in pendingAudits) + { + try + { + await SpeakerIdentityTranscriptAudit.AppendIdentifiedAsync( + audit.Identity.References, + audit.Speaker, + audit.Name, + cancellationToken); + } + catch (Exception exception) when (exception is not OperationCanceledException) + { + logger.LogError( + exception, + "Resemblyzer identity {IdentityId} was saved, but its transcript identification audit could not be written", + audit.Identity.Id); + } + } + } + + private async Task> ResolveAutomaticVectorsAsync( + SpeakerIdentificationRequest request, + string speaker, + bool allowAudioFallback, + CancellationToken cancellationToken) + { + var requiredCount = resemblyzerOptions.RequiredVectorsPerSpeaker; + var samples = await ResolveWavSamplesAsync( + request, + speaker, + resemblyzerOptions.MaxVectorsPerIdentity, + allowAudioFallback, + cancellationToken); + if (samples.Count < requiredCount) + { + return []; + } + + return await encoder.EncodeAsync(samples, cancellationToken); + } + + private async Task PersistMappedSpeakerEvidenceAsync( + SpeakerIdentityDbContext context, + SpeakerIdentificationRequest request, + IReadOnlyDictionary knownMappings, + CancellationToken cancellationToken) + { + var identities = await LoadIdentities(context).ToListAsync(cancellationToken); + foreach (var (speaker, mappedName) in knownMappings) + { + var identity = identities + .Where(candidate => SpeakerIdentityNaming.GetAcceptedNames(candidate).Contains(mappedName)) + .OrderBy(candidate => + string.Equals(candidate.CanonicalName, mappedName, StringComparison.OrdinalIgnoreCase) ? 0 : 1) + .ThenBy(candidate => candidate.Id) + .FirstOrDefault(); + if (identity is null) + { + logger.LogWarning( + "Could not retain final Resemblyzer evidence for mapped speaker {Speaker}: identity {MappedName} was not found", + speaker, + mappedName); + continue; + } + + var vectors = await ResolveAvailableVectorsAsync(request, speaker, cancellationToken); + if (vectors.Count == 0) + { + continue; + } + + var now = DateTimeOffset.UtcNow; + var previousReferenceCount = identity.References.Count; + var vectorUpdate = outlierPruner.AddAndPrune( + identity, + vectors, + now); + SpeakerIdentityReferences.AddIfMissing( + identity, + CreateReference(request.MeetingNote, now), + now); + if (identity.References.Count != previousReferenceCount) + { + identity.UpdatedAt = now; + } + + logger.LogInformation( + "Retained {AddedVectorCount} new final Resemblyzer vector(s) for mapped speaker {Speaker} as identity {IdentityId}", + vectorUpdate.AddedCount, + speaker, + identity.Id); + } + } + + private async Task> ResolveAvailableVectorsAsync( + SpeakerIdentificationRequest request, + string speaker, + CancellationToken cancellationToken) + { + var samples = await ResolveWavSamplesAsync( + request, + speaker, + resemblyzerOptions.MaxVectorsPerIdentity, + allowAudioFallback: true, + cancellationToken); + if (samples.Count == 0) + { + return []; + } + + return await encoder.EncodeAsync(samples, cancellationToken); + } + + private async Task> ResolveWavSamplesAsync( + SpeakerIdentificationRequest request, + string speaker, + int maxSamples, + bool allowAudioFallback, + CancellationToken cancellationToken) + { + var suppliedSamples = request.Samples? + .Where(sample => string.Equals(sample.Speaker, speaker, StringComparison.OrdinalIgnoreCase)) + .Where(sample => sample.WavBytes.Length > 0) + .OrderByDescending(sample => sample.Score) + .Take(maxSamples) + .ToList() + ?? []; + var wavSamples = suppliedSamples + .Select(sample => sample.WavBytes) + .ToList(); + if (!allowAudioFallback || wavSamples.Count >= maxSamples) + { + return wavSamples; + } + + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + request.Segments, + speaker, + options.MinimumSampleSpeechDuration, + options.MaximumSampleDuration, + request.Segments.Count); + foreach (var span in spans.Where(span => !OverlapsSuppliedSample(span, suppliedSamples))) + { + var wavBytes = await snippetExtractor.ExtractSnippetAsync( + request.AudioPath, + span, + cancellationToken); + if (wavBytes.Length > 0) + { + wavSamples.Add(wavBytes); + } + + if (wavSamples.Count >= maxSamples) + { + break; + } + } + + logger.LogInformation( + "Resolved {SampleCount}/{RequestedSampleCount} Resemblyzer WAV samples for {Speaker}: {SuppliedSampleCount} supplied, {ExtractedSampleCount} extracted from completed audio", + wavSamples.Count, + maxSamples, + speaker, + suppliedSamples.Count, + wavSamples.Count - suppliedSamples.Count); + return wavSamples; + } + + private static bool OverlapsSuppliedSample( + IReadOnlyList span, + IReadOnlyList suppliedSamples) + { + if (span.Count == 0) + { + return false; + } + + var start = span[0].Start; + var end = span[^1].End; + return suppliedSamples.Any(sample => sample.Segment.Start < end && sample.Segment.End > start); + } + + private static IQueryable LoadIdentities(SpeakerIdentityDbContext context) + { + return context.SpeakerIdentities + .AsSplitQuery() + .Include(identity => identity.Aliases) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.References); + } + + private void UpdateMatchedIdentity(SpeakerIdentity identity, IReadOnlyList attendees) + { + identity.UpdatedAt = DateTimeOffset.UtcNow; + if (!string.IsNullOrWhiteSpace(identity.CanonicalName) || attendees.Count == 0) + { + return; + } + + SpeakerIdentityNaming.UpdateCandidateNames(identity, attendees); + } + + private static SpeakerIdentityReference CreateReference(MeetingNote meetingNote, DateTimeOffset timestamp) + { + return SpeakerIdentityReferences.Create(meetingNote.Path, meetingNote.Frontmatter.Transcript, timestamp); + } + + private static SpeakerIdentificationResult EmptyResult(IReadOnlyList segments) + { + return new SpeakerIdentificationResult(segments, new Dictionary()); + } + + private sealed record PendingIdentificationAudit( + SpeakerIdentity Identity, + string Speaker, + string Name); + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentityMergeService.cs b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentityMergeService.cs new file mode 100644 index 0000000..b2c5d40 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentityMergeService.cs @@ -0,0 +1,172 @@ +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed class ResemblyzerSpeakerIdentityMergeService : ISpeakerIdentityMergeService +{ + private readonly IDbContextFactory dbContextFactory; + private readonly ResemblyzerVoiceClusterMatcher matcher; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; + private readonly SpeakerIdentificationOptions options; + private readonly ResemblyzerSpeakerRecognitionOptions resemblyzerOptions; + private readonly ILogger logger; + + public ResemblyzerSpeakerIdentityMergeService( + IDbContextFactory dbContextFactory, + ResemblyzerVoiceClusterMatcher matcher, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, + IOptions options, + ILogger logger) + { + this.dbContextFactory = dbContextFactory; + this.matcher = matcher; + this.outlierPruner = outlierPruner; + this.options = options.Value.SpeakerIdentification; + resemblyzerOptions = this.options.Resemblyzer; + this.logger = logger; + } + + public async Task MergeRecentIdentitiesAsync( + TimeSpan? recentIdentityAge, + CancellationToken cancellationToken) + { + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var identities = await context.SpeakerIdentities + .AsSplitQuery() + .Include(identity => identity.Aliases) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.References) + .OrderByDescending(identity => identity.References.Count) + .ThenBy(identity => identity.Id) + .ToListAsync(cancellationToken); + var cutoff = DateTimeOffset.UtcNow - (recentIdentityAge ?? options.MergeRecentIdentityAge); + var recentIds = identities + .Where(identity => identity.CreatedAt >= cutoff) + .Select(identity => identity.Id) + .ToList(); + var required = resemblyzerOptions.RequiredVectorsPerSpeaker; + var attempts = 0; + var mergedPairs = 0; + var pendingAudits = new List(); + + foreach (var sourceId in recentIds) + { + var source = identities.SingleOrDefault(identity => identity.Id == sourceId); + if (source is null) + { + continue; + } + + var sourceVectors = SpeakerVoiceVectors.DecodeCompatible( + source, + resemblyzerOptions.ModelId, + logger) + .Take(required * 2) + .ToList(); + if (sourceVectors.Count < required * 2) + { + logger.LogInformation( + "Skipping Resemblyzer merge source identity {SourceIdentityId}: {VectorCount}/{RequiredVectorCount} compatible vectors", + source.Id, + sourceVectors.Count, + required * 2); + continue; + } + + var candidates = identities + .Where(identity => identity.Id != source.Id) + .Where(identity => identity.VoiceVectors.Any(vector => + string.Equals(vector.ModelId, resemblyzerOptions.ModelId, StringComparison.Ordinal))) + .Take(Math.Max(1, options.MaxMatchCandidates)) + .Select(identity => new ResemblyzerVoiceVectorCandidate( + identity.Id, + SpeakerVoiceVectors.DecodeCompatible( + identity, + resemblyzerOptions.ModelId, + logger))) + .Where(candidate => candidate.Vectors.Count > 0) + .ToList(); + if (candidates.Count == 0) + { + continue; + } + + attempts++; + var first = matcher.Match(sourceVectors.Take(required).ToList(), candidates); + if (!first.Accepted || first.IdentityId is not { } targetId) + { + continue; + } + + attempts++; + var second = matcher.Match(sourceVectors.Skip(required).Take(required).ToList(), candidates); + if (!second.Accepted || second.IdentityId != targetId) + { + logger.LogInformation( + "Rejected Resemblyzer merge for source identity {SourceIdentityId}: disjoint clusters selected {FirstIdentityId} and {SecondIdentityId}", + source.Id, + first.IdentityId, + second.IdentityId); + continue; + } + + var target = identities.SingleOrDefault(identity => identity.Id == targetId); + if (target is null) + { + continue; + } + + var targetName = target.GetDisplayName() ?? $"identity-{target.Id}"; + var sourceName = source.GetDisplayName() ?? $"identity-{source.Id}"; + SpeakerIdentityMerger.MergeIntoAndPrune( + target, + source, + options.MaxSnippetsPerSpeaker, + resemblyzerOptions.MaxVectorsPerIdentity, + outlierPruner); + pendingAudits.Add(new PendingMergeAudit( + target, + targetName, + sourceName)); + context.SpeakerIdentities.Remove(source); + identities.Remove(source); + mergedPairs++; + } + + await context.SaveChangesAsync(cancellationToken); + foreach (var audit in pendingAudits) + { + try + { + await SpeakerIdentityTranscriptAudit.AppendMergedAsync( + audit.Identity.References, + audit.TargetName, + audit.SourceName, + cancellationToken); + } + catch (Exception exception) when (exception is not OperationCanceledException) + { + logger.LogError( + exception, + "Resemblyzer identity merge was saved for target {IdentityId}, but its transcript audit could not be written", + audit.Identity.Id); + } + } + + return new SpeakerIdentityMergeResult( + recentIds.Count, + identities.Count, + attempts, + mergedPairs); + } + + private sealed record PendingMergeAudit( + SpeakerIdentity Identity, + string TargetName, + string SourceName); + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerVoiceClusterMatcher.cs b/MeetingAssistant/Speakers/ResemblyzerVoiceClusterMatcher.cs new file mode 100644 index 0000000..a5450c0 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerVoiceClusterMatcher.cs @@ -0,0 +1,204 @@ +namespace MeetingAssistant.Speakers; + +public sealed record ResemblyzerVoiceVectorCandidate( + int IdentityId, + IReadOnlyList Vectors); + +public sealed record ResemblyzerVoiceClusterMatchResult( + int? IdentityId, + string Reason, + double Cohesion, + double? BestSimilarity, + double? RunnerUpSimilarity) +{ + public bool Accepted => IdentityId.HasValue; +} + +public sealed class ResemblyzerVoiceClusterMatcher +{ + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + + public ResemblyzerVoiceClusterMatcher( + ResemblyzerSpeakerRecognitionOptions options, + ILogger logger) + { + this.options = options; + this.logger = logger; + } + + public ResemblyzerVoiceClusterMatchResult Match( + IReadOnlyList queryVectors, + IReadOnlyList candidates) + { + var requiredCount = options.RequiredVectorsPerSpeaker; + if (queryVectors.Count < requiredCount) + { + return Reject( + $"insufficient query vectors ({queryVectors.Count}/{requiredCount})", + cohesion: 0, + bestSimilarity: null, + runnerUpSimilarity: null); + } + + var normalizedQuery = queryVectors + .Take(requiredCount) + .Select(vector => SpeakerVoiceVectors.Normalize(vector, "Voice vector")) + .ToList(); + var cohesion = MeanPairwiseCosine(normalizedQuery); + if (cohesion < options.MinimumClusterCohesion) + { + return Reject( + $"query cohesion {cohesion:F4} is below {options.MinimumClusterCohesion:F4}", + cohesion, + bestSimilarity: null, + runnerUpSimilarity: null); + } + + var scored = candidates + .Select(candidate => ScoreCandidate(normalizedQuery, candidate)) + .Where(candidate => candidate is not null) + .Select(candidate => candidate!.Value) + .OrderByDescending(candidate => candidate.Similarity) + .ThenBy(candidate => candidate.IdentityId) + .ToList(); + if (scored.Count == 0) + { + return Reject("no compatible identity vectors were available", cohesion, null, null); + } + + var best = scored[0]; + var runnerUp = scored.Count > 1 ? scored[1].Similarity : (double?)null; + if (best.Similarity < options.MinimumIdentitySimilarity) + { + return Reject( + $"best similarity {best.Similarity:F4} is below {options.MinimumIdentitySimilarity:F4}", + cohesion, + best.Similarity, + runnerUp); + } + + if (runnerUp is { } runnerUpSimilarity && + best.Similarity - runnerUpSimilarity < options.MinimumSimilarityMargin) + { + return Reject( + $"similarity margin {best.Similarity - runnerUpSimilarity:F4} is below {options.MinimumSimilarityMargin:F4}", + cohesion, + best.Similarity, + runnerUpSimilarity); + } + + logger.LogInformation( + "Resemblyzer cluster accepted identity {IdentityId}: cohesion {Cohesion:F4}, similarity {Similarity:F4}, runner-up {RunnerUpSimilarity}", + best.IdentityId, + cohesion, + best.Similarity, + runnerUp); + return new ResemblyzerVoiceClusterMatchResult( + best.IdentityId, + "accepted", + cohesion, + best.Similarity, + runnerUp); + } + + private ResemblyzerVoiceClusterMatchResult Reject( + string reason, + double cohesion, + double? bestSimilarity, + double? runnerUpSimilarity) + { + logger.LogInformation( + "Resemblyzer cluster rejected: {Reason}; cohesion {Cohesion:F4}, best {BestSimilarity}, runner-up {RunnerUpSimilarity}", + reason, + cohesion, + bestSimilarity, + runnerUpSimilarity); + return new ResemblyzerVoiceClusterMatchResult( + null, + reason, + cohesion, + bestSimilarity, + runnerUpSimilarity); + } + + private (int IdentityId, double Similarity)? ScoreCandidate( + IReadOnlyList normalizedQuery, + ResemblyzerVoiceVectorCandidate candidate) + { + if (candidate.Vectors.Count == 0) + { + return null; + } + + try + { + var normalizedCandidate = candidate.Vectors + .Select(vector => SpeakerVoiceVectors.Normalize(vector, "Voice vector")) + .ToList(); + var centroid = SpeakerVoiceVectors.Normalize(Centroid(normalizedCandidate), "Voice-vector centroid"); + var similarities = normalizedQuery + .Select(vector => SpeakerVoiceVectors.Cosine(vector, centroid)) + .Order() + .ToList(); + return (candidate.IdentityId, Median(similarities)); + } + catch (InvalidDataException exception) + { + logger.LogWarning( + exception, + "Skipping invalid Resemblyzer evidence for identity {IdentityId}", + candidate.IdentityId); + return null; + } + } + + private static float[] Centroid(IReadOnlyList vectors) + { + var centroid = new float[ResemblyzerVectorContract.Dimensions]; + foreach (var vector in vectors) + { + for (var index = 0; index < centroid.Length; index++) + { + centroid[index] += vector[index]; + } + } + + for (var index = 0; index < centroid.Length; index++) + { + centroid[index] /= vectors.Count; + } + + return centroid; + } + + private static double MeanPairwiseCosine(IReadOnlyList vectors) + { + if (vectors.Count < 2) + { + return 1; + } + + double total = 0; + var pairs = 0; + for (var first = 0; first < vectors.Count - 1; first++) + { + for (var second = first + 1; second < vectors.Count; second++) + { + total += SpeakerVoiceVectors.Cosine(vectors[first], vectors[second]); + pairs++; + } + } + + return total / pairs; + } + + private static double Median(IReadOnlyList sortedValues) + { + var middle = sortedValues.Count / 2; + return sortedValues.Count % 2 == 0 + ? (sortedValues[middle - 1] + sortedValues[middle]) / 2 + : sortedValues[middle]; + } + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerVoiceVectorOutlierPruner.cs b/MeetingAssistant/Speakers/ResemblyzerVoiceVectorOutlierPruner.cs new file mode 100644 index 0000000..e8102db --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerVoiceVectorOutlierPruner.cs @@ -0,0 +1,208 @@ +namespace MeetingAssistant.Speakers; + +public sealed record ResemblyzerVoiceVectorUpdateResult( + int AddedCount, + int RemovedCount) +{ + public bool Changed => AddedCount > 0 || RemovedCount > 0; +} + +public sealed class ResemblyzerVoiceVectorOutlierPruner +{ + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + + public ResemblyzerVoiceVectorOutlierPruner( + ResemblyzerSpeakerRecognitionOptions options, + ILogger logger) + { + this.options = options; + this.logger = logger; + } + + public int Prune(SpeakerIdentity identity) + { + var compatible = new List(); + foreach (var entry in SpeakerVoiceVectors.DecodeCompatibleEntries( + identity, + options.ModelId, + logger)) + { + try + { + compatible.Add(new DecodedSpeakerVoiceVector( + entry.Stored, + SpeakerVoiceVectors.Normalize( + entry.Vector, + $"Stored voice vector {entry.Stored.Id}"))); + } + catch (InvalidDataException exception) + { + logger.LogWarning( + exception, + "Preserving invalid stored voice vector {VectorId} while pruning identity {IdentityId}", + entry.Stored.Id, + identity.Id); + } + } + var minimumVectors = options.OutlierPruningMinimumVectors; + if (compatible.Count < minimumVectors) + { + return 0; + } + + var clusters = FindDensityClusters(compatible.Select(item => item.Vector).ToList()); + var ranked = clusters + .GroupBy(cluster => cluster) + .Where(group => group.Key > 0) + .Select(group => new { Id = group.Key, Count = group.Count() }) + .OrderByDescending(group => group.Count) + .ThenBy(group => group.Id) + .ToList(); + if (ranked.Count == 0) + { + return Skip(identity, "no dense cluster was found"); + } + + if (ranked.Count > 1 && ranked[0].Count == ranked[1].Count) + { + return Skip(identity, "no uniquely largest dense cluster was found"); + } + + var dominant = ranked[0]; + var ratio = (double)dominant.Count / compatible.Count; + if (ratio < options.OutlierPruningMinimumClusterRatio) + { + return Skip( + identity, + $"largest dense cluster ratio {ratio:F4} is below {options.OutlierPruningMinimumClusterRatio:F4}"); + } + + var removed = compatible + .Where((_, index) => clusters[index] != dominant.Id) + .Select(item => item.Stored) + .ToList(); + foreach (var vector in removed) + { + identity.VoiceVectors.Remove(vector); + } + + if (removed.Count > 0) + { + identity.UpdatedAt = DateTimeOffset.UtcNow; + } + + logger.LogInformation( + "Pruned {RemovedCount} Resemblyzer voice-vector outliers from identity {IdentityId}; retained dominant cluster of {RetainedCount}/{CompatibleCount} vectors", + removed.Count, + identity.Id, + dominant.Count, + compatible.Count); + return removed.Count; + } + + public ResemblyzerVoiceVectorUpdateResult AddAndPrune( + SpeakerIdentity identity, + IReadOnlyList vectors, + DateTimeOffset createdAt) + { + var removedBeforeAdding = Prune(identity); + var added = SpeakerVoiceVectors.AddDistinct( + identity, + vectors, + options.ModelId, + options.MaxVectorsPerIdentity, + createdAt); + var removedAfterAdding = Prune(identity); + var result = new ResemblyzerVoiceVectorUpdateResult( + added, + removedBeforeAdding + removedAfterAdding); + if (result.Changed) + { + identity.UpdatedAt = createdAt; + } + + return result; + } + + private int[] FindDensityClusters(IReadOnlyList vectors) + { + var neighbors = Enumerable.Range(0, vectors.Count) + .Select(index => Enumerable.Range(0, vectors.Count) + .Where(candidate => SpeakerVoiceVectors.Cosine(vectors[index], vectors[candidate]) >= options.OutlierPruningNeighborSimilarity) + .ToList()) + .ToList(); + var minimumNeighbors = options.OutlierPruningMinimumNeighbors; + var labels = new int[vectors.Count]; + var clusterId = 0; + + for (var point = 0; point < vectors.Count; point++) + { + if (labels[point] != 0) + { + continue; + } + + if (neighbors[point].Count < minimumNeighbors) + { + labels[point] = -1; + continue; + } + + clusterId++; + ExpandCluster(point, clusterId, neighbors, labels, minimumNeighbors); + } + + return labels; + } + + private static void ExpandCluster( + int seed, + int clusterId, + IReadOnlyList> neighbors, + int[] labels, + int minimumNeighbors) + { + labels[seed] = clusterId; + var pending = new Queue(neighbors[seed]); + var queued = neighbors[seed].ToHashSet(); + while (pending.TryDequeue(out var point)) + { + if (labels[point] == -1) + { + labels[point] = clusterId; + } + + if (labels[point] != 0) + { + continue; + } + + labels[point] = clusterId; + if (neighbors[point].Count < minimumNeighbors) + { + continue; + } + + foreach (var neighbor in neighbors[point]) + { + if (queued.Add(neighbor)) + { + pending.Enqueue(neighbor); + } + } + } + } + + private int Skip( + SpeakerIdentity identity, + string reason) + { + logger.LogInformation( + "Skipped Resemblyzer voice-vector pruning for identity {IdentityId}: {Reason}", + identity.Id, + reason); + return 0; + } + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerWarmupHostedService.cs b/MeetingAssistant/Speakers/ResemblyzerWarmupHostedService.cs new file mode 100644 index 0000000..69607f9 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerWarmupHostedService.cs @@ -0,0 +1,54 @@ +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed class ResemblyzerWarmupHostedService : BackgroundService +{ + private readonly IResemblyzerVoiceEncoder encoder; + private readonly bool speakerIdentificationEnabled; + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + public ResemblyzerWarmupHostedService( + IResemblyzerVoiceEncoder encoder, + IOptions options, + ILogger logger) + { + this.encoder = encoder; + speakerIdentificationEnabled = options.Value.SpeakerIdentification.Enabled; + this.options = options.Value.SpeakerIdentification.Resemblyzer; + this.logger = logger; + } + + protected override async Task ExecuteAsync(CancellationToken stoppingToken) + { + if (!speakerIdentificationEnabled || !options.Enabled) + { + return; + } + + await Task.Yield(); + try + { + logger.LogInformation( + "Starting Resemblyzer warm-up in runtime folder {RuntimeFolder} for model {ModelId}", + VaultPath.Resolve(options.RuntimeFolder), + options.ModelId); + await encoder.WarmUpAsync(stoppingToken); + logger.LogInformation( + "Finished Resemblyzer warm-up in runtime folder {RuntimeFolder} for model {ModelId}", + VaultPath.Resolve(options.RuntimeFolder), + options.ModelId); + } + catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) + { + logger.LogInformation("Resemblyzer warm-up was cancelled during application shutdown"); + } + catch (Exception exception) + { + logger.LogError( + exception, + "Resemblyzer warm-up failed in runtime folder {RuntimeFolder}; voice encoding can still retry on demand", + VaultPath.Resolve(options.RuntimeFolder)); + } + } +} diff --git a/MeetingAssistant/Speakers/SpeakerIdentity.cs b/MeetingAssistant/Speakers/SpeakerIdentity.cs index 554ecd6..653f7e9 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentity.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentity.cs @@ -21,6 +21,8 @@ public sealed class SpeakerIdentity public List Snippets { get; set; } = []; + public List VoiceVectors { get; set; } = []; + public List References { get; set; } = []; [NotMapped] @@ -93,7 +95,7 @@ public static class SpeakerIdentityReferences string.IsNullOrWhiteSpace(reference.TranscriptPath); } - private static bool IsSame(SpeakerIdentityReference first, SpeakerIdentityReference second) + public static bool IsSame(SpeakerIdentityReference first, SpeakerIdentityReference second) { return string.Equals(first.MeetingNotePath, second.MeetingNotePath, StringComparison.OrdinalIgnoreCase) && string.Equals(first.TranscriptPath, second.TranscriptPath, StringComparison.OrdinalIgnoreCase); @@ -124,6 +126,25 @@ public sealed class SpeakerSnippet public DateTimeOffset CreatedAt { get; set; } } +public sealed class SpeakerVoiceVector +{ + public int Id { get; set; } + + public int SpeakerIdentityId { get; set; } + + public SpeakerIdentity? SpeakerIdentity { get; set; } + + public string ModelId { get; set; } = ""; + + public int Dimensions { get; set; } + + public byte[] VectorBytes { get; set; } = []; + + public string Fingerprint { get; set; } = ""; + + public DateTimeOffset CreatedAt { get; set; } +} + public sealed class SpeakerAlias { public int Id { get; set; } @@ -150,6 +171,8 @@ public sealed class SpeakerIdentityDbContext : DbContext public DbSet SpeakerSnippets => Set(); + public DbSet SpeakerVoiceVectors => Set(); + public DbSet SpeakerIdentityReferences => Set(); protected override void OnModelCreating(ModelBuilder modelBuilder) @@ -175,6 +198,11 @@ public sealed class SpeakerIdentityDbContext : DbContext .HasForeignKey(snippet => snippet.SpeakerIdentityId) .OnDelete(DeleteBehavior.Cascade); + entity.HasMany(identity => identity.VoiceVectors) + .WithOne(vector => vector.SpeakerIdentity) + .HasForeignKey(vector => vector.SpeakerIdentityId) + .OnDelete(DeleteBehavior.Cascade); + entity.HasMany(identity => identity.References) .WithOne(reference => reference.SpeakerIdentity) .HasForeignKey(reference => reference.SpeakerIdentityId) @@ -189,6 +217,13 @@ public sealed class SpeakerIdentityDbContext : DbContext .HasIndex(alias => new { alias.SpeakerIdentityId, alias.Name }) .IsUnique(); + modelBuilder.Entity() + .HasIndex(vector => new { vector.SpeakerIdentityId, vector.Fingerprint }) + .IsUnique(); + + modelBuilder.Entity() + .HasIndex(vector => new { vector.SpeakerIdentityId, vector.ModelId }); + modelBuilder.Entity() .HasIndex(reference => new { reference.SpeakerIdentityId, reference.MeetingNotePath, reference.TranscriptPath }) .IsUnique(); diff --git a/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs b/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs index 187cefc..539ae29 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs @@ -20,17 +20,20 @@ public sealed class SpeakerIdentityMergeService : ISpeakerIdentityMergeService { private readonly IDbContextFactory dbContextFactory; private readonly ISpeakerIdentityMatcher matcher; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; private readonly SpeakerIdentificationOptions options; private readonly ILogger logger; public SpeakerIdentityMergeService( IDbContextFactory dbContextFactory, ISpeakerIdentityMatcher matcher, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, IOptions options, ILogger logger) { this.dbContextFactory = dbContextFactory; this.matcher = matcher; + this.outlierPruner = outlierPruner; this.options = options.Value.SpeakerIdentification; this.logger = logger; } @@ -44,9 +47,11 @@ public sealed class SpeakerIdentityMergeService : ISpeakerIdentityMergeService var cutoff = DateTimeOffset.UtcNow - (recentIdentityAge ?? options.MergeRecentIdentityAge); var identities = await context.SpeakerIdentities + .AsSplitQuery() .Include(identity => identity.Aliases) .Include(identity => identity.CandidateNames) .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) .Include(identity => identity.References) .OrderByDescending(identity => identity.References.Count) .ThenBy(identity => identity.Id) @@ -154,10 +159,12 @@ public sealed class SpeakerIdentityMergeService : ISpeakerIdentityMergeService var targetName = target.GetDisplayName() ?? $"identity-{target.Id}"; var sourceName = source.GetDisplayName() ?? $"identity-{source.Id}"; - SpeakerIdentityMerger.MergeInto( + SpeakerIdentityMerger.MergeIntoAndPrune( target, source, - options.MaxSnippetsPerSpeaker); + options.MaxSnippetsPerSpeaker, + options.Resemblyzer.MaxVectorsPerIdentity, + outlierPruner); logger.LogInformation( "Speaker identity merge diagnostics merging source identity {SourceIdentityId} ({SourceName}) into target identity {TargetIdentityId} ({TargetName})", source.Id, diff --git a/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs b/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs index 640055c..81d8c0d 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs @@ -2,10 +2,24 @@ namespace MeetingAssistant.Speakers; internal static class SpeakerIdentityMerger { + public static void MergeIntoAndPrune( + SpeakerIdentity target, + SpeakerIdentity source, + int maxSnippets, + int maxVoiceVectors, + ResemblyzerVoiceVectorOutlierPruner outlierPruner) + { + outlierPruner.Prune(target); + outlierPruner.Prune(source); + MergeInto(target, source, maxSnippets, maxVoiceVectors); + outlierPruner.Prune(target); + } + public static void MergeInto( SpeakerIdentity target, SpeakerIdentity source, - int maxSnippets) + int maxSnippets, + int maxVoiceVectors = int.MaxValue) { AddAlias(target, source.CanonicalName); foreach (var alias in source.Aliases) @@ -35,6 +49,26 @@ internal static class SpeakerIdentityMerger .ToList(); target.Snippets.Clear(); target.Snippets.AddRange(retainedSnippets); + + var retainedVectors = target.VoiceVectors + .Concat(source.VoiceVectors) + .GroupBy(vector => vector.Fingerprint, StringComparer.Ordinal) + .Select(group => group.OrderByDescending(vector => vector.CreatedAt).First()) + .OrderByDescending(vector => vector.CreatedAt) + .ThenBy(vector => vector.Fingerprint, StringComparer.Ordinal) + .Take(Math.Max(1, maxVoiceVectors)) + .Select(vector => new SpeakerVoiceVector + { + SpeakerIdentity = target, + ModelId = vector.ModelId, + Dimensions = vector.Dimensions, + VectorBytes = vector.VectorBytes.ToArray(), + Fingerprint = vector.Fingerprint, + CreatedAt = vector.CreatedAt + }) + .ToList(); + target.VoiceVectors.Clear(); + target.VoiceVectors.AddRange(retainedVectors); target.UpdatedAt = DateTimeOffset.UtcNow; } diff --git a/MeetingAssistant/Speakers/SpeakerIdentityNaming.cs b/MeetingAssistant/Speakers/SpeakerIdentityNaming.cs new file mode 100644 index 0000000..dce3718 --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerIdentityNaming.cs @@ -0,0 +1,103 @@ +using MeetingAssistant.MeetingNotes; + +namespace MeetingAssistant.Speakers; + +internal static class SpeakerIdentityNaming +{ + public static IReadOnlySet GetAcceptedNames(SpeakerIdentity identity) + { + return new[] { identity.CanonicalName } + .Concat(identity.Aliases.Select(alias => alias.Name)) + .Concat(identity.CandidateNames.Select(candidate => candidate.Name)) + .Where(name => !string.IsNullOrWhiteSpace(name)) + .Select(name => name!.Trim()) + .ToHashSet(StringComparer.OrdinalIgnoreCase); + } + + public static bool MatchesAnyAcceptedName( + SpeakerIdentity identity, + IReadOnlySet names) + { + return GetAcceptedNames(identity).Any(names.Contains); + } + + public static bool MatchesAttendees( + SpeakerIdentity identity, + IReadOnlyList attendees) + { + var attendeeSet = attendees.ToHashSet(StringComparer.OrdinalIgnoreCase); + return GetAcceptedNames(identity).Any(attendeeSet.Contains); + } + + public static IReadOnlyList NormalizeAttendees(IEnumerable attendees) + { + return attendees + .Select(MeetingAttendeeNames.NormalizeDisplayName) + .Where(attendee => !string.IsNullOrWhiteSpace(attendee)) + .Distinct(StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .ToList(); + } + + public static bool IsDiarizedSpeakerLabel(string speaker) + { + var normalized = speaker.Trim(); + return normalized.Equals("Unknown", StringComparison.OrdinalIgnoreCase) || + normalized.StartsWith("Guest", StringComparison.OrdinalIgnoreCase) || + normalized.StartsWith("Speaker", StringComparison.OrdinalIgnoreCase); + } + + public static bool UpdateCandidateNames( + SpeakerIdentity identity, + IReadOnlyList attendees) + { + if (!string.IsNullOrWhiteSpace(identity.CanonicalName) || attendees.Count == 0) + { + return false; + } + + var currentCandidates = identity.CandidateNames + .Select(candidate => candidate.Name) + .ToHashSet(StringComparer.OrdinalIgnoreCase); + var fallbackAliasCandidate = currentCandidates.Count == 1 ? currentCandidates.Single() : null; + var aliasToCandidate = identity.Aliases + .Where(alias => !string.IsNullOrWhiteSpace(alias.Name)) + .SelectMany(alias => currentCandidates.Select(candidate => new { Alias = alias.Name, Candidate = candidate })) + .Where(pair => string.Equals(pair.Alias, pair.Candidate, StringComparison.OrdinalIgnoreCase) || + pair.Alias.Contains(pair.Candidate, StringComparison.OrdinalIgnoreCase) || + pair.Candidate.Contains(pair.Alias, StringComparison.OrdinalIgnoreCase)) + .ToDictionary(pair => pair.Alias, pair => pair.Candidate, StringComparer.OrdinalIgnoreCase); + var intersection = attendees + .Select(attendee => currentCandidates.Contains(attendee) + ? attendee + : aliasToCandidate.GetValueOrDefault(attendee) ?? + (identity.Aliases.Any(alias => string.Equals(alias.Name, attendee, StringComparison.OrdinalIgnoreCase)) + ? fallbackAliasCandidate + : null)) + .Where(candidate => !string.IsNullOrWhiteSpace(candidate)) + .Select(candidate => candidate!) + .Distinct(StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .ToList(); + var resetToAttendees = intersection.Count == 0; + SetCandidates(identity, resetToAttendees ? attendees : intersection); + if (intersection.Count == 1) + { + identity.CanonicalName = intersection[0]; + } + + return resetToAttendees; + } + + public static void SetCandidates( + SpeakerIdentity identity, + IEnumerable candidates) + { + identity.CandidateNames.Clear(); + identity.CandidateNames.AddRange(candidates + .Where(candidate => !string.IsNullOrWhiteSpace(candidate)) + .Distinct(StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .Select(candidate => new SpeakerCandidateName { Name = candidate })); + } +} diff --git a/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs b/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs index 34d4e47..aab3c15 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs @@ -46,6 +46,33 @@ internal static class SpeakerIdentitySchema ON "SpeakerIdentityReferences" ("SpeakerIdentityId", "MeetingNotePath", "TranscriptPath"); """, cancellationToken); + await context.Database.ExecuteSqlRawAsync( + """ + CREATE TABLE IF NOT EXISTS "SpeakerVoiceVectors" ( + "Id" INTEGER NOT NULL CONSTRAINT "PK_SpeakerVoiceVectors" PRIMARY KEY AUTOINCREMENT, + "SpeakerIdentityId" INTEGER NOT NULL, + "ModelId" TEXT NOT NULL, + "Dimensions" INTEGER NOT NULL, + "VectorBytes" BLOB NOT NULL, + "Fingerprint" TEXT NOT NULL, + "CreatedAt" TEXT NOT NULL, + CONSTRAINT "FK_SpeakerVoiceVectors_SpeakerIdentities_SpeakerIdentityId" + FOREIGN KEY ("SpeakerIdentityId") REFERENCES "SpeakerIdentities" ("Id") ON DELETE CASCADE + ); + """, + cancellationToken); + await context.Database.ExecuteSqlRawAsync( + """ + CREATE UNIQUE INDEX IF NOT EXISTS "IX_SpeakerVoiceVectors_SpeakerIdentityId_Fingerprint" + ON "SpeakerVoiceVectors" ("SpeakerIdentityId", "Fingerprint"); + """, + cancellationToken); + await context.Database.ExecuteSqlRawAsync( + """ + CREATE INDEX IF NOT EXISTS "IX_SpeakerVoiceVectors_SpeakerIdentityId_ModelId" + ON "SpeakerVoiceVectors" ("SpeakerIdentityId", "ModelId"); + """, + cancellationToken); } private static async Task EnsureSpeakerIdentityTimestampColumnsAsync( diff --git a/MeetingAssistant/Speakers/SpeakerIdentityService.cs b/MeetingAssistant/Speakers/SpeakerIdentityService.cs index 37e2500..de316d9 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentityService.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentityService.cs @@ -143,7 +143,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService target.CanonicalName = targetName; target.UpdatedAt = now; - ResetCandidates(target, [targetName]); + SpeakerIdentityNaming.SetCandidates(target, [targetName]); AddMeetingReference(target, meetingReference); var snippetAdded = AddSnippetIfNeeded(target, snippet); await context.SaveChangesAsync(cancellationToken); @@ -207,7 +207,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); - var attendees = NormalizeAttendees(request.MeetingNote.Frontmatter.Attendees); + var attendees = SpeakerIdentityNaming.NormalizeAttendees(request.MeetingNote.Frontmatter.Attendees); var meetingReference = CreateReference(request.MeetingNote, DateTimeOffset.UtcNow); var speakerMappings = new Dictionary(StringComparer.OrdinalIgnoreCase); var attendeeMatches = new List(); @@ -222,7 +222,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .ToHashSet(StringComparer.OrdinalIgnoreCase); foreach (var speaker in request.Segments .Select(segment => segment.Speaker) - .Where(speaker => !string.IsNullOrWhiteSpace(speaker) && !IsDiarizedSpeakerLabel(speaker))) + .Where(speaker => !string.IsNullOrWhiteSpace(speaker) && !SpeakerIdentityNaming.IsDiarizedSpeakerLabel(speaker))) { alreadyIdentifiedNames.Add(speaker.Trim()); } @@ -360,7 +360,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService cancellationToken); } - foreach (var acceptedName in GetAcceptedNames(identity)) + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) { matchedAcceptedNames.Add(acceptedName); } @@ -371,14 +371,14 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService { speakerMappings[speaker] = speakerName; alreadyIdentifiedNames.Add(speakerName); - foreach (var acceptedName in GetAcceptedNames(identity)) + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) { alreadyIdentifiedNames.Add(acceptedName); } attendeeMatches.Add(new SpeakerIdentityAttendeeMatch( speakerName, - GetAcceptedNames(identity).ToList())); + SpeakerIdentityNaming.GetAcceptedNames(identity).ToList())); } } @@ -440,7 +440,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .Select(identity => new { Identity = identity, - IsAttendee = MatchesAttendees(identity, attendees), + IsAttendee = SpeakerIdentityNaming.MatchesAttendees(identity, attendees), IsActive = identity.UpdatedAt >= activeCutoff }) .Where(candidate => candidate.IsAttendee || candidate.IsActive) @@ -541,11 +541,11 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService string operation, CancellationToken cancellationToken) { - var span = SpeakerSampleSpanSelector.SelectBestContinuousSpan( + var span = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpan( request.Segments, speaker, - options.MaximumSampleSegmentGap, - options.MinimumSampleSpeechDuration); + options.MinimumSampleSpeechDuration, + options.MaximumSampleDuration); logger.LogInformation( "{Operation} extracting fallback sample for {Speaker}: selected {SegmentCount} segment(s), span {SpanDuration}, minimum {MinimumDuration}", operation, @@ -569,7 +569,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .Include(identity => identity.References) .ToListAsync(cancellationToken); return identities - .Where(identity => GetAcceptedNames(identity).Contains(name)) + .Where(identity => SpeakerIdentityNaming.GetAcceptedNames(identity).Contains(name)) .OrderBy(identity => string.Equals(identity.CanonicalName, name, StringComparison.OrdinalIgnoreCase) ? 0 : 1) .ThenBy(identity => identity.Id) .FirstOrDefault(); @@ -589,7 +589,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .Where(identity => string.IsNullOrWhiteSpace(identity.CanonicalName)) .ToListAsync(cancellationToken); return candidates - .Where(identity => identity.References.Any(existing => IsSameReference(existing, reference))) + .Where(identity => identity.References.Any(existing => SpeakerIdentityReferences.IsSame(existing, reference))) .Where(identity => identity.CandidateNames.Any(candidate => string.Equals(candidate.Name, targetName, StringComparison.OrdinalIgnoreCase)) || identity.Aliases.Any(alias => @@ -646,27 +646,11 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService identity.Aliases.Add(new SpeakerAlias { Name = alias.Trim() }); } - private static bool IsSameReference( - SpeakerIdentityReference first, - SpeakerIdentityReference second) - { - return string.Equals(first.MeetingNotePath, second.MeetingNotePath, StringComparison.OrdinalIgnoreCase) && - string.Equals(first.TranscriptPath, second.TranscriptPath, StringComparison.OrdinalIgnoreCase); - } - private static bool MatchesAcceptedNames( SpeakerIdentity identity, IReadOnlySet names) { - return GetAcceptedNames(identity).Any(names.Contains); - } - - private static bool MatchesAttendees( - SpeakerIdentity identity, - IReadOnlyList attendees) - { - var attendeeSet = attendees.ToHashSet(StringComparer.OrdinalIgnoreCase); - return GetAcceptedNames(identity).Any(attendeeSet.Contains); + return SpeakerIdentityNaming.MatchesAnyAcceptedName(identity, names); } private static Task LoadIdentityAsync( @@ -694,48 +678,25 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService var currentCandidates = identity.CandidateNames .Select(candidate => candidate.Name) .ToHashSet(StringComparer.OrdinalIgnoreCase); - var fallbackAliasCandidate = currentCandidates.Count == 1 ? currentCandidates.Single() : null; - var aliasToCandidate = identity.Aliases - .Where(alias => !string.IsNullOrWhiteSpace(alias.Name)) - .SelectMany(alias => currentCandidates.Select(candidate => new { Alias = alias.Name, Candidate = candidate })) - .Where(pair => string.Equals(pair.Alias, pair.Candidate, StringComparison.OrdinalIgnoreCase) || - pair.Alias.Contains(pair.Candidate, StringComparison.OrdinalIgnoreCase) || - pair.Candidate.Contains(pair.Alias, StringComparison.OrdinalIgnoreCase)) - .ToDictionary(pair => pair.Alias, pair => pair.Candidate, StringComparer.OrdinalIgnoreCase); - var intersection = attendees - .Select(attendee => currentCandidates.Contains(attendee) - ? attendee - : aliasToCandidate.GetValueOrDefault(attendee) ?? - (identity.Aliases.Any(alias => string.Equals(alias.Name, attendee, StringComparison.OrdinalIgnoreCase)) - ? fallbackAliasCandidate - : null)) - .Where(candidate => !string.IsNullOrWhiteSpace(candidate)) - .Select(candidate => candidate!) - .Distinct(StringComparer.OrdinalIgnoreCase) - .Order(StringComparer.OrdinalIgnoreCase) - .ToList(); - - if (intersection.Count == 0) + var resetToAttendees = SpeakerIdentityNaming.UpdateCandidateNames(identity, attendees); + if (resetToAttendees) { logger.LogInformation( "Speaker identity candidate elimination for identity {IdentityId} had empty intersection; resetting candidates to attendees {Attendees} and replacing oldest snippet", identity.Id, FormatNames(attendees)); - ResetCandidates(identity, attendees); ReplaceOldestSnippet(identity, snippet); return; } logger.LogInformation( - "Speaker identity candidate elimination for identity {IdentityId}: candidates {CurrentCandidates}, attendees {Attendees}, intersection {Intersection}", + "Speaker identity candidate elimination for identity {IdentityId}: candidates {CurrentCandidates}, attendees {Attendees}, remaining {RemainingCandidates}", identity.Id, FormatNames(currentCandidates), FormatNames(attendees), - FormatNames(intersection)); - ResetCandidates(identity, intersection); - if (intersection.Count == 1) + FormatNames(identity.CandidateNames.Select(candidate => candidate.Name))); + if (!string.IsNullOrWhiteSpace(identity.CanonicalName)) { - identity.CanonicalName = intersection[0]; logger.LogInformation( "Speaker identity candidate elimination promoted identity {IdentityId} to canonical name {CanonicalName}", identity.Id, @@ -872,51 +833,6 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService } } - private static void ResetCandidates(SpeakerIdentity identity, IReadOnlyList candidates) - { - identity.CandidateNames.Clear(); - identity.CandidateNames.AddRange(candidates - .Distinct(StringComparer.OrdinalIgnoreCase) - .Order(StringComparer.OrdinalIgnoreCase) - .Select(candidate => new SpeakerCandidateName { Name = candidate })); - } - - private static IReadOnlySet GetAcceptedNames(SpeakerIdentity identity) - { - return new[] - { - identity.CanonicalName - } - .Concat(identity.Aliases.Select(alias => alias.Name)) - .Concat(identity.CandidateNames.Select(candidate => candidate.Name)) - .Where(name => !string.IsNullOrWhiteSpace(name)) - .Select(name => name!.Trim()) - .ToHashSet(StringComparer.OrdinalIgnoreCase); - } - - private static IReadOnlyList NormalizeAttendees(IEnumerable attendees) - { - return attendees - .Select(NormalizeAttendee) - .Where(attendee => !string.IsNullOrWhiteSpace(attendee)) - .Distinct(StringComparer.OrdinalIgnoreCase) - .Order(StringComparer.OrdinalIgnoreCase) - .ToList(); - } - - private static string NormalizeAttendee(string attendee) - { - return MeetingAttendeeNames.NormalizeDisplayName(attendee); - } - - private static bool IsDiarizedSpeakerLabel(string speaker) - { - var normalized = speaker.Trim(); - return normalized.Equals("Unknown", StringComparison.OrdinalIgnoreCase) || - normalized.StartsWith("Guest", StringComparison.OrdinalIgnoreCase) || - normalized.StartsWith("Speaker", StringComparison.OrdinalIgnoreCase); - } - private static SpeakerIdentityReference CreateReference(MeetingNote meetingNote, DateTimeOffset timestamp) { return SpeakerIdentityReferences.Create( diff --git a/MeetingAssistant/Speakers/SpeakerSampleCollectionPolicy.cs b/MeetingAssistant/Speakers/SpeakerSampleCollectionPolicy.cs new file mode 100644 index 0000000..1968f43 --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerSampleCollectionPolicy.cs @@ -0,0 +1,154 @@ +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed record SpeakerSampleCollectionPolicy( + int MinimumRetainedSamples, + bool RequireNonOverlappingSamples) +{ + public static SpeakerSampleCollectionPolicy ExistingBackend { get; } = new(1, false); + + public static SpeakerSampleCollectionPolicy IndependentVectors(int maximumVectors) + { + return new SpeakerSampleCollectionPolicy(Math.Max(1, maximumVectors), true); + } + + public int ResolveRetainedSampleLimit(int configuredLimit) + { + return Math.Max(Math.Max(1, configuredLimit), MinimumRetainedSamples); + } +} + +internal static class SpeakerSampleDurationConfiguration +{ + public static void ValidateOrThrow( + SpeakerIdentificationOptions options, + string context) + { + ValidateOrThrow( + options.MinimumSampleSpeechDuration, + options.MaximumSampleDuration, + context); + } + + public static void ValidateOrThrow( + TimeSpan minimumDuration, + TimeSpan maximumDuration, + string context) + { + var error = GetError(minimumDuration, maximumDuration); + if (error is not null) + { + throw new InvalidOperationException($"{context}: {error}"); + } + } + + public static string? GetError( + TimeSpan minimumDuration, + TimeSpan maximumDuration) + { + if (minimumDuration < TimeSpan.Zero) + { + return "SpeakerIdentification:MinimumSampleSpeechDuration must not be negative."; + } + + if (maximumDuration <= TimeSpan.Zero) + { + return "SpeakerIdentification:MaximumSampleDuration must be greater than zero."; + } + + return minimumDuration > maximumDuration + ? "SpeakerIdentification:MinimumSampleSpeechDuration must not exceed SpeakerIdentification:MaximumSampleDuration." + : null; + } +} + +internal static class ResemblyzerSpeakerRecognitionConfiguration +{ + public static void ValidateOrThrow( + ResemblyzerSpeakerRecognitionOptions options, + string context) + { + var error = GetError(options); + if (error is not null) + { + throw new InvalidOperationException($"{context}: {error}"); + } + } + + public static string? GetError(ResemblyzerSpeakerRecognitionOptions options) + { + if (options.RequiredVectorsPerSpeaker < 1) + { + return "SpeakerIdentification:Resemblyzer:RequiredVectorsPerSpeaker must be at least one."; + } + + if (options.MaxVectorsPerIdentity < options.RequiredVectorsPerSpeaker) + { + return "SpeakerIdentification:Resemblyzer:MaxVectorsPerIdentity must not be less than RequiredVectorsPerSpeaker."; + } + + if (options.OutlierPruningMinimumVectors < options.RequiredVectorsPerSpeaker) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumVectors must not be less than RequiredVectorsPerSpeaker."; + } + + if (options.OutlierPruningMinimumVectors > options.MaxVectorsPerIdentity) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumVectors must not exceed MaxVectorsPerIdentity."; + } + + if (options.OutlierPruningMinimumNeighbors < 1 || + options.OutlierPruningMinimumNeighbors > options.OutlierPruningMinimumVectors) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumNeighbors must be between one and OutlierPruningMinimumVectors."; + } + + if (!IsCosine(options.OutlierPruningNeighborSimilarity)) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningNeighborSimilarity must be finite and between -1 and 1."; + } + + if (!double.IsFinite(options.OutlierPruningMinimumClusterRatio) || + options.OutlierPruningMinimumClusterRatio is <= 0 or > 1) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumClusterRatio must be finite, greater than zero, and at most one."; + } + + if (!IsCosine(options.MinimumClusterCohesion)) + { + return "SpeakerIdentification:Resemblyzer:MinimumClusterCohesion must be finite and between -1 and 1."; + } + + if (!IsCosine(options.MinimumIdentitySimilarity)) + { + return "SpeakerIdentification:Resemblyzer:MinimumIdentitySimilarity must be finite and between -1 and 1."; + } + + return !double.IsFinite(options.MinimumSimilarityMargin) || + options.MinimumSimilarityMargin is < 0 or > 2 + ? "SpeakerIdentification:Resemblyzer:MinimumSimilarityMargin must be finite and between 0 and 2." + : null; + } + + private static bool IsCosine(double value) + { + return double.IsFinite(value) && value is >= -1 and <= 1; + } +} + +internal sealed class MeetingAssistantSpeakerSampleOptionsValidator + : IValidateOptions +{ + public ValidateOptionsResult Validate(string? name, MeetingAssistantOptions options) + { + var error = SpeakerSampleDurationConfiguration.GetError( + options.SpeakerIdentification.MinimumSampleSpeechDuration, + options.SpeakerIdentification.MaximumSampleDuration) ?? + ResemblyzerSpeakerRecognitionConfiguration.GetError( + options.SpeakerIdentification.Resemblyzer); + return error is null + ? ValidateOptionsResult.Success + : ValidateOptionsResult.Fail(error); + } +} diff --git a/MeetingAssistant/Speakers/SpeakerSampleSpan.cs b/MeetingAssistant/Speakers/SpeakerSampleSpan.cs new file mode 100644 index 0000000..b859c7c --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerSampleSpan.cs @@ -0,0 +1,82 @@ +using MeetingAssistant.Transcription; + +namespace MeetingAssistant.Speakers; + +internal sealed record SpeakerSampleSpan( + string Speaker, + TimeSpan Start, + TimeSpan End, + TimeSpan SpeechDuration, + TimeSpan MaximumDuration, + IReadOnlyList Segments) +{ + public static SpeakerSampleSpan? Create( + TranscriptionSegment segment, + TimeSpan maximumDuration) + { + if (maximumDuration <= TimeSpan.Zero) + { + throw new ArgumentOutOfRangeException( + nameof(maximumDuration), + maximumDuration, + "Speaker sample maximum duration must be greater than zero."); + } + + var boundedEnd = Min(segment.End, segment.Start + maximumDuration); + return boundedEnd > segment.Start + ? new SpeakerSampleSpan( + segment.Speaker, + segment.Start, + boundedEnd, + boundedEnd - segment.Start, + maximumDuration, + [segment with { End = boundedEnd }]) + : null; + } + + public bool TryExtend( + TranscriptionSegment segment, + out SpeakerSampleSpan extended) + { + extended = this; + if (!string.Equals(Speaker, segment.Speaker, StringComparison.OrdinalIgnoreCase) || + segment.Start >= Start + MaximumDuration) + { + return false; + } + + var boundedEnd = Min(segment.End, Start + MaximumDuration); + if (boundedEnd <= segment.Start) + { + return false; + } + + var uncoveredStart = segment.Start > End ? segment.Start : End; + var additionalSpeechDuration = boundedEnd > uncoveredStart + ? boundedEnd - uncoveredStart + : TimeSpan.Zero; + extended = this with + { + End = boundedEnd > End ? boundedEnd : End, + SpeechDuration = SpeechDuration + additionalSpeechDuration, + Segments = Segments.Append(segment with { End = boundedEnd }).ToList() + }; + return true; + } + + public TranscriptionSegment ToSegment() + { + return new TranscriptionSegment( + Start, + End, + Speaker, + string.Join(' ', Segments + .Select(segment => segment.Text) + .Where(text => !string.IsNullOrWhiteSpace(text)))); + } + + private static TimeSpan Min(TimeSpan left, TimeSpan right) + { + return left < right ? left : right; + } +} diff --git a/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs b/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs index 5e3bf52..4f16538 100644 --- a/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs +++ b/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs @@ -4,56 +4,84 @@ namespace MeetingAssistant.Speakers; internal static class SpeakerSampleSpanSelector { - public static bool CanExtend( - string currentSpeaker, - TimeSpan currentEnd, - TranscriptionSegment nextSegment, - TimeSpan maximumSegmentGap) - { - return string.Equals(currentSpeaker, nextSegment.Speaker, StringComparison.OrdinalIgnoreCase) && - nextSegment.Start - currentEnd <= maximumSegmentGap; - } - - public static IReadOnlyList SelectBestContinuousSpan( + public static IReadOnlyList SelectBestSameSpeakerSpan( IReadOnlyList segments, string speaker, - TimeSpan maximumSegmentGap, - TimeSpan minimumDuration) + TimeSpan minimumDuration, + TimeSpan maximumSampleDuration) { - var best = new List(); - var current = new List(); - + SpeakerSampleSpan? best = null; + SpeakerSampleSpan? current = null; foreach (var segment in segments.OrderBy(segment => segment.Start)) { - if (current.Count == 0) + if (!IsSpeaker(segment, speaker)) { - if (IsSpeaker(segment, speaker)) - { - current.Add(segment); - best = LongerSpan(current, best); - } - + current = null; continue; } - if (!CanExtend(speaker, current[^1].End, segment, maximumSegmentGap)) + current = ExtendOrStart(current, segment, maximumSampleDuration); + if (current is not null && + (best is null || current.SpeechDuration > best.SpeechDuration)) { - current.Clear(); - if (!IsSpeaker(segment, speaker)) - { - continue; - } + best = current; } - - current.Add(segment); - best = LongerSpan(current, best); } - return SpanDuration(best) >= minimumDuration - ? best + return best is not null && best.SpeechDuration >= minimumDuration + ? best.Segments : []; } + public static IReadOnlyList> SelectBestSameSpeakerSpans( + IReadOnlyList segments, + string speaker, + TimeSpan minimumDuration, + TimeSpan maximumSampleDuration, + int maxSpans) + { + if (maxSpans <= 0) + { + return []; + } + + var completed = new List(); + SpeakerSampleSpan? current = null; + TimeSpan? lastCompletedEnd = null; + foreach (var segment in segments.OrderBy(segment => segment.Start)) + { + if (!IsSpeaker(segment, speaker)) + { + current = null; + continue; + } + + if (lastCompletedEnd is { } end && segment.Start < end) + { + current = null; + continue; + } + + current = ExtendOrStart(current, segment, maximumSampleDuration); + if (current is null || current.SpeechDuration < minimumDuration) + { + continue; + } + + completed.Add(current); + lastCompletedEnd = current.End; + current = null; + } + + return completed + .OrderByDescending(span => span.SpeechDuration) + .ThenBy(span => span.Start) + .Take(maxSpans) + .OrderBy(span => span.Start) + .Select(span => span.Segments) + .ToList(); + } + public static TimeSpan SpanDuration(IReadOnlyList segments) { return segments.Count == 0 @@ -61,19 +89,20 @@ internal static class SpeakerSampleSpanSelector : segments[^1].End - segments[0].Start; } + private static SpeakerSampleSpan? ExtendOrStart( + SpeakerSampleSpan? current, + TranscriptionSegment segment, + TimeSpan maximumSampleDuration) + { + return current is not null && current.TryExtend(segment, out var extended) + ? extended + : SpeakerSampleSpan.Create(segment, maximumSampleDuration); + } + private static bool IsSpeaker( TranscriptionSegment segment, string speaker) { return string.Equals(segment.Speaker, speaker, StringComparison.OrdinalIgnoreCase); } - - private static List LongerSpan( - List current, - List best) - { - return SpanDuration(current) > SpanDuration(best) - ? current.ToList() - : best; - } } diff --git a/MeetingAssistant/Speakers/SpeakerVoiceVectors.cs b/MeetingAssistant/Speakers/SpeakerVoiceVectors.cs new file mode 100644 index 0000000..9735e39 --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerVoiceVectors.cs @@ -0,0 +1,191 @@ +using System.Buffers.Binary; +using System.Security.Cryptography; +using System.Text; + +namespace MeetingAssistant.Speakers; + +internal sealed record DecodedSpeakerVoiceVector( + SpeakerVoiceVector Stored, + float[] Vector); + +internal static class ResemblyzerVectorContract +{ + public const int Dimensions = 256; +} + +internal static class SpeakerVoiceVectors +{ + public static float[] Decode(SpeakerVoiceVector stored) + { + if (stored.Dimensions != ResemblyzerVectorContract.Dimensions || + stored.VectorBytes.Length != stored.Dimensions * sizeof(float)) + { + throw new InvalidDataException( + $"Stored voice vector {stored.Id} has invalid dimension or byte length metadata."); + } + + var vector = new float[stored.Dimensions]; + for (var index = 0; index < vector.Length; index++) + { + vector[index] = BinaryPrimitives.ReadSingleLittleEndian( + stored.VectorBytes.AsSpan(index * sizeof(float), sizeof(float))); + if (!float.IsFinite(vector[index])) + { + throw new InvalidDataException($"Stored voice vector {stored.Id} contains a non-finite value."); + } + } + + return vector; + } + + public static int AddDistinct( + SpeakerIdentity identity, + IEnumerable vectors, + string modelId, + int maximumCount, + DateTimeOffset createdAt) + { + var limit = Math.Max(1, maximumCount); + var fingerprints = identity.VoiceVectors + .Select(vector => vector.Fingerprint) + .ToHashSet(StringComparer.Ordinal); + var added = 0; + foreach (var vector in vectors) + { + if (identity.VoiceVectors.Count >= limit) + { + break; + } + + var bytes = Encode(vector); + var fingerprint = Fingerprint(modelId, bytes); + if (!fingerprints.Add(fingerprint)) + { + continue; + } + + identity.VoiceVectors.Add(new SpeakerVoiceVector + { + ModelId = modelId, + Dimensions = vector.Length, + VectorBytes = bytes, + Fingerprint = fingerprint, + CreatedAt = createdAt + }); + added++; + } + + return added; + } + + public static IReadOnlyList DecodeCompatible( + SpeakerIdentity identity, + string modelId, + ILogger logger) + { + return DecodeCompatibleEntries(identity, modelId, logger) + .Select(entry => entry.Vector) + .ToList(); + } + + public static IReadOnlyList DecodeCompatibleEntries( + SpeakerIdentity identity, + string modelId, + ILogger logger) + { + var vectors = new List(); + foreach (var stored in identity.VoiceVectors + .Where(vector => string.Equals(vector.ModelId, modelId, StringComparison.Ordinal)) + .OrderBy(vector => vector.CreatedAt) + .ThenBy(vector => vector.Id)) + { + try + { + vectors.Add(new DecodedSpeakerVoiceVector(stored, Decode(stored))); + } + catch (InvalidDataException exception) + { + logger.LogWarning( + exception, + "Ignoring invalid stored voice vector {VectorId} for identity {IdentityId}", + stored.Id, + identity.Id); + } + } + + return vectors; + } + + public static double Cosine(float[] first, float[] second) + { + double dot = 0; + for (var index = 0; index < first.Length; index++) + { + dot += first[index] * second[index]; + } + + return dot; + } + + public static float[] Normalize(float[]? vector, string description) + { + if (vector is null || vector.Length != ResemblyzerVectorContract.Dimensions) + { + throw new InvalidDataException( + $"{description} has {vector?.Length ?? 0} dimensions; expected {ResemblyzerVectorContract.Dimensions}."); + } + + double magnitudeSquared = 0; + foreach (var value in vector) + { + if (!float.IsFinite(value)) + { + throw new InvalidDataException($"{description} contains a non-finite value."); + } + + magnitudeSquared += value * value; + } + + var magnitude = Math.Sqrt(magnitudeSquared); + if (magnitude <= double.Epsilon) + { + throw new InvalidDataException($"{description} has zero magnitude."); + } + + return vector.Select(value => (float)(value / magnitude)).ToArray(); + } + + private static byte[] Encode(float[] vector) + { + if (vector.Length != ResemblyzerVectorContract.Dimensions) + { + throw new InvalidDataException( + $"Voice vector has {vector.Length} dimensions; expected {ResemblyzerVectorContract.Dimensions}."); + } + + var bytes = new byte[vector.Length * sizeof(float)]; + for (var index = 0; index < vector.Length; index++) + { + if (!float.IsFinite(vector[index])) + { + throw new InvalidDataException("Voice vector contains a non-finite value."); + } + + BinaryPrimitives.WriteSingleLittleEndian( + bytes.AsSpan(index * sizeof(float), sizeof(float)), + vector[index]); + } + + return bytes; + } + + private static string Fingerprint(string modelId, byte[] vectorBytes) + { + var modelBytes = Encoding.UTF8.GetBytes(modelId); + var input = new byte[modelBytes.Length + 1 + vectorBytes.Length]; + modelBytes.CopyTo(input, 0); + input[modelBytes.Length] = 0; + vectorBytes.CopyTo(input, modelBytes.Length + 1); + return Convert.ToHexString(SHA256.HashData(input)); + } +} diff --git a/MeetingAssistant/Speakers/VenvResemblyzerVoiceEncoder.cs b/MeetingAssistant/Speakers/VenvResemblyzerVoiceEncoder.cs new file mode 100644 index 0000000..81159cf --- /dev/null +++ b/MeetingAssistant/Speakers/VenvResemblyzerVoiceEncoder.cs @@ -0,0 +1,391 @@ +using System.Globalization; +using System.Security.Cryptography; +using System.Text; +using System.Text.Json; +using System.Text.RegularExpressions; +using MeetingAssistant.Transcription; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public interface IResemblyzerVoiceEncoder +{ + Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken); + + Task WarmUpAsync(CancellationToken cancellationToken); +} + +public sealed partial class VenvResemblyzerVoiceEncoder : IResemblyzerVoiceEncoder +{ + private const string EnvironmentSchemaVersion = "venv-v2"; + private const string JsonStart = "__MEETING_ASSISTANT_RESEMBLYZER_JSON_START__"; + private const string JsonEnd = "__MEETING_ASSISTANT_RESEMBLYZER_JSON_END__"; + private const string NumpyBeforePython313Requirement = "numpy<2; python_version < '3.13'"; + private const string NumpyFromPython313Requirement = "numpy>=2,<3; python_version >= '3.13'"; + private const string LibrosaRequirement = "librosa>=0.9.1"; + private const string ScipyRequirement = "scipy>=1.2.1"; + private readonly SemaphoreSlim commandLock = new(1, 1); + private readonly ICommandRunner commandRunner; + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + private string? verifiedEnvironmentPythonPath; + + public VenvResemblyzerVoiceEncoder( + ICommandRunner commandRunner, + IOptions options, + ILogger logger) + { + this.commandRunner = commandRunner; + this.options = options.Value.SpeakerIdentification.Resemblyzer; + this.logger = logger; + } + + public async Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + if (wavSamples.Count == 0) + { + return []; + } + + if (wavSamples.Any(sample => sample.Length == 0)) + { + throw new InvalidDataException("Resemblyzer cannot encode an empty WAV sample."); + } + + var runtimeFolder = VaultPath.Resolve(options.RuntimeFolder); + var inputFolder = Path.Combine(runtimeFolder, "input", Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(inputFolder); + try + { + for (var index = 0; index < wavSamples.Count; index++) + { + await File.WriteAllBytesAsync( + Path.Combine(inputFolder, $"{index:D4}.wav"), + wavSamples[index], + cancellationToken); + } + + await commandLock.WaitAsync(cancellationToken); + try + { + return await RunEncodingAsync(inputFolder, wavSamples.Count, cancellationToken); + } + finally + { + commandLock.Release(); + } + } + finally + { + try + { + Directory.Delete(inputFolder, recursive: true); + } + catch (DirectoryNotFoundException) + { + } + catch (IOException exception) + { + logger.LogWarning(exception, "Could not remove Resemblyzer temporary input folder {InputFolder}", inputFolder); + } + catch (UnauthorizedAccessException exception) + { + logger.LogWarning(exception, "Could not remove Resemblyzer temporary input folder {InputFolder}", inputFolder); + } + } + } + + private async Task> RunEncodingAsync( + string inputFolder, + int expectedCount, + CancellationToken cancellationToken) + { + return await RunWithTimeoutAsync(async token => + { + var pythonPath = await EnsureEnvironmentAsync(token); + var scriptPath = Path.Combine(VaultPath.Resolve(options.RuntimeFolder), "encode.py"); + await File.WriteAllTextAsync(scriptPath, BuildEncodingScript(), token); + var result = await RunRequiredAsync( + pythonPath, + [scriptPath, Path.GetFullPath(inputFolder)], + "encoding", + token); + return ParseAndValidateVectors(result.StandardOutput, expectedCount); + }, cancellationToken); + } + + public async Task WarmUpAsync(CancellationToken cancellationToken) + { + await commandLock.WaitAsync(cancellationToken); + try + { + await RunWithTimeoutAsync(async token => + { + await EnsureEnvironmentAsync(token); + return true; + }, cancellationToken); + } + finally + { + commandLock.Release(); + } + } + + private async Task EnsureEnvironmentAsync(CancellationToken cancellationToken) + { + ValidateDependencySettings(); + var runtimeFolder = VaultPath.Resolve(options.RuntimeFolder); + var environmentFolder = Path.Combine(runtimeFolder, "venv", BuildEnvironmentKey()); + var pythonPath = GetEnvironmentPythonPath(environmentFolder); + var readyPath = Path.Combine(environmentFolder, ".ready"); + if (string.Equals(verifiedEnvironmentPythonPath, pythonPath, StringComparison.OrdinalIgnoreCase) + && File.Exists(pythonPath) + && File.Exists(readyPath)) + { + return pythonPath; + } + + if (File.Exists(pythonPath) && File.Exists(readyPath)) + { + var verification = await commandRunner.RunAsync( + pythonPath, + WarmupArguments, + cancellationToken); + if (verification.ExitCode == 0) + { + verifiedEnvironmentPythonPath = pythonPath; + return pythonPath; + } + + logger.LogWarning( + "Existing Resemblyzer environment verification failed with exit code {ExitCode}: {Error}. Reprovisioning it.", + verification.ExitCode, + verification.StandardError); + File.Delete(readyPath); + } + + Directory.CreateDirectory(environmentFolder); + await RunRequiredAsync( + options.PythonCommand, + ["-m", "venv", "--clear", environmentFolder], + "virtual environment creation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + ["-m", "pip", "install", "--upgrade", "pip"], + "pip upgrade", + cancellationToken); + await RunRequiredAsync( + pythonPath, + [ + "-m", "pip", "install", + "--index-url", options.TorchIndexUrl, + TorchRequirement + ], + "CPU PyTorch installation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + [ + "-m", "pip", "install", + .. RuntimeDependencyRequirements + ], + "Resemblyzer dependency installation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + [ + "-m", "pip", "install", "--no-deps", + ResemblyzerRequirement + ], + "Resemblyzer installation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + WarmupArguments, + "environment verification", + cancellationToken); + await File.WriteAllTextAsync(readyPath, BuildEnvironmentKey(), cancellationToken); + verifiedEnvironmentPythonPath = pythonPath; + return pythonPath; + } + + private async Task RunRequiredAsync( + string fileName, + IReadOnlyList arguments, + string operation, + CancellationToken cancellationToken) + { + var result = await commandRunner.RunAsync(fileName, arguments, cancellationToken); + ThrowIfFailed(result, operation); + return result; + } + + private static string BuildEncodingScript() + { + return + "import json\n" + + "import sys\n" + + "import numpy as np\n" + + "from pathlib import Path\n" + + "from resemblyzer import VoiceEncoder, preprocess_wav\n" + + "encoder = VoiceEncoder('cpu', verbose=False)\n" + + "vectors = []\n" + + "for path in sorted(Path(sys.argv[1]).glob('*.wav')):\n" + + " wav = preprocess_wav(path)\n" + + " vector = encoder.embed_utterance(wav)\n" + + " vectors.append(np.asarray(vector, dtype=np.float32).tolist())\n" + + $"print('{JsonStart}')\n" + + "print(json.dumps(vectors, allow_nan=False))\n" + + $"print('{JsonEnd}')\n"; + } + + private static IReadOnlyList ParseAndValidateVectors(string output, int expectedCount) + { + var json = ExtractJson(output); + float[][]? vectors; + try + { + vectors = JsonSerializer.Deserialize(json); + } + catch (JsonException exception) + { + throw new InvalidDataException("Resemblyzer returned malformed vector JSON.", exception); + } + + if (vectors is null || vectors.Length != expectedCount) + { + throw new InvalidDataException( + $"Resemblyzer returned {vectors?.Length ?? 0} vectors for {expectedCount} WAV samples."); + } + + return vectors + .Select((vector, index) => SpeakerVoiceVectors.Normalize(vector, $"Resemblyzer vector {index}")) + .ToList(); + } + + private static string ExtractJson(string output) + { + var start = output.IndexOf(JsonStart, StringComparison.Ordinal); + if (start < 0) + { + throw new InvalidDataException("Resemblyzer output did not contain the JSON start marker."); + } + + start += JsonStart.Length; + var end = output.IndexOf(JsonEnd, start, StringComparison.Ordinal); + if (end < 0) + { + throw new InvalidDataException("Resemblyzer output did not contain the JSON end marker."); + } + + return output[start..end].Trim(); + } + + private string BuildEnvironmentKey() + { + var settings = string.Join('\n', EnvironmentManifest); + var hash = SHA256.HashData(Encoding.UTF8.GetBytes(settings)); + return Convert.ToHexString(hash)[..16].ToLowerInvariant(); + } + + private string TorchRequirement => $"torch=={options.TorchVersion}"; + + private string ResemblyzerRequirement => $"Resemblyzer=={options.PackageVersion}"; + + private static string[] WarmupArguments => + [ + "-c", + "from resemblyzer import VoiceEncoder; VoiceEncoder('cpu', verbose=False); print('Resemblyzer warm-up complete')" + ]; + + private string[] RuntimeDependencyRequirements => + [ + NumpyBeforePython313Requirement, + NumpyFromPython313Requirement, + LibrosaRequirement, + ScipyRequirement, + $"webrtcvad-wheels=={options.WebRtcVadVersion}" + ]; + + private IEnumerable EnvironmentManifest => + [ + EnvironmentSchemaVersion, + options.PythonCommand, + options.TorchIndexUrl, + TorchRequirement, + .. RuntimeDependencyRequirements, + ResemblyzerRequirement + ]; + + private static string GetEnvironmentPythonPath(string environmentFolder) + { + return OperatingSystem.IsWindows() + ? Path.Combine(environmentFolder, "Scripts", "python.exe") + : Path.Combine(environmentFolder, "bin", "python"); + } + + private void ValidateDependencySettings() + { + if (string.IsNullOrWhiteSpace(options.PythonCommand)) + { + throw new InvalidOperationException("The Resemblyzer Python command cannot be empty."); + } + + ValidateVersion(options.PackageVersion, "Resemblyzer package"); + ValidateVersion(options.TorchVersion, "PyTorch"); + ValidateVersion(options.WebRtcVadVersion, "webrtcvad-wheels"); + if (!Uri.TryCreate(options.TorchIndexUrl, UriKind.Absolute, out var indexUri) + || indexUri.Scheme != Uri.UriSchemeHttps) + { + throw new InvalidOperationException( + $"Invalid Resemblyzer PyTorch index URL '{options.TorchIndexUrl}'."); + } + } + + private static void ValidateVersion(string version, string dependency) + { + if (!PackageVersionPattern().IsMatch(version)) + { + throw new InvalidOperationException($"Invalid {dependency} version '{version}'."); + } + } + + private async Task RunWithTimeoutAsync( + Func> operation, + CancellationToken cancellationToken) + { + using var timeoutSource = options.CommandTimeout > TimeSpan.Zero + ? new CancellationTokenSource(options.CommandTimeout) + : null; + using var linkedSource = timeoutSource is null + ? null + : CancellationTokenSource.CreateLinkedTokenSource(cancellationToken, timeoutSource.Token); + try + { + return await operation(linkedSource?.Token ?? cancellationToken); + } + catch (OperationCanceledException) when ( + !cancellationToken.IsCancellationRequested && timeoutSource?.IsCancellationRequested == true) + { + throw new TimeoutException( + $"Resemblyzer command timed out after {options.CommandTimeout.ToString(null, CultureInfo.InvariantCulture)}."); + } + } + + private static void ThrowIfFailed(CommandResult result, string operation) + { + if (result.ExitCode != 0) + { + throw new InvalidOperationException( + $"Resemblyzer {operation} failed with exit code {result.ExitCode}: {result.StandardError}"); + } + } + + [GeneratedRegex("^[0-9A-Za-z.+-]+$", RegexOptions.CultureInvariant)] + private static partial Regex PackageVersionPattern(); +} diff --git a/MeetingAssistant/Summary/LiteLlmResponsesChatClient.cs b/MeetingAssistant/Summary/LiteLlmResponsesChatClient.cs index b571edb..136d34a 100644 --- a/MeetingAssistant/Summary/LiteLlmResponsesChatClient.cs +++ b/MeetingAssistant/Summary/LiteLlmResponsesChatClient.cs @@ -7,7 +7,6 @@ using System.Text.Json.Nodes; using Microsoft.Agents.AI.Compaction; using Microsoft.Extensions.AI; using Microsoft.Extensions.Logging; -using OpenAI; using OpenAI.Responses; namespace MeetingAssistant.Summary; @@ -798,7 +797,7 @@ public sealed class LiteLlmResponsesChatClient : IChatClient string apiKey, AsyncLocal requestInitiator) { - var options = new OpenAIClientOptions + var options = new ResponsesClientOptions { Endpoint = httpClient.BaseAddress ?? throw new InvalidOperationException("LiteLLM HTTP client requires a base address."), diff --git a/MeetingAssistant/Taskbar/MeetingTaskbarMenu.cs b/MeetingAssistant/Taskbar/MeetingTaskbarMenu.cs index a59a485..5ac55f3 100644 --- a/MeetingAssistant/Taskbar/MeetingTaskbarMenu.cs +++ b/MeetingAssistant/Taskbar/MeetingTaskbarMenu.cs @@ -9,6 +9,8 @@ public enum MeetingTaskbarAction OpenSubmenu, StartRecording, StopRecording, + PauseTranscription, + UnpauseTranscription, AbortRecording, SwitchProfile, SelectMicrophone, @@ -51,6 +53,15 @@ public static class MeetingTaskbarMenuBuilder } var secondaryControls = new List(); + if (status.IsRecording) + { + secondaryControls.Add(new MeetingTaskbarMenuItem( + status.IsPaused ? "Unpause transcription" : "Pause transcription", + status.IsPaused + ? MeetingTaskbarAction.UnpauseTranscription + : MeetingTaskbarAction.PauseTranscription)); + } + if (microphones is { Count: > 0 }) { secondaryControls.Add(BuildMicrophoneMenu(microphones, currentMicrophoneDeviceId)); diff --git a/MeetingAssistant/Taskbar/UnoTaskbarIconService.Windows.cs b/MeetingAssistant/Taskbar/UnoTaskbarIconService.Windows.cs index cf481c6..33f5659 100644 --- a/MeetingAssistant/Taskbar/UnoTaskbarIconService.Windows.cs +++ b/MeetingAssistant/Taskbar/UnoTaskbarIconService.Windows.cs @@ -249,6 +249,12 @@ public sealed class UnoTaskbarIconService : IHostedService, IDisposable case MeetingTaskbarAction.StopRecording: await coordinator.StopAsync(CancellationToken.None); break; + case MeetingTaskbarAction.PauseTranscription: + await coordinator.SetTranscriptionPausedAsync(true, CancellationToken.None); + break; + case MeetingTaskbarAction.UnpauseTranscription: + await coordinator.SetTranscriptionPausedAsync(false, CancellationToken.None); + break; case MeetingTaskbarAction.AbortRecording: await coordinator.AbortAsync(CancellationToken.None); break; diff --git a/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs b/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs index e7fbb23..8e28d40 100644 --- a/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs +++ b/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs @@ -96,14 +96,18 @@ public sealed class PyannoteDiarizationWarmupHostedService : IHostedService } } - private IEnumerable GetEnabledDiarizationOptions() + private IEnumerable GetEnabledDiarizationOptions() { - return launchProfiles.GetProfiles() - .SelectMany(profile => GetEnabledDiarizationOptions(profile.Options)) + var transcriptionRuntimes = launchProfiles.GetProfiles() + .SelectMany(profile => GetEnabledTranscriptionDiarizationOptions(profile.Options)); + var speakerValidationRuntimes = GetEnabledSpeakerValidationDiarizationOptions( + launchProfiles.GetRequiredProfile(null).Options); + return transcriptionRuntimes + .Concat(speakerValidationRuntimes) .DistinctBy(CreateWarmUpKey); } - private static IEnumerable GetEnabledDiarizationOptions( + private static IEnumerable GetEnabledTranscriptionDiarizationOptions( MeetingAssistantOptions options) { if (options.Recording.TranscriptionProvider.Equals("whisper-local", StringComparison.OrdinalIgnoreCase) && @@ -111,15 +115,19 @@ public sealed class PyannoteDiarizationWarmupHostedService : IHostedService { yield return options.WhisperLocal.Diarization; } + } - if (options.SpeakerIdentification.PyannoteValidation.Enabled && - options.SpeakerIdentification.PyannoteValidation.Diarization.Enabled) + private static IEnumerable GetEnabledSpeakerValidationDiarizationOptions( + MeetingAssistantOptions options) + { + if (!options.SpeakerIdentification.Resemblyzer.Enabled && + options.SpeakerIdentification.PyannoteValidation.Enabled) { yield return options.SpeakerIdentification.PyannoteValidation.Diarization; } } - private static string CreateWarmUpKey(PyannoteDiarizationOptions diarization) + private static string CreateWarmUpKey(PyannoteRuntimeOptions diarization) { return string.Join( '\u001f', diff --git a/MeetingAssistant/Transcription/PyannoteTranscriptFinalizer.cs b/MeetingAssistant/Transcription/PyannoteTranscriptFinalizer.cs index ef4a4a0..2877262 100644 --- a/MeetingAssistant/Transcription/PyannoteTranscriptFinalizer.cs +++ b/MeetingAssistant/Transcription/PyannoteTranscriptFinalizer.cs @@ -28,7 +28,12 @@ public sealed class PyannoteTranscriptFinalizer SpeechRecognitionPipelineOptions pipelineOptions, CancellationToken cancellationToken) { - return await FinalizeAsync( + if (!options.WhisperLocal.Diarization.Enabled) + { + return []; + } + + return await FinalizeEnabledAsync( audioPath, liveSegments, options.WhisperLocal.Diarization, @@ -36,14 +41,14 @@ public sealed class PyannoteTranscriptFinalizer cancellationToken); } - public async Task> FinalizeAsync( + internal async Task> FinalizeEnabledAsync( string audioPath, IReadOnlyList liveSegments, - PyannoteDiarizationOptions diarization, + PyannoteRuntimeOptions diarization, SpeechRecognitionPipelineOptions pipelineOptions, CancellationToken cancellationToken) { - if (!diarization.Enabled || liveSegments.Count == 0) + if (liveSegments.Count == 0) { return []; } @@ -83,14 +88,9 @@ public sealed class PyannoteTranscriptFinalizer } public async Task WarmUpAsync( - PyannoteDiarizationOptions diarization, + PyannoteRuntimeOptions diarization, CancellationToken cancellationToken) { - if (!diarization.Enabled) - { - return; - } - var token = ResolveToken(diarization); if (string.IsNullOrWhiteSpace(token)) { @@ -137,7 +137,7 @@ public sealed class PyannoteTranscriptFinalizer private async Task RunDiarizationAsync( string fullAudioPath, string token, - PyannoteDiarizationOptions diarization, + PyannoteRuntimeOptions diarization, SpeechRecognitionPipelineOptions pipelineOptions, CancellationToken cancellationToken) { @@ -171,7 +171,7 @@ public sealed class PyannoteTranscriptFinalizer } private async Task EnsureDockerImageAsync( - PyannoteDiarizationOptions diarization, + PyannoteRuntimeOptions diarization, string modelsFolder, CancellationToken cancellationToken) { @@ -198,7 +198,7 @@ public sealed class PyannoteTranscriptFinalizer } } - private static string BuildDockerfile(PyannoteDiarizationOptions diarization) + private static string BuildDockerfile(PyannoteRuntimeOptions diarization) { return $"FROM {diarization.BaseImage}\n" @@ -211,7 +211,7 @@ public sealed class PyannoteTranscriptFinalizer } private string[] BuildDockerArguments( - PyannoteDiarizationOptions diarization, + PyannoteRuntimeOptions diarization, string fullAudioPath, string modelsFolder, SpeechRecognitionPipelineOptions pipelineOptions) @@ -242,7 +242,7 @@ public sealed class PyannoteTranscriptFinalizer } private string[] BuildWarmUpDockerArguments( - PyannoteDiarizationOptions diarization, + PyannoteRuntimeOptions diarization, string modelsFolder) { return @@ -269,7 +269,7 @@ public sealed class PyannoteTranscriptFinalizer } private static string BuildPythonCommand( - PyannoteDiarizationOptions diarization, + PyannoteRuntimeOptions diarization, SpeechRecognitionPipelineOptions pipelineOptions) { var model = diarization.Model; @@ -296,7 +296,7 @@ public sealed class PyannoteTranscriptFinalizer + "python /tmp/meeting_assistant_pyannote.py"; } - private static string BuildWarmUpPythonCommand(PyannoteDiarizationOptions diarization) + private static string BuildWarmUpPythonCommand(PyannoteRuntimeOptions diarization) { var model = diarization.Model; return @@ -475,7 +475,7 @@ public sealed class PyannoteTranscriptFinalizer return turns; } - private static string ResolveToken(PyannoteDiarizationOptions options) + private static string ResolveToken(PyannoteRuntimeOptions options) { if (!string.IsNullOrWhiteSpace(options.Token)) { diff --git a/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs b/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs index 8167ffe..dd32d95 100644 --- a/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs +++ b/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs @@ -24,6 +24,7 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi private readonly IMeetingMetadataProvider meetingMetadataProvider; private readonly ILaunchProfileOptionsProvider launchProfiles; private readonly ISpeakerIdentityMergeService identityMergeService; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; private readonly AsrDiagnosticService asrDiagnosticService; private readonly IConfiguration configuration; @@ -38,6 +39,7 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi IMeetingMetadataProvider meetingMetadataProvider, ILaunchProfileOptionsProvider launchProfiles, ISpeakerIdentityMergeService identityMergeService, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, AsrDiagnosticService asrDiagnosticService, IConfiguration configuration) { @@ -51,6 +53,7 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi this.meetingMetadataProvider = meetingMetadataProvider; this.launchProfiles = launchProfiles; this.identityMergeService = identityMergeService; + this.outlierPruner = outlierPruner; this.asrDiagnosticService = asrDiagnosticService; this.configuration = configuration; } @@ -78,7 +81,8 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi identityMergeService, asrDiagnosticService, configuration, - logger: logger); + logger: logger, + outlierPruner: outlierPruner); var messages = conversation .Select(ToChatMessage) .Append(new ChatMessage(ChatRole.User, userMessage.Trim())) diff --git a/MeetingAssistant/Workflow/WorkflowRulesEditorSamplePlaybackQueue.cs b/MeetingAssistant/Workflow/WorkflowRulesEditorSamplePlaybackQueue.cs index c024ac2..6f073fa 100644 --- a/MeetingAssistant/Workflow/WorkflowRulesEditorSamplePlaybackQueue.cs +++ b/MeetingAssistant/Workflow/WorkflowRulesEditorSamplePlaybackQueue.cs @@ -8,6 +8,7 @@ public interface IWorkflowRulesEditorSamplePlaybackQueue Task QueueAsync(int sampleId, byte[] wavBytes, CancellationToken cancellationToken = default); } +#if WINDOWS public sealed class WorkflowRulesEditorSamplePlaybackQueue : IWorkflowRulesEditorSamplePlaybackQueue, IDisposable { private readonly Channel channel = Channel.CreateUnbounded(); @@ -98,3 +99,13 @@ public sealed class WorkflowRulesEditorSamplePlaybackQueue : IWorkflowRulesEdito private sealed record QueuedSample(int SampleId, byte[] WavBytes); } +#else +public sealed class UnavailableWorkflowRulesEditorSamplePlaybackQueue : IWorkflowRulesEditorSamplePlaybackQueue +{ + public Task QueueAsync(int sampleId, byte[] wavBytes, CancellationToken cancellationToken = default) + { + cancellationToken.ThrowIfCancellationRequested(); + return Task.FromResult("Speaker sample playback requires the Windows build of Meeting Assistant."); + } +} +#endif diff --git a/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs b/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs index 70405e3..f772973 100644 --- a/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs +++ b/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs @@ -10,6 +10,7 @@ using MeetingAssistant.Transcription; using Microsoft.EntityFrameworkCore; using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; using YamlDotNet.Core; using YamlDotNet.Serialization; @@ -42,6 +43,7 @@ public sealed class WorkflowRulesEditorTools private readonly IMeetingMetadataProvider? meetingMetadataProvider; private readonly ILaunchProfileOptionsProvider? launchProfiles; private readonly ISpeakerIdentityMergeService? identityMergeService; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; private readonly AsrDiagnosticService? asrDiagnosticService; private readonly IConfiguration? configuration; private readonly ILogger? logger; @@ -61,7 +63,8 @@ public sealed class WorkflowRulesEditorTools string? logDirectory = null, string? specRootPath = null, string? projectAgentsTemplatePath = null, - ILogger? logger = null) + ILogger? logger = null, + ResemblyzerVoiceVectorOutlierPruner? outlierPruner = null) { this.options = options; rulesPath = WorkflowRulesPathResolver.Resolve(options.Automation.RulesPath); @@ -82,6 +85,9 @@ public sealed class WorkflowRulesEditorTools this.meetingMetadataProvider = meetingMetadataProvider; this.launchProfiles = launchProfiles; this.identityMergeService = identityMergeService; + this.outlierPruner = outlierPruner ?? new ResemblyzerVoiceVectorOutlierPruner( + speakerOptions.Resemblyzer, + NullLogger.Instance); this.asrDiagnosticService = asrDiagnosticService; this.configuration = configuration; this.logger = logger; @@ -743,8 +749,8 @@ public sealed class WorkflowRulesEditorTools string[]? candidateNames = null) { return Task.FromResult( - "Refused: speaker identities require at least one audio sample. " + - "Use a speaker override from an existing transcript speaker sample, or update/merge an existing sampled identity."); + "Refused: speaker identities require audio evidence (a WAV sample or voice vector). " + + "Use a speaker override from an existing transcript speaker, or update/merge an existing identity."); } public async Task UpdateIdentity( @@ -826,10 +832,12 @@ public sealed class WorkflowRulesEditorTools return $"Could not find target {targetIdentityId} or source {sourceIdentityId}."; } - SpeakerIdentityMerger.MergeInto( + SpeakerIdentityMerger.MergeIntoAndPrune( target, source, - speakerOptions.MaxSnippetsPerSpeaker); + speakerOptions.MaxSnippetsPerSpeaker, + speakerOptions.Resemblyzer.MaxVectorsPerIdentity, + outlierPruner); context.SpeakerIdentities.Remove(source); await context.SaveChangesAsync(); return ToJson(ToIdentityDetail(target)); @@ -1567,9 +1575,11 @@ public sealed class WorkflowRulesEditorTools private static IQueryable LoadIdentities(SpeakerIdentityDbContext context) { return context.SpeakerIdentities + .AsSplitQuery() .Include(identity => identity.Aliases) .Include(identity => identity.CandidateNames) .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) .Include(identity => identity.References); } @@ -1591,6 +1601,7 @@ public sealed class WorkflowRulesEditorTools identity.Aliases.Select(alias => alias.Name).Order(StringComparer.OrdinalIgnoreCase).ToArray(), identity.CandidateNames.Select(candidate => candidate.Name).Order(StringComparer.OrdinalIgnoreCase).ToArray(), identity.Snippets.Count, + identity.VoiceVectors.Count, identity.References.Count, identity.UpdatedAt); } @@ -1650,6 +1661,7 @@ public sealed class WorkflowRulesEditorTools IReadOnlyList Aliases, IReadOnlyList CandidateNames, int SampleCount, + int VoiceVectorCount, int ReferenceCount, DateTimeOffset UpdatedAt); diff --git a/MeetingAssistant/appsettings.json b/MeetingAssistant/appsettings.json index 3219cce..585e142 100644 --- a/MeetingAssistant/appsettings.json +++ b/MeetingAssistant/appsettings.json @@ -36,6 +36,7 @@ "00:10:00" ], "AutoStopAfter": "00:30:00", + "MaximumPauseDuration": "04:00:00", "InferredEndPadding": "00:01:00", "CheckInterval": "00:00:15" } @@ -119,18 +120,17 @@ "MaxMatchCandidates": 100, "MatchIdentityActiveAge": "365.00:00:00", "MaxSnippetsPerSpeaker": 3, - "MinimumSampleSpeechDuration": "00:00:30", - "MaximumSampleSegmentGap": "00:00:01", + "MinimumSampleSpeechDuration": "00:00:10", + "MaximumSampleDuration": "00:01:00", "SilenceBetweenSnippetsSeconds": 1, "LiveSampleBufferDuration": "00:10:00", "MergeRecentIdentityAge": "14.00:00:00", "MatchTimeout": "00:03:00", "PyannoteValidation": { - "Enabled": true, + "Enabled": false, "MinimumSingleSpeakerCoverage": 0.9, "MinimumMatchingKnownSnippetRatio": 0.5, "Diarization": { - "Enabled": false, "DockerCommand": "docker", "BaseImage": "python:3.11-slim", "Image": "meeting-assistant-pyannote:local", @@ -142,6 +142,26 @@ "TokenEnv": "HF_TOKEN", "CommandTimeout": "01:00:00" } + }, + "Resemblyzer": { + "Enabled": false, + "RequiredVectorsPerSpeaker": 5, + "MaxVectorsPerIdentity": 1000, + "OutlierPruningMinimumVectors": 20, + "OutlierPruningNeighborSimilarity": 0.75, + "OutlierPruningMinimumNeighbors": 3, + "OutlierPruningMinimumClusterRatio": 0.60, + "MinimumClusterCohesion": 0.75, + "MinimumIdentitySimilarity": 0.75, + "MinimumSimilarityMargin": 0.05, + "ModelId": "resemblyzer-0.1.4-pretrained", + "PackageVersion": "0.1.4", + "PythonCommand": "python", + "TorchVersion": "2.14.0+cpu", + "TorchIndexUrl": "https://download.pytorch.org/whl/cpu", + "WebRtcVadVersion": "2.0.14", + "RuntimeFolder": "%LOCALAPPDATA%\\MeetingAssistant\\Resemblyzer", + "CommandTimeout": "00:15:00" } }, "Automation": { diff --git a/README.md b/README.md index f6131bc..be0b8a1 100644 --- a/README.md +++ b/README.md @@ -61,7 +61,7 @@ Recording can be controlled through global hotkeys, the Windows tray icon, the m - `Ctrl+Alt+Z`: abort the active run and delete its artifacts. - `Ctrl+Alt+S`: capture the active window into the meeting context. -The Windows tray and macOS menu-bar menu present `Finish meeting` as the primary action during capture; microphone selection, cancel/discard, and profile switching remain separate controls. `Exit` is always available and requires confirmation while any meeting is recording or processing. +The Windows tray presents `Finish meeting` as the primary action during capture; microphone selection, cancel/discard, profile switching, and pause/unpause transcription remain separate controls. A paused meeting stays active and can still be finished or canceled normally. Windows `Exit` requires confirmation while any meeting is recording or processing. The macOS menu offers profile start actions while idle and `Stop meeting recording and transcribe` plus cancel/discard during capture; its `Exit` action currently exits directly. The loopback HTTP surface has no application authentication, so port `5090` must remain a trusted loopback-only control surface. The main endpoints are: @@ -82,6 +82,8 @@ Some generated retry links use `GET` while starting work. Stopping capture lets buffered transcription, speaker work, meeting-note image OCR, screenshot OCR, and summary generation finish. Another meeting can start while an older stopped run finalizes; each run retains isolated options and artifact paths. +Pausing transcription keeps the active recognition pipeline and meeting session alive, including Azure conversation transcription and the run's speaker context. Captured audio is replaced with equal-length silence before it reaches the temporary WAV or transcription backend, so real audio from the paused interval is discarded while provider continuity and meeting-relative timing are preserved. Normal transcript-inactivity notifications and auto-stop are suppressed during pause, while a separate four-hour maximum continuous pause prevents a forgotten paused meeting from running indefinitely. The recording status response exposes pause state as `isPaused`. + During an active run, microphone creation failures and disconnects are retried every second with fresh endpoint selection. The meeting and system-loopback capture stay active, with microphone silence mixed in until capture resumes. This recovery does not cover a failed system-loopback source. Outlook enrichment selects an unambiguous current or imminent appointment. A scheduled prompt shown during an active recording can apply that exact appointment's title, eligible attendees, agenda, and scheduled end without interrupting capture; explicit prompt metadata wins over a slower background lookup. @@ -95,7 +97,8 @@ Agents are intentionally stateful. Depending on the invoked tools, they can chan Local runtime state outside the vault includes: - `%LOCALAPPDATA%\MeetingAssistant\Recordings`: mixed WAV files are normally deleted after completion, and unqueued stale files are deleted at startup. If an Azure stop cannot drain within `Recording:StopProcessingTimeout`, the WAV plus a JSON item under `offline-transcription-backlog` are retained and retried every minute. Both are removed only after successful replay, transcript finalization, and summary processing. -- `%LOCALAPPDATA%\MeetingAssistant\SpeakerIdentity\speaker-identities.db`: SQLite identities, aliases, meeting references, and bounded voice snippets. +- `%LOCALAPPDATA%\MeetingAssistant\SpeakerIdentity\speaker-identities.db`: SQLite identities, aliases, meeting references, bounded WAV snippets for the existing matcher, and separate versioned voice vectors for the optional Resemblyzer matcher. +- `%LOCALAPPDATA%\MeetingAssistant\Resemblyzer`: content-versioned managed Python environments, the local encoder script, and temporary encoder inputs. Per-meeting WAV inputs are deleted after each encoding command. - `%LOCALAPPDATA%\MeetingAssistant\FunASR\models` and `%LOCALAPPDATA%\MeetingAssistant\Pyannote\models`: persistent model, hotword, Hugging Face, and torch caches for optional local backends. - `%TEMP%\MeetingAssistant\Logs\meeting-assistant.log`: application log with four rotated predecessors. Paths, transcript text, agent diagnostics, and provider errors can make these logs sensitive. @@ -108,7 +111,7 @@ Abort is destructive: it removes the active run's note, transcript, context, sum - The default `azure-speech` provider sends mixed meeting audio and dictation phrase hints to Azure AI Speech. Azure-backed speaker matching also sends selected voice audio. - Summary, screenshot OCR, and interactive-agent requests go to the configured OpenAI-compatible Responses endpoint. They can include meeting/transcript/project text, screenshots, configuration, logs, and speaker samples when corresponding tools are used. The checked-in endpoint is a loopback proxy; its ultimate provider, data path, and retention policy are outside this repository. - Outlook Classic access on Windows is local COM. EventKit access on macOS reads calendars synchronized into the Calendar app. Neither provides the primary capture path. -- A managed FunASR run pulls its configured image, removes any same-named container, starts a disposable container privileged by default, publishes the configured host port, and mounts the model/hotword cache. Pyannote may build a local image and starts disposable containers with the input WAV mounted read-only and its model cache read/write. These paths require Docker Desktop or a compatible Docker CLI and may download images/models from external registries. +- A managed FunASR run pulls its configured image, removes any same-named container, starts a disposable container privileged by default, publishes the configured host port, and mounts the model/hotword cache. Pyannote may build a local image and start disposable containers with audio inputs mounted read-only; those paths require Docker Desktop or a compatible Docker CLI. The opt-in Resemblyzer speaker matcher instead provisions an isolated local Python venv with CPU-only dependencies. First use may download images, Python packages, or models from external registries. ## Configuration @@ -121,6 +124,7 @@ The settings with the largest operational effect are: - `Recording:MicrophoneDeviceId`, mix gains, stop timeout, minimum duration, and temporary folder: control capture selection, audio, cleanup, and Azure backlog behavior. Microphone-device selection is Windows-only; macOS follows the system default input device. - `Recording:InactivitySafeguard`: prompts and can auto-finish a run after no new transcript text; it is not an audio-silence detector. - `LaunchProfiles`: overlay named recording/ASR/agent settings and require distinct hotkeys. +- `SpeakerIdentification:Resemblyzer:Enabled`: selects the local, managed-Python-venv vector matcher for the whole application; when disabled, the existing WAV/Azure path stays active. Five vectors unlock matching by default without capping retained evidence, and mature profiles use configurable fail-safe density clustering to remove likely mixed-speaker outliers. - `Automation:RulesPath`: points to the local YAML workflow-rules file, normally ignored `meeting-rules.local.yaml`. - `CalendarRecordingPrompts` and `Screenshots`: control Outlook prompts on Windows, EventKit prompts on macOS, capture, attachments, and configured OCR. - `Agent` and `WorkflowRulesEditor`: select the Responses endpoint/model, streaming or non-streaming transport, reasoning, retry, output, and compaction behavior; the available tools are defined by the application. diff --git a/docs/meeting-assistant-configuration.md b/docs/meeting-assistant-configuration.md index dd4f558..983c7aa 100644 --- a/docs/meeting-assistant-configuration.md +++ b/docs/meeting-assistant-configuration.md @@ -38,6 +38,7 @@ This example is abbreviated so the most common shape is readable. The checked-in "FirstPromptAfter": "00:02:00", "ReminderPromptAfter": [ "00:05:00", "00:10:00" ], "AutoStopAfter": "00:30:00", + "MaximumPauseDuration": "04:00:00", "InferredEndPadding": "00:01:00", "CheckInterval": "00:00:15" } @@ -133,20 +134,23 @@ During recording, Meeting Assistant captures microphone and system loopback sepa On macOS, the portable target captures the default microphone through AVFoundation and computer output through ScreenCaptureKit. The build compiles the Swift audio and desktop-integration helpers into the application output and publish `Native` folder; build or publish on macOS with Xcode Command Line Tools installed. The first capture requires both **Microphone** and **Screen & System Audio Recording** permissions under System Settings > Privacy & Security. Restart the application after granting a newly requested permission. `Recording:MicrophoneDeviceId` and runtime microphone selection remain Windows-only; macOS follows the system default input device. +The tray's fine-grained controls expose `Pause transcription` while a meeting is active and `Unpause transcription` while it is paused. Pause does not stop audio devices, finish the meeting, replace the speech-recognition pipeline, or reset run-local speaker mappings and collected samples. Instead, each captured mixed chunk is replaced with equal-length PCM silence before it reaches the temporary WAV, live speaker buffer, or configured transcription provider. This discards real audio from the paused interval while preserving the provider session and meeting-relative timing. In particular, Azure keeps the same active `ConversationTranscriber` and push stream; pause therefore does not suspend Azure connection time or billing. `Finish meeting` and cancel/discard remain available while paused, and `/recording/status` reports the state in `isPaused`. + On Windows, `Recording:MicrophoneDeviceId` can pin capture to a specific active microphone endpoint id. Leave it blank to follow the Windows default capture endpoint. The tray icon menu also exposes `Microphone`, listing active microphone endpoints with the effective endpoint checked. Selecting a microphone there overrides the configured/default microphone for later recording starts until another microphone is selected or the process exits. `Recording:MicrophoneMixGain` and `Recording:SystemAudioMixGain` are applied during the final mix and default to `1`. `Recording:TemporaryRecordingsFolder` controls where the temporary mixed WAV is written while the run is active. Temporary WAV files are deleted after the run completes, and stale temporary recordings from interrupted runs are deleted when the application starts. If an Azure Speech meeting cannot drain transcription before `Recording:StopProcessingTimeout`, Meeting Assistant keeps the WAV and writes a durable backlog item under `TemporaryRecordingsFolder\offline-transcription-backlog`. The background backlog worker retries those queued meetings, replays each WAV through a fresh speech pipeline, rewrites the original transcript, completes meeting metadata and summary generation, then removes the backlog item and WAV. `Recording:MaxMetadataAttendeeImportCount` limits how many attendees calendar metadata enrichment imports into meeting-note frontmatter. The default is `30`; when an appointment has more attendees than that, Meeting Assistant still imports title, agenda, and scheduled end time, but leaves attendees empty because large invites are usually presentation-style meetings. Windows reads Outlook Classic through COM. macOS reads EventKit calendars, including Outlook accounts synchronized into macOS Calendar, and requires **Calendar Full Access**. -`Recording:InactivitySafeguard` watches active recordings for long periods without transcript text. The timer starts at meeting start and resets whenever a live transcript segment with text arrives. By default the app asks whether to stop after 2, 5, and 10 minutes of inactivity through native Windows app notifications with action buttons, requests reminder-style toast behavior, keeps each stop reminder actionable for 1 minute, and automatically stops normally after 30 minutes. Ignoring a notification does not block later checks or auto-stop. Safeguard stops are not aborts: transcription drain, speaker processing, screenshots, and summary generation continue through the normal stop flow. When the safeguard stops a run, the meeting end time is inferred as the last transcript segment timestamp plus `InferredEndPadding`; if no transcript text arrived, it uses meeting start plus the same padding. +`Recording:InactivitySafeguard` watches active recordings for long periods without transcript text. The timer starts at meeting start and resets whenever a live transcript segment with text is written. Writing new transcript text also dismisses every outstanding inactivity notification and invalidates its actions. By default the app asks whether to stop after 2, 5, and 10 minutes of inactivity through native Windows app notifications with Yes, No, and `Pause transcription` actions, requests reminder-style toast behavior, keeps each stop reminder actionable for 1 minute, and automatically stops normally after 30 minutes. While transcription is intentionally paused, those prompts and the ordinary transcript-inactivity stop are fully suspended. A separate `MaximumPauseDuration`, defaulting to 4 hours, normally stops a meeting that remains continuously paused without showing an inactivity notification; it remains active when the ordinary inactivity safeguard is disabled, while a non-positive value disables the paused-session cutoff. Unpausing clears the continuous-pause timer and restarts transcript-inactivity timing from that moment. Ignoring a notification does not block later checks or auto-stop. Safeguard stops are not aborts: transcription drain, speaker processing, screenshots, and summary generation continue through the normal stop flow. When transcript inactivity stops a run, the meeting end time is inferred as the last transcript segment timestamp plus `InferredEndPadding`; if no transcript text arrived, it uses meeting start plus the same padding. | Setting | Purpose | | --- | --- | -| `Enabled` | Enables the inactivity safeguard for active recordings. | +| `Enabled` | Enables transcript-inactivity prompts and `AutoStopAfter`; `MaximumPauseDuration` remains independent. | | `FirstPromptAfter` | First transcript-inactivity duration before showing the stop prompt. | | `ReminderPromptAfter` | Additional transcript-inactivity durations before showing another stop prompt. | | `AutoStopAfter` | Transcript-inactivity duration after which Meeting Assistant stops the recording normally without prompting again. | +| `MaximumPauseDuration` | Maximum continuous transcription pause before Meeting Assistant stops the meeting normally without an inactivity notification; defaults to 4 hours, and a non-positive value disables it. | | `InferredEndPadding` | Padding added to the last transcript timestamp, or meeting start when no transcript arrived, for safeguard-triggered end times. | | `CheckInterval` | Polling interval for checking the active recording inactivity state. | @@ -221,11 +225,11 @@ When `WhisperLocal:Diarization:Enabled` is true, the final post-processing pass Active pyannote runtimes are warmed up on application start so image setup and model download do not wait for the first diarization request. -Pyannote diarization settings are shared by local Whisper finalization and speaker-identification validation: +Pyannote runtime settings are shared by local Whisper finalization and speaker-identification validation. `WhisperLocal:Diarization:Enabled` controls the optional Whisper finalization pass; speaker-identification validation instead uses its single outer `SpeakerIdentification:PyannoteValidation:Enabled` switch. | Setting | Purpose | | --- | --- | -| `Enabled` | Enables the pyannote-backed pass. | +| `Enabled` | Available under `WhisperLocal:Diarization` to enable the pyannote-backed Whisper finalization pass. It is not part of the nested speaker-validation runtime settings. | | `DockerCommand` | Docker executable name or path. | | `BaseImage` | Python base image used when building the local pyannote image. | | `Image` | Local pyannote Docker image tag. | @@ -266,7 +270,9 @@ Azure returns generic speaker IDs such as `Guest-1` and `Guest-2`, which Meeting ## Speaker Identification -Speaker identity matching keeps candidate samples only after a diarized speaker has at least `SpeakerIdentification:MinimumSampleSpeechDuration` of continuous speech, defaulting to 30 seconds. Adjacent same-speaker segments may be combined when the gap is no larger than `MaximumSampleSegmentGap`, but a different speaker resets the pending span. +Speaker identity matching keeps candidate samples only after a diarized speaker has at least `SpeakerIdentification:MinimumSampleSpeechDuration` of same-speaker audio, defaulting to 10 seconds. Consecutive lines with the same diarized speaker are combined even when any configured STT backend splits them around pauses. This applies to both live and final-only diarization. A different speaker ends the pending span, and `MaximumSampleDuration` caps every recognition WAV at 60 seconds by default. + +Configuration is rejected when the minimum duration is negative, the maximum is not positive, or the minimum exceeds the maximum. The same validation applies to launch-profile overrides. | Setting | Purpose | | --- | --- | @@ -274,12 +280,12 @@ Speaker identity matching keeps candidate samples only after a diarized speaker | `DatabasePath` | SQLite database path for identities, aliases, references, and samples. | | `InitialDelay` | Delay after recording starts before the first live identity pass. | | `Interval` | Interval between live identity passes. | -| `MatchBatchSize` | Number of identities processed per model matching batch. | +| `MatchBatchSize` | Number of identities processed per Azure model matching batch. Resemblyzer scores the full capped candidate set together so its ambiguity margin includes the global runner-up. | | `MaxMatchCandidates` | Maximum known identities considered during one match pass. | | `MatchIdentityActiveAge` | Age window for identities considered active enough for automatic matching. | -| `MaxSnippetsPerSpeaker` | Maximum stored voice snippets retained per speaker identity. | -| `MinimumSampleSpeechDuration` | Minimum uninterrupted same-speaker speech span needed before storing a sample. | -| `MaximumSampleSegmentGap` | Maximum gap allowed when combining adjacent same-speaker segments into one sample. | +| `MaxSnippetsPerSpeaker` | Maximum stored WAV snippets retained per speaker identity by the existing backend. | +| `MinimumSampleSpeechDuration` | Minimum total diarized speaker audio needed in a same-speaker sample; provider-created pauses do not count toward it. | +| `MaximumSampleDuration` | Maximum duration of an extracted speaker-recognition WAV. Defaults to 60 seconds. | | `SilenceBetweenSnippetsSeconds` | Silence padding inserted between snippets during Azure identity matching. | | `LiveSampleBufferDuration` | How long live transcript/audio material is retained for extracting samples. | | `MergeRecentIdentityAge` | Age window used by diagnostics that merge recent duplicate identities. | @@ -287,14 +293,46 @@ Speaker identity matching keeps candidate samples only after a diarized speaker `SpeakerIdentification:AzureSpeech` is an advanced nested override for speaker identity matching. It uses the same shape as `AzureSpeech` and lets identity matching use different Azure language, endpoint, or key settings than live transcription when needed. If it is left unset, the normal Azure Speech settings remain the practical default. -`SpeakerIdentification:PyannoteValidation` is an optional secondary confidence layer. When enabled, pyannote rejects multi-speaker samples and must confirm an Azure-confirmed identity match before Meeting Assistant accepts it. It uses the same Docker-based pyannote runtime shape as local Whisper finalization and defaults the validation command timeout to 1 hour because local model setup can take substantial time. +`SpeakerIdentification:PyannoteValidation` is an optional application-level secondary confidence layer and defaults to disabled. Its outer `Enabled` setting is the only validation toggle; the nested `Diarization` block contains runtime settings but no second enable switch. Launch profiles do not override speaker validation or its runtime settings. When enabled, pyannote rejects multi-speaker samples and must confirm an Azure-confirmed identity match before Meeting Assistant accepts it. It uses the same Docker-based pyannote runtime shape as local Whisper finalization and defaults the validation command timeout to 1 hour because local model setup can take substantial time. | Setting | Purpose | | --- | --- | +| `Enabled` | Sole switch for speaker-identity pyannote validation and its startup warm-up. | | `MinimumSingleSpeakerCoverage` | Required fraction of the tested sample that pyannote must attribute to a single speaker. | | `MinimumMatchingKnownSnippetRatio` | Required pyannote agreement ratio between the new sample and known snippets for an accepted identity. | | `Diarization` | Nested pyannote settings used for this validation pass. | +`SpeakerIdentification:Resemblyzer` is a separate, application-level speaker-recognition backend and defaults to disabled. Setting its single `Enabled` flag to `true` selects the Resemblyzer identification and merge services for the process lifetime; Azure Speech and pyannote are then not used for identity matching or identity-match validation. Launch profiles do not override this choice. The normal `SpeakerIdentification:Enabled` setting remains the master switch for speaker identification as a whole. + +The selected service reuses temporary WAV samples from diarized same-speaker runs, but starts a fresh span after every accepted sample so one speaker's vector inputs do not overlap. It waits for five samples by default before automatic matching; five is an eligibility threshold, not a storage cap. Additional samples for an unresolved speaker trigger another attempt at the configured interval, even when attendees and speaker labels have not changed. All distinct qualifying vectors from the run are retained up to the per-identity limit, including later evidence collected after a live match has already renamed the transcript. Providers that only diarize during finalization can extract missing non-overlapping samples from the completed mixed WAV. Resemblyzer runs locally in an application-managed Python virtual environment with CPU-only PyTorch, produces 256-value embeddings, and stores only versioned float32 vectors in the identity database; it does not persist the temporary WAV inputs as identity evidence. Explicit speaker assignments from the summary agent may save fewer than five available vectors. + +For a query, the matcher first requires the mean pairwise cosine similarity of its vectors to meet `MinimumClusterCohesion`. It represents each compatible known identity by a normalized centroid, scores it with the median query-to-centroid cosine similarity, and accepts only if the best score meets `MinimumIdentitySimilarity` and beats the runner-up by `MinimumSimilarityMargin`. The runner-up margin is waived when only one identity is scoreable. Logs include measured scores and rejection reasons so these initial thresholds can be calibrated. + +Once an identity has at least `OutlierPruningMinimumVectors` compatible vectors, a separate DBSCAN-style cosine-density pass protects the profile from mixed-speaker diarization errors. A vector is a neighbor when its cosine similarity meets `OutlierPruningNeighborSimilarity`, and a dense point requires `OutlierPruningMinimumNeighbors` neighbors including itself. Vectors outside the uniquely largest dense cluster are removed only when that cluster contains at least `OutlierPruningMinimumClusterRatio` of all compatible vectors. If there is no dense cluster, the largest clusters tie, or the ratio is too low, pruning fails safe and retains every vector. Invalid vectors and vectors produced by other model versions are preserved. + +| Resemblyzer setting | Purpose | +| --- | --- | +| `Enabled` | Selects the complete local Resemblyzer identity backend. Defaults to `false`. | +| `RequiredVectorsPerSpeaker` | Independent vectors required for automatic matching and per-cluster merge validation. Defaults to `5`; this does not cap collection or persistence. | +| `MaxVectorsPerIdentity` | Maximum distinct vectors retained per identity. Defaults to `1000`; merges prefer the newest vectors. | +| `OutlierPruningMinimumVectors` | Compatible vectors required before density-cluster pruning runs. Defaults to `20`. | +| `OutlierPruningNeighborSimilarity` | Minimum cosine similarity for two vectors to count as density neighbors. Defaults to `0.75`. | +| `OutlierPruningMinimumNeighbors` | Neighbors required for a dense point, including the point itself. Defaults to `3`. | +| `OutlierPruningMinimumClusterRatio` | Minimum fraction of compatible vectors that the uniquely largest cluster must contain before other vectors are removed. Defaults to `0.60`. | +| `MinimumClusterCohesion` | Minimum mean pairwise cosine similarity within a query cluster. | +| `MinimumIdentitySimilarity` | Minimum median similarity from query vectors to a known identity centroid. | +| `MinimumSimilarityMargin` | Required difference between the best and runner-up identity scores. | +| `ModelId` | Version identifier persisted with vectors and required for compatible comparisons. | +| `PackageVersion` | Resemblyzer package version installed in the managed virtual environment. | +| `PythonCommand` | Python executable used to create the managed virtual environment. | +| `TorchVersion` | CPU-only PyTorch wheel version installed in the managed environment. | +| `TorchIndexUrl` | HTTPS package index used for the CPU-only PyTorch wheel. | +| `WebRtcVadVersion` | Version of the prebuilt Windows-compatible `webrtcvad-wheels` package. | +| `RuntimeFolder` | Local folder for versioned virtual environments, the encoder script, and short-lived WAV input batches. | +| `CommandTimeout` | Bound for first-time environment provisioning, warm-up, and encoder commands. | + +The first enabled startup creates a content-versioned virtual environment under `RuntimeFolder`. Dependency-version changes select a new environment automatically. The workstation's global Python packages are not modified, and Docker is not required for Resemblyzer speaker recognition. + ## Automation `Automation:RulesPath` points to an optional local YAML rules file. The default `meeting-rules.local.yaml` is ignored by git. Rules can trigger on meeting creation, assistant-context state transitions, identified speakers, or transcript line writes; conditions are evaluated with NCalc-style expressions and step values can use Razor syntax against `Model.Meeting`, `Model.Event`, `Model.Speaker`, and `Model.Transcript`. diff --git a/docs/meeting-workflow-engine.md b/docs/meeting-workflow-engine.md index 9f6fc40..27a9d14 100644 --- a/docs/meeting-workflow-engine.md +++ b/docs/meeting-workflow-engine.md @@ -1,5 +1,7 @@ # Meeting Workflow Engine +The interactive rules and identities editor can queue speaker samples for playback in the Windows build. The neutral build reports that playback requires the Windows build; it does not enqueue audio or initialize Windows audio devices. + Meeting Assistant has a small local workflow engine for meeting-specific automation. It is intended for rules that are too personal or environment-specific to hard-code, such as adding default attendees, cleaning meeting titles, binding projects, or adding context notes when known speakers are detected. The workflow engine is deliberately narrow. It runs from a local YAML file, evaluates rules against the current meeting note read from disk, and applies a small set of meeting-safe mutations. diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/.openspec.yaml b/openspec/changes/add-resemblyzer-speaker-recognition/.openspec.yaml new file mode 100644 index 0000000..032461f --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-02 diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/design.md b/openspec/changes/add-resemblyzer-speaker-recognition/design.md new file mode 100644 index 0000000..c23e794 --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/design.md @@ -0,0 +1,102 @@ +## Context + +The existing speaker-identification implementation stores bounded WAV snippets and sends composite audio to a dedicated Azure Speech diarization verifier, optionally followed by pyannote validation. Recording already maintains timestamped mixed audio and creates candidate WAV samples for diarized speaker labels. Identity names, aliases, candidate names, meeting references, transcript relabeling, summarizer overrides, deletion, and merges are stored locally in SQLite. + +Resemblyzer 0.1.4 exposes a local `VoiceEncoder` that produces L2-normalized 256-value embeddings. Its upstream examples compare embeddings with dot products, which are cosine similarities for normalized vectors. The package includes its pretrained model but has Python, PyTorch, audio, and native VAD dependencies, so the application isolates them in a managed virtual environment rather than mutating the workstation Python installation. + +The feature must remain opt-in and must not silently mix Resemblyzer vectors with evidence from the WAV/Azure backend. Existing identities remain shared because their names, aliases, references, and downstream behavior are backend-independent, but each backend reads and writes only its own voice evidence. + +## Goals / Non-Goals + +**Goals:** + +- Select a separate local Resemblyzer recognition path with one application-level feature flag that defaults off. +- Convert temporary per-speaker WAV samples to versioned 256-float embeddings and persist only the embeddings for this path. +- Require five independent, coherent query embeddings before automatic matching. +- Make cohesion, acceptance similarity, ambiguity margin, required-vector count, runtime, and per-identity limit configurable. +- Preserve existing naming, attendee, relabeling, override, deletion, reference, and merge outcomes. +- Bound persisted embeddings to 1,000 per identity by default. + +**Non-Goals:** + +- Convert existing WAV snippets to Resemblyzer vectors automatically. +- Use Resemblyzer for ASR speaker diarization; diarized labels still come from the configured transcription backend. +- Run the Azure or pyannote speaker verifier as a second opinion when the Resemblyzer path is selected. +- Guarantee calibrated production thresholds before real meeting data has been observed. + +## Decisions + +### Select a complete backend at the application boundary + +Add `SpeakerIdentification:Resemblyzer:Enabled`, defaulting to `false`. Dependency injection selects either the existing `SpeakerIdentityService`/`SpeakerIdentityMergeService` pair or a separate Resemblyzer identification/merge pair for the process lifetime. Resemblyzer configuration is application-level and launch profiles do not override it because the identity database and selected singleton backend are application-wide. + +The alternative of adding conditional vector branches throughout the existing WAV service was rejected because it would make it easy to mix evidence types or accidentally invoke Azure/pyannote while the local backend is enabled. + +### Reuse temporary sample capture but make vector samples independent + +The recording run retains the configured number of best WAV samples in memory. Consecutive transcript lines with the same diarized speaker are treated as one same-speaker run regardless of which STT backend emitted them, even when that backend splits them around a pause; only an intervening different speaker or the configured maximum sample duration ends the run. Provider-created pauses remain inside the extracted time range but do not count toward its minimum speaker-audio duration. Samples are capped at 60 seconds by default. With Resemblyzer enabled, the collector starts a fresh span after each accepted sample so the five embeddings are based on non-overlapping speech. The WAV bytes are temporary inputs only and are not written to the identity database by the Resemblyzer service. + +For providers that only yield speaker labels during finalization, the service extracts disjoint qualifying spans from the completed mixed WAV. Explicit summarizer assignments may learn from fewer than five valid vectors, but automatic matching and automatic unnamed-candidate learning wait for the configured required count. + +`RequiredVectorsPerSpeaker` is an eligibility threshold, not a collection or persistence cap. The collector retains qualifying non-overlapping samples up to `MaxVectorsPerIdentity`, matching scores the configured minimum high-quality vectors, and an accepted live or final assignment persists every distinct compatible vector available for that meeting. A speaker matched during live transcription receives a final evidence-accumulation pass so samples collected after the initial match are not lost. + +### Persist versioned float32 embeddings in a separate table + +Add a `SpeakerVoiceVectors` table related to `SpeakerIdentities` with cascade deletion. Each row stores a little-endian float32 blob, dimension count, model identifier, SHA-256 fingerprint, and creation timestamp. The model identifier prevents comparisons across incompatible encoder versions. A unique identity/fingerprint index makes retrying the same evidence idempotent. + +Vector rows are capped by `MaxVectorsPerIdentity`, default 1,000. Normal additions stop at the cap. Merges combine distinct rows and keep the most recently created vectors when the combined set exceeds the cap. WAV snippets and voice vectors remain independent collections. + +### Run Resemblyzer in an application-managed Python virtual environment + +`VenvResemblyzerVoiceEncoder` batches WAV files into one invocation of the managed virtual environment's Python executable, preprocesses each file with `preprocess_wav`, calls `VoiceEncoder("cpu").embed_utterance`, and returns marked JSON. The environment is content-versioned from its dependency settings, pins Resemblyzer and a CPU-only PyTorch wheel, and uses `webrtcvad-wheels` on Windows to avoid requiring Visual C++ build tooling. NumPy stays below 2 on Python versions where a compatible NumPy 1.x wheel exists and uses NumPy 2 on Python 3.13 or later. A non-blocking startup warm-up provisions and verifies the environment only when the feature is enabled. + +The encoder validates result count, dimension, finite values, and nonzero magnitude before returning normalized vectors. Temporary input directories are deleted after each bounded invocation. + +Installing packages into the workstation Python environment was rejected because it creates dependency conflicts and upstream `webrtcvad` requires native build tooling on clean Windows systems. The managed venv avoids both issues, while the compatible VAD wheel removes the compiler requirement. Docker was rejected because it adds an unnecessary VM/runtime dependency and caused CPU inference to pull multi-gigabyte CUDA packages from the default Linux PyTorch distribution. A long-lived inference service was deferred until measured process-start overhead warrants the extra lifecycle complexity. + +### Use a coherent-query centroid heuristic with ambiguity rejection + +All input vectors are normalized before scoring. + +1. Query cohesion is the mean pairwise cosine similarity among the required query vectors. A query below `MinimumClusterCohesion` is rejected before identity comparison. +2. Each identity is represented by the normalized centroid of all stored vectors having the configured model identifier. +3. Candidate similarity is the median cosine similarity from the query vectors to that identity centroid. The median limits the effect of one noisy query sample. +4. The best candidate must meet `MinimumIdentitySimilarity` and exceed the runner-up by `MinimumSimilarityMargin`. The margin is waived when there is no runner-up. + +Defaults are five vectors, `0.75` minimum cohesion, `0.75` minimum identity similarity, and `0.05` minimum margin. These are initial conservative values between the same-speaker and different-speaker similarities shown in Resemblyzer's upstream demonstrations; every threshold is configurable for calibration from local logs. + +### Prune mature identity evidence with fail-safe density clustering + +Once an identity has at least `OutlierPruningMinimumVectors` valid vectors for the configured model, defaulting to 20, run a separate DBSCAN-style clustering pass using cosine similarity. Two vectors are neighbors when their similarity meets `OutlierPruningNeighborSimilarity`; a dense point requires `OutlierPruningMinimumNeighbors`, including itself. This distinguishes isolated or small foreign-speaker groups without forcing every vector toward the matching centroid. + +Pruning keeps a uniquely largest dense cluster only when it contains at least `OutlierPruningMinimumClusterRatio` of the compatible evidence, defaulting to 60%. Vectors outside that dominant cluster are removed, while incompatible-model and malformed rows are left untouched. If no dense cluster dominates, retain all evidence and log the ambiguity rather than arbitrarily selecting one voice. Run pruning after vector additions and identity merges; also prune before additions so a full identity can recover capacity previously occupied by outliers. + +### Persist accepted evidence while keeping downstream identity behavior + +When live or finished matching accepts a known identity, the query vectors and meeting reference are added immediately, bounded and deduplicated, and the existing canonical name is used for transcript relabeling and attendee updates. Final processing still performs candidate-name intersection/promotion and creates unmatched candidates using summary-refined attendees. + +Summarizer overrides attach all available valid current-run vectors to the named identity, merge a current-run unnamed candidate when present, and create a named identity only when evidence or such a candidate exists. Identity deletion cascades to both evidence types. + +Diagnostic automatic merge uses two disjoint query clusters and requires both to select the same target, preserving the existing two-pass confirmation rule. Manual merges always move bounded vector evidence along with aliases, candidates, references, and WAV snippets. + +## Risks / Trade-offs + +- [Initial thresholds may be too strict or permissive for mixed microphone/system audio] → Log cohesion, best similarity, runner-up similarity, margin, sample count, and rejection reason; expose every decision threshold in configuration. +- [Five independent 10-second samples can delay recognition] → Keep required count and minimum speech duration configurable; explicit summarizer assignments can seed an identity with fewer vectors. +- [Provider-created pauses can add silence to a same-speaker sample] → Bound every sample to 60 seconds and rely on Resemblyzer preprocessing to remove non-speech before embedding. +- [Existing identities have no vector evidence] → Do not cross-use WAV evidence automatically; identities become matchable after an explicit assignment or new vector-backed learning. +- [First-time virtual-environment provisioning and model startup add latency] → Pin CPU-only dependencies, content-version and reuse the environment, batch samples, warm non-blockingly, bound commands, and serialize encoder invocations to avoid concurrent model memory spikes. +- [A false positive can contaminate an identity with five vectors] → Require query cohesion, an absolute similarity threshold, an ambiguity margin, and two independent clusters for automatic merges. +- [Density clustering could discard a legitimate secondary acoustic mode] → Do not prune below 20 vectors or without a uniquely dominant 60% cluster; expose the neighborhood and dominance settings and log every decision. +- [Changing the encoder model invalidates comparisons] → Store and filter by model identifier; require an explicit configuration/migration decision for future model upgrades. + +## Migration Plan + +1. Apply the additive SQLite table/index migration while the flag remains disabled. +2. Provision and warm the configured local virtual environment, then enable Resemblyzer explicitly. +3. Calibrate thresholds from decision logs and corrected summarizer assignments. +4. Roll back by disabling the feature flag; the existing WAV/Azure backend and its stored snippets remain intact, while vector rows stay dormant. + +## Open Questions + +None. diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/proposal.md b/openspec/changes/add-resemblyzer-speaker-recognition/proposal.md new file mode 100644 index 0000000..0370aae --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/proposal.md @@ -0,0 +1,31 @@ +## Why + +The current speaker-recognition path persists WAV snippets and depends on Azure Speech plus optional pyannote validation. Meeting Assistant needs an opt-in, fully local alternative that persists compact voice embeddings and can accumulate stronger identity evidence over time without replacing the existing backend. + +## What Changes + +- Add an application-level feature flag that selects a separate Resemblyzer speaker-recognition backend while leaving the current WAV/Azure backend unchanged when disabled. +- Create temporary WAV samples during recording, encode each retained sample locally into a 256-value Resemblyzer voice vector, and persist vectors rather than WAV data for this backend. +- Require a configurable minimum of five coherent vectors for automatic recognition, compare their cluster with known identity vector clusters using configurable cosine-similarity, cohesion, and ambiguity thresholds, and learn the accepted vectors. +- Store at most a configurable 1,000 vectors per identity and retain vectors through identity naming, summarizer overrides, deletion, and merge operations. +- Keep transcript relabeling, attendee updates, candidate-name learning, meeting references, and identity-management behavior consistent with the existing speaker-identification flow. +- Add a managed local Python virtual environment for Resemblyzer with CPU-only PyTorch and document its configuration and tuning parameters. +- Merge consecutive transcript lines from any STT backend for the same diarized speaker into recognition samples despite provider-created pauses, while capping every sample at 60 seconds by default. +- Treat five vectors only as the default automatic-decision threshold, retain all qualifying current-run vectors up to the identity limit, and prune accumulated outliers with a configurable density-clustering pass once an identity has at least 20 compatible vectors. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `meeting-transcription`: Add an opt-in local voice-vector speaker-recognition backend and define its collection, matching, persistence, and lifecycle behavior. + +## Impact + +- Speaker identity options, dependency registration, recording sample retention, and live/final identification orchestration. +- SQLite schema and identity merge/management tools gain a separate voice-vector collection. +- A local Python installation is required only when the feature is enabled; Resemblyzer and CPU-only PyTorch are isolated in an application-managed virtual environment. +- Canonical configuration and speaker-identification documentation gain the feature flag and tunable matching thresholds. diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/specs/meeting-transcription/spec.md b/openspec/changes/add-resemblyzer-speaker-recognition/specs/meeting-transcription/spec.md new file mode 100644 index 0000000..3207201 --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/specs/meeting-transcription/spec.md @@ -0,0 +1,361 @@ +## ADDED Requirements + +### Requirement: Speaker recognition can use local Resemblyzer voice vectors +Meeting Assistant SHALL expose `SpeakerIdentification:Resemblyzer:Enabled` as an application-level feature flag that defaults to disabled. + +When the feature is disabled, Meeting Assistant SHALL use the existing WAV-snippet, Azure Speech, and optional pyannote speaker-identification backend without reading or writing Resemblyzer voice vectors. + +When the feature is enabled, Meeting Assistant SHALL use a separate local Resemblyzer speaker-identification backend and SHALL NOT invoke the Azure Speech or pyannote speaker-identity matchers. + +The Resemblyzer backend SHALL create temporary WAV samples from diarized same-speaker runs during recording, SHALL encode each retained sample locally as a versioned 256-value voice vector, and SHALL NOT persist those temporary WAV samples as identity evidence. + +Resemblyzer sample spans retained for one speaker SHALL not overlap. For transcription providers that only produce diarized speakers during finalization, Meeting Assistant SHALL extract qualifying non-overlapping samples from the completed mixed recording. + +Automatic matching SHALL wait until the configured required number of valid vectors is available for a diarized speaker. The default required count SHALL be five. + +The required vector count SHALL be an automatic-decision threshold and SHALL NOT cap collection, encoding, or persistence. After the threshold is met, Meeting Assistant SHALL retain every distinct qualifying current-run vector up to the configured per-identity limit. When a speaker was assigned during live transcription, final processing SHALL attach qualifying vectors collected after that assignment to the same identity. + +The matcher SHALL reject a query cluster whose mean pairwise cosine similarity is below the configured minimum cluster cohesion. For a coherent query, it SHALL represent each known identity by the normalized centroid of compatible stored vectors, SHALL score that identity using the median cosine similarity from query vectors to the centroid, and SHALL select an identity only when the best score meets the configured minimum identity similarity and exceeds the runner-up by the configured minimum similarity margin. The runner-up margin SHALL be waived when only one candidate can be scored. + +The required vector count, minimum cluster cohesion, minimum identity similarity, minimum runner-up margin, encoder model identifier, local runtime settings, and command timeout SHALL be configurable. + +When an identity has at least the configured outlier-pruning minimum number of valid vectors for the active model, defaulting to 20, Meeting Assistant SHALL run a separate cosine-density clustering pass. Neighbor similarity, minimum neighbors, and minimum dominant-cluster ratio SHALL be configurable. + +Meeting Assistant SHALL remove vectors outside the uniquely largest dense cluster only when that cluster meets the configured minimum ratio of compatible evidence, defaulting to 60%. When no cluster qualifies or the largest cluster is tied, Meeting Assistant SHALL retain the evidence and log that pruning was skipped. Vectors for other model identifiers SHALL NOT be removed by this pass. + +When enabled, the local encoder SHALL provision and reuse an application-managed Python virtual environment under the configured runtime folder. It SHALL install a pinned CPU-only PyTorch distribution and Windows-compatible VAD wheel without requiring Docker or a system-wide Python package installation. + +The local encoder SHALL reject missing, malformed, non-finite, zero-magnitude, wrong-count, and wrong-dimension results without persisting them or falling back to the existing remote matcher. + +#### Scenario: Disabled feature preserves existing backend +- **GIVEN** Resemblyzer speaker recognition is disabled +- **WHEN** Meeting Assistant tries to identify a diarized speaker +- **THEN** it uses the existing WAV-snippet speaker-identification backend +- **AND** does not create or compare Resemblyzer voice vectors + +#### Scenario: Automatic matching waits for five vectors +- **GIVEN** Resemblyzer speaker recognition requires five vectors +- **AND** an unresolved diarized speaker has four valid samples +- **WHEN** live speaker identification runs +- **THEN** Meeting Assistant does not compare that speaker with known identities +- **WHEN** a fifth valid sample becomes available +- **THEN** Meeting Assistant can encode and compare the coherent five-vector cluster + +#### Scenario: Five vectors do not cap retained evidence +- **GIVEN** Resemblyzer automatic matching requires five vectors +- **AND** a meeting yields eight distinct qualifying vectors for one speaker +- **WHEN** Meeting Assistant accepts or creates that speaker identity +- **THEN** it stores all eight vectors within the configured identity limit + +#### Scenario: Final processing retains evidence collected after a live match +- **GIVEN** a diarized speaker was matched after five vectors during live transcription +- **AND** three more qualifying vectors were collected later in the meeting +- **AND** the finished transcript already uses the matched speaker's name while retained samples use the original diarized label +- **WHEN** final speaker processing runs with the existing mapping +- **THEN** the three later vectors are attached to the matched identity + +#### Scenario: Mature identity outliers are pruned +- **GIVEN** an identity has at least 20 compatible vectors +- **AND** a uniquely largest cosine-density cluster contains at least 60% of them +- **WHEN** vector evidence is added or identities are merged +- **THEN** vectors outside the dominant cluster are removed +- **AND** the pruning decision and removed count are logged + +#### Scenario: Ambiguous clusters are retained +- **GIVEN** an identity has at least 20 compatible vectors split between equally large or non-dominant dense clusters +- **WHEN** outlier pruning runs +- **THEN** Meeting Assistant removes no vectors +- **AND** logs that no uniquely dominant cluster qualified + +#### Scenario: Incoherent query cluster is rejected +- **GIVEN** five query vectors have mean pairwise cosine similarity below the configured cohesion threshold +- **WHEN** Resemblyzer speaker identification runs +- **THEN** Meeting Assistant does not assign the speaker to a known identity +- **AND** logs the measured cohesion and rejection reason + +#### Scenario: Similar and unambiguous cluster is accepted +- **GIVEN** a coherent five-vector query cluster +- **AND** its median similarity to Chris's vector centroid meets the configured identity threshold +- **AND** its score exceeds every other scored identity by the configured margin +- **WHEN** Resemblyzer speaker identification runs +- **THEN** Meeting Assistant identifies the diarized speaker as Chris +- **AND** adds the five query vectors to Chris's identity within the configured limit + +#### Scenario: Ambiguous best cluster is rejected +- **GIVEN** a coherent five-vector query cluster meets the identity similarity threshold for Chris +- **AND** another identity's score is within the configured runner-up margin +- **WHEN** Resemblyzer speaker identification runs +- **THEN** Meeting Assistant leaves the diarized speaker unresolved +- **AND** logs both candidate scores and the insufficient margin + +#### Scenario: Encoder failure preserves diarized labels +- **GIVEN** Resemblyzer speaker recognition is enabled +- **WHEN** the local encoder fails or returns invalid vectors +- **THEN** Meeting Assistant does not invoke the existing Azure or pyannote identity matcher as a fallback +- **AND** keeps the available diarized speaker labels + +#### Scenario: Encoder provisions an isolated CPU environment +- **GIVEN** Resemblyzer speaker recognition is enabled +- **AND** its versioned virtual environment is not ready +- **WHEN** encoder warm-up runs +- **THEN** Meeting Assistant creates the virtual environment with the configured Python command +- **AND** installs the configured CPU-only PyTorch, Windows-compatible VAD, and Resemblyzer versions inside that environment +- **AND** does not invoke Docker + +## MODIFIED Requirements + +### Requirement: Speaker identity samples require uninterrupted speech +Meeting Assistant SHALL only retain speaker identity samples after a diarized speaker has produced a same-speaker sample span meeting the configured minimum duration. + +The default minimum sample duration SHALL be 10 seconds. + +Meeting Assistant SHALL combine consecutive transcript segments for the same diarized speaker into one sample span even when the transcription provider splits those segments around pauses. Provider-created pauses SHALL remain in the bounded extracted WAV but SHALL NOT count toward the configured minimum speaker-audio duration. + +This aggregation behavior SHALL apply uniformly to diarized segments from every configured STT backend, whether segments arrive during live transcription or become available during finalization. + +Meeting Assistant SHALL end the pending span when a different diarized speaker interrupts it or when the configured maximum sample duration is reached. The default maximum sample duration SHALL be 60 seconds, and no extracted recognition WAV SHALL exceed it. + +When Resemblyzer recognition is enabled, Meeting Assistant SHALL start a new non-overlapping sample after accepting the previous sample from the same speaker. + +#### Scenario: Short speaker span is not retained +- **GIVEN** the configured minimum sample duration is 10 seconds +- **WHEN** a diarized speaker produces only 8 seconds of uninterrupted speech +- **THEN** Meeting Assistant does not retain a speaker identity sample for that span + +#### Scenario: Adjacent same-speaker segments form a sample +- **GIVEN** the configured minimum sample duration is 10 seconds +- **WHEN** a diarized speaker produces consecutive provider segments containing at least 10 seconds of speaker audio without another speaker interrupting +- **THEN** Meeting Assistant retains one speaker identity sample covering the continuous span + +#### Scenario: Provider pause does not split a same-speaker sample +- **GIVEN** any configured STT backend emits consecutive lines for `Guest01` with a pause longer than the former segment-gap threshold +- **WHEN** no differently labeled speaker appears between those lines +- **THEN** Meeting Assistant combines the lines into one speaker-recognition sample span +- **AND** counts only their diarized speaker-audio durations toward the minimum + +#### Scenario: Speaker sample is capped at 60 seconds +- **GIVEN** the maximum sample duration is 60 seconds +- **WHEN** consecutive transcript lines for one speaker span more than 60 seconds +- **THEN** every extracted speaker-recognition WAV is at most 60 seconds long + +#### Scenario: Different speaker interrupts pending span +- **GIVEN** the configured minimum sample duration is 10 seconds +- **WHEN** `Guest01` speaks for 8 seconds and then `Guest02` speaks +- **THEN** Meeting Assistant discards the pending `Guest01` span instead of retaining or later extending it + +### Requirement: Meeting Assistant learns speaker identities locally +Meeting Assistant SHALL maintain a local SQLite speaker identity database in the user's application data folder. + +The speaker identity database SHALL store speaker identities, optional canonical names, aliases, candidate names, meeting file references, a bounded set of WAV snippets per identity for the existing backend, and a separate bounded set of versioned voice vectors per identity for the Resemblyzer backend. + +Each persisted voice vector SHALL store its model identifier, dimension, creation time, and a fingerprint that makes adding the same vector to the same identity idempotent. + +Meeting file references SHALL include the meeting note file address and the transcript file address. + +Meeting Assistant SHALL calculate speaker identity participation counts from meeting file references when needed instead of persisting a denormalized transcript count. + +Each speaker identity SHALL store a last-modified timestamp used by active-age filtering, and Meeting Assistant SHALL update it whenever the identity is created or modified by identification, candidate updates, snippet changes, voice-vector changes, reference changes, or merge operations. + +The configured maximum snippet count and maximum voice-vector count per identity SHALL prevent unbounded growth. The default maximum voice-vector count SHALL be 1,000. + +Except for adding newly accepted Resemblyzer match evidence and its meeting reference, final candidate elimination, canonical promotion, and new unmatched identity creation SHALL happen only after transcription is finished and after automatic summary generation has completed, using the latest meeting note frontmatter. + +When the summary agent records a speaker override from a diarized transcript label to a named speaker, final speaker identity processing SHALL attach the current run's evidence to an existing identity with that name when one exists, or create a new canonical speaker identity with that name when none exists. For the existing backend that evidence SHALL be the resolved WAV snippet; for the Resemblyzer backend it SHALL be all available valid current-run voice vectors up to the configured per-run count. Meeting Assistant SHALL NOT create a new speaker identity for an override when no current run evidence or current run candidate can be resolved for the source speaker label. + +When a speaker override maps a current-run unnamed candidate to an existing named identity, Meeting Assistant SHALL merge the candidate's meeting reference and useful backend-specific evidence into the named identity instead of leaving a duplicate candidate. + +When the summary agent records that a speaker identity was wrongfully matched, final speaker identity processing SHALL delete the matching identity and all of its WAV and voice-vector evidence from the local speaker identity database so it cannot be matched again unless it is newly created in the future. + +#### Scenario: Unknown speaker is learned from meeting attendees +- **WHEN** a finished transcript contains an unmatched diarized speaker and the meeting note has attendees +- **THEN** Meeting Assistant stores a new unnamed speaker identity with candidate names from the attendees that were not already matched in that meeting +- **AND** stores a meeting file reference for that identity + +#### Scenario: Speaker snippets are bounded +- **WHEN** Meeting Assistant adds a snippet for an identity that already has the configured maximum number of snippets +- **THEN** Meeting Assistant does not store more snippets for that identity + +#### Scenario: Speaker voice vectors are bounded +- **GIVEN** the Resemblyzer vector limit is 1,000 +- **WHEN** Meeting Assistant adds vectors to an identity that already has 1,000 stored vectors +- **THEN** Meeting Assistant does not store more than 1,000 vectors for that identity + +#### Scenario: Retried vector evidence is idempotent +- **GIVEN** an identity already contains a voice vector +- **WHEN** Meeting Assistant retries adding the same vector to that identity +- **THEN** it stores only one copy of that vector + +#### Scenario: Identity modification updates active-age timestamp +- **WHEN** Meeting Assistant creates, identifies, updates candidates for, stores snippets or voice vectors for, stores references for, or merges a speaker identity +- **THEN** Meeting Assistant updates that identity's last-modified timestamp + +#### Scenario: Final speaker identity learning uses summary-refined attendees +- **GIVEN** the summary agent changes meeting note attendees during automatic summary generation +- **WHEN** Meeting Assistant performs final speaker identity learning and candidate creation +- **THEN** it uses the attendee list from the meeting note after the summary agent changes + +#### Scenario: Speaker override attaches to existing identity +- **GIVEN** the speaker identity database contains canonical speaker `Sabrina` +- **AND** the summary agent records that transcript speaker `Guest-01` is `Sabrina` +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant stores the meeting reference and current backend-specific speaker evidence on Sabrina's identity +- **AND** does not create a separate unnamed candidate for `Guest-01` + +#### Scenario: Resemblyzer override stores available vectors +- **GIVEN** Resemblyzer speaker recognition is enabled +- **AND** the current run has three valid vectors for `Guest-01` +- **WHEN** the summary agent assigns `Guest-01` to `Sabrina` +- **THEN** Meeting Assistant attaches those three vectors to Sabrina's identity +- **AND** does not require five vectors for the explicit assignment + +#### Scenario: Speaker override creates named identity +- **GIVEN** the speaker identity database has no accepted name `Sabrina` +- **AND** the summary agent records that transcript speaker `Guest-01` is `Sabrina` +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant creates a canonical speaker identity named `Sabrina` +- **AND** stores the meeting reference and current backend-specific speaker evidence on that identity + +#### Scenario: Speaker override with missing source sample is skipped +- **GIVEN** the speaker identity database has no accepted name `Sabrina` +- **AND** the summary agent records that transcript speaker `Guest-5` is `Sabrina` +- **AND** final speaker identity processing has no sample, vector, or segment for `Guest-5` +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant does not create a speaker identity for `Sabrina` + +#### Scenario: Speaker identity deletion removes a wrong match +- **GIVEN** the speaker identity database contains canonical speaker `Sabrina` +- **AND** the summary agent records that `Sabrina` was wrongfully matched +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant removes Sabrina's identity and backend-specific evidence from the speaker identity database +- **AND** the relabeled transcript uses `Removed-1` instead of `Sabrina` + +### Requirement: Speaker identities can be merged diagnostically +Meeting Assistant SHALL expose a diagnostic endpoint that merges duplicate speaker identities. + +The merge process SHALL compare recently-created identities, using a configurable recent age that defaults to two weeks, against all other identities using the selected backend's candidate-scoring strategy. + +For the existing WAV backend, the merge process SHALL require a match and a second validation match using a different source sample. For the Resemblyzer backend, it SHALL require two disjoint coherent source-vector clusters to select the same target identity. + +When identities are merged, Meeting Assistant SHALL retain one identity, move useful names from the merged identity into aliases, combine meeting file references, retain bounded sets of snippets and voice vectors from both identities, and append an audit line to each referenced transcript in the form ` and were merged`. + +When combined Resemblyzer evidence exceeds the configured vector limit, Meeting Assistant SHALL keep no more than that limit, preferring the most recently created distinct vectors. + +After combining Resemblyzer evidence, Meeting Assistant SHALL apply the configured mature-identity outlier-pruning policy. + +#### Scenario: Recently-created duplicate identity is merged +- **GIVEN** a recently-created identity and an older identity have matching backend-specific speaker evidence +- **WHEN** the diagnostic merge endpoint is triggered +- **THEN** Meeting Assistant validates the match twice with different source evidence +- **AND** merges the recent identity into the older identity +- **AND** stores the recent identity name as an alias on the retained identity +- **AND** keeps meeting file references and bounded backend-specific evidence from both identities +- **AND** appends the merge audit line to the referenced transcripts + +#### Scenario: Resemblyzer merge needs two clusters +- **GIVEN** Resemblyzer speaker recognition requires five vectors per cluster +- **AND** a recent identity has ten vectors split into two coherent clusters +- **WHEN** both clusters independently match the same target identity +- **THEN** Meeting Assistant merges the recent identity into that target + +#### Scenario: Old identities are not used as merge sources +- **GIVEN** two identities older than the configured recent age +- **WHEN** the diagnostic merge endpoint is triggered +- **THEN** Meeting Assistant does not compare them as source identities + +### Requirement: Speaker identity matches relabel transcripts +Meeting Assistant SHALL attempt to match unknown diarized speaker evidence against known speaker identities ordered by calculated meeting reference count. + +When Resemblyzer speaker recognition is disabled, matching SHALL use the existing dedicated Azure Speech diarization verifier and optional pyannote validator with WAV snippets. When Resemblyzer speaker recognition is enabled, matching SHALL instead use only compatible locally calculated Resemblyzer voice-vector clusters. + +For the existing WAV backend, the matcher SHALL test at most the configured batch size of known people per matching round and continue with later batches until a match is found or no candidates remain. The Resemblyzer backend SHALL score the capped candidate set together so ambiguity is measured against the global runner-up. + +The matcher SHALL prioritize identities whose canonical name or aliases match current meeting attendees. + +After attendee-matched identities, the matcher SHALL order identities by calculated meeting reference count, filter out non-attendee identities whose last update is older than the configured active age, and cap the candidate set at the configured maximum match candidate count. + +When a match is confirmed, Meeting Assistant SHALL store a meeting file reference and the accepted backend-specific evidence for that identity within its configured limit. + +When a match is confirmed and the identity has a canonical name, Meeting Assistant SHALL rewrite finished transcript segments for that diarized speaker with the canonical name. + +When a match is confirmed and the matched speaker is not already listed in meeting note attendees by display name or alias, Meeting Assistant SHALL add the speaker display name to the attendee list. + +When a match is confirmed and the meeting note attendees contain both the speaker display name and one or more accepted aliases for that same speaker, Meeting Assistant SHALL remove the alias attendee entries and keep the display name entry. + +When Meeting Assistant writes attendees from calendar metadata, it SHALL match attendee display names exactly against known identity canonical names and aliases, replace matches with the identity display name, and deduplicate attendees that map to the same identity. + +#### Scenario: Finished transcript is relabeled after a confirmed match +- **GIVEN** the speaker identity database contains canonical speaker `Chris` +- **WHEN** a finished transcript has diarized speaker `Guest03` and the selected matching backend confirms it is `Chris` +- **THEN** Meeting Assistant rewrites `Guest03` segments in the transcript as `Chris` + +#### Scenario: Confirmed match stores meeting reference +- **GIVEN** the speaker identity database contains canonical speaker `Chris` +- **WHEN** a finished transcript has diarized speaker `Guest03` and the selected matching backend confirms it is `Chris` +- **THEN** Meeting Assistant stores the meeting note and transcript file addresses as a reference for `Chris` +- **AND** stores the accepted backend-specific evidence within its configured limit + +#### Scenario: Confirmed match removes duplicate aliases +- **GIVEN** the speaker identity database contains canonical speaker `Christopher` with alias `Chris` +- **AND** the meeting note attendees contain both `Christopher` and `Chris ` +- **WHEN** live or final speaker matching confirms a diarized speaker is `Christopher` +- **THEN** Meeting Assistant keeps `Christopher` in the meeting note attendees +- **AND** removes `Chris ` from the meeting note attendees + +### Requirement: Speaker matching runs during active transcription +Meeting Assistant SHALL start speaker identity matching only after the configured initial transcription duration has elapsed. + +For backends that emit live diarized transcript segments, Meeting Assistant SHALL keep a bounded in-memory sliding audio buffer with chunk timestamps and extract candidate WAV samples from that buffer when live diarized segments arrive. + +Meeting Assistant SHALL keep only the configured best candidate samples per diarized speaker in memory. Better samples SHALL be preferred when the segment looks like a continuous medium-length sentence. When Resemblyzer is enabled, accepted samples for one speaker SHALL be non-overlapping and the retained count SHALL be at least the configured required vector count. + +Meeting Assistant SHALL periodically match unresolved diarized speaker evidence while transcription is active and attempt to match it against the local identity database. + +Meeting Assistant SHALL run live matching incrementally at the configured interval only when at least one new unmapped diarized speaker sample appears or the meeting note attendee frontmatter changes while unmapped speaker samples still exist. For Resemblyzer, additional samples for an existing unresolved speaker SHALL also trigger another attempt so an earlier insufficient-vector result does not suppress matching when the required count becomes available. + +When a speaker is matched during transcription, Meeting Assistant SHALL rewrite already-written live transcript segments for that diarized speaker and write future transcript segments using the canonical name. + +For the existing WAV backend, live speaker matching SHALL be read-only with respect to the speaker identity database. For the Resemblyzer backend, a confirmed live match SHALL persist the accepted deduplicated voice vectors and meeting reference immediately so an assignment made during transcription is learned. Candidate elimination, canonical promotion, and new unmatched identity creation SHALL still happen only after transcription is finished and after automatic summary generation has completed, using the latest meeting note frontmatter. + +For backends that only provide diarization after finalization, Meeting Assistant SHALL defer speaker identity matching until finished diarization is available, extract candidate samples from the completed temporary recording, complete identity matching, and only then allow summary generation to start. + +#### Scenario: Matching waits for useful speech duration +- **WHEN** transcription has been active for less than the configured speaker identification initial delay +- **THEN** Meeting Assistant does not run speaker identity matching yet + +#### Scenario: Live matching uses in-memory speaker samples +- **WHEN** a live diarized transcript segment identifies an unresolved speaker +- **THEN** Meeting Assistant extracts a temporary WAV sample for that segment from the in-memory sliding audio buffer +- **AND** uses retained speaker evidence for live identity matching without reading the temporary recording file + +#### Scenario: Live match rewrites current and future transcript writes +- **WHEN** periodic matching confirms that diarized speaker `Guest03` is canonical speaker `Chris` +- **THEN** already-written live transcript segments for `Guest03` are rewritten as `Chris` +- **AND** later live transcript segments for `Guest03` are written as `Chris` + +#### Scenario: Resemblyzer live match persists vectors +- **GIVEN** Resemblyzer speaker recognition is enabled +- **WHEN** periodic matching confirms a coherent five-vector cluster for `Guest03` as canonical speaker `Chris` +- **THEN** Meeting Assistant stores those vectors and the current meeting reference on Chris's identity +- **AND** does not persist the temporary WAV samples + +#### Scenario: New live speaker triggers another identification round +- **GIVEN** live matching already checked the current unresolved speaker samples +- **WHEN** a new unmapped diarized speaker sample appears +- **THEN** Meeting Assistant runs another live matching round at the next configured interval + +#### Scenario: Additional samples unlock live Resemblyzer matching +- **GIVEN** an earlier live attempt had fewer than five samples for an unresolved speaker +- **AND** no new speaker or attendee change occurs +- **WHEN** that speaker accumulates five qualifying samples +- **THEN** Meeting Assistant attempts matching again at the next configured interval +- **AND** does not repeatedly match unchanged evidence + +#### Scenario: Attendee changes trigger another identification round +- **GIVEN** live matching already checked unresolved speaker samples +- **WHEN** the meeting note attendee frontmatter changes +- **THEN** Meeting Assistant runs another live matching round at the next configured interval using the latest attendees + +#### Scenario: Final decisions use summary-refined attendees +- **WHEN** live matching finds or does not find a possible speaker identity during transcription +- **THEN** Meeting Assistant does not eliminate candidate names, promote canonical names, or create unmatched identities during that live pass +- **AND** the final speaker identity pass uses the latest meeting note attendees after transcription finishes diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/tasks.md b/openspec/changes/add-resemblyzer-speaker-recognition/tasks.md new file mode 100644 index 0000000..967a772 --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/tasks.md @@ -0,0 +1,73 @@ +## 1. Voice-vector persistence + +- [x] 1.1 Add a failing database behavior test for versioned, deduplicated voice vectors and cascade deletion +- [x] 1.2 Add the voice-vector entity, EF mapping, and additive SQLite schema migration + +## 2. Local Resemblyzer encoding + +- [x] 2.1 Add a failing encoder behavior test for batching WAV samples into validated 256-value vectors +- [x] 2.2 Implement the bounded local Resemblyzer encoder and non-blocking feature-gated warm-up +- [x] 2.3 Add behavior coverage for malformed, wrong-dimension, non-finite, and failed encoder results + +## 3. Tunable cluster matching + +- [x] 3.1 Add a failing behavior test for accepting a coherent, similar, unambiguous five-vector cluster +- [x] 3.2 Implement normalized-centroid, median-cosine, cohesion, threshold, and runner-up-margin scoring +- [x] 3.3 Add behavior coverage for insufficient, incoherent, below-threshold, and ambiguous clusters + +## 4. Resemblyzer identity lifecycle + +- [x] 4.1 Add a failing service behavior test proving a live vector match relabels the speaker and persists five vectors without WAV snippets +- [x] 4.2 Implement the separate Resemblyzer identification service with existing candidate ordering, naming, attendee, reference, and transcript outcomes +- [x] 4.3 Add and pass behavior tests for summary overrides with fewer than five vectors, unmatched learning, deduplication, and the 1,000-vector cap + +## 5. Recording and merge integration + +- [x] 5.1 Add behavior tests and implement non-overlapping Resemblyzer sample collection with at least the configured required count +- [x] 5.2 Add behavior tests and implement application-level feature selection without Azure/pyannote identity fallback +- [x] 5.3 Add behavior tests and implement two-cluster Resemblyzer diagnostic merging plus bounded vector retention in manual merges +- [x] 5.4 Expose vector counts in identity-management tools while keeping WAV playback operations separate + +## 6. Configuration and verification + +- [x] 6.1 Add the disabled-by-default canonical configuration and document runtime, persistence, and tuning behavior +- [x] 6.2 Run focused speaker, schema, encoder, matching, merge, recording, and workflow-tool tests +- [x] 6.3 Run the full solution test suite and validate the OpenSpec change strictly + +## 7. Local virtual-environment correction + +- [x] 7.1 Add a failing behavior test for provisioning a versioned venv with CPU-only PyTorch and no Docker command +- [x] 7.2 Replace the Docker encoder with managed-venv provisioning and direct venv Python batch encoding +- [x] 7.3 Replace Docker-specific Resemblyzer configuration and documentation with Python/venv settings +- [x] 7.4 Run focused and full tests, validate OpenSpec strictly, and verify enabled application warm-up through logs + +## 8. Sample-duration tuning + +- [x] 8.1 Add a failing configuration-default test for a 10-second minimum speaker sample +- [x] 8.2 Change the sample-duration default and canonical configuration to 10 seconds and update documentation +- [x] 8.3 Run focused tests, validate OpenSpec strictly, restart the enabled application, and verify health + +## 9. STT segment aggregation and sample cap + +- [x] 9.1 Add a failing live-collector behavior test proving consecutive same-speaker STT lines survive provider-created pauses +- [x] 9.2 Implement same-speaker aggregation for live and finalized samples while excluding provider pauses from the minimum speech duration +- [x] 9.3 Add a failing behavior test proving recognition WAVs never exceed the configurable 60-second default +- [x] 9.4 Implement and document the maximum sample duration across live and finalized collection +- [x] 9.5 Run focused and full tests, refactor, validate OpenSpec strictly, and verify operational readiness without interrupting active work + +## 10. Complete evidence retention and outlier pruning + +- [x] 10.1 Add a failing service behavior test proving five vectors unlock a decision without capping all qualifying current-run evidence +- [x] 10.2 Retain, encode, and persist all qualifying current-run vectors up to the configured identity limit, including evidence collected after a live match +- [x] 10.3 Add failing behavior tests for dominant density-cluster pruning and fail-safe ambiguous-cluster retention at the 20-vector floor +- [x] 10.4 Implement configurable cosine-density outlier pruning after vector additions and identity merges +- [x] 10.5 Document tuning settings, run focused/full tests and sequential refactor passes, validate OpenSpec strictly, and verify the enabled application without interrupting active work + +## 11. Release verification corrections + +- [x] 11.1 Reproduce and fix final evidence retention after live transcript relabeling +- [x] 11.2 Reproduce and fix live matching when an existing speaker reaches the required sample count +- [x] 11.3 Run focused/full tests, strictly validate OpenSpec, and verify operational readiness +- [x] 11.4 Split evidence-loading queries to avoid multiplying stored WAV blobs across vector, reference, and name rows + +Release verification (2026-09-11): 128 initial focused tests passed. Both lifecycle regressions were reproduced and fixed; the full suite passed all 519 tests with compilation complete after two timing-sensitive audio tests failed during the concurrent Windows build and passed in isolation. The Windows target built successfully and strict OpenSpec validation passed. Local `/health` returned `ok`, recording status was idle, and application logs showed successful Resemblyzer warm-up, local encoding, and vector pruning. The release corrections were verified through behavior tests; the running workstation process was not restarted. diff --git a/openspec/changes/add-transcription-pause-controls/.openspec.yaml b/openspec/changes/add-transcription-pause-controls/.openspec.yaml new file mode 100644 index 0000000..032461f --- /dev/null +++ b/openspec/changes/add-transcription-pause-controls/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-02 diff --git a/openspec/changes/add-transcription-pause-controls/design.md b/openspec/changes/add-transcription-pause-controls/design.md new file mode 100644 index 0000000..f48c356 --- /dev/null +++ b/openspec/changes/add-transcription-pause-controls/design.md @@ -0,0 +1,69 @@ +## Context + +Meeting Assistant currently has one active `RecordingRun` that owns its speech-recognition pipeline, transcript session, temporary WAV, speaker mappings, and live speaker samples. Mixed audio chunks are written to both the WAV and the active pipeline. Transcript inactivity is tracked separately, and Windows inactivity toasts share a notification group but expose only stop/continue callbacks. + +The Azure Speech SDK `ConversationTranscriber` has start and stop operations but no pause operation. Stopping ends the ongoing real-time recognition session, and recreating or restarting recognition can reset backend speaker IDs. Continuous recognition does support an open push stream containing silence, so the provider session can remain alive without receiving the meeting's actual audio during a pause. + +## Goals / Non-Goals + +**Goals:** + +- Pause the transcription of an active meeting without finishing its recording run. +- Prevent actual paused audio from reaching either durable temporary audio or a remote/local transcription backend. +- Preserve the same recognition pipeline, Azure conversation transcriber, transcript session, speaker mappings, collected speaker samples, and artifacts. +- Keep normal finish, cancel/discard, and profile-switch controls usable while paused. +- Treat intentional pause as distinct from inactivity and dismiss notifications that become obsolete when transcript text resumes. + +**Non-Goals:** + +- Suspend microphone or loopback device capture at the operating-system layer. +- Disconnect or stop the configured speech-recognition backend while paused. +- Reduce Azure connection time or billing during a pause. +- Persist pause state across process restarts or stopped meeting backlog replay. +- Add a pause hotkey. + +## Decisions + +### Preserve the pipeline by substituting silence at the recording-run boundary + +For each mixed audio chunk captured while paused, the coordinator will create an equal-length zeroed PCM chunk and route that chunk to the temporary WAV, live speaker audio buffer, and current speech-recognition pipeline. Real captured samples are discarded at that boundary. + +This keeps audio duration and transcript timestamps aligned while keeping the Azure push stream active. Dropping chunks entirely was rejected because a sufficiently long input gap can stop or reconnect a provider session. Calling `StopTranscribingAsync` was rejected because it terminates the ongoing Azure operation and cannot guarantee stable backend speaker IDs when started again. + +### Keep pause state on the active recording run + +The run will expose a thread-safe paused flag and coordinator pause/unpause operations. `RecordingStatus` will expose the flag so tray rendering and the loopback status endpoint observe the same state. Pause requests outside an active capture will be harmless, and normal stop/abort paths will remain authoritative. + +The paused flag will not become another post-recording process state: a paused run is still an active recording and continues to offer Finish and Cancel. + +### Separate transcript inactivity from a maximum continuous pause + +The inactivity safeguard will skip transcript-inactivity prompting and its ordinary auto-stop while the run is paused. Pausing will dismiss all outstanding inactivity notifications. A separate `MaximumPauseDuration`, defaulting to four hours, will normally stop a run that remains continuously paused for that duration without showing inactivity notifications. Unpausing will clear the paused-duration timer, set a new transcript-inactivity baseline, and advance the activity version so prompt thresholds start over rather than firing immediately. + +### Make notification lifecycle part of the prompt-service contract + +`IMeetingInactivityPromptService` will gain an operation to dismiss all active inactivity prompts. The Windows implementation will remove the notification group from Action Center and clear its pending callback registry. The no-op and test implementations will implement the same public contract. + +The coordinator will call dismissal only after a non-empty live segment has been durably appended, matching the user's observable meaning of a new transcription being written. Notification callbacks will verify that their originating run is still current before changing recording state. + +Transcript processing requests notification dismissal without awaiting it. Profile switching holds the coordinator gate while draining the old transcript reader, so awaiting dismissal from that reader would create a circular wait on the same gate. Deferred cleanup retains the gate and current-run check, uses the capture cancellation token, and handles cancellation when capture stops. Durable transcript activity still invalidates stale notification actions immediately, before deferred cleanup runs. + +### Model pause as one toggling tray action + +The active-recording tray menu will show `Pause transcription` while running and `Unpause transcription` while paused. It will stay in the fine-grained controls section below the dedicated `Finish meeting` section. The inactivity toast will add `Pause transcription` alongside the existing Yes/No stop controls. + +## Risks / Trade-offs + +- **Azure may still reconnect for unrelated transport failures during a long pause** → Reuse the existing Azure reconnect behavior; the application preserves meeting artifacts and resets backend speaker assumptions only if Azure actually creates a new SDK session. +- **Silence consumes provider connection time and temporary WAV space** → Accept this to preserve provider/session continuity and timestamp alignment; document that pause is not a cost-suspension mechanism. +- **A forgotten pause could otherwise keep capture and provider resources alive indefinitely** → Apply the separate four-hour maximum continuous pause while keeping the shorter transcript-inactivity prompts fully suppressed. +- **A transcript result already buffered by the provider can arrive just after pause** → Keep and write it because it represents audio submitted before the pause boundary. +- **A stale notification action could affect a later meeting** → Clear pending callbacks on dismissal and verify the originating run before applying stop or pause. + +## Migration Plan + +No data or configuration migration is required. Deploy the updated executable normally. Rollback restores the previous controls; existing meeting artifacts and speaker identities remain compatible. + +## Open Questions + +None. diff --git a/openspec/changes/add-transcription-pause-controls/proposal.md b/openspec/changes/add-transcription-pause-controls/proposal.md new file mode 100644 index 0000000..95e3ead --- /dev/null +++ b/openspec/changes/add-transcription-pause-controls/proposal.md @@ -0,0 +1,28 @@ +## Why + +Transcript-inactivity notifications remain visible after transcription has resumed, and there is no way to intentionally suspend transcription during a break without ending the meeting and losing the active recognition session. Meeting Assistant should distinguish intentional pauses from accidental inactivity while preserving the meeting run and speaker context. + +## What Changes + +- Dismiss every outstanding transcript-inactivity notification as soon as a new non-empty transcript segment is written. +- Allow an active meeting transcription to be paused and unpaused without ending the recording run, replacing captured audio with silence so paused audio is neither persisted nor sent to the transcription backend. +- Keep the active speech-recognition pipeline, including the Azure conversation transcriber session, run-local speaker mappings, artifacts, and normal finish/cancel controls intact across a pause. +- Suspend transcript-inactivity prompts and the ordinary inactivity stop while paused, but normally stop a meeting that remains continuously paused for four hours by default; restart inactivity timing when transcription is unpaused. +- Add `Pause transcription` / `Unpause transcription` to the active-recording tray menu and add `Pause transcription` to transcript-inactivity notifications. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `meeting-recording`: Add pause state and controls, intentional-pause inactivity behavior, and automatic dismissal of obsolete inactivity notifications. +- `meeting-transcription`: Preserve one streaming recognition session and speaker context while paused audio is replaced with silence. + +## Impact + +- Affects recording status and coordinator behavior, inactivity prompt abstractions and the Windows toast implementation, tray menu modeling/rendering, and loopback status output. +- Affects the audio routed to temporary recordings, live speaker sampling, and configured streaming speech-recognition providers during an intentional pause. +- Adds one paused-session safety-timeout setting under the existing inactivity safeguard, adds no external dependency, and does not change normal finish, cancel/discard, profile-switch, or post-recording processing behavior. diff --git a/openspec/changes/add-transcription-pause-controls/specs/meeting-recording/spec.md b/openspec/changes/add-transcription-pause-controls/specs/meeting-recording/spec.md new file mode 100644 index 0000000..714f069 --- /dev/null +++ b/openspec/changes/add-transcription-pause-controls/specs/meeting-recording/spec.md @@ -0,0 +1,224 @@ +## ADDED Requirements + +### Requirement: Active transcription can be paused without ending the meeting +Meeting Assistant SHALL allow transcription for an active meeting recording to be paused and unpaused without stopping the meeting run. + +While transcription is paused, Meeting Assistant SHALL keep the recording status active, SHALL preserve the meeting artifacts and run-local speaker context, and SHALL keep the normal finish and cancel/discard controls available. + +When transcription is unpaused, Meeting Assistant SHALL resume transcribing newly captured audio through the existing meeting run. + +Pausing or unpausing when no active recording exists SHALL leave recording state unchanged. + +#### Scenario: Active transcription is paused and unpaused +- **GIVEN** a meeting is actively recording +- **WHEN** the user pauses transcription +- **THEN** the meeting remains active and reports that transcription is paused +- **AND** keeps its existing artifacts and speaker context +- **WHEN** the user unpauses transcription +- **THEN** newly captured audio is transcribed in the same meeting run + +#### Scenario: Paused meeting can still be finished +- **GIVEN** an active meeting transcription is paused +- **WHEN** the user finishes the meeting +- **THEN** Meeting Assistant follows the normal stop, transcription drain, speaker processing, and summary flow + +#### Scenario: Pause request while idle is harmless +- **GIVEN** no meeting recording is active +- **WHEN** transcription pause is requested +- **THEN** Meeting Assistant remains idle + +## MODIFIED Requirements + +### Requirement: Windows taskbar icon controls recording +Meeting Assistant SHALL show a Windows taskbar notification icon when running on Windows. + +The taskbar icon SHALL indicate whether the newest meeting process is idle, actively recording, or post-recording processing/summarizing. + +When a new meeting is actively recording while an older stopped meeting is still transcribing, recognizing speakers, or summarizing, the taskbar icon SHALL show the new active recording state. + +The taskbar icon right-click menu SHALL expose recording controls based on the current state and configured launch profiles. + +The taskbar icon right-click menu SHALL expose an Exit action in every recording state. + +When Meeting Assistant is idle or only processing older stopped meetings, the menu SHALL allow starting a meeting recording for each configured launch profile. + +When a meeting is actively recording, the menu SHALL allow stopping the recording and continuing transcription/summary generation. + +During an active recording, the normal stop action SHALL be labeled `Finish meeting` and SHALL be the only action in a dedicated menu section immediately below the `Open agent` section. + +During an active recording, pause/unpause, microphone selection, cancel/discard, and profile-switch actions SHALL appear in a separate fine-grained controls section below `Finish meeting`. + +When a meeting is actively recording and transcription is running, the menu SHALL expose `Pause transcription`. + +When a meeting is actively recording and transcription is paused, the menu SHALL expose `Unpause transcription` and SHALL continue exposing `Finish meeting`. + +When a meeting is actively recording, the menu SHALL allow canceling the recording and discarding that run's artifacts. + +When a meeting is actively recording, the menu SHALL allow switching to each configured launch profile other than the current active profile. + +Selecting Exit while Meeting Assistant is idle SHALL stop the application without an additional confirmation prompt. + +Selecting Exit while Meeting Assistant is recording, transcribing, recognizing speakers, or summarizing SHALL show a confirmation dialog before stopping the application. + +#### Scenario: Idle tray menu can start configured profiles +- **GIVEN** launch profiles `default` and `english` are configured +- **AND** no meeting recording is active +- **WHEN** the taskbar menu is opened +- **THEN** it offers start recording actions for `default` and `english` + +#### Scenario: Recording tray menu prioritizes finishing the meeting +- **GIVEN** launch profiles `default` and `english` are configured +- **AND** a meeting is actively recording with profile `default` +- **WHEN** the taskbar menu is opened +- **THEN** `Finish meeting` is the only action in the section immediately below `Open agent` +- **AND** pause, microphone selection, cancel/discard, and switching to `english` appear in a separate following section +- **AND** the menu does not offer switching to `default` + +#### Scenario: Tray pause action changes to unpause +- **GIVEN** a meeting is actively recording with transcription running +- **WHEN** the taskbar menu is opened +- **THEN** it offers `Pause transcription` +- **WHEN** transcription is paused and the taskbar menu is opened again +- **THEN** it offers `Unpause transcription` +- **AND** still offers `Finish meeting` + +#### Scenario: Active recording has priority over older summarizing runs +- **GIVEN** an older meeting is still summarizing +- **WHEN** a newer meeting is actively recording +- **THEN** the taskbar icon indicates recording + +#### Scenario: Tray menu always exposes Exit +- **GIVEN** Meeting Assistant is running +- **WHEN** the taskbar menu is opened +- **THEN** it offers an Exit action + +#### Scenario: Idle Exit stops immediately +- **GIVEN** no recording, transcription, speaker recognition, or summary work is running +- **WHEN** the user selects Exit from the taskbar menu +- **THEN** Meeting Assistant stops the application without an additional confirmation prompt + +#### Scenario: In-progress Exit asks for confirmation +- **GIVEN** Meeting Assistant is recording, transcribing, recognizing speakers, or summarizing +- **WHEN** the user selects Exit from the taskbar menu +- **THEN** Meeting Assistant asks for confirmation before stopping the application + +### Requirement: Recording inactivity safeguard stops forgotten meetings +Meeting Assistant SHALL track transcript inactivity during an active recording from the later of meeting start, the most recent unpause, or the most recent live transcript segment that contains text. + +Meeting Assistant SHALL show a stop prompt when transcript inactivity reaches configured prompt thresholds. The default thresholds SHALL be 2 minutes, 5 minutes, and 10 minutes. + +On Windows, the stop prompt SHALL use a native Windows app notification with stop, continue, and pause-transcription action buttons. + +On Windows, the stop prompt notification SHALL request reminder-style toast behavior and remain actionable for 1 minute. + +The stop prompt SHALL ask whether to stop the meeting and SHALL provide affirmative, negative, and pause-transcription actions. + +Showing or ignoring the stop prompt SHALL NOT block later inactivity checks, reminder prompts, or automatic stop. + +If the user accepts the stop prompt, Meeting Assistant SHALL stop the recording normally, allowing transcription, speaker recognition, and summary generation to continue as for a normal stop. + +If the user selects pause from the stop prompt, Meeting Assistant SHALL pause transcription for that same active meeting without finishing it. + +When a new non-empty live transcript segment is written, Meeting Assistant SHALL dismiss all outstanding transcript-inactivity notifications and invalidate their pending actions. + +Notification dismissal SHALL NOT block processing the final transcript segments emitted while switching launch profiles. Any deferred dismissal SHALL recheck that its originating run is still the active recording before dismissing notifications. + +While transcription is intentionally paused, Meeting Assistant SHALL NOT show transcript-inactivity prompts and SHALL NOT apply the ordinary transcript-inactivity auto-stop threshold. + +Meeting Assistant SHALL normally stop a meeting that remains continuously paused for the configured maximum pause duration, defaulting to 4 hours, without first showing a transcript-inactivity notification. This maximum continuous-pause cutoff SHALL remain active when ordinary transcript-inactivity prompting and auto-stop are disabled. + +When transcription is unpaused, Meeting Assistant SHALL clear the continuous-pause timer and restart transcript-inactivity timing from the unpause time. + +Meeting Assistant SHALL automatically stop the recording normally when transcript inactivity reaches the configured auto-stop threshold, defaulting to 30 minutes. + +When the inactivity safeguard stops a recording, Meeting Assistant SHALL infer the meeting end time from the most recent transcript segment timestamp plus configured padding, defaulting to 1 minute. If no transcript segment has arrived, Meeting Assistant SHALL infer the end time from the meeting start time plus the same padding. + +When a recording stops normally and its meeting note, transcript, and assistant context contain no user-authored or captured content beyond generated default headings and metadata, Meeting Assistant SHALL delete the run artifacts instead of running summary generation. + +#### Scenario: Inactive recording prompts the user +- **GIVEN** a recording is active +- **AND** no transcript text has arrived for the first configured inactivity prompt threshold +- **WHEN** the inactivity safeguard checks the active recording +- **THEN** Meeting Assistant prompts the user whether to stop the meeting with a native Windows app notification when running on Windows +- **AND** the notification offers pause transcription +- **AND** the notification remains actionable for 1 minute +- **AND** does not abort or discard meeting artifacts + +#### Scenario: Ignored inactivity prompt does not block auto-stop +- **GIVEN** a recording is active +- **AND** the inactivity safeguard prompt was shown +- **WHEN** the user ignores the prompt until the automatic stop threshold is reached +- **THEN** Meeting Assistant stops the recording normally without waiting for a prompt response + +#### Scenario: User accepts inactivity stop prompt +- **GIVEN** a recording is active +- **AND** the inactivity safeguard prompt is shown +- **WHEN** the user chooses to stop the meeting +- **THEN** Meeting Assistant stops the recording normally +- **AND** continues normal transcription and summary processing +- **AND** writes the inferred meeting end time to meeting artifacts + +#### Scenario: User pauses from inactivity prompt +- **GIVEN** a recording is active +- **AND** the inactivity safeguard prompt is shown +- **WHEN** the user chooses to pause transcription +- **THEN** Meeting Assistant dismisses the outstanding inactivity notifications +- **AND** pauses transcription without finishing the meeting + +#### Scenario: Inactive recording automatically stops +- **GIVEN** a recording is active +- **AND** no transcript text has arrived through the configured automatic stop threshold +- **WHEN** the inactivity safeguard checks the active recording +- **THEN** Meeting Assistant stops the recording normally without aborting artifacts +- **AND** writes the inferred meeting end time to meeting artifacts + +#### Scenario: New transcript text resets inactivity prompts +- **GIVEN** a recording is active +- **AND** one or more inactivity prompts were shown +- **WHEN** a new transcript segment with text is written +- **AND** the segment belongs to that active recording +- **THEN** Meeting Assistant dismisses every outstanding inactivity notification +- **AND** invalidates their pending actions +- **AND** resets the inactivity prompt schedule from that transcript arrival + +#### Scenario: Paused transcription suppresses inactivity safeguard +- **GIVEN** an active meeting transcription is paused +- **WHEN** configured transcript-inactivity prompt or automatic-stop thresholds pass before the maximum pause duration +- **THEN** Meeting Assistant does not show an inactivity prompt +- **AND** does not stop the meeting for transcript inactivity +- **WHEN** transcription is unpaused +- **THEN** inactivity timing restarts from the unpause time + +#### Scenario: Continuously paused meeting is stopped after four hours +- **GIVEN** an active meeting transcription is paused continuously +- **WHEN** the configured maximum pause duration of 4 hours is reached +- **THEN** Meeting Assistant does not show a transcript-inactivity notification +- **AND** stops the meeting normally + +#### Scenario: Maximum pause remains active when transcript-inactivity handling is disabled +- **GIVEN** ordinary transcript-inactivity prompting and auto-stop are disabled +- **AND** an active meeting transcription is paused continuously +- **WHEN** the configured maximum pause duration is reached +- **THEN** Meeting Assistant stops the meeting normally without an inactivity notification + +#### Scenario: An older run cannot dismiss a current run's notification +- **GIVEN** an older stopped meeting is still draining transcription +- **AND** a newer active meeting has an outstanding transcript-inactivity notification +- **WHEN** the older meeting writes a late transcript segment +- **THEN** the newer meeting's notification and pending actions remain active + +#### Scenario: Final transcript during a profile switch dismisses notifications without blocking the switch +- **GIVEN** a meeting is actively recording with either the `default` or `english` profile +- **WHEN** the user switches to the other profile +- **AND** the old recognizer emits a final non-empty transcript segment while draining +- **THEN** Meeting Assistant writes that final segment before the profile-switch marker +- **AND** completes the switch and transcribes buffered audio in the same meeting +- **AND** dismisses the originating active run's outstanding inactivity notifications +- **AND** subsequent profile switches and normal meeting completion remain available + +#### Scenario: Empty stopped recording is cleaned up +- **GIVEN** a recording is active +- **AND** the meeting note, transcript, and assistant context only contain generated default content +- **WHEN** the recording stops normally +- **THEN** Meeting Assistant deletes the run artifacts +- **AND** does not run summary generation diff --git a/openspec/changes/add-transcription-pause-controls/specs/meeting-transcription/spec.md b/openspec/changes/add-transcription-pause-controls/specs/meeting-transcription/spec.md new file mode 100644 index 0000000..1354994 --- /dev/null +++ b/openspec/changes/add-transcription-pause-controls/specs/meeting-transcription/spec.md @@ -0,0 +1,28 @@ +## ADDED Requirements + +### Requirement: Streaming recognition sessions survive intentional transcription pauses +Meeting Assistant SHALL preserve the active streaming speech-recognition pipeline and transcript session while transcription is intentionally paused. + +For every mixed audio chunk captured while paused, Meeting Assistant SHALL discard the captured sample values and SHALL send an equal-duration PCM silence chunk to the temporary recording, live speaker audio buffer, and configured speech-recognition pipeline. + +Meeting Assistant SHALL NOT clear run-local speaker mappings or previously collected speaker samples when transcription is paused or unpaused. + +When Azure Speech is the configured provider, Meeting Assistant SHALL keep the same active `ConversationTranscriber` operation and push audio stream across the pause rather than stopping and recreating the Azure recognition session. + +#### Scenario: Paused audio is replaced with silence +- **GIVEN** an active meeting transcription is paused +- **WHEN** the audio source captures a non-silent mixed audio chunk +- **THEN** the temporary recording and speech-recognition pipeline receive an equal-length silent chunk +- **AND** neither receives the captured sample values + +#### Scenario: Azure session remains active across pause +- **GIVEN** Azure Speech is transcribing an active meeting with speaker attribution +- **WHEN** transcription is paused and later unpaused +- **THEN** Meeting Assistant keeps the same conversation transcriber and push stream active +- **AND** preserves run-local speaker mappings and collected speaker samples +- **AND** newly captured audio after unpause continues through that session + +#### Scenario: Buffered pre-pause result is retained +- **GIVEN** the transcription backend accepted audio before transcription was paused +- **WHEN** the corresponding transcript result arrives after the pause begins +- **THEN** Meeting Assistant writes that transcript result to the same meeting transcript diff --git a/openspec/changes/add-transcription-pause-controls/tasks.md b/openspec/changes/add-transcription-pause-controls/tasks.md new file mode 100644 index 0000000..81c9d73 --- /dev/null +++ b/openspec/changes/add-transcription-pause-controls/tasks.md @@ -0,0 +1,49 @@ +## 1. Inactivity notification lifecycle + +- [x] 1.1 Add a failing coordinator behavior test proving that a newly written non-empty transcript segment dismisses every outstanding inactivity prompt. +- [x] 1.2 Extend the inactivity prompt service contract and Windows implementation to invalidate callbacks and remove the complete inactivity notification group. + +## 2. Pause and unpause behavior + +- [x] 2.1 Add a failing coordinator behavior test proving that pause keeps the run active, routes equal-duration silence instead of captured audio, and unpause resumes real audio through the same pipeline. +- [x] 2.2 Implement thread-safe pause state, coordinator pause/unpause controls, status reporting, and silence substitution without resetting speaker state or the active pipeline. +- [x] 2.3 Add a failing behavior test proving that inactivity prompts and auto-stop are suspended while paused and restart from the unpause time. +- [x] 2.4 Implement pause-aware inactivity timing and guard prompt callbacks so stale notifications cannot control another run. + +## 3. User controls + +- [x] 3.1 Add a failing tray-menu behavior test for `Pause transcription` / `Unpause transcription` while `Finish meeting` remains available. +- [x] 3.2 Add the tray pause toggle action in the fine-grained controls section and wire it to the coordinator. +- [x] 3.3 Add a failing inactivity-prompt behavior test proving the pause response pauses the originating active meeting. +- [x] 3.4 Add the `Pause transcription` Windows notification action and route its response through the guarded coordinator pause path. + +## 4. Documentation and verification + +- [x] 4.1 Update the operational documentation with pause behavior, Azure session continuity, silence substitution, and the tray/notification controls. +- [x] 4.2 Refactor the touched paths for DRYness, SOLID boundaries, and simplicity while preserving behavior, including pause/inactivity race handling found during independent review. +- [x] 4.3 Run focused recording/taskbar tests, the Windows application build, the full solution tests, and `openspec validate add-transcription-pause-controls --strict`. +- [x] 4.4 Inspect the local health and recording-status surfaces without restarting or interrupting an active meeting run. + +## 5. Maximum continuous pause + +- [x] 5.1 Add a failing coordinator behavior test proving that ordinary inactivity prompts remain suppressed while paused and the meeting stops normally after four continuous paused hours. +- [x] 5.2 Track the continuous pause start and implement the configurable four-hour paused-session safety stop without allowing an unpause race to stop the meeting. +- [x] 5.3 Update configuration documentation and rerun focused tests, the Windows build, the full suite, and strict OpenSpec validation. + +## 6. Review follow-up + +- [x] 6.1 Keep the maximum continuous-pause cutoff active when ordinary inactivity handling is disabled, with a failing public behavior test. +- [x] 6.2 Scope notification dismissal to the active run and move activity invalidation after durable transcript append, with overlapping-run and blocked-write regression tests. +- [x] 6.3 Make pause-notification activity validation atomic and make tray pause/unpause actions intent-specific, with race and stale-menu tests. +- [x] 6.4 Complete an independent simplification review and rerun all required validation before commit. + +## 7. Profile-switch deadlock repair + +- [x] 7.1 Reproduce final transcript delivery during profile switching in both directions through the coordinator's public interface. +- [x] 7.2 Decouple notification dismissal from transcript draining while preserving current-run checks and cancellation handling. +- [x] 7.3 Verify buffered transcription, subsequent controls, notification lifecycle tests, the full suite, Windows build, and strict OpenSpec validation. +- [x] 7.4 Deploy with explicit restart authorization, verify the live health/control endpoints, and record the operational verification limits. + +Verification on 2026-09-16: both profile-switch regression cases failed with the original circular wait and passed after the repair. All 520 solution tests passed on the final run. An unchanged audio-mixing timing test failed during the first full run, then passed alone and in the final full run. Strict OpenSpec validation and the Windows Release publish passed. The executable is staged at `tmp/meeting-assistant-runtime/run-20260916-142827-profile-switch-fix`. + +Operational verification on 2026-09-16: after explicit user authorization, the deadlocked process was killed and the staged Windows release started as PID 48716. `/health` returned `ok`, `/recording/status` returned idle, and the process path confirmed the fixed release. The old meeting's summary was requested through `/meetings/summary/retry`. Profile-switch behavior was verified through the public coordinator regression tests rather than recording a new live meeting. The existing 25,092,144-byte WAV and meeting artifacts were backed up under `tmp/profile-switch-recovery-20260916`; audio held only in the killed process after the switch was not recovered. The meeting context records that transcription gap. diff --git a/openspec/changes/unify-pyannote-validation-toggle/.openspec.yaml b/openspec/changes/unify-pyannote-validation-toggle/.openspec.yaml new file mode 100644 index 0000000..032461f --- /dev/null +++ b/openspec/changes/unify-pyannote-validation-toggle/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-02 diff --git a/openspec/changes/unify-pyannote-validation-toggle/design.md b/openspec/changes/unify-pyannote-validation-toggle/design.md new file mode 100644 index 0000000..af2fdcb --- /dev/null +++ b/openspec/changes/unify-pyannote-validation-toggle/design.md @@ -0,0 +1,52 @@ +## Context + +Speaker identity validation reuses the general pyannote diarization option type. That type includes an `Enabled` property for the optional Whisper finalization feature, while speaker validation already has its own outer `Enabled` property. The checked-in and deployed configuration currently sets those properties to opposite values. The validator observes the outer value, calls the finalizer, and receives an empty result because the finalizer observes the nested value, causing every sample to be rejected before Azure matching. + +## Goals / Non-Goals + +**Goals:** + +- Expose one authoritative enable switch for speaker identity pyannote validation. +- Keep Whisper diarization independently configurable. +- Make validation, matching, and startup warm-up observe the same speaker-validation state. +- Preserve the existing pyannote runtime settings and validation thresholds. + +**Non-Goals:** + +- Change speaker sample duration or gap thresholds. +- Change Azure Speech matching semantics. +- Enable pyannote validation without the required Docker runtime and Hugging Face token. + +## Decisions + +### Give speaker validation a runtime-options type without an enable flag + +Extract the shared pyannote runtime properties into `PyannoteRuntimeOptions`. Keep `PyannoteDiarizationOptions` as the Whisper-facing derived type that adds `Enabled`, and type `SpeakerIdentityPyannoteValidationOptions.Diarization` as `PyannoteRuntimeOptions`. + +This makes contradictory speaker-validation state unrepresentable through the typed configuration model. The alternative—retaining the nested flag and overriding it at runtime—would leave a misleading configuration surface and permit the same mistake to recur. + +### Gate once at the feature boundary + +`PyannoteSpeakerIdentityMatchValidator` bypasses pyannote when the application-level outer validation switch is off. When it is on, the validator calls an enabled-runtime finalization path that does not evaluate another toggle. The warm-up service selects the same application-level speaker-validation runtime instead of resolving speaker validation independently for each launch profile. + +Whisper finalization continues checking `WhisperLocal:Diarization:Enabled` before it invokes the shared runtime path. + +### Migrate configuration by removing the nested key + +Remove `SpeakerIdentification:PyannoteValidation:Diarization:Enabled` from the canonical configuration and document `SpeakerIdentification:PyannoteValidation:Enabled` as the sole switch. Existing unknown nested keys are ignored by .NET configuration binding after the typed property is removed; deployments should republish from the canonical configuration to remove the stale key. + +## Risks / Trade-offs + +- [Enabling the outer switch now really invokes Docker/pyannote] → Preserve the existing token/runtime error reporting and document that disabling the outer switch is the supported bypass. +- [The shared options refactor touches Whisper code] → Preserve its independent `Enabled` property on the derived type and run focused Whisper/pyannote tests plus the full suite. +- [Old deployed appsettings may retain the removed nested key] → The binder ignores it, so runtime behavior remains controlled by the outer switch; republishing removes it from the canonical deployed file. + +## Migration Plan + +1. Publish the updated application configuration with the nested key removed. +2. Keep `SpeakerIdentification:PyannoteValidation:Enabled` on only where Docker, the pyannote model, and `HF_TOKEN` are available. +3. Roll back by deploying the prior build and configuration together if necessary. + +## Open Questions + +None. diff --git a/openspec/changes/unify-pyannote-validation-toggle/proposal.md b/openspec/changes/unify-pyannote-validation-toggle/proposal.md new file mode 100644 index 0000000..2eabd4a --- /dev/null +++ b/openspec/changes/unify-pyannote-validation-toggle/proposal.md @@ -0,0 +1,27 @@ +## Why + +Speaker identity matching currently exposes two independent pyannote validation switches. Enabling the outer validation switch while disabling the nested diarization switch silently rejects every otherwise usable speaker sample, so the default configuration can prevent all automatic identity matches. + +## What Changes + +- Make `SpeakerIdentification:PyannoteValidation:Enabled` the only switch controlling secondary pyannote validation. +- Remove the nested `SpeakerIdentification:PyannoteValidation:Diarization:Enabled` configuration setting. +- Ensure enabling validation also enables its pyannote runtime and startup warm-up, while disabling validation preserves primary Azure identity matching without invoking pyannote. +- Add regression coverage and configuration documentation for both toggle states. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `meeting-transcription`: Clarify that secondary pyannote validation has one authoritative enable setting and cannot be partially enabled. + +## Impact + +- Speaker identification options and pyannote runtime invocation. +- Pyannote startup warm-up selection. +- Checked-in application configuration and configuration reference. +- Speaker validation and warm-up behavior tests. diff --git a/openspec/changes/unify-pyannote-validation-toggle/specs/meeting-transcription/spec.md b/openspec/changes/unify-pyannote-validation-toggle/specs/meeting-transcription/spec.md new file mode 100644 index 0000000..513d91e --- /dev/null +++ b/openspec/changes/unify-pyannote-validation-toggle/specs/meeting-transcription/spec.md @@ -0,0 +1,60 @@ +## MODIFIED Requirements + +### Requirement: Speaker identity matching can use pyannote secondary validation +Meeting Assistant SHALL support an optional configurable pyannote secondary validation layer for speaker identity matching. + +`SpeakerIdentification:PyannoteValidation:Enabled` SHALL be the only enable setting for speaker identity pyannote validation. The nested pyannote runtime settings SHALL NOT expose or honor a second enable setting. + +Speaker identity pyannote validation SHALL use the application-level setting consistently for matching and startup warm-up. Launch profiles SHALL NOT override this validation setting or its runtime configuration. + +When pyannote secondary validation is enabled, Meeting Assistant SHALL verify candidate speaker samples before retaining them for identity matching. Samples that pyannote reports as containing multiple speakers SHALL be rejected. + +When pyannote secondary validation is enabled, Meeting Assistant SHALL verify speaker-override samples before retaining them on speaker identities. Speaker overrides whose source samples are rejected SHALL NOT create a new speaker identity from that rejected sample. + +When pyannote secondary validation is enabled and the primary identity matcher confirms a speaker, Meeting Assistant SHALL run a second validation pass through pyannote before accepting the match. + +If pyannote secondary validation cannot confirm that the unknown live sample and matched identity samples belong to one speaker, Meeting Assistant SHALL reject the match. + +When pyannote secondary validation is disabled, Meeting Assistant SHALL preserve the primary identity matching behavior. + +When pyannote secondary validation is enabled, Meeting Assistant SHALL start a non-blocking startup warm-up that builds or verifies the configured pyannote runtime image and downloads the configured model into the persistent model cache before the first validation request when possible. + +#### Scenario: Multi-speaker sample is rejected +- **GIVEN** pyannote secondary validation is enabled +- **WHEN** pyannote reports multiple speakers in a candidate sample +- **THEN** Meeting Assistant does not retain that sample for identity matching + +#### Scenario: Multi-speaker speaker-override sample is rejected +- **GIVEN** pyannote secondary validation is enabled +- **WHEN** a summary speaker override resolves a source sample that pyannote reports as containing multiple speakers +- **THEN** Meeting Assistant does not retain that sample on a speaker identity +- **AND** does not create a new speaker identity from that rejected sample + +#### Scenario: Pyannote rejects primary match +- **GIVEN** pyannote secondary validation is enabled +- **AND** the primary identity matcher confirms `Guest03` as `Chris` +- **WHEN** pyannote reports that the unknown `Guest03` sample and known `Chris` samples contain different speakers +- **THEN** Meeting Assistant rejects the match + +#### Scenario: Enabled pyannote validation invokes its runtime +- **GIVEN** speaker identity pyannote validation is enabled +- **WHEN** Meeting Assistant validates a readable speaker sample +- **THEN** it invokes the configured pyannote runtime without requiring another enable setting + +#### Scenario: Launch profile cannot override speaker validation +- **GIVEN** application-level speaker identity pyannote validation is disabled +- **AND** a named launch profile contains different speaker-validation settings +- **WHEN** Meeting Assistant starts or matches a speaker for that profile +- **THEN** it keeps application-level validation disabled +- **AND** does not warm or invoke the named profile's speaker-validation runtime + +#### Scenario: Disabled pyannote validation preserves primary match +- **GIVEN** pyannote secondary validation is disabled +- **WHEN** the primary identity matcher confirms `Guest03` as `Chris` +- **THEN** Meeting Assistant accepts the match without running pyannote secondary validation + +#### Scenario: Pyannote validation warms up on startup +- **GIVEN** pyannote secondary validation is enabled +- **WHEN** Meeting Assistant starts +- **THEN** it begins preparing the configured pyannote runtime image and model cache without waiting for the first validation request +- **AND** application startup is not blocked by the warm-up task diff --git a/openspec/changes/unify-pyannote-validation-toggle/tasks.md b/openspec/changes/unify-pyannote-validation-toggle/tasks.md new file mode 100644 index 0000000..ba7de50 --- /dev/null +++ b/openspec/changes/unify-pyannote-validation-toggle/tasks.md @@ -0,0 +1,15 @@ +## 1. Single-toggle behavior + +- [x] 1.1 Add a failing behavior test proving enabled speaker validation invokes pyannote without a nested enable setting +- [x] 1.2 Introduce toggle-free speaker-validation runtime options and make the validator use them +- [x] 1.3 Add or update behavior coverage proving disabled validation bypasses pyannote and enabled validation warms the runtime +- [x] 1.4 Keep application-level validation and warm-up consistent when named launch profiles contain speaker-validation overrides + +## 2. Configuration migration + +- [x] 2.1 Remove the nested speaker-validation diarization toggle from canonical configuration and document the outer toggle as authoritative + +## 3. Verification + +- [x] 3.1 Run focused speaker validation, warm-up, and pyannote finalizer tests +- [x] 3.2 Run the full solution test suite and validate the OpenSpec change strictly diff --git a/openspec/specs/meeting-session/spec.md b/openspec/specs/meeting-session/spec.md index 85e0381..4b49c49 100644 --- a/openspec/specs/meeting-session/spec.md +++ b/openspec/specs/meeting-session/spec.md @@ -285,6 +285,8 @@ The rules and identities editor agent SHALL receive speaker identity tools to se The rules and identities editor agent SHALL receive speaker sample tools to list, read, delete, and queue playback of samples linked to identities. The delete sample tool SHALL refuse to delete the last remaining sample for an identity. +Speaker sample playback SHALL use the Windows audio implementation only in the Windows-targeted build. The neutral build SHALL return an actionable unavailable response instead of reporting a sample as queued for playback. + The first model request caused by each user-submitted chat turn SHALL send the `X-Initiator: user` header, while follow-up model requests within that same turn, such as tool-call continuations, SHALL send `X-Initiator: agent`. Meeting Assistant SHALL provide a diagnostic endpoint that opens the workflow rules editor through the same window service used by the tray menu. @@ -337,6 +339,12 @@ Meeting Assistant SHALL provide a diagnostic endpoint that opens the workflow ru - **AND** it can list, read, delete, and queue playback of identity samples - **AND** deleting the last sample for an identity is refused +#### Scenario: Neutral build refuses speaker playback +- **GIVEN** Meeting Assistant runs from its neutral target build +- **WHEN** an agent requests playback of a speaker sample +- **THEN** the application reports that playback requires the Windows build +- **AND** does not report that the sample was queued + #### Scenario: User sends a rules-editing chat turn - **GIVEN** the rules editor chat window is open - **WHEN** the user types a prompt and presses Enter