diff --git a/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs b/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs index a39d570..bfa622d 100644 --- a/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs +++ b/MeetingAssistant.Tests/LaunchProfileOptionsProviderTests.cs @@ -58,6 +58,64 @@ public sealed class LaunchProfileOptionsProviderTests Assert.Equal("Ctrl+Alt+L", profile.Options.Hotkey.Toggle); } + [Fact] + public void CheckedInConfigurationKeepsResemblyzerOptInWithDocumentedDefaults() + { + var profile = CreateProviderFromAppsettings().GetRequiredProfile(null); + + Assert.False(profile.Options.SpeakerIdentification.Resemblyzer.Enabled); + Assert.Equal(5, profile.Options.SpeakerIdentification.Resemblyzer.RequiredVectorsPerSpeaker); + Assert.Equal(1000, profile.Options.SpeakerIdentification.Resemblyzer.MaxVectorsPerIdentity); + Assert.Equal(20, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningMinimumVectors); + Assert.Equal(0.75, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningNeighborSimilarity); + Assert.Equal(3, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningMinimumNeighbors); + Assert.Equal(0.60, profile.Options.SpeakerIdentification.Resemblyzer.OutlierPruningMinimumClusterRatio); + } + + [Fact] + public void InvalidResemblyzerOutlierPruningSettingsAreRejected() + { + var provider = CreateProvider(new Dictionary + { + ["MeetingAssistant:SpeakerIdentification:Resemblyzer:OutlierPruningMinimumClusterRatio"] = "1.1" + }); + + var exception = Assert.Throws( + () => provider.GetRequiredProfile(null)); + + Assert.Contains("OutlierPruningMinimumClusterRatio", exception.Message); + } + + [Fact] + public void NonFiniteResemblyzerSimilaritySettingsAreRejected() + { + var provider = CreateProvider(new Dictionary + { + ["MeetingAssistant:SpeakerIdentification:Resemblyzer:MinimumIdentitySimilarity"] = "NaN" + }); + + var exception = Assert.Throws( + () => provider.GetRequiredProfile(null)); + + Assert.Contains("MinimumIdentitySimilarity", exception.Message); + } + + [Fact] + public void ContradictorySpeakerSampleDurationsAreRejected() + { + var provider = CreateProvider(new Dictionary + { + ["MeetingAssistant:SpeakerIdentification:MinimumSampleSpeechDuration"] = "00:01:01", + ["MeetingAssistant:SpeakerIdentification:MaximumSampleDuration"] = "00:01:00" + }); + + var exception = Assert.Throws( + () => provider.GetRequiredProfile(null)); + + Assert.Contains("MinimumSampleSpeechDuration", exception.Message); + Assert.Contains("MaximumSampleDuration", exception.Message); + } + [Fact] public void DuplicateProfileHotkeysAreRejected() { diff --git a/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs b/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs index 7afe7b7..a7306a1 100644 --- a/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs +++ b/MeetingAssistant.Tests/PyannoteDiarizationWarmupHostedServiceTests.cs @@ -59,6 +59,28 @@ public sealed class PyannoteDiarizationWarmupHostedServiceTests Assert.DoesNotContain(commandRunner.Commands, command => command.Arguments.Contains("meeting-assistant-pyannote-named:local")); } + [Fact] + public async Task HostedServiceDoesNotWarmIdentityValidationWhenResemblyzerIsSelected() + { + var commandRunner = new BlockingCommandRunner(); + var finalizer = new PyannoteTranscriptFinalizer( + commandRunner, + Options.Create(new MeetingAssistantOptions()), + NullLogger.Instance); + var options = CreateValidationOptions("meeting-assistant-pyannote-validation:local"); + options.SpeakerIdentification.Resemblyzer.Enabled = true; + var service = new PyannoteDiarizationWarmupHostedService( + finalizer, + new FakeLaunchProfileOptionsProvider(options), + NullLogger.Instance); + + await service.StartAsync(CancellationToken.None); + await Task.Delay(TimeSpan.FromMilliseconds(50)); + await service.StopAsync(CancellationToken.None); + + Assert.Empty(commandRunner.Commands); + } + private static MeetingAssistantOptions CreateValidationOptions(string image) { return new MeetingAssistantOptions diff --git a/MeetingAssistant.Tests/RecordingCoordinatorTests.cs b/MeetingAssistant.Tests/RecordingCoordinatorTests.cs index 9e89090..3bd994a 100644 --- a/MeetingAssistant.Tests/RecordingCoordinatorTests.cs +++ b/MeetingAssistant.Tests/RecordingCoordinatorTests.cs @@ -2899,6 +2899,65 @@ public sealed class RecordingCoordinatorTests Assert.Equal(["Chris"], speakerIdentification.Requests.Last().MeetingNote.Frontmatter.Attendees); } + [Fact] + public async Task LiveResemblyzerMatchingRetriesWhenSameSpeakerCollectsRequiredSamples() + { + var audioSource = new ControlledAudioSource(); + var transcriptStore = new InMemoryTranscriptStore(); + var speakerIdentification = new CountingSpeakerIdentificationService(); + var coordinator = new MeetingRecordingCoordinator( + audioSource, + new TestSpeechRecognitionPipelineFactory(new OrderedChunkProvider()), + transcriptStore, + new InMemoryMeetingNoteStore(), + new CapturingMeetingNoteOpener(), + new InMemoryMeetingArtifactStore(), + new InMemoryRecordedAudioStore(), + new CapturingMeetingSummaryPipeline(), + Options.Create(new MeetingAssistantOptions + { + SpeakerIdentification = new SpeakerIdentificationOptions + { + InitialDelay = TimeSpan.Zero, + Interval = TimeSpan.FromMilliseconds(20), + MinimumSampleSpeechDuration = TimeSpan.Zero, + Resemblyzer = new ResemblyzerSpeakerRecognitionOptions + { + Enabled = true, + RequiredVectorsPerSpeaker = 5 + } + } + }), + NullLogger.Instance, + speakerIdentificationService: speakerIdentification, + speakerSampleCollectionPolicy: SpeakerSampleCollectionPolicy.IndependentVectors(1000)); + + await coordinator.StartAsync(CancellationToken.None); + try + { + await audioSource.WriteAsync(new AudioChunk(Samples(0, 1, 2, 3, 4, 5, 6, 7), 4, 1), CancellationToken.None); + await WaitUntilAsync(() => speakerIdentification.Requests.Count == 1); + Assert.Single(speakerIdentification.Requests.Single().Samples!); + + // The provider emits overlapping two-second segments, so every second + // segment supplies another independent recognition sample. + for (var index = 1; index < 9; index++) + { + await audioSource.WriteAsync(new AudioChunk(Samples(0, 1, 2, 3, 4, 5, 6, 7), 4, 1), CancellationToken.None); + } + + await WaitUntilAsync(() => speakerIdentification.Requests.Last().Samples?.Count == 5); + var attempts = speakerIdentification.Requests.Count; + await Task.Delay(100); + Assert.Equal(attempts, speakerIdentification.Requests.Count); + Assert.All(speakerIdentification.Requests.Last().Samples!, sample => Assert.Equal("Guest03", sample.Speaker)); + } + finally + { + await coordinator.StopAsync(CancellationToken.None); + } + } + [Fact] public async Task LiveSpeakerIdentificationRetriesWhenNewUnmappedSpeakerAppears() { diff --git a/MeetingAssistant.Tests/ResemblyzerSpeakerIdentificationServiceTests.cs b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentificationServiceTests.cs new file mode 100644 index 0000000..c81bba7 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentificationServiceTests.cs @@ -0,0 +1,596 @@ +using MeetingAssistant; +using MeetingAssistant.MeetingNotes; +using MeetingAssistant.Speakers; +using MeetingAssistant.Transcription; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerSpeakerIdentificationServiceTests +{ + [Fact] + public async Task LiveMatchRelabelsSpeakerAndStoresFiveVectorsWithoutWavSnippets() + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync("Chris", [UnitVector(0), UnitVector(0, 20, 0.02f)]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + var result = await service.IdentifyKnownSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 5), + CancellationToken.None); + + Assert.Equal("Chris", result.Segments.Single().Speaker); + Assert.Equal("Chris", result.SpeakerMappings["Guest03"]); + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal(7, saved.VoiceVectors.Count); + Assert.Empty(saved.Snippets); + Assert.Single(saved.References); + Assert.Single(fixture.Encoder.Requests); + Assert.Equal(5, fixture.Encoder.Requests[0].Count); + } + + [Fact] + public async Task SummaryOverrideCreatesNamedIdentityFromThreeAvailableVectors() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 3) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + var request = fixture.CreateRequest("Guest-01", sampleCount: 3); + + await service.ApplySpeakerOverrideAsync( + request, + "Guest-01", + "Sabrina", + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal("Sabrina", saved.CanonicalName); + Assert.Equal(3, saved.VoiceVectors.Count); + Assert.Empty(saved.Snippets); + Assert.Single(saved.References); + Assert.Equal(3, fixture.Encoder.Requests.Single().Count); + } + + [Fact] + public async Task FinalProcessingLearnsUnmatchedSpeakerFromFiveVectorsAndAttendees() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 5, attendees: ["John", "Mike"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Null(saved.CanonicalName); + Assert.Equal(["John", "Mike"], saved.CandidateNames.Select(candidate => candidate.Name).Order()); + Assert.Equal(5, saved.VoiceVectors.Count); + Assert.Empty(saved.Snippets); + Assert.Single(saved.References); + } + + [Fact] + public async Task FiveVectorThresholdDoesNotCapQualifyingCurrentRunEvidence() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 8) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 8, attendees: ["John", "Mike"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal(8, saved.VoiceVectors.Count); + Assert.Equal(8, fixture.Encoder.Requests.Single().Count); + } + + [Fact] + public async Task FinalProcessingPrunesForeignVectorsFromNewIdentityAtMinimum() + { + await using var fixture = await Fixture.CreateAsync(options => + { + options.OutlierPruningMinimumVectors = 20; + options.OutlierPruningNeighborSimilarity = 0.90; + options.OutlierPruningMinimumNeighbors = 3; + options.OutlierPruningMinimumClusterRatio = 0.60; + }); + fixture.Encoder.Vectors = Enumerable.Range(0, 16) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 20, attendees: ["John", "Mike"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal(16, saved.VoiceVectors.Count); + Assert.All( + saved.VoiceVectors, + stored => Assert.True(SpeakerVoiceVectors.Decode(stored)[0] > 0.9f)); + } + + [Theory] + [InlineData(false)] + [InlineData(true)] + public async Task FinalProcessingStoresVectorsCollectedAfterLiveMatch(bool transcriptAlreadyRelabeled) + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync( + "Chris", + [UnitVector(0), UnitVector(0, 20, 0.02f)]); + var currentRunVectors = Enumerable.Range(0, 8) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + fixture.Encoder.Vectors = currentRunVectors.Take(5).ToList(); + var service = fixture.CreateService(); + + var liveResult = await service.IdentifyKnownSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 5), + CancellationToken.None); + fixture.Encoder.Vectors = currentRunVectors; + var finalRequest = fixture.CreateRequest("Guest03", sampleCount: 8) with + { + KnownSpeakerMappings = liveResult.SpeakerMappings + }; + if (transcriptAlreadyRelabeled) + { + finalRequest = finalRequest with { Segments = liveResult.Segments }; + } + + await service.ProcessFinishedTranscriptAsync(finalRequest, CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal(10, saved.VoiceVectors.Count); + Assert.Equal([5, 8], fixture.Encoder.Requests.Select(request => request.Count)); + } + + [Fact] + public async Task RepeatedMatchDeduplicatesVectorsAndHonorsConfiguredLimit() + { + await using var fixture = await Fixture.CreateAsync(options => options.MaxVectorsPerIdentity = 6); + var identity = await fixture.AddIdentityAsync("Chris", [UnitVector(0), UnitVector(0, 20, 0.02f)]); + fixture.Encoder.Vectors = + [ + UnitVector(0), + UnitVector(0, 2, 0.01f), + UnitVector(0, 3, 0.02f), + UnitVector(0, 4, 0.03f), + UnitVector(0, 5, 0.04f) + ]; + var service = fixture.CreateService(); + var request = fixture.CreateRequest("Guest03", sampleCount: 5); + + await service.IdentifyKnownSpeakersAsync(request, CancellationToken.None); + await service.IdentifyKnownSpeakersAsync(request, CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal(6, saved.VoiceVectors.Count); + Assert.Single(saved.References); + } + + [Fact] + public async Task FinishedMatchingExtractsFiveNonOverlappingSamplesWhenLiveSamplesAreMissing() + { + await using var fixture = await Fixture.CreateAsync(); + await fixture.AddIdentityAsync("Chris", [UnitVector(0), UnitVector(0, 20, 0.02f)]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + var result = await service.IdentifyFinishedSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 0, segmentCount: 5), + CancellationToken.None); + + Assert.Equal("Chris", result.Segments[0].Speaker); + Assert.Equal(5, fixture.SnippetExtractor.Requests.Count); + Assert.All( + fixture.SnippetExtractor.Requests.Zip(fixture.SnippetExtractor.Requests.Skip(1)), + pair => Assert.True(pair.First[^1].End <= pair.Second[0].Start)); + Assert.Equal(5, fixture.Encoder.Requests.Single().Count); + } + + [Fact] + public async Task FinalMatchPromotesTheRemainingCandidateAndAuditsTheTranscript() + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync( + null, + [UnitVector(0), UnitVector(0, 20, 0.02f)], + candidates: ["John", "Mike"]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + var result = await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest03", sampleCount: 5, attendees: ["Jane", "John", "Chris"]), + CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Equal("John", saved.CanonicalName); + Assert.Equal(["John"], saved.CandidateNames.Select(candidate => candidate.Name)); + Assert.Equal("John", result.Segments.Single().Speaker); + Assert.All(saved.References, reference => + Assert.Contains("Guest03 was identified as John", File.ReadAllText(reference.TranscriptPath))); + } + + [Fact] + public async Task FinalMatchResetsCandidatesWhenAttendeesDoNotIntersect() + { + await using var fixture = await Fixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync( + null, + [UnitVector(0), UnitVector(0, 20, 0.02f)], + candidates: ["John", "Mike"]); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest03", sampleCount: 5, attendees: ["Jane", "Chris"]), + CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(identity.Id); + Assert.Null(saved.CanonicalName); + Assert.Equal(["Chris", "Jane"], saved.CandidateNames.Select(candidate => candidate.Name).Order()); + } + + [Fact] + public async Task FinalUnmatchedSpeakerWithOneCandidateIsAuditedWhenLearned() + { + await using var fixture = await Fixture.CreateAsync(); + fixture.Encoder.Vectors = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, index + 2, 0.03f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ProcessFinishedTranscriptAsync( + fixture.CreateRequest("Guest-01", sampleCount: 5, attendees: ["Manuel"]), + CancellationToken.None); + + var saved = await fixture.LoadOnlyIdentityAsync(); + Assert.Equal("Manuel", saved.CanonicalName); + Assert.Contains("Guest-01 was identified as Manuel", File.ReadAllText(saved.References.Single().TranscriptPath)); + } + + [Fact] + public async Task CandidateLimitCountsOnlyIdentitiesWithCompatibleVectors() + { + await using var fixture = await Fixture.CreateAsync( + configureSpeaker: options => options.MaxMatchCandidates = 1); + await fixture.AddIdentityAsync("Legacy WAV identity", []); + await fixture.AddIdentityAsync("Chris", [UnitVector(0)]); + fixture.Encoder.Vectors = Enumerable.Repeat(UnitVector(0), 5).ToList(); + var service = fixture.CreateService(); + + var result = await service.IdentifyKnownSpeakersAsync( + fixture.CreateRequest("Guest03", sampleCount: 5, attendees: []), + CancellationToken.None); + + Assert.Equal("Chris", result.Segments.Single().Speaker); + } + + [Fact] + public async Task SummaryOverrideMergePreservesAllEvidenceFromCurrentRunCandidate() + { + await using var fixture = await Fixture.CreateAsync(); + var target = await fixture.AddIdentityAsync("Sabrina", [UnitVector(0)]); + var request = fixture.CreateRequest("Guest-01", sampleCount: 3, attendees: ["Sabrina"]); + await fixture.AddCurrentRunCandidateAsync(request, "Sabrina"); + fixture.Encoder.Vectors = Enumerable.Range(0, 3) + .Select(index => UnitVector(0, index + 2, 0.02f)) + .ToList(); + var service = fixture.CreateService(); + + await service.ApplySpeakerOverrideAsync( + request, + "Guest-01", + "Sabrina", + CancellationToken.None); + + var saved = await fixture.LoadIdentityAsync(target.Id); + Assert.Single(saved.Snippets); + Assert.Equal(5, saved.VoiceVectors.Count); + Assert.Contains(saved.VoiceVectors, vector => vector.ModelId == "older-model"); + await using var context = new TestDbContextFactory(fixture.DatabasePath).CreateDbContext(); + Assert.Single(await context.SpeakerIdentities.ToListAsync()); + } + + private static float[] UnitVector( + int primaryDimension, + int? secondaryDimension = null, + float secondaryValue = 0) + { + var vector = new float[256]; + vector[primaryDimension] = 1; + if (secondaryDimension is { } dimension) + { + vector[dimension] = secondaryValue; + } + + return vector; + } + + private static byte[] ToBytes(float[] vector) + { + var bytes = new byte[vector.Length * sizeof(float)]; + Buffer.BlockCopy(vector, 0, bytes, 0, bytes.Length); + return bytes; + } + + private sealed class Fixture : IAsyncDisposable + { + private readonly string directory; + private readonly string databasePath; + private readonly SpeakerIdentificationOptions speakerOptions; + + private Fixture(string directory, string databasePath, SpeakerIdentificationOptions speakerOptions) + { + this.directory = directory; + this.databasePath = databasePath; + this.speakerOptions = speakerOptions; + } + + public FakeEncoder Encoder { get; } = new(); + + public FakeSnippetExtractor SnippetExtractor { get; } = new(); + + public string DatabasePath => databasePath; + + public static async Task CreateAsync( + Action? configure = null, + Action? configureSpeaker = null) + { + var directory = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + var databasePath = Path.Combine(directory, "speaker-identities.db"); + var speakerOptions = new SpeakerIdentificationOptions + { + DatabasePath = databasePath, + MatchBatchSize = 6, + MaxMatchCandidates = 100, + MatchIdentityActiveAge = TimeSpan.FromDays(365), + MinimumSampleSpeechDuration = TimeSpan.Zero, + Resemblyzer = new ResemblyzerSpeakerRecognitionOptions + { + Enabled = true, + RequiredVectorsPerSpeaker = 5, + MaxVectorsPerIdentity = 1000, + MinimumClusterCohesion = 0.75, + MinimumIdentitySimilarity = 0.75, + MinimumSimilarityMargin = 0.05, + ModelId = "resemblyzer-0.1.4-pretrained" + } + }; + configure?.Invoke(speakerOptions.Resemblyzer); + configureSpeaker?.Invoke(speakerOptions); + await using var context = new SpeakerIdentityDbContext( + new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath};Pooling=False") + .Options); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + return new Fixture(directory, databasePath, speakerOptions); + } + + public ResemblyzerSpeakerIdentificationService CreateService() + { + var appOptions = new MeetingAssistantOptions { SpeakerIdentification = speakerOptions }; + return new ResemblyzerSpeakerIdentificationService( + new TestDbContextFactory(databasePath), + SnippetExtractor, + Encoder, + new ResemblyzerVoiceClusterMatcher( + speakerOptions.Resemblyzer, + NullLogger.Instance), + new ResemblyzerVoiceVectorOutlierPruner( + speakerOptions.Resemblyzer, + NullLogger.Instance), + Options.Create(appOptions), + NullLogger.Instance); + } + + public SpeakerIdentificationRequest CreateRequest( + string speaker, + int sampleCount, + IReadOnlyList? attendees = null, + int segmentCount = 1) + { + var transcriptPath = Path.Combine(directory, "transcript.md"); + File.WriteAllText(transcriptPath, "Transcript"); + var segments = Enumerable.Range(0, segmentCount) + .Select(index => new TranscriptionSegment( + TimeSpan.FromSeconds(index * 30), + TimeSpan.FromSeconds((index + 1) * 30), + speaker, + "enough useful words for a speaker sample")) + .ToList(); + var segment = segments[0]; + return new SpeakerIdentificationRequest( + Path.Combine(directory, "meeting.wav"), + new MeetingNote( + Path.Combine(directory, "meeting.md"), + new MeetingNoteFrontmatter + { + Title = "Test", + Attendees = attendees?.ToList() ?? ["Chris"], + Transcript = transcriptPath, + AssistantContext = Path.Combine(directory, "context.md"), + Summary = Path.Combine(directory, "summary.md") + }, + ""), + segments, + Enumerable.Range(0, sampleCount) + .Select(index => new SpeakerAudioSample(speaker, segment, [(byte)(index + 1)], 100 - index)) + .ToList()); + } + + public async Task AddIdentityAsync( + string? name, + IReadOnlyList vectors, + IReadOnlyList? candidates = null, + IReadOnlyList? aliases = null) + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + var now = DateTimeOffset.UtcNow; + var identity = new SpeakerIdentity + { + CanonicalName = name, + CreatedAt = now, + UpdatedAt = now, + CandidateNames = candidates?.Select(candidate => new SpeakerCandidateName { Name = candidate }).ToList() ?? [], + Aliases = aliases?.Select(alias => new SpeakerAlias { Name = alias }).ToList() ?? [], + VoiceVectors = vectors.Select((vector, index) => new SpeakerVoiceVector + { + ModelId = speakerOptions.Resemblyzer.ModelId, + Dimensions = 256, + VectorBytes = ToBytes(vector), + Fingerprint = $"known-{index}", + CreatedAt = now.AddMinutes(index) + }).ToList() + }; + context.SpeakerIdentities.Add(identity); + await context.SaveChangesAsync(); + return identity; + } + + public async Task AddCurrentRunCandidateAsync( + SpeakerIdentificationRequest request, + string candidateName) + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + var now = DateTimeOffset.UtcNow; + context.SpeakerIdentities.Add(new SpeakerIdentity + { + CreatedAt = now, + UpdatedAt = now, + CandidateNames = [new SpeakerCandidateName { Name = candidateName }], + Snippets = [new SpeakerSnippet { WavBytes = [9, 8, 7], CreatedAt = now }], + VoiceVectors = + [ + new SpeakerVoiceVector + { + ModelId = "older-model", + Dimensions = 256, + VectorBytes = ToBytes(UnitVector(1)), + Fingerprint = "older-model-vector", + CreatedAt = now + } + ], + References = + [ + SpeakerIdentityReferences.Create( + request.MeetingNote.Path, + request.MeetingNote.Frontmatter.Transcript, + now) + ] + }); + await context.SaveChangesAsync(); + } + + public async Task LoadIdentityAsync(int id) + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + return await context.SpeakerIdentities + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.Aliases) + .Include(identity => identity.References) + .SingleAsync(identity => identity.Id == id); + } + + public async Task LoadOnlyIdentityAsync() + { + await using var context = new TestDbContextFactory(databasePath).CreateDbContext(); + return await context.SpeakerIdentities + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.References) + .SingleAsync(); + } + + public ValueTask DisposeAsync() + { + if (Directory.Exists(directory)) + { + Directory.Delete(directory, recursive: true); + } + + return ValueTask.CompletedTask; + } + } + + private sealed class FakeEncoder : IResemblyzerVoiceEncoder + { + public IReadOnlyList Vectors { get; set; } = []; + + public List> Requests { get; } = []; + + public Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + Requests.Add(wavSamples.Select(sample => sample.ToArray()).ToList()); + return Task.FromResult>(Vectors.Take(wavSamples.Count).ToList()); + } + + public Task WarmUpAsync(CancellationToken cancellationToken) + { + return Task.CompletedTask; + } + } + + private sealed class FakeSnippetExtractor : ISpeakerSnippetExtractor + { + public List> Requests { get; } = []; + + public Task ExtractSnippetAsync( + string audioPath, + IReadOnlyList speakerSegments, + CancellationToken cancellationToken) + { + Requests.Add(speakerSegments.ToList()); + return Task.FromResult([checked((byte)Requests.Count)]); + } + } + + private sealed class TestDbContextFactory : IDbContextFactory + { + private readonly string databasePath; + + public TestDbContextFactory(string databasePath) + { + this.databasePath = databasePath; + } + + public SpeakerIdentityDbContext CreateDbContext() + { + return new SpeakerIdentityDbContext( + new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath};Pooling=False") + .Options); + } + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerSpeakerIdentityMergeServiceTests.cs b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentityMergeServiceTests.cs new file mode 100644 index 0000000..5711692 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerSpeakerIdentityMergeServiceTests.cs @@ -0,0 +1,174 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerSpeakerIdentityMergeServiceTests +{ + [Fact] + public async Task MergeRecentIdentitiesRequiresTwoDisjointMatchingVectorClusters() + { + var directory = Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + try + { + var databasePath = Path.Combine(directory, "identities.db"); + var factory = new TestDbContextFactory(databasePath); + await using (var context = factory.CreateDbContext()) + { + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + context.SpeakerIdentities.Add(CreateIdentity( + "Chris", + DateTimeOffset.UtcNow.AddMonths(-2), + Enumerable.Range(0, 10) + .Select(index => UnitVector(0, index + 20, 0.02f)) + .ToList(), + directory)); + context.SpeakerIdentities.Add(CreateIdentity( + "Chris duplicate", + DateTimeOffset.UtcNow.AddDays(-1), + Enumerable.Range(0, 10) + .Select(index => UnitVector(0, index + 2, 0.02f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 40, 0.02f))) + .ToList(), + directory)); + context.SpeakerIdentities.Add(CreateIdentity( + "Decoy", + DateTimeOffset.UtcNow.AddMonths(-1), + [UnitVector(0)], + directory)); + await context.SaveChangesAsync(); + } + + var resemblyzerOptions = new ResemblyzerSpeakerRecognitionOptions + { + Enabled = true, + RequiredVectorsPerSpeaker = 5, + MaxVectorsPerIdentity = 1000, + MinimumClusterCohesion = 0.75, + MinimumIdentitySimilarity = 0.75, + MinimumSimilarityMargin = 0.05, + ModelId = "resemblyzer-0.1.4-pretrained" + }; + var service = new ResemblyzerSpeakerIdentityMergeService( + factory, + new ResemblyzerVoiceClusterMatcher( + resemblyzerOptions, + NullLogger.Instance), + new ResemblyzerVoiceVectorOutlierPruner( + resemblyzerOptions, + NullLogger.Instance), + Options.Create(new MeetingAssistantOptions + { + SpeakerIdentification = new SpeakerIdentificationOptions + { + MergeRecentIdentityAge = TimeSpan.FromDays(14), + MaxMatchCandidates = 1, + MaxSnippetsPerSpeaker = 3, + Resemblyzer = resemblyzerOptions + } + }), + NullLogger.Instance); + + var result = await service.MergeRecentIdentitiesAsync(TimeSpan.FromDays(14), CancellationToken.None); + + Assert.Equal(2, result.MatchAttempts); + Assert.Equal(1, result.MergedPairs); + await using var verification = factory.CreateDbContext(); + var saved = await verification.SpeakerIdentities + .Include(identity => identity.Aliases) + .Include(identity => identity.VoiceVectors) + .SingleAsync(identity => identity.CanonicalName == "Chris"); + Assert.Equal(2, await verification.SpeakerIdentities.CountAsync()); + Assert.Equal("Chris", saved.CanonicalName); + Assert.Contains(saved.Aliases, alias => alias.Name == "Chris duplicate"); + Assert.Equal(20, saved.VoiceVectors.Count); + Assert.All(saved.VoiceVectors, vector => Assert.True(ToVector(vector.VectorBytes)[0] > 0.9f)); + } + finally + { + Directory.Delete(directory, recursive: true); + } + } + + private static SpeakerIdentity CreateIdentity( + string name, + DateTimeOffset createdAt, + IReadOnlyList vectors, + string directory) + { + var transcriptPath = Path.Combine(directory, $"{Guid.NewGuid():N}.md"); + File.WriteAllText(transcriptPath, "Transcript"); + return new SpeakerIdentity + { + CanonicalName = name, + CreatedAt = createdAt, + UpdatedAt = createdAt, + VoiceVectors = vectors.Select((vector, index) => new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = ToBytes(vector), + Fingerprint = $"{name}-{index}", + CreatedAt = createdAt.AddMinutes(index) + }).ToList(), + References = + [ + new SpeakerIdentityReference + { + MeetingNotePath = Path.Combine(directory, $"{Guid.NewGuid():N}.md"), + TranscriptPath = transcriptPath, + CreatedAt = createdAt + } + ] + }; + } + + private static float[] UnitVector(int primary, int? secondary = null, float secondaryValue = 0) + { + var vector = new float[256]; + vector[primary] = 1; + if (secondary is { } index) + { + vector[index] = secondaryValue; + } + + return vector; + } + + private static byte[] ToBytes(float[] vector) + { + var bytes = new byte[vector.Length * sizeof(float)]; + Buffer.BlockCopy(vector, 0, bytes, 0, bytes.Length); + return bytes; + } + + private static float[] ToVector(byte[] bytes) + { + var vector = new float[bytes.Length / sizeof(float)]; + Buffer.BlockCopy(bytes, 0, vector, 0, bytes.Length); + return vector; + } + + private sealed class TestDbContextFactory : IDbContextFactory + { + private readonly string databasePath; + + public TestDbContextFactory(string databasePath) + { + this.databasePath = databasePath; + } + + public SpeakerIdentityDbContext CreateDbContext() + { + return new SpeakerIdentityDbContext( + new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath};Pooling=False") + .Options); + } + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerVoiceClusterMatcherTests.cs b/MeetingAssistant.Tests/ResemblyzerVoiceClusterMatcherTests.cs new file mode 100644 index 0000000..19f93f3 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerVoiceClusterMatcherTests.cs @@ -0,0 +1,122 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerVoiceClusterMatcherTests +{ + [Fact] + public void CoherentSimilarClusterSelectsUnambiguousIdentity() + { + var matcher = CreateMatcher(); + var query = Enumerable.Range(0, 5) + .Select(index => UnitVector(0, secondaryDimension: index + 2, secondaryValue: 0.05f)) + .ToList(); + var candidates = new[] + { + new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0), UnitVector(0, 10, 0.02f)]), + new ResemblyzerVoiceVectorCandidate(77, [UnitVector(1), UnitVector(1, 11, 0.02f)]) + }; + + var result = matcher.Match(query, candidates); + + Assert.True(result.Accepted); + Assert.Equal(42, result.IdentityId); + Assert.True(result.Cohesion >= 0.99); + Assert.True(result.BestSimilarity >= 0.99); + Assert.True(result.RunnerUpSimilarity < 0.1); + } + + [Fact] + public void FourVectorsDoNotTriggerAutomaticMatching() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 4).ToList(), + [new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)])]); + + Assert.False(result.Accepted); + Assert.Contains("4/5", result.Reason); + } + + [Fact] + public void IncoherentClusterIsRejectedBeforeIdentityScoring() + { + var result = CreateMatcher().Match( + Enumerable.Range(0, 5).Select(index => UnitVector(index)).ToList(), + [new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)])]); + + Assert.False(result.Accepted); + Assert.Contains("cohesion", result.Reason); + Assert.Null(result.BestSimilarity); + } + + [Fact] + public void SimilarityBelowThresholdIsRejected() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 5).ToList(), + [new ResemblyzerVoiceVectorCandidate(42, [UnitVector(1)])]); + + Assert.False(result.Accepted); + Assert.Contains("best similarity", result.Reason); + Assert.Equal(0, result.BestSimilarity); + } + + [Fact] + public void SimilarCandidatesWithinMarginAreRejected() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 5).ToList(), + [ + new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)]), + new ResemblyzerVoiceVectorCandidate(77, [UnitVector(0, 1, 0.01f)]) + ]); + + Assert.False(result.Accepted); + Assert.Contains("margin", result.Reason); + Assert.NotNull(result.RunnerUpSimilarity); + } + + [Fact] + public void InvalidStoredCandidateDoesNotAbortScoringOtherIdentities() + { + var result = CreateMatcher().Match( + Enumerable.Repeat(UnitVector(0), 5).ToList(), + [ + new ResemblyzerVoiceVectorCandidate(13, [new float[256]]), + new ResemblyzerVoiceVectorCandidate(42, [UnitVector(0)]) + ]); + + Assert.True(result.Accepted); + Assert.Equal(42, result.IdentityId); + } + + private static ResemblyzerVoiceClusterMatcher CreateMatcher() + { + return new ResemblyzerVoiceClusterMatcher( + new ResemblyzerSpeakerRecognitionOptions + { + RequiredVectorsPerSpeaker = 5, + MinimumClusterCohesion = 0.75, + MinimumIdentitySimilarity = 0.75, + MinimumSimilarityMargin = 0.05 + }, + NullLogger.Instance); + } + + private static float[] UnitVector( + int primaryDimension, + int? secondaryDimension = null, + float secondaryValue = 0) + { + var vector = new float[256]; + vector[primaryDimension] = 1; + if (secondaryDimension is { } dimension) + { + vector[dimension] = secondaryValue; + } + + return vector; + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerVoiceVectorOutlierPrunerTests.cs b/MeetingAssistant.Tests/ResemblyzerVoiceVectorOutlierPrunerTests.cs new file mode 100644 index 0000000..050159f --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerVoiceVectorOutlierPrunerTests.cs @@ -0,0 +1,141 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.Extensions.Logging.Abstractions; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerVoiceVectorOutlierPrunerTests +{ + [Fact] + public void DominantDensityClusterRemovesForeignSpeakerVectorsAtThreshold() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 16) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(4, result); + Assert.Equal(16, identity.VoiceVectors.Count); + Assert.All( + identity.VoiceVectors, + stored => Assert.True(SpeakerVoiceVectors.Decode(stored)[0] > 0.9f)); + } + + [Fact] + public void AmbiguousDenseClustersPreserveAllVectors() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 10) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 10) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(0, result); + Assert.Equal(20, identity.VoiceVectors.Count); + } + + [Fact] + public void BelowMinimumVectorCountPreservesAllVectorsWithoutEvaluation() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 15) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(0, result); + Assert.Equal(19, identity.VoiceVectors.Count); + } + + [Fact] + public void PruningPreservesOtherModelsAndMalformedStoredRows() + { + var identity = IdentityWithVectors( + Enumerable.Range(0, 16) + .Select(index => UnitVector(0, index + 2, 0.04f)) + .Concat(Enumerable.Range(0, 4) + .Select(index => UnitVector(1, index + 30, 0.04f))) + .ToList()); + SpeakerVoiceVectors.AddDistinct( + identity, + [UnitVector(2, 60, 0.04f)], + "older-model", + 1000, + DateTimeOffset.UtcNow); + identity.VoiceVectors.Add(new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = [1], + Fingerprint = "malformed", + CreatedAt = DateTimeOffset.UtcNow + }); + identity.VoiceVectors.Add(new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = "zero-magnitude", + CreatedAt = DateTimeOffset.UtcNow + }); + var pruner = CreatePruner(); + + var result = pruner.Prune(identity); + + Assert.Equal(4, result); + Assert.Equal(19, identity.VoiceVectors.Count); + Assert.Contains(identity.VoiceVectors, vector => vector.ModelId == "older-model"); + Assert.Contains(identity.VoiceVectors, vector => vector.Fingerprint == "malformed"); + Assert.Contains(identity.VoiceVectors, vector => vector.Fingerprint == "zero-magnitude"); + } + + private static ResemblyzerVoiceVectorOutlierPruner CreatePruner() + { + return new ResemblyzerVoiceVectorOutlierPruner( + new ResemblyzerSpeakerRecognitionOptions + { + ModelId = "resemblyzer-0.1.4-pretrained", + OutlierPruningMinimumVectors = 20, + OutlierPruningNeighborSimilarity = 0.90, + OutlierPruningMinimumNeighbors = 3, + OutlierPruningMinimumClusterRatio = 0.60 + }, + NullLogger.Instance); + } + + private static SpeakerIdentity IdentityWithVectors(IReadOnlyList vectors) + { + var identity = new SpeakerIdentity(); + SpeakerVoiceVectors.AddDistinct( + identity, + vectors, + "resemblyzer-0.1.4-pretrained", + 1000, + DateTimeOffset.UtcNow); + return identity; + } + + private static float[] UnitVector( + int primaryDimension, + int secondaryDimension, + float secondaryValue) + { + var vector = new float[256]; + vector[primaryDimension] = 1; + vector[secondaryDimension] = secondaryValue; + return vector; + } +} diff --git a/MeetingAssistant.Tests/ResemblyzerWarmupHostedServiceTests.cs b/MeetingAssistant.Tests/ResemblyzerWarmupHostedServiceTests.cs new file mode 100644 index 0000000..c61ed78 --- /dev/null +++ b/MeetingAssistant.Tests/ResemblyzerWarmupHostedServiceTests.cs @@ -0,0 +1,78 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Tests; + +public sealed class ResemblyzerWarmupHostedServiceTests +{ + [Fact] + public async Task EnabledWarmupStartsWithoutBlockingApplicationStartup() + { + var encoder = new BlockingEncoder(); + var options = new MeetingAssistantOptions(); + options.SpeakerIdentification.Resemblyzer.Enabled = true; + var service = new ResemblyzerWarmupHostedService( + encoder, + Options.Create(options), + NullLogger.Instance); + + await service.StartAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(1)); + await encoder.WaitForWarmupAsync(); + + await service.StopAsync(CancellationToken.None); + + Assert.True(encoder.CancellationObserved); + } + + [Fact] + public async Task DisabledWarmupDoesNotInvokeEncoder() + { + var encoder = new BlockingEncoder(); + var service = new ResemblyzerWarmupHostedService( + encoder, + Options.Create(new MeetingAssistantOptions()), + NullLogger.Instance); + + await service.StartAsync(CancellationToken.None); + await service.StopAsync(CancellationToken.None); + + Assert.False(encoder.WarmupStarted); + } + + private sealed class BlockingEncoder : IResemblyzerVoiceEncoder + { + private readonly TaskCompletionSource started = new(TaskCreationOptions.RunContinuationsAsynchronously); + + public bool WarmupStarted { get; private set; } + + public bool CancellationObserved { get; private set; } + + public Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + return Task.FromResult>([]); + } + + public async Task WarmUpAsync(CancellationToken cancellationToken) + { + WarmupStarted = true; + started.TrySetResult(); + try + { + await Task.Delay(Timeout.InfiniteTimeSpan, cancellationToken); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + CancellationObserved = true; + } + } + + public Task WaitForWarmupAsync() + { + return started.Task.WaitAsync(TimeSpan.FromSeconds(1)); + } + } +} diff --git a/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs b/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs index 2253534..230ebb1 100644 --- a/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs +++ b/MeetingAssistant.Tests/SpeakerAudioSampleCollectorTests.cs @@ -8,13 +8,22 @@ namespace MeetingAssistant.Tests; public sealed class SpeakerAudioSampleCollectorTests { [Fact] - public void CollectorWaitsForConfiguredUninterruptedSpeechDuration() + public void DefaultSampleDurationsAreTenAndSixtySeconds() + { + var options = new SpeakerIdentificationOptions(); + + Assert.Equal(TimeSpan.FromSeconds(10), options.MinimumSampleSpeechDuration); + Assert.Equal(TimeSpan.FromSeconds(60), options.MaximumSampleDuration); + } + + [Fact] + public void CollectorWaitsForConfiguredSpeakerAudioDuration() { var collector = new SpeakerAudioSampleCollector( TimeSpan.FromMinutes(2), maxSamplesPerSpeaker: 3, - minimumUninterruptedSpeechDuration: TimeSpan.FromSeconds(30), - maximumSegmentGap: TimeSpan.FromSeconds(1)); + minimumSampleSpeechDuration: TimeSpan.FromSeconds(30), + maximumSampleDuration: TimeSpan.FromSeconds(60)); collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(35))); collector.TryAdd(Segment(0, 12, "Guest01", "one two three four five.")); @@ -37,8 +46,8 @@ public sealed class SpeakerAudioSampleCollectorTests var collector = new SpeakerAudioSampleCollector( TimeSpan.FromMinutes(2), maxSamplesPerSpeaker: 3, - minimumUninterruptedSpeechDuration: TimeSpan.FromSeconds(30), - maximumSegmentGap: TimeSpan.FromSeconds(1)); + minimumSampleSpeechDuration: TimeSpan.FromSeconds(30), + maximumSampleDuration: TimeSpan.FromSeconds(60)); collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(50))); collector.TryAdd(Segment(0, 20, "Guest01", "one two three four five.")); @@ -48,6 +57,61 @@ public sealed class SpeakerAudioSampleCollectorTests Assert.DoesNotContain(collector.Snapshot(), sample => sample.Speaker == "Guest01"); } + [Fact] + public void CollectorCombinesConsecutiveSameSpeakerSegmentsAcrossProviderPauses() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 3, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(10), + maximumSampleDuration: TimeSpan.FromSeconds(60)); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(20))); + + collector.TryAdd(Segment(0, 4, "Guest01", "one two three four.")); + collector.TryAdd(Segment(7, 13, "Guest01", "five six seven eight.")); + + var sample = Assert.Single(collector.Snapshot()); + Assert.Equal(TimeSpan.Zero, sample.Segment.Start); + Assert.Equal(TimeSpan.FromSeconds(13), sample.Segment.End); + } + + [Fact] + public void CollectorDoesNotCountProviderPausesAsSpeakerAudio() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 3, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(10), + maximumSampleDuration: TimeSpan.FromSeconds(60)); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(30))); + + collector.TryAdd(Segment(0, 4, "Guest01", "one two three four.")); + collector.TryAdd(Segment(20, 25, "Guest01", "five six seven eight.")); + + Assert.Empty(collector.Snapshot()); + } + + [Fact] + public void CollectorCapsRecognitionWavAtConfiguredMaximumDuration() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 3, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(10), + maximumSampleDuration: TimeSpan.FromSeconds(60)); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(75))); + + var sample = collector.TryAdd(Segment( + 0, + 70, + "Guest01", + "one two three four five six seven eight nine ten.")); + + Assert.NotNull(sample); + Assert.Equal(TimeSpan.FromSeconds(60), sample.Segment.End); + Assert.True(ReadDuration(sample.WavBytes) <= TimeSpan.FromSeconds(60)); + } + [Fact] public void CollectorLogsWhenSampleIsDiscardedBecauseSpeechIsTooShort() { @@ -55,8 +119,8 @@ public sealed class SpeakerAudioSampleCollectorTests var collector = new SpeakerAudioSampleCollector( TimeSpan.FromMinutes(2), maxSamplesPerSpeaker: 3, - minimumUninterruptedSpeechDuration: TimeSpan.FromSeconds(30), - maximumSegmentGap: TimeSpan.FromSeconds(1), + minimumSampleSpeechDuration: TimeSpan.FromSeconds(30), + maximumSampleDuration: TimeSpan.FromSeconds(60), logger: logger); collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(35))); @@ -69,6 +133,47 @@ public sealed class SpeakerAudioSampleCollectorTests Assert.Contains("minimum duration", message); } + [Fact] + public void CollectorCanResetThePendingSpanAfterEachAcceptedSample() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 5, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(2), + maximumSampleDuration: TimeSpan.FromSeconds(60), + requireNonOverlappingSamples: true); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(6))); + + collector.TryAdd(Segment(0, 2, "Guest01", "one two three.")); + collector.TryAdd(Segment(2, 4, "Guest01", "four five six.")); + collector.TryAdd(Segment(4, 6, "Guest01", "seven eight nine.")); + + var samples = collector.Snapshot().OrderBy(sample => sample.Segment.Start).ToList(); + + Assert.Collection( + samples, + sample => Assert.Equal((TimeSpan.Zero, TimeSpan.FromSeconds(2)), (sample.Segment.Start, sample.Segment.End)), + sample => Assert.Equal((TimeSpan.FromSeconds(2), TimeSpan.FromSeconds(4)), (sample.Segment.Start, sample.Segment.End)), + sample => Assert.Equal((TimeSpan.FromSeconds(4), TimeSpan.FromSeconds(6)), (sample.Segment.Start, sample.Segment.End))); + } + + [Fact] + public void ResettingCollectorRejectsASegmentThatOverlapsAnAcceptedSample() + { + var collector = new SpeakerAudioSampleCollector( + TimeSpan.FromMinutes(2), + maxSamplesPerSpeaker: 5, + minimumSampleSpeechDuration: TimeSpan.FromSeconds(2), + maximumSampleDuration: TimeSpan.FromSeconds(60), + requireNonOverlappingSamples: true); + collector.AppendAudio(CreateAudio(TimeSpan.FromSeconds(4))); + + collector.TryAdd(Segment(0, 2, "Guest01", "one two three.")); + collector.TryAdd(Segment(1, 3, "Guest01", "four five six.")); + + Assert.Single(collector.Snapshot()); + } + private static TranscriptionSegment Segment( double start, double end, diff --git a/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs b/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs index 32e17d4..d0d62cf 100644 --- a/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs +++ b/MeetingAssistant.Tests/SpeakerIdentityMergeServiceTests.cs @@ -107,6 +107,9 @@ public sealed class SpeakerIdentityMergeServiceTests return new SpeakerIdentityMergeService( new TestSpeakerIdentityDbContextFactory(dbPath), Matcher, + new ResemblyzerVoiceVectorOutlierPruner( + new ResemblyzerSpeakerRecognitionOptions(), + NullLogger.Instance), Options.Create(new MeetingAssistantOptions { SpeakerIdentification = new SpeakerIdentificationOptions diff --git a/MeetingAssistant.Tests/SpeakerIdentityMergerTests.cs b/MeetingAssistant.Tests/SpeakerIdentityMergerTests.cs new file mode 100644 index 0000000..daf3fba --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerIdentityMergerTests.cs @@ -0,0 +1,47 @@ +using MeetingAssistant.Speakers; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerIdentityMergerTests +{ + [Fact] + public void MergeIntoRetainsNewestDistinctVoiceVectorsUpToConfiguredLimit() + { + var now = DateTimeOffset.UtcNow; + var target = new SpeakerIdentity + { + CanonicalName = "Chris", + VoiceVectors = + [ + Vector("oldest", 1, now.AddMinutes(-4)), + Vector("duplicate", 2, now.AddMinutes(-3)) + ] + }; + var source = new SpeakerIdentity + { + VoiceVectors = + [ + Vector("duplicate", 2, now.AddMinutes(-2)), + Vector("newer", 3, now.AddMinutes(-1)), + Vector("newest", 4, now) + ] + }; + + SpeakerIdentityMerger.MergeInto(target, source, maxSnippets: 3, maxVoiceVectors: 3); + + Assert.Equal(["duplicate", "newer", "newest"], target.VoiceVectors.Select(vector => vector.Fingerprint).Order()); + Assert.All(target.VoiceVectors, vector => Assert.Same(target, vector.SpeakerIdentity)); + } + + private static SpeakerVoiceVector Vector(string fingerprint, byte value, DateTimeOffset createdAt) + { + return new SpeakerVoiceVector + { + ModelId = "model", + Dimensions = 256, + VectorBytes = Enumerable.Repeat(value, 256 * sizeof(float)).ToArray(), + Fingerprint = fingerprint, + CreatedAt = createdAt + }; + } +} diff --git a/MeetingAssistant.Tests/SpeakerRecognitionFeatureSelectionTests.cs b/MeetingAssistant.Tests/SpeakerRecognitionFeatureSelectionTests.cs new file mode 100644 index 0000000..a62005c --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerRecognitionFeatureSelectionTests.cs @@ -0,0 +1,80 @@ +using MeetingAssistant.Speakers; +using Microsoft.Data.Sqlite; +using Microsoft.AspNetCore.Hosting; +using Microsoft.AspNetCore.Mvc.Testing; +using Microsoft.AspNetCore.TestHost; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.DependencyInjection.Extensions; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerRecognitionFeatureSelectionTests +{ + [Theory] + [InlineData(false, typeof(SpeakerIdentityService), typeof(SpeakerIdentityMergeService), 1, false)] + [InlineData(true, typeof(ResemblyzerSpeakerIdentificationService), typeof(ResemblyzerSpeakerIdentityMergeService), 1000, true)] + public async Task FeatureFlagSelectsExactlyOneIdentificationBackend( + bool enabled, + Type expectedIdentificationType, + Type expectedMergeType, + int expectedMinimumSamples, + bool expectedNonOverlappingSamples) + { + var directory = Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + try + { + await using var factory = new WebApplicationFactory() + .WithWebHostBuilder(builder => + { + builder.ConfigureAppConfiguration((_, configuration) => + { + configuration.AddInMemoryCollection(new Dictionary + { + ["MeetingAssistant:FunAsr:Backend:Enabled"] = "false", + ["MeetingAssistant:SpeakerIdentification:DatabasePath"] = Path.Combine(directory, "identities.db"), + ["MeetingAssistant:SpeakerIdentification:PyannoteValidation:Enabled"] = "false", + ["MeetingAssistant:SpeakerIdentification:Resemblyzer:Enabled"] = enabled.ToString() + }); + }); + builder.ConfigureTestServices(services => + { + services.RemoveAll(); + services.AddSingleton(); + }); + }); + + var service = factory.Services.GetRequiredService(); + + Assert.IsType(expectedIdentificationType, service); + Assert.Single(factory.Services.GetServices()); + var mergeService = factory.Services.GetRequiredService(); + Assert.IsType(expectedMergeType, mergeService); + Assert.Single(factory.Services.GetServices()); + var policy = factory.Services.GetRequiredService(); + Assert.Equal(expectedMinimumSamples, policy.MinimumRetainedSamples); + Assert.Equal(expectedNonOverlappingSamples, policy.RequireNonOverlappingSamples); + } + finally + { + SqliteConnection.ClearAllPools(); + Directory.Delete(directory, recursive: true); + } + } + + private sealed class NoopEncoder : IResemblyzerVoiceEncoder + { + public Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + return Task.FromResult>([]); + } + + public Task WarmUpAsync(CancellationToken cancellationToken) + { + return Task.CompletedTask; + } + } +} diff --git a/MeetingAssistant.Tests/SpeakerSampleSpanSelectorTests.cs b/MeetingAssistant.Tests/SpeakerSampleSpanSelectorTests.cs new file mode 100644 index 0000000..0d6d88a --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerSampleSpanSelectorTests.cs @@ -0,0 +1,142 @@ +using MeetingAssistant.Speakers; +using MeetingAssistant.Transcription; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerSampleSpanSelectorTests +{ + [Fact] + public void SelectBestSameSpeakerSpanCombinesLinesAndCapsTheClip() + { + var span = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpan( + [ + Segment(0, 4), + Segment(7, 13), + Segment(20, 70) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60)); + + Assert.Equal(3, span.Count); + Assert.Equal(TimeSpan.Zero, span[0].Start); + Assert.Equal(TimeSpan.FromSeconds(60), span[^1].End); + Assert.Equal(TimeSpan.FromSeconds(60), SpeakerSampleSpanSelector.SpanDuration(span)); + } + + [Fact] + public void SelectBestSameSpeakerSpanDoesNotCountProviderPausesAsSpeakerAudio() + { + var span = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpan( + [ + Segment(0, 4), + Segment(20, 25) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60)); + + Assert.Empty(span); + } + + [Fact] + public void SelectBestSameSpeakerSpansReturnsDistinctMinimumLengthSamples() + { + var segments = Enumerable.Range(0, 6) + .Select(index => new TranscriptionSegment( + TimeSpan.FromSeconds(index * 10), + TimeSpan.FromSeconds((index + 1) * 10), + "Guest01", + $"segment {index} has enough words")) + .ToList(); + + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + segments, + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + Assert.Equal(5, spans.Count); + Assert.All(spans, span => Assert.True(SpeakerSampleSpanSelector.SpanDuration(span) >= TimeSpan.FromSeconds(10))); + Assert.All( + spans.Zip(spans.Skip(1)), + pair => Assert.True(pair.First[^1].End <= pair.Second[0].Start)); + } + + [Fact] + public void SelectBestSameSpeakerSpansSkipsSegmentsThatOverlapACompletedSample() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [ + Segment(0, 10), + Segment(9, 19), + Segment(20, 30) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + Assert.Equal(2, spans.Count); + Assert.Equal(TimeSpan.Zero, spans[0][0].Start); + Assert.Equal(TimeSpan.FromSeconds(20), spans[1][0].Start); + } + + [Fact] + public void SelectBestSameSpeakerSpansCombinesConsecutiveLinesAcrossProviderPauses() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [ + Segment(0, 4), + Segment(7, 13) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + var span = Assert.Single(spans); + Assert.Equal(TimeSpan.Zero, span[0].Start); + Assert.Equal(TimeSpan.FromSeconds(13), span[^1].End); + } + + [Fact] + public void SelectBestSameSpeakerSpansDoesNotCountProviderPausesAsSpeakerAudio() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [ + Segment(0, 4), + Segment(20, 25) + ], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + Assert.Empty(spans); + } + + [Fact] + public void SelectBestSameSpeakerSpansCapsFinalizedSamplesAtConfiguredMaximumDuration() + { + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + [Segment(0, 70)], + "Guest01", + TimeSpan.FromSeconds(10), + TimeSpan.FromSeconds(60), + maxSpans: 5); + + var span = Assert.Single(spans); + Assert.Equal(TimeSpan.FromSeconds(60), SpeakerSampleSpanSelector.SpanDuration(span)); + } + + private static TranscriptionSegment Segment(double start, double end) + { + return new TranscriptionSegment( + TimeSpan.FromSeconds(start), + TimeSpan.FromSeconds(end), + "Guest01", + "enough words for this segment"); + } +} diff --git a/MeetingAssistant.Tests/SpeakerVoiceVectorPersistenceTests.cs b/MeetingAssistant.Tests/SpeakerVoiceVectorPersistenceTests.cs new file mode 100644 index 0000000..21464cc --- /dev/null +++ b/MeetingAssistant.Tests/SpeakerVoiceVectorPersistenceTests.cs @@ -0,0 +1,106 @@ +using MeetingAssistant.Speakers; +using Microsoft.EntityFrameworkCore; + +namespace MeetingAssistant.Tests; + +public sealed class SpeakerVoiceVectorPersistenceTests +{ + [Fact] + public async Task IdentityStoresVersionedVoiceVectorAndDeletesItWithIdentity() + { + var databasePath = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "speaker-identities.db"); + Directory.CreateDirectory(Path.GetDirectoryName(databasePath)!); + var options = new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath}") + .Options; + + await using (var context = new SpeakerIdentityDbContext(options)) + { + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + context.SpeakerIdentities.Add(new SpeakerIdentity + { + CanonicalName = "Chris", + CreatedAt = DateTimeOffset.UtcNow, + UpdatedAt = DateTimeOffset.UtcNow, + VoiceVectors = + [ + new SpeakerVoiceVector + { + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = "vector-a", + CreatedAt = DateTimeOffset.UtcNow + } + ] + }); + await context.SaveChangesAsync(); + } + + await using (var context = new SpeakerIdentityDbContext(options)) + { + var identity = await context.SpeakerIdentities + .Include(candidate => candidate.VoiceVectors) + .SingleAsync(); + var vector = Assert.Single(identity.VoiceVectors); + Assert.Equal("resemblyzer-0.1.4-pretrained", vector.ModelId); + Assert.Equal(256, vector.Dimensions); + Assert.Equal(256 * sizeof(float), vector.VectorBytes.Length); + + context.SpeakerIdentities.Remove(identity); + await context.SaveChangesAsync(); + } + + await using (var context = new SpeakerIdentityDbContext(options)) + { + Assert.Empty(await context.SpeakerVoiceVectors.ToListAsync()); + } + } + + [Fact] + public async Task IdentityRejectsDuplicateVoiceVectorFingerprint() + { + var databasePath = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "speaker-identities.db"); + Directory.CreateDirectory(Path.GetDirectoryName(databasePath)!); + var options = new DbContextOptionsBuilder() + .UseSqlite($"Data Source={databasePath}") + .Options; + await using var context = new SpeakerIdentityDbContext(options); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, CancellationToken.None); + var identity = new SpeakerIdentity + { + CanonicalName = "Chris", + CreatedAt = DateTimeOffset.UtcNow, + UpdatedAt = DateTimeOffset.UtcNow + }; + context.SpeakerIdentities.Add(identity); + await context.SaveChangesAsync(); + + context.SpeakerVoiceVectors.AddRange( + CreateVector(identity.Id, "same-vector"), + CreateVector(identity.Id, "same-vector")); + + await Assert.ThrowsAsync(() => context.SaveChangesAsync()); + } + + private static SpeakerVoiceVector CreateVector(int identityId, string fingerprint) + { + return new SpeakerVoiceVector + { + SpeakerIdentityId = identityId, + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = fingerprint, + CreatedAt = DateTimeOffset.UtcNow + }; + } +} diff --git a/MeetingAssistant.Tests/VenvResemblyzerVoiceEncoderTests.cs b/MeetingAssistant.Tests/VenvResemblyzerVoiceEncoderTests.cs new file mode 100644 index 0000000..ecdac99 --- /dev/null +++ b/MeetingAssistant.Tests/VenvResemblyzerVoiceEncoderTests.cs @@ -0,0 +1,249 @@ +using MeetingAssistant; +using MeetingAssistant.Speakers; +using MeetingAssistant.Transcription; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Options; +using System.Text.Json; + +namespace MeetingAssistant.Tests; + +public sealed class VenvResemblyzerVoiceEncoderTests +{ + [Fact] + public async Task WarmUpAsyncProvisionsVersionedCpuOnlyEnvironmentWithoutDocker() + { + using var fixture = new EncoderFixture(); + + await fixture.Encoder.WarmUpAsync(CancellationToken.None); + + Assert.DoesNotContain( + fixture.Runner.Commands, + command => command.FileName.Contains("docker", StringComparison.OrdinalIgnoreCase)); + Assert.Contains( + fixture.Runner.Commands, + command => command.FileName == "python-test" + && command.Arguments.Take(2).SequenceEqual(["-m", "venv"])); + Assert.Contains( + fixture.Runner.Commands, + command => command.Arguments.Contains("torch==2.14.0+cpu") + && command.Arguments.Contains("https://download.pytorch.org/whl/cpu")); + Assert.Contains( + fixture.Runner.Commands, + command => command.Arguments.Contains("webrtcvad-wheels==2.0.14")); + Assert.Contains( + fixture.Runner.Commands, + command => command.Arguments.Contains("Resemblyzer==0.1.4") + && command.Arguments.Contains("--no-deps")); + } + + [Fact] + public async Task WarmUpAsyncDoesNotMarkFailedEnvironmentReady() + { + using var fixture = new EncoderFixture( + warmupExitCode: 21, + warmupError: "incompatible environment"); + + await Assert.ThrowsAsync( + () => fixture.Encoder.WarmUpAsync(CancellationToken.None)); + + Assert.Empty(Directory.EnumerateFiles( + fixture.RuntimeFolder, + ".ready", + SearchOption.AllDirectories)); + } + + [Fact] + public async Task EncodeAsyncRunsBatchWithManagedEnvironmentPython() + { + var first = new float[256]; + first[0] = 2; + var second = new float[256]; + second[1] = 3; + using var fixture = new EncoderFixture(VectorOutput([first, second])); + + var vectors = await fixture.Encoder.EncodeAsync([[1, 2, 3], [4, 5, 6]], CancellationToken.None); + + Assert.Equal(2, vectors.Count); + Assert.Equal(1f, vectors[0][0], 5); + Assert.Equal(1f, vectors[1][1], 5); + var encodingCommand = Assert.Single( + fixture.Runner.Commands, + command => command.Arguments.Any( + argument => argument.EndsWith("encode.py", StringComparison.Ordinal))); + Assert.EndsWith( + Path.Combine("Scripts", "python.exe"), + encodingCommand.FileName, + StringComparison.OrdinalIgnoreCase); + Assert.Equal(2, encodingCommand.Arguments.Count); + Assert.True(Path.IsPathFullyQualified(encodingCommand.Arguments[1])); + } + + [Fact] + public async Task EncodeAsyncRejectsWrongVectorDimension() + { + var vector = new float[255]; + vector[0] = 1; + using var fixture = new EncoderFixture(VectorOutput([vector])); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("255 dimensions", error.Message); + } + + [Fact] + public async Task EncodeAsyncRejectsMalformedVectorJson() + { + using var fixture = new EncoderFixture( + "__MEETING_ASSISTANT_RESEMBLYZER_JSON_START__\n[not-json]\n__MEETING_ASSISTANT_RESEMBLYZER_JSON_END__"); + + await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + } + + [Fact] + public async Task EncodeAsyncRejectsNonFiniteVector() + { + var values = string.Join(',', new[] { "NaN" }.Concat(Enumerable.Repeat("0", 255))); + using var fixture = new EncoderFixture( + $"__MEETING_ASSISTANT_RESEMBLYZER_JSON_START__\n[[{values}]]\n__MEETING_ASSISTANT_RESEMBLYZER_JSON_END__"); + + await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + } + + [Fact] + public async Task EncodeAsyncRejectsWrongResultCount() + { + var vector = new float[256]; + vector[0] = 1; + using var fixture = new EncoderFixture(VectorOutput([vector, vector])); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("2 vectors for 1 WAV samples", error.Message); + } + + [Fact] + public async Task EncodeAsyncRejectsZeroMagnitudeVector() + { + using var fixture = new EncoderFixture(VectorOutput([new float[256]])); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("zero magnitude", error.Message); + } + + [Fact] + public async Task EncodeAsyncReportsFailedLocalCommand() + { + using var fixture = new EncoderFixture("", 17, "runtime failed"); + + var error = await Assert.ThrowsAsync( + () => fixture.Encoder.EncodeAsync([[1]], CancellationToken.None)); + + Assert.Contains("exit code 17", error.Message); + Assert.Contains("runtime failed", error.Message); + } + + private static string VectorOutput(IReadOnlyList vectors) + { + return $""" + __MEETING_ASSISTANT_RESEMBLYZER_JSON_START__ + {JsonSerializer.Serialize(vectors)} + __MEETING_ASSISTANT_RESEMBLYZER_JSON_END__ + """; + } + + private sealed class EncoderFixture : IDisposable + { + public EncoderFixture( + string encodingOutput = "", + int encodingExitCode = 0, + string encodingError = "", + int warmupExitCode = 0, + string warmupError = "") + { + RuntimeFolder = Path.Combine( + Path.GetTempPath(), + "meeting-assistant-tests", + Guid.NewGuid().ToString("N"), + "resemblyzer"); + Runner = new CapturingCommandRunner( + encodingOutput, + encodingExitCode, + encodingError, + warmupExitCode, + warmupError); + var options = new MeetingAssistantOptions(); + options.SpeakerIdentification.Resemblyzer.PythonCommand = "python-test"; + options.SpeakerIdentification.Resemblyzer.RuntimeFolder = RuntimeFolder; + Encoder = new VenvResemblyzerVoiceEncoder( + Runner, + Options.Create(options), + NullLogger.Instance); + } + + public CapturingCommandRunner Runner { get; } + + public VenvResemblyzerVoiceEncoder Encoder { get; } + + public string RuntimeFolder { get; } + + public void Dispose() + { + if (Directory.Exists(RuntimeFolder)) + { + Directory.Delete(RuntimeFolder, recursive: true); + } + } + } + + private sealed class CapturingCommandRunner : ICommandRunner + { + private readonly string encodingOutput; + private readonly int encodingExitCode; + private readonly string encodingError; + private readonly int warmupExitCode; + private readonly string warmupError; + + public CapturingCommandRunner( + string encodingOutput = "", + int encodingExitCode = 0, + string encodingError = "", + int warmupExitCode = 0, + string warmupError = "") + { + this.encodingOutput = encodingOutput; + this.encodingExitCode = encodingExitCode; + this.encodingError = encodingError; + this.warmupExitCode = warmupExitCode; + this.warmupError = warmupError; + } + + public List Commands { get; } = []; + + public Task RunAsync( + string fileName, + IReadOnlyList arguments, + CancellationToken cancellationToken, + IReadOnlyDictionary? environment = null) + { + Commands.Add(new CapturedCommand(fileName, arguments.ToList())); + var isEncoding = arguments.Any(argument => argument.EndsWith("encode.py", StringComparison.Ordinal)); + if (isEncoding) + { + return Task.FromResult(new CommandResult(encodingExitCode, encodingOutput, encodingError)); + } + + var isWarmup = arguments.Any(argument => argument.Contains("VoiceEncoder", StringComparison.Ordinal)); + return Task.FromResult(isWarmup + ? new CommandResult(warmupExitCode, string.Empty, warmupError) + : new CommandResult(0, string.Empty, string.Empty)); + } + } + + private sealed record CapturedCommand(string FileName, IReadOnlyList Arguments); +} diff --git a/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs b/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs index 109a9ce..14814e2 100644 --- a/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs +++ b/MeetingAssistant.Tests/WorkflowRulesEditorTests.cs @@ -809,6 +809,32 @@ public sealed class WorkflowRulesEditorTests Assert.Empty(await fixture.Context.SpeakerIdentities.ToListAsync()); } + [Fact] + public async Task RulesEditorToolsExposeVoiceVectorCountsSeparatelyFromWavSamples() + { + await using var fixture = await WorkflowRulesEditorIdentityFixture.CreateAsync(); + var identity = await fixture.AddIdentityAsync("Sabrina", sample: [1, 2, 3]); + fixture.Context.SpeakerVoiceVectors.Add(new SpeakerVoiceVector + { + SpeakerIdentityId = identity.Id, + ModelId = "resemblyzer-0.1.4-pretrained", + Dimensions = 256, + VectorBytes = new byte[256 * sizeof(float)], + Fingerprint = "vector-1", + CreatedAt = DateTimeOffset.UtcNow + }); + await fixture.Context.SaveChangesAsync(); + var tools = fixture.CreateTools(); + + var searchResult = await tools.SearchIdentities("Sabrina"); + var readResult = await tools.ReadIdentity(identity.Id); + + Assert.Contains("\"sampleCount\": 1", searchResult); + Assert.Contains("\"voiceVectorCount\": 1", searchResult); + Assert.Contains("\"voiceVectorCount\": 1", readResult); + Assert.DoesNotContain("vector-1", readResult); + } + [Fact] public async Task RulesEditorToolsRefusesSamplelessSpeakerIdentityCreation() { @@ -817,7 +843,7 @@ public sealed class WorkflowRulesEditorTests var result = await tools.CreateIdentity("Sabrina", ["Sabi"], ["Guest-01"]); - Assert.StartsWith("Refused: speaker identities require at least one audio sample.", result); + Assert.StartsWith("Refused: speaker identities require audio evidence", result); Assert.Empty(await fixture.Context.SpeakerIdentities.ToListAsync()); } diff --git a/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs b/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs index 45f3675..692b6c8 100644 --- a/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs +++ b/MeetingAssistant/LaunchProfiles/LaunchProfileOptionsProvider.cs @@ -1,4 +1,5 @@ using Microsoft.Extensions.Configuration; +using MeetingAssistant.Speakers; namespace MeetingAssistant.LaunchProfiles; @@ -48,6 +49,7 @@ public sealed class ConfigurationLaunchProfileOptionsProvider : ILaunchProfileOp var options = BindDefaultOptions(); if (profileName.Equals(DefaultProfileName, StringComparison.OrdinalIgnoreCase)) { + ValidateSpeakerSampleDurations(options, DefaultProfileName); return new LaunchProfile(DefaultProfileName, options); } @@ -59,6 +61,7 @@ public sealed class ConfigurationLaunchProfileOptionsProvider : ILaunchProfileOp profileSection.Bind(options); ApplyArrayOverrides(profileSection, options); + ValidateSpeakerSampleDurations(options, profileName); return new LaunchProfile(profileName, options); } @@ -149,6 +152,18 @@ public sealed class ConfigurationLaunchProfileOptionsProvider : ILaunchProfileOp : name.Trim(); } + private static void ValidateSpeakerSampleDurations( + MeetingAssistantOptions options, + string profileName) + { + SpeakerSampleDurationConfiguration.ValidateOrThrow( + options.SpeakerIdentification, + $"Launch profile '{profileName}' has invalid speaker sample durations"); + ResemblyzerSpeakerRecognitionConfiguration.ValidateOrThrow( + options.SpeakerIdentification.Resemblyzer, + $"Launch profile '{profileName}' has invalid Resemblyzer speaker-recognition settings"); + } + private static void ApplyArrayOverrides( IConfigurationSection profileSection, MeetingAssistantOptions options) diff --git a/MeetingAssistant/MeetingAssistantOptions.cs b/MeetingAssistant/MeetingAssistantOptions.cs index b9d2423..25da41b 100644 --- a/MeetingAssistant/MeetingAssistantOptions.cs +++ b/MeetingAssistant/MeetingAssistantOptions.cs @@ -345,9 +345,9 @@ public sealed class SpeakerIdentificationOptions public int MaxSnippetsPerSpeaker { get; set; } = 3; - public TimeSpan MinimumSampleSpeechDuration { get; set; } = TimeSpan.FromSeconds(30); + public TimeSpan MinimumSampleSpeechDuration { get; set; } = TimeSpan.FromSeconds(10); - public TimeSpan MaximumSampleSegmentGap { get; set; } = TimeSpan.FromSeconds(1); + public TimeSpan MaximumSampleDuration { get; set; } = TimeSpan.FromSeconds(60); public double SilenceBetweenSnippetsSeconds { get; set; } = 1; @@ -360,6 +360,47 @@ public sealed class SpeakerIdentificationOptions public AzureSpeechOptions AzureSpeech { get; set; } = new(); public SpeakerIdentityPyannoteValidationOptions PyannoteValidation { get; set; } = new(); + + public ResemblyzerSpeakerRecognitionOptions Resemblyzer { get; set; } = new(); +} + +public sealed class ResemblyzerSpeakerRecognitionOptions +{ + public bool Enabled { get; set; } + + public int RequiredVectorsPerSpeaker { get; set; } = 5; + + public int MaxVectorsPerIdentity { get; set; } = 1000; + + public int OutlierPruningMinimumVectors { get; set; } = 20; + + public double OutlierPruningNeighborSimilarity { get; set; } = 0.75; + + public int OutlierPruningMinimumNeighbors { get; set; } = 3; + + public double OutlierPruningMinimumClusterRatio { get; set; } = 0.60; + + public double MinimumClusterCohesion { get; set; } = 0.75; + + public double MinimumIdentitySimilarity { get; set; } = 0.75; + + public double MinimumSimilarityMargin { get; set; } = 0.05; + + public string ModelId { get; set; } = "resemblyzer-0.1.4-pretrained"; + + public string PackageVersion { get; set; } = "0.1.4"; + + public string PythonCommand { get; set; } = "python"; + + public string TorchVersion { get; set; } = "2.14.0+cpu"; + + public string TorchIndexUrl { get; set; } = "https://download.pytorch.org/whl/cpu"; + + public string WebRtcVadVersion { get; set; } = "2.0.14"; + + public string RuntimeFolder { get; set; } = @"%LOCALAPPDATA%\MeetingAssistant\Resemblyzer"; + + public TimeSpan CommandTimeout { get; set; } = TimeSpan.FromMinutes(15); } public sealed class SpeakerIdentityPyannoteValidationOptions diff --git a/MeetingAssistant/Program.cs b/MeetingAssistant/Program.cs index fc83750..43f3baa 100644 --- a/MeetingAssistant/Program.cs +++ b/MeetingAssistant/Program.cs @@ -16,7 +16,10 @@ using Microsoft.Extensions.Options; var builder = WebApplication.CreateBuilder(args); builder.Logging.AddProvider(new MeetingAssistantFileLoggerProvider()); -builder.Services.Configure(builder.Configuration.GetSection("MeetingAssistant")); +builder.Services.AddSingleton, MeetingAssistantSpeakerSampleOptionsValidator>(); +builder.Services.AddOptions() + .Bind(builder.Configuration.GetSection("MeetingAssistant")) + .ValidateOnStart(); builder.Services.AddSingleton(); #if WINDOWS builder.Services.AddSingleton(); @@ -91,8 +94,35 @@ builder.Services.AddSingleton(); builder.Services.AddSingleton(); builder.Services.AddSingleton(); -builder.Services.AddSingleton(); -builder.Services.AddSingleton(); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(services => new ResemblyzerVoiceClusterMatcher( + services.GetRequiredService>().Value.SpeakerIdentification.Resemblyzer, + services.GetRequiredService>())); +builder.Services.AddSingleton(services => new ResemblyzerVoiceVectorOutlierPruner( + services.GetRequiredService>().Value.SpeakerIdentification.Resemblyzer, + services.GetRequiredService>())); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(services => +{ + var resemblyzer = services.GetRequiredService>() + .Value.SpeakerIdentification.Resemblyzer; + return resemblyzer.Enabled + ? SpeakerSampleCollectionPolicy.IndependentVectors(resemblyzer.MaxVectorsPerIdentity) + : SpeakerSampleCollectionPolicy.ExistingBackend; +}); +builder.Services.AddSingleton(services => + services.GetRequiredService>() + .Value.SpeakerIdentification.Resemblyzer.Enabled + ? services.GetRequiredService() + : services.GetRequiredService()); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(); +builder.Services.AddSingleton(services => + services.GetRequiredService>() + .Value.SpeakerIdentification.Resemblyzer.Enabled + ? services.GetRequiredService() + : services.GetRequiredService()); builder.Services.AddSingleton(); builder.Services.AddSingleton(); builder.Services.AddSingleton(); @@ -134,6 +164,7 @@ builder.Services.AddHostedService(); builder.Services.AddHostedService(); builder.Services.AddHostedService(); builder.Services.AddHostedService(); +builder.Services.AddHostedService(); #if WINDOWS builder.Services.AddHostedService(); builder.Services.AddHostedService(); diff --git a/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs b/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs index 43315f1..c15c180 100644 --- a/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs +++ b/MeetingAssistant/Recording/MeetingRecordingCoordinator.cs @@ -34,6 +34,7 @@ public sealed class MeetingRecordingCoordinator private readonly IMeetingInactivityClock inactivityClock; private readonly IOfflineTranscriptionBacklog offlineTranscriptionBacklog; private readonly MeetingAssistantOptions options; + private readonly SpeakerSampleCollectionPolicy speakerSampleCollectionPolicy; private readonly ILogger logger; private readonly SemaphoreSlim gate = new(1, 1); private RecordingRun? currentRun; @@ -63,7 +64,8 @@ public sealed class MeetingRecordingCoordinator IMeetingRunArtifactCleaner? artifactCleaner = null, IMeetingInactivityPromptService? inactivityPromptService = null, IMeetingInactivityClock? inactivityClock = null, - IOfflineTranscriptionBacklog? offlineTranscriptionBacklog = null) + IOfflineTranscriptionBacklog? offlineTranscriptionBacklog = null, + SpeakerSampleCollectionPolicy? speakerSampleCollectionPolicy = null) { this.audioSource = audioSource; this.speechRecognitionPipelineFactory = speechRecognitionPipelineFactory; @@ -86,6 +88,8 @@ public sealed class MeetingRecordingCoordinator this.inactivityClock = inactivityClock ?? new SystemMeetingInactivityClock(); this.offlineTranscriptionBacklog = offlineTranscriptionBacklog ?? NoopOfflineTranscriptionBacklog.Instance; this.options = options.Value; + this.speakerSampleCollectionPolicy = speakerSampleCollectionPolicy + ?? SpeakerSampleCollectionPolicy.ExistingBackend; this.logger = logger; } @@ -290,7 +294,9 @@ public sealed class MeetingRecordingCoordinator startedAt, launchProfile.Name, runOptions.SpeakerIdentification.LiveSampleBufferDuration, - runOptions.SpeakerIdentification.MaxSnippetsPerSpeaker, + speakerSampleCollectionPolicy.ResolveRetainedSampleLimit( + runOptions.SpeakerIdentification.MaxSnippetsPerSpeaker), + speakerSampleCollectionPolicy.RequireNonOverlappingSamples, logger); run.Task = Task.Run(() => RecordAsync(run), CancellationToken.None); if (ShouldRunInactivitySafeguard(runOptions.Recording.InactivitySafeguard)) @@ -1169,11 +1175,11 @@ public sealed class MeetingRecordingCoordinator try { var meetingNote = await meetingNoteStore.ReadAsync(run.MeetingNotePath, cancellationToken); - var checkpoint = run.CreateLiveIdentificationCheckpoint(meetingNote); + var checkpoint = run.CreateLiveIdentificationCheckpoint(meetingNote, samples); if (checkpoint is null) { logger.LogInformation( - "Skipping live speaker identity matching because no new unmapped sample speakers or attendee changes were found"); + "Skipping live speaker identity matching because no new eligible speaker evidence or attendee changes were found"); return; } @@ -2079,6 +2085,7 @@ public sealed class MeetingRecordingCoordinator string launchProfileName, TimeSpan liveSampleBufferDuration, int maxSpeakerSamples, + bool requireNonOverlappingSpeakerSamples, ILogger logger) { CaptureCancellationSource = captureCancellation; @@ -2099,8 +2106,9 @@ public sealed class MeetingRecordingCoordinator liveSampleBufferDuration, maxSpeakerSamples, options.SpeakerIdentification.MinimumSampleSpeechDuration, - options.SpeakerIdentification.MaximumSampleSegmentGap, - logger); + options.SpeakerIdentification.MaximumSampleDuration, + logger, + requireNonOverlappingSpeakerSamples); } public CancellationTokenSource CaptureCancellationSource { get; } @@ -2508,9 +2516,11 @@ public sealed class MeetingRecordingCoordinator return speakerSampleCollector.Snapshot(); } - public LiveIdentificationCheckpoint? CreateLiveIdentificationCheckpoint(MeetingNote meetingNote) + public LiveIdentificationCheckpoint? CreateLiveIdentificationCheckpoint( + MeetingNote meetingNote, + IReadOnlyList samples) { - var speakers = GetUnmappedSampleSpeakers(); + var speakers = GetUnmappedSampleSpeakers(samples); if (speakers.Count == 0) { return null; @@ -2518,7 +2528,11 @@ public sealed class MeetingRecordingCoordinator var checkpoint = new LiveIdentificationCheckpoint( speakers, - BuildAttendeeSignature(meetingNote)); + BuildAttendeeSignature(meetingNote), + Options.SpeakerIdentification.Resemblyzer.Enabled + ? speakers.Select(speaker => samples.Count(sample => + string.Equals(sample.Speaker, speaker, StringComparison.OrdinalIgnoreCase))).ToArray() + : []); lock (liveIdentificationGate) { return lastLiveIdentificationCheckpoint?.Matches(checkpoint) == true @@ -2576,9 +2590,9 @@ public sealed class MeetingRecordingCoordinator StringComparer.OrdinalIgnoreCase); } - private IReadOnlyList GetUnmappedSampleSpeakers() + private IReadOnlyList GetUnmappedSampleSpeakers(IReadOnlyList samples) { - return GetSpeakerSamplesSnapshot() + return samples .Select(sample => sample.Speaker) .Where(speaker => !string.IsNullOrWhiteSpace(speaker)) .Where(speaker => !string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase)) @@ -2634,11 +2648,13 @@ public sealed class MeetingRecordingCoordinator public sealed record LiveIdentificationCheckpoint( IReadOnlyList Speakers, - string AttendeeSignature) + string AttendeeSignature, + IReadOnlyList SampleCounts) { public bool Matches(LiveIdentificationCheckpoint other) { return string.Equals(AttendeeSignature, other.AttendeeSignature, StringComparison.Ordinal) && + SampleCounts.SequenceEqual(other.SampleCounts) && Speakers.Count == other.Speakers.Count && Speakers.SequenceEqual(other.Speakers, StringComparer.OrdinalIgnoreCase); } diff --git a/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs b/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs index dccb9a3..6f1226c 100644 --- a/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs +++ b/MeetingAssistant/Recording/SpeakerAudioSampleCollector.cs @@ -9,37 +9,31 @@ internal sealed class SpeakerAudioSampleCollector private readonly object gate = new(); private readonly RollingAudioBuffer audioBuffer; private readonly Dictionary> samplesBySpeaker = new(StringComparer.OrdinalIgnoreCase); + private readonly Dictionary lastAcceptedEndBySpeaker = new(StringComparer.OrdinalIgnoreCase); private readonly int maxSamplesPerSpeaker; - private readonly TimeSpan minimumUninterruptedSpeechDuration; - private readonly TimeSpan maximumSegmentGap; + private readonly TimeSpan minimumSampleSpeechDuration; + private readonly TimeSpan maximumSampleDuration; + private readonly bool requireNonOverlappingSamples; private readonly ILogger? logger; - private PendingSpeakerSpan? pendingSpan; - - public SpeakerAudioSampleCollector(TimeSpan bufferDuration, int maxSamplesPerSpeaker) - : this( - bufferDuration, - maxSamplesPerSpeaker, - TimeSpan.FromSeconds(30), - TimeSpan.FromSeconds(1), - logger: null) - { - } + private SpeakerSampleSpan? pendingSpan; public SpeakerAudioSampleCollector( TimeSpan bufferDuration, int maxSamplesPerSpeaker, - TimeSpan minimumUninterruptedSpeechDuration, - TimeSpan maximumSegmentGap, - ILogger? logger = null) + TimeSpan minimumSampleSpeechDuration, + TimeSpan maximumSampleDuration, + ILogger? logger = null, + bool requireNonOverlappingSamples = false) { + SpeakerSampleDurationConfiguration.ValidateOrThrow( + minimumSampleSpeechDuration, + maximumSampleDuration, + "Speaker sample collector configuration is invalid"); audioBuffer = new RollingAudioBuffer(bufferDuration); this.maxSamplesPerSpeaker = Math.Max(1, maxSamplesPerSpeaker); - this.minimumUninterruptedSpeechDuration = minimumUninterruptedSpeechDuration > TimeSpan.Zero - ? minimumUninterruptedSpeechDuration - : TimeSpan.Zero; - this.maximumSegmentGap = maximumSegmentGap >= TimeSpan.Zero - ? maximumSegmentGap - : TimeSpan.Zero; + this.minimumSampleSpeechDuration = minimumSampleSpeechDuration; + this.maximumSampleDuration = maximumSampleDuration; + this.requireNonOverlappingSamples = requireNonOverlappingSamples; this.logger = logger; } @@ -53,6 +47,7 @@ internal sealed class SpeakerAudioSampleCollector lock (gate) { samplesBySpeaker.Clear(); + lastAcceptedEndBySpeaker.Clear(); pendingSpan = null; audioBuffer.Reset(); } @@ -68,34 +63,54 @@ internal sealed class SpeakerAudioSampleCollector return null; } - TranscriptionSegment sampleSegment; - PendingSpanReset? reset; + SpeakerSampleSpan? sampleSpan; + SpeakerSampleSpan? previousSpan; lock (gate) { - (sampleSegment, reset) = ExtendPendingSpan(segment); + if (requireNonOverlappingSamples && + lastAcceptedEndBySpeaker.TryGetValue(segment.Speaker, out var lastAcceptedEnd) && + segment.Start < lastAcceptedEnd) + { + logger?.LogInformation( + "Discarding speaker identity sample for {Speaker} because it overlaps an accepted sample ending at {AcceptedSampleEnd}", + segment.Speaker, + lastAcceptedEnd); + return null; + } + + (sampleSpan, previousSpan) = ExtendPendingSpan(segment); } - if (reset is not null) + if (sampleSpan is null) { logger?.LogInformation( - "Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}, gap {Gap}, maximum gap {MaximumGap}", - reset.PreviousSpeaker, - segment.Speaker, - reset.PreviousEnd, - segment.Start, - reset.Gap, - maximumSegmentGap); + "Discarding speaker identity sample for {Speaker} because the segment duration is not positive", + segment.Speaker); + return null; } - var score = Score(sampleSegment, minimumUninterruptedSpeechDuration); + var sampleSegment = sampleSpan.ToSegment(); + + if (previousSpan is not null) + { + logger?.LogInformation( + "Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}", + previousSpan.Speaker, + segment.Speaker, + previousSpan.End, + segment.Start); + } + + var score = Score(sampleSegment, sampleSpan.SpeechDuration, minimumSampleSpeechDuration); if (!score.Accepted) { logger?.LogInformation( - "Discarding speaker identity sample for {Speaker} because {Reason}: duration {Duration}, minimum duration {MinimumDuration}, word count {WordCount}", + "Discarding speaker identity sample for {Speaker} because {Reason}: speaker audio {SpeakerAudioDuration}, clip duration {ClipDuration}, minimum duration {MinimumDuration}, word count {WordCount}", sampleSegment.Speaker, score.Reason, + sampleSpan.SpeechDuration, sampleSegment.End - sampleSegment.Start, - minimumUninterruptedSpeechDuration, + minimumSampleSpeechDuration, score.WordCount); return null; } @@ -115,6 +130,13 @@ internal sealed class SpeakerAudioSampleCollector var sample = new SpeakerAudioSample(sampleSegment.Speaker, sampleSegment, wavBytes, score.Value); lock (gate) { + if (requireNonOverlappingSamples && + ReferenceEquals(pendingSpan, sampleSpan)) + { + pendingSpan = null; + lastAcceptedEndBySpeaker[sampleSegment.Speaker] = sampleSegment.End; + } + if (!samplesBySpeaker.TryGetValue(sampleSegment.Speaker, out var samples)) { samples = []; @@ -163,35 +185,26 @@ internal sealed class SpeakerAudioSampleCollector !string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase); } - private (TranscriptionSegment Segment, PendingSpanReset? Reset) ExtendPendingSpan(TranscriptionSegment segment) + private (SpeakerSampleSpan? Span, SpeakerSampleSpan? PreviousSpan) ExtendPendingSpan(TranscriptionSegment segment) { - if (pendingSpan is null || - !SpeakerSampleSpanSelector.CanExtend(pendingSpan.Speaker, pendingSpan.End, segment, maximumSegmentGap)) + if (pendingSpan is not null && pendingSpan.TryExtend(segment, out var extended)) { - var reset = pendingSpan is null - ? null - : new PendingSpanReset( - pendingSpan.Speaker, - pendingSpan.End, - segment.Start - pendingSpan.End); - pendingSpan = new PendingSpeakerSpan( - segment.Speaker, - segment.Start, - segment.End, - [segment.Text]); - return (pendingSpan.ToSegment(), reset); + pendingSpan = extended; + return (pendingSpan, null); } - pendingSpan = pendingSpan.Extend(segment); - return (pendingSpan.ToSegment(), null); + var previousSpan = pendingSpan; + pendingSpan = SpeakerSampleSpan.Create(segment, maximumSampleDuration); + return (pendingSpan, previousSpan); } private static SampleScore Score( TranscriptionSegment segment, - TimeSpan minimumUninterruptedSpeechDuration) + TimeSpan speechDuration, + TimeSpan minimumSampleSpeechDuration) { - var durationSeconds = (segment.End - segment.Start).TotalSeconds; - if (durationSeconds < minimumUninterruptedSpeechDuration.TotalSeconds) + var durationSeconds = speechDuration.TotalSeconds; + if (durationSeconds < minimumSampleSpeechDuration.TotalSeconds) { return new SampleScore(false, 0, "speech duration is below the configured minimum", WordCount(segment.Text)); } @@ -202,7 +215,7 @@ internal sealed class SpeakerAudioSampleCollector return new SampleScore(false, 0, "word count is below the minimum useful sample length", words); } - var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumUninterruptedSpeechDuration.TotalSeconds), 2); + var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumSampleSpeechDuration.TotalSeconds), 2); var wordScore = Math.Min(words / 60.0, 1); var sentenceBonus = segment.Text.TrimEnd().EndsWith('.') || segment.Text.TrimEnd().EndsWith('?') || @@ -221,30 +234,4 @@ internal sealed class SpeakerAudioSampleCollector private sealed record SampleScore(bool Accepted, double Value, string? Reason, int WordCount); - private sealed record PendingSpanReset(string PreviousSpeaker, TimeSpan PreviousEnd, TimeSpan Gap); - - private sealed record PendingSpeakerSpan( - string Speaker, - TimeSpan Start, - TimeSpan End, - IReadOnlyList TextParts) - { - public PendingSpeakerSpan Extend(TranscriptionSegment segment) - { - return this with - { - End = segment.End > End ? segment.End : End, - TextParts = TextParts.Append(segment.Text).ToList() - }; - } - - public TranscriptionSegment ToSegment() - { - return new TranscriptionSegment( - Start, - End, - Speaker, - string.Join(' ', TextParts.Where(part => !string.IsNullOrWhiteSpace(part)))); - } - } } diff --git a/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentificationService.cs b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentificationService.cs new file mode 100644 index 0000000..cafebc0 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentificationService.cs @@ -0,0 +1,631 @@ +using MeetingAssistant.MeetingNotes; +using MeetingAssistant.Transcription; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed class ResemblyzerSpeakerIdentificationService : ISpeakerIdentificationService +{ + private readonly IDbContextFactory dbContextFactory; + private readonly ISpeakerSnippetExtractor snippetExtractor; + private readonly IResemblyzerVoiceEncoder encoder; + private readonly ResemblyzerVoiceClusterMatcher clusterMatcher; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; + private readonly SpeakerIdentificationOptions options; + private readonly ResemblyzerSpeakerRecognitionOptions resemblyzerOptions; + private readonly ILogger logger; + + public ResemblyzerSpeakerIdentificationService( + IDbContextFactory dbContextFactory, + ISpeakerSnippetExtractor snippetExtractor, + IResemblyzerVoiceEncoder encoder, + ResemblyzerVoiceClusterMatcher clusterMatcher, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, + IOptions options, + ILogger logger) + { + this.dbContextFactory = dbContextFactory; + this.snippetExtractor = snippetExtractor; + this.encoder = encoder; + this.clusterMatcher = clusterMatcher; + this.outlierPruner = outlierPruner; + this.options = options.Value.SpeakerIdentification; + resemblyzerOptions = this.options.Resemblyzer; + this.logger = logger; + } + + public Task IdentifyKnownSpeakersAsync( + SpeakerIdentificationRequest request, + CancellationToken cancellationToken) + { + return ProcessTranscriptAsync(request, final: false, allowAudioFallback: false, cancellationToken); + } + + public Task IdentifyFinishedSpeakersAsync( + SpeakerIdentificationRequest request, + CancellationToken cancellationToken) + { + return ProcessTranscriptAsync(request, final: false, allowAudioFallback: true, cancellationToken); + } + + public Task ProcessFinishedTranscriptAsync( + SpeakerIdentificationRequest request, + CancellationToken cancellationToken) + { + return ProcessTranscriptAsync(request, final: true, allowAudioFallback: true, cancellationToken); + } + + public async Task ApplySpeakerOverrideAsync( + SpeakerIdentificationRequest request, + string sourceSpeaker, + string targetSpeaker, + CancellationToken cancellationToken) + { + if (!options.Enabled || + string.IsNullOrWhiteSpace(sourceSpeaker) || + string.IsNullOrWhiteSpace(targetSpeaker) || + string.Equals(sourceSpeaker, targetSpeaker, StringComparison.OrdinalIgnoreCase)) + { + return; + } + + var sourceLabel = sourceSpeaker.Trim(); + var targetName = targetSpeaker.Trim(); + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var identities = await LoadIdentities(context).ToListAsync(cancellationToken); + var target = identities + .Where(identity => SpeakerIdentityNaming.GetAcceptedNames(identity).Contains(targetName)) + .OrderBy(identity => string.Equals(identity.CanonicalName, targetName, StringComparison.OrdinalIgnoreCase) ? 0 : 1) + .ThenBy(identity => identity.Id) + .FirstOrDefault(); + var reference = CreateReference(request.MeetingNote, DateTimeOffset.UtcNow); + var sourceCandidate = identities + .Where(identity => string.IsNullOrWhiteSpace(identity.CanonicalName)) + .Where(identity => identity.References.Any(existing => SpeakerIdentityReferences.IsSame(existing, reference))) + .Where(identity => SpeakerIdentityNaming.GetAcceptedNames(identity).Contains(targetName)) + .OrderBy(identity => identity.Id) + .FirstOrDefault(); + var vectors = await ResolveAvailableVectorsAsync(request, sourceLabel, cancellationToken); + if (target is null && sourceCandidate is null && vectors.Count == 0) + { + logger.LogWarning( + "Skipping Resemblyzer speaker override from {SourceSpeaker} to {TargetSpeaker} because no source evidence was available", + sourceLabel, + targetName); + return; + } + + var now = DateTimeOffset.UtcNow; + if (target is null) + { + target = sourceCandidate ?? new SpeakerIdentity + { + CreatedAt = now + }; + if (target.Id == 0) + { + context.SpeakerIdentities.Add(target); + } + } + else if (sourceCandidate is not null && sourceCandidate.Id != target.Id) + { + SpeakerIdentityMerger.MergeIntoAndPrune( + target, + sourceCandidate, + options.MaxSnippetsPerSpeaker, + resemblyzerOptions.MaxVectorsPerIdentity, + outlierPruner); + context.SpeakerIdentities.Remove(sourceCandidate); + } + + target.CanonicalName = targetName; + SpeakerIdentityNaming.SetCandidates(target, [targetName]); + SpeakerIdentityReferences.AddIfMissing(target, reference, now); + outlierPruner.AddAndPrune( + target, + vectors, + now); + target.UpdatedAt = now; + await context.SaveChangesAsync(cancellationToken); + await SpeakerIdentityTranscriptAudit.AppendIdentifiedAsync( + target.References, + sourceLabel, + targetName, + cancellationToken); + } + + public async Task DeleteSpeakerIdentityAsync( + string identity, + CancellationToken cancellationToken) + { + if (!options.Enabled || string.IsNullOrWhiteSpace(identity)) + { + return; + } + + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var identities = await LoadIdentities(context).ToListAsync(cancellationToken); + var target = identities.FirstOrDefault(candidate => + SpeakerIdentityNaming.GetAcceptedNames(candidate).Contains(identity.Trim())); + if (target is null) + { + return; + } + + context.SpeakerIdentities.Remove(target); + await context.SaveChangesAsync(cancellationToken); + } + + private async Task ProcessTranscriptAsync( + SpeakerIdentificationRequest request, + bool final, + bool allowAudioFallback, + CancellationToken cancellationToken) + { + if (!options.Enabled || request.Segments.Count == 0) + { + return EmptyResult(request.Segments); + } + + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var attendees = SpeakerIdentityNaming.NormalizeAttendees(request.MeetingNote.Frontmatter.Attendees); + var knownMappings = request.KnownSpeakerMappings ?? + new Dictionary(StringComparer.OrdinalIgnoreCase); + var knownLabels = knownMappings.Keys.ToHashSet(StringComparer.OrdinalIgnoreCase); + if (final && knownMappings.Count > 0) + { + await PersistMappedSpeakerEvidenceAsync( + context, + request, + knownMappings, + cancellationToken); + } + + var identifiedNames = knownMappings.Values + .Where(name => !string.IsNullOrWhiteSpace(name)) + .Select(name => name.Trim()) + .ToHashSet(StringComparer.OrdinalIgnoreCase); + foreach (var segmentSpeaker in request.Segments + .Select(segment => segment.Speaker) + .Where(speaker => !string.IsNullOrWhiteSpace(speaker) && !SpeakerIdentityNaming.IsDiarizedSpeakerLabel(speaker))) + { + identifiedNames.Add(segmentSpeaker.Trim()); + } + + var mappings = new Dictionary(StringComparer.OrdinalIgnoreCase); + var attendeeMatches = new List(); + var pendingAudits = new List(); + var matchedAcceptedNames = identifiedNames.ToHashSet(StringComparer.OrdinalIgnoreCase); + var unmatchedSpeakers = new List<(string Speaker, IReadOnlyList Vectors)>(); + foreach (var speaker in request.Segments + .Select(segment => segment.Speaker) + .Where(speaker => !string.IsNullOrWhiteSpace(speaker)) + .Distinct(StringComparer.OrdinalIgnoreCase)) + { + if (knownLabels.Contains(speaker) || identifiedNames.Contains(speaker) || !SpeakerIdentityNaming.IsDiarizedSpeakerLabel(speaker)) + { + continue; + } + + var vectors = await ResolveAutomaticVectorsAsync( + request, + speaker, + allowAudioFallback, + cancellationToken); + if (vectors.Count < resemblyzerOptions.RequiredVectorsPerSpeaker) + { + logger.LogInformation( + "Resemblyzer matching waits for more vectors for {Speaker}: {VectorCount}/{RequiredVectorCount}", + speaker, + vectors.Count, + resemblyzerOptions.RequiredVectorsPerSpeaker); + continue; + } + + var (identity, decision) = await FindMatchAsync( + context, + attendees, + identifiedNames, + vectors, + cancellationToken); + if (identity is null) + { + if (final && decision.Cohesion >= resemblyzerOptions.MinimumClusterCohesion) + { + unmatchedSpeakers.Add((speaker, vectors)); + } + + continue; + } + + var now = DateTimeOffset.UtcNow; + var previousCanonicalName = identity.CanonicalName; + var previousReferenceCount = identity.References.Count; + outlierPruner.AddAndPrune( + identity, + vectors, + now); + SpeakerIdentityReferences.AddIfMissing( + identity, + CreateReference(request.MeetingNote, now), + now); + if (identity.References.Count != previousReferenceCount) + { + identity.UpdatedAt = now; + } + + if (final) + { + UpdateMatchedIdentity(identity, attendees); + if (string.IsNullOrWhiteSpace(previousCanonicalName) && + !string.IsNullOrWhiteSpace(identity.CanonicalName)) + { + pendingAudits.Add(new PendingIdentificationAudit( + identity, + speaker, + identity.CanonicalName)); + } + + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) + { + matchedAcceptedNames.Add(acceptedName); + } + } + + var displayName = identity.GetDisplayName(); + if (!string.IsNullOrWhiteSpace(displayName)) + { + mappings[speaker] = displayName; + identifiedNames.Add(displayName); + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) + { + identifiedNames.Add(acceptedName); + } + attendeeMatches.Add(new SpeakerIdentityAttendeeMatch( + displayName, + SpeakerIdentityNaming.GetAcceptedNames(identity).ToList())); + } + } + + if (final) + { + LearnUnmatchedSpeakers( + context, + request.MeetingNote, + attendees, + matchedAcceptedNames, + unmatchedSpeakers, + pendingAudits); + } + + await context.SaveChangesAsync(cancellationToken); + await AppendAuditsAsync(pendingAudits, cancellationToken); + var relabeled = request.Segments + .Select(segment => mappings.TryGetValue(segment.Speaker, out var name) + ? segment with { Speaker = name } + : segment) + .ToList(); + return new SpeakerIdentificationResult(relabeled, mappings, attendeeMatches); + } + + private async Task<(SpeakerIdentity? Identity, ResemblyzerVoiceClusterMatchResult Decision)> FindMatchAsync( + SpeakerIdentityDbContext context, + IReadOnlyList attendees, + IReadOnlySet identifiedNames, + IReadOnlyList queryVectors, + CancellationToken cancellationToken) + { + var activeCutoff = DateTimeOffset.UtcNow - options.MatchIdentityActiveAge; + var identities = await LoadIdentities(context) + .OrderByDescending(identity => identity.References.Count) + .ThenBy(identity => identity.Id) + .ToListAsync(cancellationToken); + var candidates = identities + .Select(identity => new + { + Identity = identity, + IsAttendee = SpeakerIdentityNaming.MatchesAttendees(identity, attendees), + IsActive = identity.UpdatedAt >= activeCutoff + }) + .Where(candidate => candidate.IsAttendee || candidate.IsActive) + .Where(candidate => !SpeakerIdentityNaming.MatchesAnyAcceptedName(candidate.Identity, identifiedNames)) + .Where(candidate => candidate.Identity.VoiceVectors.Any(vector => + string.Equals(vector.ModelId, resemblyzerOptions.ModelId, StringComparison.Ordinal))) + .OrderByDescending(candidate => candidate.IsAttendee) + .ThenByDescending(candidate => candidate.Identity.ReferenceCount) + .ThenBy(candidate => candidate.Identity.Id) + .Take(Math.Max(1, options.MaxMatchCandidates)) + .Select(candidate => new ResemblyzerVoiceVectorCandidate( + candidate.Identity.Id, + SpeakerVoiceVectors.DecodeCompatible( + candidate.Identity, + resemblyzerOptions.ModelId, + logger))) + .Where(candidate => candidate.Vectors.Count > 0) + .ToList(); + var match = clusterMatcher.Match(queryVectors, candidates); + return ( + match.IdentityId is { } identityId + ? identities.Single(identity => identity.Id == identityId) + : null, + match); + } + + private void LearnUnmatchedSpeakers( + SpeakerIdentityDbContext context, + MeetingNote meetingNote, + IReadOnlyList attendees, + IReadOnlySet matchedAcceptedNames, + IReadOnlyList<(string Speaker, IReadOnlyList Vectors)> unmatchedSpeakers, + ICollection pendingAudits) + { + var candidates = attendees + .Except(matchedAcceptedNames, StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .ToList(); + if (candidates.Count == 0) + { + return; + } + + foreach (var (speaker, vectors) in unmatchedSpeakers) + { + var now = DateTimeOffset.UtcNow; + var identity = new SpeakerIdentity + { + CanonicalName = candidates.Count == 1 ? candidates[0] : null, + CreatedAt = now, + UpdatedAt = now, + CandidateNames = candidates + .Select(name => new SpeakerCandidateName { Name = name }) + .ToList(), + References = [CreateReference(meetingNote, now)] + }; + outlierPruner.AddAndPrune( + identity, + vectors, + now); + context.SpeakerIdentities.Add(identity); + logger.LogInformation( + "Created Resemblyzer identity candidate for {Speaker} with {VectorCount} vector(s) and candidates {Candidates}", + speaker, + identity.VoiceVectors.Count, + string.Join(", ", candidates)); + if (!string.IsNullOrWhiteSpace(identity.CanonicalName)) + { + pendingAudits.Add(new PendingIdentificationAudit( + identity, + speaker, + identity.CanonicalName)); + } + } + } + + private async Task AppendAuditsAsync( + IEnumerable pendingAudits, + CancellationToken cancellationToken) + { + foreach (var audit in pendingAudits) + { + try + { + await SpeakerIdentityTranscriptAudit.AppendIdentifiedAsync( + audit.Identity.References, + audit.Speaker, + audit.Name, + cancellationToken); + } + catch (Exception exception) when (exception is not OperationCanceledException) + { + logger.LogError( + exception, + "Resemblyzer identity {IdentityId} was saved, but its transcript identification audit could not be written", + audit.Identity.Id); + } + } + } + + private async Task> ResolveAutomaticVectorsAsync( + SpeakerIdentificationRequest request, + string speaker, + bool allowAudioFallback, + CancellationToken cancellationToken) + { + var requiredCount = resemblyzerOptions.RequiredVectorsPerSpeaker; + var samples = await ResolveWavSamplesAsync( + request, + speaker, + resemblyzerOptions.MaxVectorsPerIdentity, + allowAudioFallback, + cancellationToken); + if (samples.Count < requiredCount) + { + return []; + } + + return await encoder.EncodeAsync(samples, cancellationToken); + } + + private async Task PersistMappedSpeakerEvidenceAsync( + SpeakerIdentityDbContext context, + SpeakerIdentificationRequest request, + IReadOnlyDictionary knownMappings, + CancellationToken cancellationToken) + { + var identities = await LoadIdentities(context).ToListAsync(cancellationToken); + foreach (var (speaker, mappedName) in knownMappings) + { + var identity = identities + .Where(candidate => SpeakerIdentityNaming.GetAcceptedNames(candidate).Contains(mappedName)) + .OrderBy(candidate => + string.Equals(candidate.CanonicalName, mappedName, StringComparison.OrdinalIgnoreCase) ? 0 : 1) + .ThenBy(candidate => candidate.Id) + .FirstOrDefault(); + if (identity is null) + { + logger.LogWarning( + "Could not retain final Resemblyzer evidence for mapped speaker {Speaker}: identity {MappedName} was not found", + speaker, + mappedName); + continue; + } + + var vectors = await ResolveAvailableVectorsAsync(request, speaker, cancellationToken); + if (vectors.Count == 0) + { + continue; + } + + var now = DateTimeOffset.UtcNow; + var previousReferenceCount = identity.References.Count; + var vectorUpdate = outlierPruner.AddAndPrune( + identity, + vectors, + now); + SpeakerIdentityReferences.AddIfMissing( + identity, + CreateReference(request.MeetingNote, now), + now); + if (identity.References.Count != previousReferenceCount) + { + identity.UpdatedAt = now; + } + + logger.LogInformation( + "Retained {AddedVectorCount} new final Resemblyzer vector(s) for mapped speaker {Speaker} as identity {IdentityId}", + vectorUpdate.AddedCount, + speaker, + identity.Id); + } + } + + private async Task> ResolveAvailableVectorsAsync( + SpeakerIdentificationRequest request, + string speaker, + CancellationToken cancellationToken) + { + var samples = await ResolveWavSamplesAsync( + request, + speaker, + resemblyzerOptions.MaxVectorsPerIdentity, + allowAudioFallback: true, + cancellationToken); + if (samples.Count == 0) + { + return []; + } + + return await encoder.EncodeAsync(samples, cancellationToken); + } + + private async Task> ResolveWavSamplesAsync( + SpeakerIdentificationRequest request, + string speaker, + int maxSamples, + bool allowAudioFallback, + CancellationToken cancellationToken) + { + var suppliedSamples = request.Samples? + .Where(sample => string.Equals(sample.Speaker, speaker, StringComparison.OrdinalIgnoreCase)) + .Where(sample => sample.WavBytes.Length > 0) + .OrderByDescending(sample => sample.Score) + .Take(maxSamples) + .ToList() + ?? []; + var wavSamples = suppliedSamples + .Select(sample => sample.WavBytes) + .ToList(); + if (!allowAudioFallback || wavSamples.Count >= maxSamples) + { + return wavSamples; + } + + var spans = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpans( + request.Segments, + speaker, + options.MinimumSampleSpeechDuration, + options.MaximumSampleDuration, + request.Segments.Count); + foreach (var span in spans.Where(span => !OverlapsSuppliedSample(span, suppliedSamples))) + { + var wavBytes = await snippetExtractor.ExtractSnippetAsync( + request.AudioPath, + span, + cancellationToken); + if (wavBytes.Length > 0) + { + wavSamples.Add(wavBytes); + } + + if (wavSamples.Count >= maxSamples) + { + break; + } + } + + logger.LogInformation( + "Resolved {SampleCount}/{RequestedSampleCount} Resemblyzer WAV samples for {Speaker}: {SuppliedSampleCount} supplied, {ExtractedSampleCount} extracted from completed audio", + wavSamples.Count, + maxSamples, + speaker, + suppliedSamples.Count, + wavSamples.Count - suppliedSamples.Count); + return wavSamples; + } + + private static bool OverlapsSuppliedSample( + IReadOnlyList span, + IReadOnlyList suppliedSamples) + { + if (span.Count == 0) + { + return false; + } + + var start = span[0].Start; + var end = span[^1].End; + return suppliedSamples.Any(sample => sample.Segment.Start < end && sample.Segment.End > start); + } + + private static IQueryable LoadIdentities(SpeakerIdentityDbContext context) + { + return context.SpeakerIdentities + .AsSplitQuery() + .Include(identity => identity.Aliases) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.References); + } + + private void UpdateMatchedIdentity(SpeakerIdentity identity, IReadOnlyList attendees) + { + identity.UpdatedAt = DateTimeOffset.UtcNow; + if (!string.IsNullOrWhiteSpace(identity.CanonicalName) || attendees.Count == 0) + { + return; + } + + SpeakerIdentityNaming.UpdateCandidateNames(identity, attendees); + } + + private static SpeakerIdentityReference CreateReference(MeetingNote meetingNote, DateTimeOffset timestamp) + { + return SpeakerIdentityReferences.Create(meetingNote.Path, meetingNote.Frontmatter.Transcript, timestamp); + } + + private static SpeakerIdentificationResult EmptyResult(IReadOnlyList segments) + { + return new SpeakerIdentificationResult(segments, new Dictionary()); + } + + private sealed record PendingIdentificationAudit( + SpeakerIdentity Identity, + string Speaker, + string Name); + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentityMergeService.cs b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentityMergeService.cs new file mode 100644 index 0000000..b2c5d40 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerSpeakerIdentityMergeService.cs @@ -0,0 +1,172 @@ +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed class ResemblyzerSpeakerIdentityMergeService : ISpeakerIdentityMergeService +{ + private readonly IDbContextFactory dbContextFactory; + private readonly ResemblyzerVoiceClusterMatcher matcher; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; + private readonly SpeakerIdentificationOptions options; + private readonly ResemblyzerSpeakerRecognitionOptions resemblyzerOptions; + private readonly ILogger logger; + + public ResemblyzerSpeakerIdentityMergeService( + IDbContextFactory dbContextFactory, + ResemblyzerVoiceClusterMatcher matcher, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, + IOptions options, + ILogger logger) + { + this.dbContextFactory = dbContextFactory; + this.matcher = matcher; + this.outlierPruner = outlierPruner; + this.options = options.Value.SpeakerIdentification; + resemblyzerOptions = this.options.Resemblyzer; + this.logger = logger; + } + + public async Task MergeRecentIdentitiesAsync( + TimeSpan? recentIdentityAge, + CancellationToken cancellationToken) + { + await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); + await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); + var identities = await context.SpeakerIdentities + .AsSplitQuery() + .Include(identity => identity.Aliases) + .Include(identity => identity.CandidateNames) + .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) + .Include(identity => identity.References) + .OrderByDescending(identity => identity.References.Count) + .ThenBy(identity => identity.Id) + .ToListAsync(cancellationToken); + var cutoff = DateTimeOffset.UtcNow - (recentIdentityAge ?? options.MergeRecentIdentityAge); + var recentIds = identities + .Where(identity => identity.CreatedAt >= cutoff) + .Select(identity => identity.Id) + .ToList(); + var required = resemblyzerOptions.RequiredVectorsPerSpeaker; + var attempts = 0; + var mergedPairs = 0; + var pendingAudits = new List(); + + foreach (var sourceId in recentIds) + { + var source = identities.SingleOrDefault(identity => identity.Id == sourceId); + if (source is null) + { + continue; + } + + var sourceVectors = SpeakerVoiceVectors.DecodeCompatible( + source, + resemblyzerOptions.ModelId, + logger) + .Take(required * 2) + .ToList(); + if (sourceVectors.Count < required * 2) + { + logger.LogInformation( + "Skipping Resemblyzer merge source identity {SourceIdentityId}: {VectorCount}/{RequiredVectorCount} compatible vectors", + source.Id, + sourceVectors.Count, + required * 2); + continue; + } + + var candidates = identities + .Where(identity => identity.Id != source.Id) + .Where(identity => identity.VoiceVectors.Any(vector => + string.Equals(vector.ModelId, resemblyzerOptions.ModelId, StringComparison.Ordinal))) + .Take(Math.Max(1, options.MaxMatchCandidates)) + .Select(identity => new ResemblyzerVoiceVectorCandidate( + identity.Id, + SpeakerVoiceVectors.DecodeCompatible( + identity, + resemblyzerOptions.ModelId, + logger))) + .Where(candidate => candidate.Vectors.Count > 0) + .ToList(); + if (candidates.Count == 0) + { + continue; + } + + attempts++; + var first = matcher.Match(sourceVectors.Take(required).ToList(), candidates); + if (!first.Accepted || first.IdentityId is not { } targetId) + { + continue; + } + + attempts++; + var second = matcher.Match(sourceVectors.Skip(required).Take(required).ToList(), candidates); + if (!second.Accepted || second.IdentityId != targetId) + { + logger.LogInformation( + "Rejected Resemblyzer merge for source identity {SourceIdentityId}: disjoint clusters selected {FirstIdentityId} and {SecondIdentityId}", + source.Id, + first.IdentityId, + second.IdentityId); + continue; + } + + var target = identities.SingleOrDefault(identity => identity.Id == targetId); + if (target is null) + { + continue; + } + + var targetName = target.GetDisplayName() ?? $"identity-{target.Id}"; + var sourceName = source.GetDisplayName() ?? $"identity-{source.Id}"; + SpeakerIdentityMerger.MergeIntoAndPrune( + target, + source, + options.MaxSnippetsPerSpeaker, + resemblyzerOptions.MaxVectorsPerIdentity, + outlierPruner); + pendingAudits.Add(new PendingMergeAudit( + target, + targetName, + sourceName)); + context.SpeakerIdentities.Remove(source); + identities.Remove(source); + mergedPairs++; + } + + await context.SaveChangesAsync(cancellationToken); + foreach (var audit in pendingAudits) + { + try + { + await SpeakerIdentityTranscriptAudit.AppendMergedAsync( + audit.Identity.References, + audit.TargetName, + audit.SourceName, + cancellationToken); + } + catch (Exception exception) when (exception is not OperationCanceledException) + { + logger.LogError( + exception, + "Resemblyzer identity merge was saved for target {IdentityId}, but its transcript audit could not be written", + audit.Identity.Id); + } + } + + return new SpeakerIdentityMergeResult( + recentIds.Count, + identities.Count, + attempts, + mergedPairs); + } + + private sealed record PendingMergeAudit( + SpeakerIdentity Identity, + string TargetName, + string SourceName); + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerVoiceClusterMatcher.cs b/MeetingAssistant/Speakers/ResemblyzerVoiceClusterMatcher.cs new file mode 100644 index 0000000..a5450c0 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerVoiceClusterMatcher.cs @@ -0,0 +1,204 @@ +namespace MeetingAssistant.Speakers; + +public sealed record ResemblyzerVoiceVectorCandidate( + int IdentityId, + IReadOnlyList Vectors); + +public sealed record ResemblyzerVoiceClusterMatchResult( + int? IdentityId, + string Reason, + double Cohesion, + double? BestSimilarity, + double? RunnerUpSimilarity) +{ + public bool Accepted => IdentityId.HasValue; +} + +public sealed class ResemblyzerVoiceClusterMatcher +{ + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + + public ResemblyzerVoiceClusterMatcher( + ResemblyzerSpeakerRecognitionOptions options, + ILogger logger) + { + this.options = options; + this.logger = logger; + } + + public ResemblyzerVoiceClusterMatchResult Match( + IReadOnlyList queryVectors, + IReadOnlyList candidates) + { + var requiredCount = options.RequiredVectorsPerSpeaker; + if (queryVectors.Count < requiredCount) + { + return Reject( + $"insufficient query vectors ({queryVectors.Count}/{requiredCount})", + cohesion: 0, + bestSimilarity: null, + runnerUpSimilarity: null); + } + + var normalizedQuery = queryVectors + .Take(requiredCount) + .Select(vector => SpeakerVoiceVectors.Normalize(vector, "Voice vector")) + .ToList(); + var cohesion = MeanPairwiseCosine(normalizedQuery); + if (cohesion < options.MinimumClusterCohesion) + { + return Reject( + $"query cohesion {cohesion:F4} is below {options.MinimumClusterCohesion:F4}", + cohesion, + bestSimilarity: null, + runnerUpSimilarity: null); + } + + var scored = candidates + .Select(candidate => ScoreCandidate(normalizedQuery, candidate)) + .Where(candidate => candidate is not null) + .Select(candidate => candidate!.Value) + .OrderByDescending(candidate => candidate.Similarity) + .ThenBy(candidate => candidate.IdentityId) + .ToList(); + if (scored.Count == 0) + { + return Reject("no compatible identity vectors were available", cohesion, null, null); + } + + var best = scored[0]; + var runnerUp = scored.Count > 1 ? scored[1].Similarity : (double?)null; + if (best.Similarity < options.MinimumIdentitySimilarity) + { + return Reject( + $"best similarity {best.Similarity:F4} is below {options.MinimumIdentitySimilarity:F4}", + cohesion, + best.Similarity, + runnerUp); + } + + if (runnerUp is { } runnerUpSimilarity && + best.Similarity - runnerUpSimilarity < options.MinimumSimilarityMargin) + { + return Reject( + $"similarity margin {best.Similarity - runnerUpSimilarity:F4} is below {options.MinimumSimilarityMargin:F4}", + cohesion, + best.Similarity, + runnerUpSimilarity); + } + + logger.LogInformation( + "Resemblyzer cluster accepted identity {IdentityId}: cohesion {Cohesion:F4}, similarity {Similarity:F4}, runner-up {RunnerUpSimilarity}", + best.IdentityId, + cohesion, + best.Similarity, + runnerUp); + return new ResemblyzerVoiceClusterMatchResult( + best.IdentityId, + "accepted", + cohesion, + best.Similarity, + runnerUp); + } + + private ResemblyzerVoiceClusterMatchResult Reject( + string reason, + double cohesion, + double? bestSimilarity, + double? runnerUpSimilarity) + { + logger.LogInformation( + "Resemblyzer cluster rejected: {Reason}; cohesion {Cohesion:F4}, best {BestSimilarity}, runner-up {RunnerUpSimilarity}", + reason, + cohesion, + bestSimilarity, + runnerUpSimilarity); + return new ResemblyzerVoiceClusterMatchResult( + null, + reason, + cohesion, + bestSimilarity, + runnerUpSimilarity); + } + + private (int IdentityId, double Similarity)? ScoreCandidate( + IReadOnlyList normalizedQuery, + ResemblyzerVoiceVectorCandidate candidate) + { + if (candidate.Vectors.Count == 0) + { + return null; + } + + try + { + var normalizedCandidate = candidate.Vectors + .Select(vector => SpeakerVoiceVectors.Normalize(vector, "Voice vector")) + .ToList(); + var centroid = SpeakerVoiceVectors.Normalize(Centroid(normalizedCandidate), "Voice-vector centroid"); + var similarities = normalizedQuery + .Select(vector => SpeakerVoiceVectors.Cosine(vector, centroid)) + .Order() + .ToList(); + return (candidate.IdentityId, Median(similarities)); + } + catch (InvalidDataException exception) + { + logger.LogWarning( + exception, + "Skipping invalid Resemblyzer evidence for identity {IdentityId}", + candidate.IdentityId); + return null; + } + } + + private static float[] Centroid(IReadOnlyList vectors) + { + var centroid = new float[ResemblyzerVectorContract.Dimensions]; + foreach (var vector in vectors) + { + for (var index = 0; index < centroid.Length; index++) + { + centroid[index] += vector[index]; + } + } + + for (var index = 0; index < centroid.Length; index++) + { + centroid[index] /= vectors.Count; + } + + return centroid; + } + + private static double MeanPairwiseCosine(IReadOnlyList vectors) + { + if (vectors.Count < 2) + { + return 1; + } + + double total = 0; + var pairs = 0; + for (var first = 0; first < vectors.Count - 1; first++) + { + for (var second = first + 1; second < vectors.Count; second++) + { + total += SpeakerVoiceVectors.Cosine(vectors[first], vectors[second]); + pairs++; + } + } + + return total / pairs; + } + + private static double Median(IReadOnlyList sortedValues) + { + var middle = sortedValues.Count / 2; + return sortedValues.Count % 2 == 0 + ? (sortedValues[middle - 1] + sortedValues[middle]) / 2 + : sortedValues[middle]; + } + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerVoiceVectorOutlierPruner.cs b/MeetingAssistant/Speakers/ResemblyzerVoiceVectorOutlierPruner.cs new file mode 100644 index 0000000..e8102db --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerVoiceVectorOutlierPruner.cs @@ -0,0 +1,208 @@ +namespace MeetingAssistant.Speakers; + +public sealed record ResemblyzerVoiceVectorUpdateResult( + int AddedCount, + int RemovedCount) +{ + public bool Changed => AddedCount > 0 || RemovedCount > 0; +} + +public sealed class ResemblyzerVoiceVectorOutlierPruner +{ + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + + public ResemblyzerVoiceVectorOutlierPruner( + ResemblyzerSpeakerRecognitionOptions options, + ILogger logger) + { + this.options = options; + this.logger = logger; + } + + public int Prune(SpeakerIdentity identity) + { + var compatible = new List(); + foreach (var entry in SpeakerVoiceVectors.DecodeCompatibleEntries( + identity, + options.ModelId, + logger)) + { + try + { + compatible.Add(new DecodedSpeakerVoiceVector( + entry.Stored, + SpeakerVoiceVectors.Normalize( + entry.Vector, + $"Stored voice vector {entry.Stored.Id}"))); + } + catch (InvalidDataException exception) + { + logger.LogWarning( + exception, + "Preserving invalid stored voice vector {VectorId} while pruning identity {IdentityId}", + entry.Stored.Id, + identity.Id); + } + } + var minimumVectors = options.OutlierPruningMinimumVectors; + if (compatible.Count < minimumVectors) + { + return 0; + } + + var clusters = FindDensityClusters(compatible.Select(item => item.Vector).ToList()); + var ranked = clusters + .GroupBy(cluster => cluster) + .Where(group => group.Key > 0) + .Select(group => new { Id = group.Key, Count = group.Count() }) + .OrderByDescending(group => group.Count) + .ThenBy(group => group.Id) + .ToList(); + if (ranked.Count == 0) + { + return Skip(identity, "no dense cluster was found"); + } + + if (ranked.Count > 1 && ranked[0].Count == ranked[1].Count) + { + return Skip(identity, "no uniquely largest dense cluster was found"); + } + + var dominant = ranked[0]; + var ratio = (double)dominant.Count / compatible.Count; + if (ratio < options.OutlierPruningMinimumClusterRatio) + { + return Skip( + identity, + $"largest dense cluster ratio {ratio:F4} is below {options.OutlierPruningMinimumClusterRatio:F4}"); + } + + var removed = compatible + .Where((_, index) => clusters[index] != dominant.Id) + .Select(item => item.Stored) + .ToList(); + foreach (var vector in removed) + { + identity.VoiceVectors.Remove(vector); + } + + if (removed.Count > 0) + { + identity.UpdatedAt = DateTimeOffset.UtcNow; + } + + logger.LogInformation( + "Pruned {RemovedCount} Resemblyzer voice-vector outliers from identity {IdentityId}; retained dominant cluster of {RetainedCount}/{CompatibleCount} vectors", + removed.Count, + identity.Id, + dominant.Count, + compatible.Count); + return removed.Count; + } + + public ResemblyzerVoiceVectorUpdateResult AddAndPrune( + SpeakerIdentity identity, + IReadOnlyList vectors, + DateTimeOffset createdAt) + { + var removedBeforeAdding = Prune(identity); + var added = SpeakerVoiceVectors.AddDistinct( + identity, + vectors, + options.ModelId, + options.MaxVectorsPerIdentity, + createdAt); + var removedAfterAdding = Prune(identity); + var result = new ResemblyzerVoiceVectorUpdateResult( + added, + removedBeforeAdding + removedAfterAdding); + if (result.Changed) + { + identity.UpdatedAt = createdAt; + } + + return result; + } + + private int[] FindDensityClusters(IReadOnlyList vectors) + { + var neighbors = Enumerable.Range(0, vectors.Count) + .Select(index => Enumerable.Range(0, vectors.Count) + .Where(candidate => SpeakerVoiceVectors.Cosine(vectors[index], vectors[candidate]) >= options.OutlierPruningNeighborSimilarity) + .ToList()) + .ToList(); + var minimumNeighbors = options.OutlierPruningMinimumNeighbors; + var labels = new int[vectors.Count]; + var clusterId = 0; + + for (var point = 0; point < vectors.Count; point++) + { + if (labels[point] != 0) + { + continue; + } + + if (neighbors[point].Count < minimumNeighbors) + { + labels[point] = -1; + continue; + } + + clusterId++; + ExpandCluster(point, clusterId, neighbors, labels, minimumNeighbors); + } + + return labels; + } + + private static void ExpandCluster( + int seed, + int clusterId, + IReadOnlyList> neighbors, + int[] labels, + int minimumNeighbors) + { + labels[seed] = clusterId; + var pending = new Queue(neighbors[seed]); + var queued = neighbors[seed].ToHashSet(); + while (pending.TryDequeue(out var point)) + { + if (labels[point] == -1) + { + labels[point] = clusterId; + } + + if (labels[point] != 0) + { + continue; + } + + labels[point] = clusterId; + if (neighbors[point].Count < minimumNeighbors) + { + continue; + } + + foreach (var neighbor in neighbors[point]) + { + if (queued.Add(neighbor)) + { + pending.Enqueue(neighbor); + } + } + } + } + + private int Skip( + SpeakerIdentity identity, + string reason) + { + logger.LogInformation( + "Skipped Resemblyzer voice-vector pruning for identity {IdentityId}: {Reason}", + identity.Id, + reason); + return 0; + } + +} diff --git a/MeetingAssistant/Speakers/ResemblyzerWarmupHostedService.cs b/MeetingAssistant/Speakers/ResemblyzerWarmupHostedService.cs new file mode 100644 index 0000000..69607f9 --- /dev/null +++ b/MeetingAssistant/Speakers/ResemblyzerWarmupHostedService.cs @@ -0,0 +1,54 @@ +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed class ResemblyzerWarmupHostedService : BackgroundService +{ + private readonly IResemblyzerVoiceEncoder encoder; + private readonly bool speakerIdentificationEnabled; + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + public ResemblyzerWarmupHostedService( + IResemblyzerVoiceEncoder encoder, + IOptions options, + ILogger logger) + { + this.encoder = encoder; + speakerIdentificationEnabled = options.Value.SpeakerIdentification.Enabled; + this.options = options.Value.SpeakerIdentification.Resemblyzer; + this.logger = logger; + } + + protected override async Task ExecuteAsync(CancellationToken stoppingToken) + { + if (!speakerIdentificationEnabled || !options.Enabled) + { + return; + } + + await Task.Yield(); + try + { + logger.LogInformation( + "Starting Resemblyzer warm-up in runtime folder {RuntimeFolder} for model {ModelId}", + VaultPath.Resolve(options.RuntimeFolder), + options.ModelId); + await encoder.WarmUpAsync(stoppingToken); + logger.LogInformation( + "Finished Resemblyzer warm-up in runtime folder {RuntimeFolder} for model {ModelId}", + VaultPath.Resolve(options.RuntimeFolder), + options.ModelId); + } + catch (OperationCanceledException) when (stoppingToken.IsCancellationRequested) + { + logger.LogInformation("Resemblyzer warm-up was cancelled during application shutdown"); + } + catch (Exception exception) + { + logger.LogError( + exception, + "Resemblyzer warm-up failed in runtime folder {RuntimeFolder}; voice encoding can still retry on demand", + VaultPath.Resolve(options.RuntimeFolder)); + } + } +} diff --git a/MeetingAssistant/Speakers/SpeakerIdentity.cs b/MeetingAssistant/Speakers/SpeakerIdentity.cs index 554ecd6..653f7e9 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentity.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentity.cs @@ -21,6 +21,8 @@ public sealed class SpeakerIdentity public List Snippets { get; set; } = []; + public List VoiceVectors { get; set; } = []; + public List References { get; set; } = []; [NotMapped] @@ -93,7 +95,7 @@ public static class SpeakerIdentityReferences string.IsNullOrWhiteSpace(reference.TranscriptPath); } - private static bool IsSame(SpeakerIdentityReference first, SpeakerIdentityReference second) + public static bool IsSame(SpeakerIdentityReference first, SpeakerIdentityReference second) { return string.Equals(first.MeetingNotePath, second.MeetingNotePath, StringComparison.OrdinalIgnoreCase) && string.Equals(first.TranscriptPath, second.TranscriptPath, StringComparison.OrdinalIgnoreCase); @@ -124,6 +126,25 @@ public sealed class SpeakerSnippet public DateTimeOffset CreatedAt { get; set; } } +public sealed class SpeakerVoiceVector +{ + public int Id { get; set; } + + public int SpeakerIdentityId { get; set; } + + public SpeakerIdentity? SpeakerIdentity { get; set; } + + public string ModelId { get; set; } = ""; + + public int Dimensions { get; set; } + + public byte[] VectorBytes { get; set; } = []; + + public string Fingerprint { get; set; } = ""; + + public DateTimeOffset CreatedAt { get; set; } +} + public sealed class SpeakerAlias { public int Id { get; set; } @@ -150,6 +171,8 @@ public sealed class SpeakerIdentityDbContext : DbContext public DbSet SpeakerSnippets => Set(); + public DbSet SpeakerVoiceVectors => Set(); + public DbSet SpeakerIdentityReferences => Set(); protected override void OnModelCreating(ModelBuilder modelBuilder) @@ -175,6 +198,11 @@ public sealed class SpeakerIdentityDbContext : DbContext .HasForeignKey(snippet => snippet.SpeakerIdentityId) .OnDelete(DeleteBehavior.Cascade); + entity.HasMany(identity => identity.VoiceVectors) + .WithOne(vector => vector.SpeakerIdentity) + .HasForeignKey(vector => vector.SpeakerIdentityId) + .OnDelete(DeleteBehavior.Cascade); + entity.HasMany(identity => identity.References) .WithOne(reference => reference.SpeakerIdentity) .HasForeignKey(reference => reference.SpeakerIdentityId) @@ -189,6 +217,13 @@ public sealed class SpeakerIdentityDbContext : DbContext .HasIndex(alias => new { alias.SpeakerIdentityId, alias.Name }) .IsUnique(); + modelBuilder.Entity() + .HasIndex(vector => new { vector.SpeakerIdentityId, vector.Fingerprint }) + .IsUnique(); + + modelBuilder.Entity() + .HasIndex(vector => new { vector.SpeakerIdentityId, vector.ModelId }); + modelBuilder.Entity() .HasIndex(reference => new { reference.SpeakerIdentityId, reference.MeetingNotePath, reference.TranscriptPath }) .IsUnique(); diff --git a/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs b/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs index 187cefc..539ae29 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentityMergeService.cs @@ -20,17 +20,20 @@ public sealed class SpeakerIdentityMergeService : ISpeakerIdentityMergeService { private readonly IDbContextFactory dbContextFactory; private readonly ISpeakerIdentityMatcher matcher; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; private readonly SpeakerIdentificationOptions options; private readonly ILogger logger; public SpeakerIdentityMergeService( IDbContextFactory dbContextFactory, ISpeakerIdentityMatcher matcher, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, IOptions options, ILogger logger) { this.dbContextFactory = dbContextFactory; this.matcher = matcher; + this.outlierPruner = outlierPruner; this.options = options.Value.SpeakerIdentification; this.logger = logger; } @@ -44,9 +47,11 @@ public sealed class SpeakerIdentityMergeService : ISpeakerIdentityMergeService var cutoff = DateTimeOffset.UtcNow - (recentIdentityAge ?? options.MergeRecentIdentityAge); var identities = await context.SpeakerIdentities + .AsSplitQuery() .Include(identity => identity.Aliases) .Include(identity => identity.CandidateNames) .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) .Include(identity => identity.References) .OrderByDescending(identity => identity.References.Count) .ThenBy(identity => identity.Id) @@ -154,10 +159,12 @@ public sealed class SpeakerIdentityMergeService : ISpeakerIdentityMergeService var targetName = target.GetDisplayName() ?? $"identity-{target.Id}"; var sourceName = source.GetDisplayName() ?? $"identity-{source.Id}"; - SpeakerIdentityMerger.MergeInto( + SpeakerIdentityMerger.MergeIntoAndPrune( target, source, - options.MaxSnippetsPerSpeaker); + options.MaxSnippetsPerSpeaker, + options.Resemblyzer.MaxVectorsPerIdentity, + outlierPruner); logger.LogInformation( "Speaker identity merge diagnostics merging source identity {SourceIdentityId} ({SourceName}) into target identity {TargetIdentityId} ({TargetName})", source.Id, diff --git a/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs b/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs index 640055c..81d8c0d 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentityMerger.cs @@ -2,10 +2,24 @@ namespace MeetingAssistant.Speakers; internal static class SpeakerIdentityMerger { + public static void MergeIntoAndPrune( + SpeakerIdentity target, + SpeakerIdentity source, + int maxSnippets, + int maxVoiceVectors, + ResemblyzerVoiceVectorOutlierPruner outlierPruner) + { + outlierPruner.Prune(target); + outlierPruner.Prune(source); + MergeInto(target, source, maxSnippets, maxVoiceVectors); + outlierPruner.Prune(target); + } + public static void MergeInto( SpeakerIdentity target, SpeakerIdentity source, - int maxSnippets) + int maxSnippets, + int maxVoiceVectors = int.MaxValue) { AddAlias(target, source.CanonicalName); foreach (var alias in source.Aliases) @@ -35,6 +49,26 @@ internal static class SpeakerIdentityMerger .ToList(); target.Snippets.Clear(); target.Snippets.AddRange(retainedSnippets); + + var retainedVectors = target.VoiceVectors + .Concat(source.VoiceVectors) + .GroupBy(vector => vector.Fingerprint, StringComparer.Ordinal) + .Select(group => group.OrderByDescending(vector => vector.CreatedAt).First()) + .OrderByDescending(vector => vector.CreatedAt) + .ThenBy(vector => vector.Fingerprint, StringComparer.Ordinal) + .Take(Math.Max(1, maxVoiceVectors)) + .Select(vector => new SpeakerVoiceVector + { + SpeakerIdentity = target, + ModelId = vector.ModelId, + Dimensions = vector.Dimensions, + VectorBytes = vector.VectorBytes.ToArray(), + Fingerprint = vector.Fingerprint, + CreatedAt = vector.CreatedAt + }) + .ToList(); + target.VoiceVectors.Clear(); + target.VoiceVectors.AddRange(retainedVectors); target.UpdatedAt = DateTimeOffset.UtcNow; } diff --git a/MeetingAssistant/Speakers/SpeakerIdentityNaming.cs b/MeetingAssistant/Speakers/SpeakerIdentityNaming.cs new file mode 100644 index 0000000..dce3718 --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerIdentityNaming.cs @@ -0,0 +1,103 @@ +using MeetingAssistant.MeetingNotes; + +namespace MeetingAssistant.Speakers; + +internal static class SpeakerIdentityNaming +{ + public static IReadOnlySet GetAcceptedNames(SpeakerIdentity identity) + { + return new[] { identity.CanonicalName } + .Concat(identity.Aliases.Select(alias => alias.Name)) + .Concat(identity.CandidateNames.Select(candidate => candidate.Name)) + .Where(name => !string.IsNullOrWhiteSpace(name)) + .Select(name => name!.Trim()) + .ToHashSet(StringComparer.OrdinalIgnoreCase); + } + + public static bool MatchesAnyAcceptedName( + SpeakerIdentity identity, + IReadOnlySet names) + { + return GetAcceptedNames(identity).Any(names.Contains); + } + + public static bool MatchesAttendees( + SpeakerIdentity identity, + IReadOnlyList attendees) + { + var attendeeSet = attendees.ToHashSet(StringComparer.OrdinalIgnoreCase); + return GetAcceptedNames(identity).Any(attendeeSet.Contains); + } + + public static IReadOnlyList NormalizeAttendees(IEnumerable attendees) + { + return attendees + .Select(MeetingAttendeeNames.NormalizeDisplayName) + .Where(attendee => !string.IsNullOrWhiteSpace(attendee)) + .Distinct(StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .ToList(); + } + + public static bool IsDiarizedSpeakerLabel(string speaker) + { + var normalized = speaker.Trim(); + return normalized.Equals("Unknown", StringComparison.OrdinalIgnoreCase) || + normalized.StartsWith("Guest", StringComparison.OrdinalIgnoreCase) || + normalized.StartsWith("Speaker", StringComparison.OrdinalIgnoreCase); + } + + public static bool UpdateCandidateNames( + SpeakerIdentity identity, + IReadOnlyList attendees) + { + if (!string.IsNullOrWhiteSpace(identity.CanonicalName) || attendees.Count == 0) + { + return false; + } + + var currentCandidates = identity.CandidateNames + .Select(candidate => candidate.Name) + .ToHashSet(StringComparer.OrdinalIgnoreCase); + var fallbackAliasCandidate = currentCandidates.Count == 1 ? currentCandidates.Single() : null; + var aliasToCandidate = identity.Aliases + .Where(alias => !string.IsNullOrWhiteSpace(alias.Name)) + .SelectMany(alias => currentCandidates.Select(candidate => new { Alias = alias.Name, Candidate = candidate })) + .Where(pair => string.Equals(pair.Alias, pair.Candidate, StringComparison.OrdinalIgnoreCase) || + pair.Alias.Contains(pair.Candidate, StringComparison.OrdinalIgnoreCase) || + pair.Candidate.Contains(pair.Alias, StringComparison.OrdinalIgnoreCase)) + .ToDictionary(pair => pair.Alias, pair => pair.Candidate, StringComparer.OrdinalIgnoreCase); + var intersection = attendees + .Select(attendee => currentCandidates.Contains(attendee) + ? attendee + : aliasToCandidate.GetValueOrDefault(attendee) ?? + (identity.Aliases.Any(alias => string.Equals(alias.Name, attendee, StringComparison.OrdinalIgnoreCase)) + ? fallbackAliasCandidate + : null)) + .Where(candidate => !string.IsNullOrWhiteSpace(candidate)) + .Select(candidate => candidate!) + .Distinct(StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .ToList(); + var resetToAttendees = intersection.Count == 0; + SetCandidates(identity, resetToAttendees ? attendees : intersection); + if (intersection.Count == 1) + { + identity.CanonicalName = intersection[0]; + } + + return resetToAttendees; + } + + public static void SetCandidates( + SpeakerIdentity identity, + IEnumerable candidates) + { + identity.CandidateNames.Clear(); + identity.CandidateNames.AddRange(candidates + .Where(candidate => !string.IsNullOrWhiteSpace(candidate)) + .Distinct(StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .Select(candidate => new SpeakerCandidateName { Name = candidate })); + } +} diff --git a/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs b/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs index 34d4e47..aab3c15 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentitySchema.cs @@ -46,6 +46,33 @@ internal static class SpeakerIdentitySchema ON "SpeakerIdentityReferences" ("SpeakerIdentityId", "MeetingNotePath", "TranscriptPath"); """, cancellationToken); + await context.Database.ExecuteSqlRawAsync( + """ + CREATE TABLE IF NOT EXISTS "SpeakerVoiceVectors" ( + "Id" INTEGER NOT NULL CONSTRAINT "PK_SpeakerVoiceVectors" PRIMARY KEY AUTOINCREMENT, + "SpeakerIdentityId" INTEGER NOT NULL, + "ModelId" TEXT NOT NULL, + "Dimensions" INTEGER NOT NULL, + "VectorBytes" BLOB NOT NULL, + "Fingerprint" TEXT NOT NULL, + "CreatedAt" TEXT NOT NULL, + CONSTRAINT "FK_SpeakerVoiceVectors_SpeakerIdentities_SpeakerIdentityId" + FOREIGN KEY ("SpeakerIdentityId") REFERENCES "SpeakerIdentities" ("Id") ON DELETE CASCADE + ); + """, + cancellationToken); + await context.Database.ExecuteSqlRawAsync( + """ + CREATE UNIQUE INDEX IF NOT EXISTS "IX_SpeakerVoiceVectors_SpeakerIdentityId_Fingerprint" + ON "SpeakerVoiceVectors" ("SpeakerIdentityId", "Fingerprint"); + """, + cancellationToken); + await context.Database.ExecuteSqlRawAsync( + """ + CREATE INDEX IF NOT EXISTS "IX_SpeakerVoiceVectors_SpeakerIdentityId_ModelId" + ON "SpeakerVoiceVectors" ("SpeakerIdentityId", "ModelId"); + """, + cancellationToken); } private static async Task EnsureSpeakerIdentityTimestampColumnsAsync( diff --git a/MeetingAssistant/Speakers/SpeakerIdentityService.cs b/MeetingAssistant/Speakers/SpeakerIdentityService.cs index 37e2500..de316d9 100644 --- a/MeetingAssistant/Speakers/SpeakerIdentityService.cs +++ b/MeetingAssistant/Speakers/SpeakerIdentityService.cs @@ -143,7 +143,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService target.CanonicalName = targetName; target.UpdatedAt = now; - ResetCandidates(target, [targetName]); + SpeakerIdentityNaming.SetCandidates(target, [targetName]); AddMeetingReference(target, meetingReference); var snippetAdded = AddSnippetIfNeeded(target, snippet); await context.SaveChangesAsync(cancellationToken); @@ -207,7 +207,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService await using var context = await dbContextFactory.CreateDbContextAsync(cancellationToken); await SpeakerIdentitySchema.EnsureCreatedOrUpdatedAsync(context, cancellationToken); - var attendees = NormalizeAttendees(request.MeetingNote.Frontmatter.Attendees); + var attendees = SpeakerIdentityNaming.NormalizeAttendees(request.MeetingNote.Frontmatter.Attendees); var meetingReference = CreateReference(request.MeetingNote, DateTimeOffset.UtcNow); var speakerMappings = new Dictionary(StringComparer.OrdinalIgnoreCase); var attendeeMatches = new List(); @@ -222,7 +222,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .ToHashSet(StringComparer.OrdinalIgnoreCase); foreach (var speaker in request.Segments .Select(segment => segment.Speaker) - .Where(speaker => !string.IsNullOrWhiteSpace(speaker) && !IsDiarizedSpeakerLabel(speaker))) + .Where(speaker => !string.IsNullOrWhiteSpace(speaker) && !SpeakerIdentityNaming.IsDiarizedSpeakerLabel(speaker))) { alreadyIdentifiedNames.Add(speaker.Trim()); } @@ -360,7 +360,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService cancellationToken); } - foreach (var acceptedName in GetAcceptedNames(identity)) + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) { matchedAcceptedNames.Add(acceptedName); } @@ -371,14 +371,14 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService { speakerMappings[speaker] = speakerName; alreadyIdentifiedNames.Add(speakerName); - foreach (var acceptedName in GetAcceptedNames(identity)) + foreach (var acceptedName in SpeakerIdentityNaming.GetAcceptedNames(identity)) { alreadyIdentifiedNames.Add(acceptedName); } attendeeMatches.Add(new SpeakerIdentityAttendeeMatch( speakerName, - GetAcceptedNames(identity).ToList())); + SpeakerIdentityNaming.GetAcceptedNames(identity).ToList())); } } @@ -440,7 +440,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .Select(identity => new { Identity = identity, - IsAttendee = MatchesAttendees(identity, attendees), + IsAttendee = SpeakerIdentityNaming.MatchesAttendees(identity, attendees), IsActive = identity.UpdatedAt >= activeCutoff }) .Where(candidate => candidate.IsAttendee || candidate.IsActive) @@ -541,11 +541,11 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService string operation, CancellationToken cancellationToken) { - var span = SpeakerSampleSpanSelector.SelectBestContinuousSpan( + var span = SpeakerSampleSpanSelector.SelectBestSameSpeakerSpan( request.Segments, speaker, - options.MaximumSampleSegmentGap, - options.MinimumSampleSpeechDuration); + options.MinimumSampleSpeechDuration, + options.MaximumSampleDuration); logger.LogInformation( "{Operation} extracting fallback sample for {Speaker}: selected {SegmentCount} segment(s), span {SpanDuration}, minimum {MinimumDuration}", operation, @@ -569,7 +569,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .Include(identity => identity.References) .ToListAsync(cancellationToken); return identities - .Where(identity => GetAcceptedNames(identity).Contains(name)) + .Where(identity => SpeakerIdentityNaming.GetAcceptedNames(identity).Contains(name)) .OrderBy(identity => string.Equals(identity.CanonicalName, name, StringComparison.OrdinalIgnoreCase) ? 0 : 1) .ThenBy(identity => identity.Id) .FirstOrDefault(); @@ -589,7 +589,7 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService .Where(identity => string.IsNullOrWhiteSpace(identity.CanonicalName)) .ToListAsync(cancellationToken); return candidates - .Where(identity => identity.References.Any(existing => IsSameReference(existing, reference))) + .Where(identity => identity.References.Any(existing => SpeakerIdentityReferences.IsSame(existing, reference))) .Where(identity => identity.CandidateNames.Any(candidate => string.Equals(candidate.Name, targetName, StringComparison.OrdinalIgnoreCase)) || identity.Aliases.Any(alias => @@ -646,27 +646,11 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService identity.Aliases.Add(new SpeakerAlias { Name = alias.Trim() }); } - private static bool IsSameReference( - SpeakerIdentityReference first, - SpeakerIdentityReference second) - { - return string.Equals(first.MeetingNotePath, second.MeetingNotePath, StringComparison.OrdinalIgnoreCase) && - string.Equals(first.TranscriptPath, second.TranscriptPath, StringComparison.OrdinalIgnoreCase); - } - private static bool MatchesAcceptedNames( SpeakerIdentity identity, IReadOnlySet names) { - return GetAcceptedNames(identity).Any(names.Contains); - } - - private static bool MatchesAttendees( - SpeakerIdentity identity, - IReadOnlyList attendees) - { - var attendeeSet = attendees.ToHashSet(StringComparer.OrdinalIgnoreCase); - return GetAcceptedNames(identity).Any(attendeeSet.Contains); + return SpeakerIdentityNaming.MatchesAnyAcceptedName(identity, names); } private static Task LoadIdentityAsync( @@ -694,48 +678,25 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService var currentCandidates = identity.CandidateNames .Select(candidate => candidate.Name) .ToHashSet(StringComparer.OrdinalIgnoreCase); - var fallbackAliasCandidate = currentCandidates.Count == 1 ? currentCandidates.Single() : null; - var aliasToCandidate = identity.Aliases - .Where(alias => !string.IsNullOrWhiteSpace(alias.Name)) - .SelectMany(alias => currentCandidates.Select(candidate => new { Alias = alias.Name, Candidate = candidate })) - .Where(pair => string.Equals(pair.Alias, pair.Candidate, StringComparison.OrdinalIgnoreCase) || - pair.Alias.Contains(pair.Candidate, StringComparison.OrdinalIgnoreCase) || - pair.Candidate.Contains(pair.Alias, StringComparison.OrdinalIgnoreCase)) - .ToDictionary(pair => pair.Alias, pair => pair.Candidate, StringComparer.OrdinalIgnoreCase); - var intersection = attendees - .Select(attendee => currentCandidates.Contains(attendee) - ? attendee - : aliasToCandidate.GetValueOrDefault(attendee) ?? - (identity.Aliases.Any(alias => string.Equals(alias.Name, attendee, StringComparison.OrdinalIgnoreCase)) - ? fallbackAliasCandidate - : null)) - .Where(candidate => !string.IsNullOrWhiteSpace(candidate)) - .Select(candidate => candidate!) - .Distinct(StringComparer.OrdinalIgnoreCase) - .Order(StringComparer.OrdinalIgnoreCase) - .ToList(); - - if (intersection.Count == 0) + var resetToAttendees = SpeakerIdentityNaming.UpdateCandidateNames(identity, attendees); + if (resetToAttendees) { logger.LogInformation( "Speaker identity candidate elimination for identity {IdentityId} had empty intersection; resetting candidates to attendees {Attendees} and replacing oldest snippet", identity.Id, FormatNames(attendees)); - ResetCandidates(identity, attendees); ReplaceOldestSnippet(identity, snippet); return; } logger.LogInformation( - "Speaker identity candidate elimination for identity {IdentityId}: candidates {CurrentCandidates}, attendees {Attendees}, intersection {Intersection}", + "Speaker identity candidate elimination for identity {IdentityId}: candidates {CurrentCandidates}, attendees {Attendees}, remaining {RemainingCandidates}", identity.Id, FormatNames(currentCandidates), FormatNames(attendees), - FormatNames(intersection)); - ResetCandidates(identity, intersection); - if (intersection.Count == 1) + FormatNames(identity.CandidateNames.Select(candidate => candidate.Name))); + if (!string.IsNullOrWhiteSpace(identity.CanonicalName)) { - identity.CanonicalName = intersection[0]; logger.LogInformation( "Speaker identity candidate elimination promoted identity {IdentityId} to canonical name {CanonicalName}", identity.Id, @@ -872,51 +833,6 @@ public sealed class SpeakerIdentityService : ISpeakerIdentificationService } } - private static void ResetCandidates(SpeakerIdentity identity, IReadOnlyList candidates) - { - identity.CandidateNames.Clear(); - identity.CandidateNames.AddRange(candidates - .Distinct(StringComparer.OrdinalIgnoreCase) - .Order(StringComparer.OrdinalIgnoreCase) - .Select(candidate => new SpeakerCandidateName { Name = candidate })); - } - - private static IReadOnlySet GetAcceptedNames(SpeakerIdentity identity) - { - return new[] - { - identity.CanonicalName - } - .Concat(identity.Aliases.Select(alias => alias.Name)) - .Concat(identity.CandidateNames.Select(candidate => candidate.Name)) - .Where(name => !string.IsNullOrWhiteSpace(name)) - .Select(name => name!.Trim()) - .ToHashSet(StringComparer.OrdinalIgnoreCase); - } - - private static IReadOnlyList NormalizeAttendees(IEnumerable attendees) - { - return attendees - .Select(NormalizeAttendee) - .Where(attendee => !string.IsNullOrWhiteSpace(attendee)) - .Distinct(StringComparer.OrdinalIgnoreCase) - .Order(StringComparer.OrdinalIgnoreCase) - .ToList(); - } - - private static string NormalizeAttendee(string attendee) - { - return MeetingAttendeeNames.NormalizeDisplayName(attendee); - } - - private static bool IsDiarizedSpeakerLabel(string speaker) - { - var normalized = speaker.Trim(); - return normalized.Equals("Unknown", StringComparison.OrdinalIgnoreCase) || - normalized.StartsWith("Guest", StringComparison.OrdinalIgnoreCase) || - normalized.StartsWith("Speaker", StringComparison.OrdinalIgnoreCase); - } - private static SpeakerIdentityReference CreateReference(MeetingNote meetingNote, DateTimeOffset timestamp) { return SpeakerIdentityReferences.Create( diff --git a/MeetingAssistant/Speakers/SpeakerSampleCollectionPolicy.cs b/MeetingAssistant/Speakers/SpeakerSampleCollectionPolicy.cs new file mode 100644 index 0000000..1968f43 --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerSampleCollectionPolicy.cs @@ -0,0 +1,154 @@ +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public sealed record SpeakerSampleCollectionPolicy( + int MinimumRetainedSamples, + bool RequireNonOverlappingSamples) +{ + public static SpeakerSampleCollectionPolicy ExistingBackend { get; } = new(1, false); + + public static SpeakerSampleCollectionPolicy IndependentVectors(int maximumVectors) + { + return new SpeakerSampleCollectionPolicy(Math.Max(1, maximumVectors), true); + } + + public int ResolveRetainedSampleLimit(int configuredLimit) + { + return Math.Max(Math.Max(1, configuredLimit), MinimumRetainedSamples); + } +} + +internal static class SpeakerSampleDurationConfiguration +{ + public static void ValidateOrThrow( + SpeakerIdentificationOptions options, + string context) + { + ValidateOrThrow( + options.MinimumSampleSpeechDuration, + options.MaximumSampleDuration, + context); + } + + public static void ValidateOrThrow( + TimeSpan minimumDuration, + TimeSpan maximumDuration, + string context) + { + var error = GetError(minimumDuration, maximumDuration); + if (error is not null) + { + throw new InvalidOperationException($"{context}: {error}"); + } + } + + public static string? GetError( + TimeSpan minimumDuration, + TimeSpan maximumDuration) + { + if (minimumDuration < TimeSpan.Zero) + { + return "SpeakerIdentification:MinimumSampleSpeechDuration must not be negative."; + } + + if (maximumDuration <= TimeSpan.Zero) + { + return "SpeakerIdentification:MaximumSampleDuration must be greater than zero."; + } + + return minimumDuration > maximumDuration + ? "SpeakerIdentification:MinimumSampleSpeechDuration must not exceed SpeakerIdentification:MaximumSampleDuration." + : null; + } +} + +internal static class ResemblyzerSpeakerRecognitionConfiguration +{ + public static void ValidateOrThrow( + ResemblyzerSpeakerRecognitionOptions options, + string context) + { + var error = GetError(options); + if (error is not null) + { + throw new InvalidOperationException($"{context}: {error}"); + } + } + + public static string? GetError(ResemblyzerSpeakerRecognitionOptions options) + { + if (options.RequiredVectorsPerSpeaker < 1) + { + return "SpeakerIdentification:Resemblyzer:RequiredVectorsPerSpeaker must be at least one."; + } + + if (options.MaxVectorsPerIdentity < options.RequiredVectorsPerSpeaker) + { + return "SpeakerIdentification:Resemblyzer:MaxVectorsPerIdentity must not be less than RequiredVectorsPerSpeaker."; + } + + if (options.OutlierPruningMinimumVectors < options.RequiredVectorsPerSpeaker) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumVectors must not be less than RequiredVectorsPerSpeaker."; + } + + if (options.OutlierPruningMinimumVectors > options.MaxVectorsPerIdentity) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumVectors must not exceed MaxVectorsPerIdentity."; + } + + if (options.OutlierPruningMinimumNeighbors < 1 || + options.OutlierPruningMinimumNeighbors > options.OutlierPruningMinimumVectors) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumNeighbors must be between one and OutlierPruningMinimumVectors."; + } + + if (!IsCosine(options.OutlierPruningNeighborSimilarity)) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningNeighborSimilarity must be finite and between -1 and 1."; + } + + if (!double.IsFinite(options.OutlierPruningMinimumClusterRatio) || + options.OutlierPruningMinimumClusterRatio is <= 0 or > 1) + { + return "SpeakerIdentification:Resemblyzer:OutlierPruningMinimumClusterRatio must be finite, greater than zero, and at most one."; + } + + if (!IsCosine(options.MinimumClusterCohesion)) + { + return "SpeakerIdentification:Resemblyzer:MinimumClusterCohesion must be finite and between -1 and 1."; + } + + if (!IsCosine(options.MinimumIdentitySimilarity)) + { + return "SpeakerIdentification:Resemblyzer:MinimumIdentitySimilarity must be finite and between -1 and 1."; + } + + return !double.IsFinite(options.MinimumSimilarityMargin) || + options.MinimumSimilarityMargin is < 0 or > 2 + ? "SpeakerIdentification:Resemblyzer:MinimumSimilarityMargin must be finite and between 0 and 2." + : null; + } + + private static bool IsCosine(double value) + { + return double.IsFinite(value) && value is >= -1 and <= 1; + } +} + +internal sealed class MeetingAssistantSpeakerSampleOptionsValidator + : IValidateOptions +{ + public ValidateOptionsResult Validate(string? name, MeetingAssistantOptions options) + { + var error = SpeakerSampleDurationConfiguration.GetError( + options.SpeakerIdentification.MinimumSampleSpeechDuration, + options.SpeakerIdentification.MaximumSampleDuration) ?? + ResemblyzerSpeakerRecognitionConfiguration.GetError( + options.SpeakerIdentification.Resemblyzer); + return error is null + ? ValidateOptionsResult.Success + : ValidateOptionsResult.Fail(error); + } +} diff --git a/MeetingAssistant/Speakers/SpeakerSampleSpan.cs b/MeetingAssistant/Speakers/SpeakerSampleSpan.cs new file mode 100644 index 0000000..b859c7c --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerSampleSpan.cs @@ -0,0 +1,82 @@ +using MeetingAssistant.Transcription; + +namespace MeetingAssistant.Speakers; + +internal sealed record SpeakerSampleSpan( + string Speaker, + TimeSpan Start, + TimeSpan End, + TimeSpan SpeechDuration, + TimeSpan MaximumDuration, + IReadOnlyList Segments) +{ + public static SpeakerSampleSpan? Create( + TranscriptionSegment segment, + TimeSpan maximumDuration) + { + if (maximumDuration <= TimeSpan.Zero) + { + throw new ArgumentOutOfRangeException( + nameof(maximumDuration), + maximumDuration, + "Speaker sample maximum duration must be greater than zero."); + } + + var boundedEnd = Min(segment.End, segment.Start + maximumDuration); + return boundedEnd > segment.Start + ? new SpeakerSampleSpan( + segment.Speaker, + segment.Start, + boundedEnd, + boundedEnd - segment.Start, + maximumDuration, + [segment with { End = boundedEnd }]) + : null; + } + + public bool TryExtend( + TranscriptionSegment segment, + out SpeakerSampleSpan extended) + { + extended = this; + if (!string.Equals(Speaker, segment.Speaker, StringComparison.OrdinalIgnoreCase) || + segment.Start >= Start + MaximumDuration) + { + return false; + } + + var boundedEnd = Min(segment.End, Start + MaximumDuration); + if (boundedEnd <= segment.Start) + { + return false; + } + + var uncoveredStart = segment.Start > End ? segment.Start : End; + var additionalSpeechDuration = boundedEnd > uncoveredStart + ? boundedEnd - uncoveredStart + : TimeSpan.Zero; + extended = this with + { + End = boundedEnd > End ? boundedEnd : End, + SpeechDuration = SpeechDuration + additionalSpeechDuration, + Segments = Segments.Append(segment with { End = boundedEnd }).ToList() + }; + return true; + } + + public TranscriptionSegment ToSegment() + { + return new TranscriptionSegment( + Start, + End, + Speaker, + string.Join(' ', Segments + .Select(segment => segment.Text) + .Where(text => !string.IsNullOrWhiteSpace(text)))); + } + + private static TimeSpan Min(TimeSpan left, TimeSpan right) + { + return left < right ? left : right; + } +} diff --git a/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs b/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs index 5e3bf52..4f16538 100644 --- a/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs +++ b/MeetingAssistant/Speakers/SpeakerSampleSpanSelector.cs @@ -4,56 +4,84 @@ namespace MeetingAssistant.Speakers; internal static class SpeakerSampleSpanSelector { - public static bool CanExtend( - string currentSpeaker, - TimeSpan currentEnd, - TranscriptionSegment nextSegment, - TimeSpan maximumSegmentGap) - { - return string.Equals(currentSpeaker, nextSegment.Speaker, StringComparison.OrdinalIgnoreCase) && - nextSegment.Start - currentEnd <= maximumSegmentGap; - } - - public static IReadOnlyList SelectBestContinuousSpan( + public static IReadOnlyList SelectBestSameSpeakerSpan( IReadOnlyList segments, string speaker, - TimeSpan maximumSegmentGap, - TimeSpan minimumDuration) + TimeSpan minimumDuration, + TimeSpan maximumSampleDuration) { - var best = new List(); - var current = new List(); - + SpeakerSampleSpan? best = null; + SpeakerSampleSpan? current = null; foreach (var segment in segments.OrderBy(segment => segment.Start)) { - if (current.Count == 0) + if (!IsSpeaker(segment, speaker)) { - if (IsSpeaker(segment, speaker)) - { - current.Add(segment); - best = LongerSpan(current, best); - } - + current = null; continue; } - if (!CanExtend(speaker, current[^1].End, segment, maximumSegmentGap)) + current = ExtendOrStart(current, segment, maximumSampleDuration); + if (current is not null && + (best is null || current.SpeechDuration > best.SpeechDuration)) { - current.Clear(); - if (!IsSpeaker(segment, speaker)) - { - continue; - } + best = current; } - - current.Add(segment); - best = LongerSpan(current, best); } - return SpanDuration(best) >= minimumDuration - ? best + return best is not null && best.SpeechDuration >= minimumDuration + ? best.Segments : []; } + public static IReadOnlyList> SelectBestSameSpeakerSpans( + IReadOnlyList segments, + string speaker, + TimeSpan minimumDuration, + TimeSpan maximumSampleDuration, + int maxSpans) + { + if (maxSpans <= 0) + { + return []; + } + + var completed = new List(); + SpeakerSampleSpan? current = null; + TimeSpan? lastCompletedEnd = null; + foreach (var segment in segments.OrderBy(segment => segment.Start)) + { + if (!IsSpeaker(segment, speaker)) + { + current = null; + continue; + } + + if (lastCompletedEnd is { } end && segment.Start < end) + { + current = null; + continue; + } + + current = ExtendOrStart(current, segment, maximumSampleDuration); + if (current is null || current.SpeechDuration < minimumDuration) + { + continue; + } + + completed.Add(current); + lastCompletedEnd = current.End; + current = null; + } + + return completed + .OrderByDescending(span => span.SpeechDuration) + .ThenBy(span => span.Start) + .Take(maxSpans) + .OrderBy(span => span.Start) + .Select(span => span.Segments) + .ToList(); + } + public static TimeSpan SpanDuration(IReadOnlyList segments) { return segments.Count == 0 @@ -61,19 +89,20 @@ internal static class SpeakerSampleSpanSelector : segments[^1].End - segments[0].Start; } + private static SpeakerSampleSpan? ExtendOrStart( + SpeakerSampleSpan? current, + TranscriptionSegment segment, + TimeSpan maximumSampleDuration) + { + return current is not null && current.TryExtend(segment, out var extended) + ? extended + : SpeakerSampleSpan.Create(segment, maximumSampleDuration); + } + private static bool IsSpeaker( TranscriptionSegment segment, string speaker) { return string.Equals(segment.Speaker, speaker, StringComparison.OrdinalIgnoreCase); } - - private static List LongerSpan( - List current, - List best) - { - return SpanDuration(current) > SpanDuration(best) - ? current.ToList() - : best; - } } diff --git a/MeetingAssistant/Speakers/SpeakerVoiceVectors.cs b/MeetingAssistant/Speakers/SpeakerVoiceVectors.cs new file mode 100644 index 0000000..9735e39 --- /dev/null +++ b/MeetingAssistant/Speakers/SpeakerVoiceVectors.cs @@ -0,0 +1,191 @@ +using System.Buffers.Binary; +using System.Security.Cryptography; +using System.Text; + +namespace MeetingAssistant.Speakers; + +internal sealed record DecodedSpeakerVoiceVector( + SpeakerVoiceVector Stored, + float[] Vector); + +internal static class ResemblyzerVectorContract +{ + public const int Dimensions = 256; +} + +internal static class SpeakerVoiceVectors +{ + public static float[] Decode(SpeakerVoiceVector stored) + { + if (stored.Dimensions != ResemblyzerVectorContract.Dimensions || + stored.VectorBytes.Length != stored.Dimensions * sizeof(float)) + { + throw new InvalidDataException( + $"Stored voice vector {stored.Id} has invalid dimension or byte length metadata."); + } + + var vector = new float[stored.Dimensions]; + for (var index = 0; index < vector.Length; index++) + { + vector[index] = BinaryPrimitives.ReadSingleLittleEndian( + stored.VectorBytes.AsSpan(index * sizeof(float), sizeof(float))); + if (!float.IsFinite(vector[index])) + { + throw new InvalidDataException($"Stored voice vector {stored.Id} contains a non-finite value."); + } + } + + return vector; + } + + public static int AddDistinct( + SpeakerIdentity identity, + IEnumerable vectors, + string modelId, + int maximumCount, + DateTimeOffset createdAt) + { + var limit = Math.Max(1, maximumCount); + var fingerprints = identity.VoiceVectors + .Select(vector => vector.Fingerprint) + .ToHashSet(StringComparer.Ordinal); + var added = 0; + foreach (var vector in vectors) + { + if (identity.VoiceVectors.Count >= limit) + { + break; + } + + var bytes = Encode(vector); + var fingerprint = Fingerprint(modelId, bytes); + if (!fingerprints.Add(fingerprint)) + { + continue; + } + + identity.VoiceVectors.Add(new SpeakerVoiceVector + { + ModelId = modelId, + Dimensions = vector.Length, + VectorBytes = bytes, + Fingerprint = fingerprint, + CreatedAt = createdAt + }); + added++; + } + + return added; + } + + public static IReadOnlyList DecodeCompatible( + SpeakerIdentity identity, + string modelId, + ILogger logger) + { + return DecodeCompatibleEntries(identity, modelId, logger) + .Select(entry => entry.Vector) + .ToList(); + } + + public static IReadOnlyList DecodeCompatibleEntries( + SpeakerIdentity identity, + string modelId, + ILogger logger) + { + var vectors = new List(); + foreach (var stored in identity.VoiceVectors + .Where(vector => string.Equals(vector.ModelId, modelId, StringComparison.Ordinal)) + .OrderBy(vector => vector.CreatedAt) + .ThenBy(vector => vector.Id)) + { + try + { + vectors.Add(new DecodedSpeakerVoiceVector(stored, Decode(stored))); + } + catch (InvalidDataException exception) + { + logger.LogWarning( + exception, + "Ignoring invalid stored voice vector {VectorId} for identity {IdentityId}", + stored.Id, + identity.Id); + } + } + + return vectors; + } + + public static double Cosine(float[] first, float[] second) + { + double dot = 0; + for (var index = 0; index < first.Length; index++) + { + dot += first[index] * second[index]; + } + + return dot; + } + + public static float[] Normalize(float[]? vector, string description) + { + if (vector is null || vector.Length != ResemblyzerVectorContract.Dimensions) + { + throw new InvalidDataException( + $"{description} has {vector?.Length ?? 0} dimensions; expected {ResemblyzerVectorContract.Dimensions}."); + } + + double magnitudeSquared = 0; + foreach (var value in vector) + { + if (!float.IsFinite(value)) + { + throw new InvalidDataException($"{description} contains a non-finite value."); + } + + magnitudeSquared += value * value; + } + + var magnitude = Math.Sqrt(magnitudeSquared); + if (magnitude <= double.Epsilon) + { + throw new InvalidDataException($"{description} has zero magnitude."); + } + + return vector.Select(value => (float)(value / magnitude)).ToArray(); + } + + private static byte[] Encode(float[] vector) + { + if (vector.Length != ResemblyzerVectorContract.Dimensions) + { + throw new InvalidDataException( + $"Voice vector has {vector.Length} dimensions; expected {ResemblyzerVectorContract.Dimensions}."); + } + + var bytes = new byte[vector.Length * sizeof(float)]; + for (var index = 0; index < vector.Length; index++) + { + if (!float.IsFinite(vector[index])) + { + throw new InvalidDataException("Voice vector contains a non-finite value."); + } + + BinaryPrimitives.WriteSingleLittleEndian( + bytes.AsSpan(index * sizeof(float), sizeof(float)), + vector[index]); + } + + return bytes; + } + + private static string Fingerprint(string modelId, byte[] vectorBytes) + { + var modelBytes = Encoding.UTF8.GetBytes(modelId); + var input = new byte[modelBytes.Length + 1 + vectorBytes.Length]; + modelBytes.CopyTo(input, 0); + input[modelBytes.Length] = 0; + vectorBytes.CopyTo(input, modelBytes.Length + 1); + return Convert.ToHexString(SHA256.HashData(input)); + } +} diff --git a/MeetingAssistant/Speakers/VenvResemblyzerVoiceEncoder.cs b/MeetingAssistant/Speakers/VenvResemblyzerVoiceEncoder.cs new file mode 100644 index 0000000..81159cf --- /dev/null +++ b/MeetingAssistant/Speakers/VenvResemblyzerVoiceEncoder.cs @@ -0,0 +1,391 @@ +using System.Globalization; +using System.Security.Cryptography; +using System.Text; +using System.Text.Json; +using System.Text.RegularExpressions; +using MeetingAssistant.Transcription; +using Microsoft.Extensions.Options; + +namespace MeetingAssistant.Speakers; + +public interface IResemblyzerVoiceEncoder +{ + Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken); + + Task WarmUpAsync(CancellationToken cancellationToken); +} + +public sealed partial class VenvResemblyzerVoiceEncoder : IResemblyzerVoiceEncoder +{ + private const string EnvironmentSchemaVersion = "venv-v2"; + private const string JsonStart = "__MEETING_ASSISTANT_RESEMBLYZER_JSON_START__"; + private const string JsonEnd = "__MEETING_ASSISTANT_RESEMBLYZER_JSON_END__"; + private const string NumpyBeforePython313Requirement = "numpy<2; python_version < '3.13'"; + private const string NumpyFromPython313Requirement = "numpy>=2,<3; python_version >= '3.13'"; + private const string LibrosaRequirement = "librosa>=0.9.1"; + private const string ScipyRequirement = "scipy>=1.2.1"; + private readonly SemaphoreSlim commandLock = new(1, 1); + private readonly ICommandRunner commandRunner; + private readonly ResemblyzerSpeakerRecognitionOptions options; + private readonly ILogger logger; + private string? verifiedEnvironmentPythonPath; + + public VenvResemblyzerVoiceEncoder( + ICommandRunner commandRunner, + IOptions options, + ILogger logger) + { + this.commandRunner = commandRunner; + this.options = options.Value.SpeakerIdentification.Resemblyzer; + this.logger = logger; + } + + public async Task> EncodeAsync( + IReadOnlyList wavSamples, + CancellationToken cancellationToken) + { + if (wavSamples.Count == 0) + { + return []; + } + + if (wavSamples.Any(sample => sample.Length == 0)) + { + throw new InvalidDataException("Resemblyzer cannot encode an empty WAV sample."); + } + + var runtimeFolder = VaultPath.Resolve(options.RuntimeFolder); + var inputFolder = Path.Combine(runtimeFolder, "input", Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(inputFolder); + try + { + for (var index = 0; index < wavSamples.Count; index++) + { + await File.WriteAllBytesAsync( + Path.Combine(inputFolder, $"{index:D4}.wav"), + wavSamples[index], + cancellationToken); + } + + await commandLock.WaitAsync(cancellationToken); + try + { + return await RunEncodingAsync(inputFolder, wavSamples.Count, cancellationToken); + } + finally + { + commandLock.Release(); + } + } + finally + { + try + { + Directory.Delete(inputFolder, recursive: true); + } + catch (DirectoryNotFoundException) + { + } + catch (IOException exception) + { + logger.LogWarning(exception, "Could not remove Resemblyzer temporary input folder {InputFolder}", inputFolder); + } + catch (UnauthorizedAccessException exception) + { + logger.LogWarning(exception, "Could not remove Resemblyzer temporary input folder {InputFolder}", inputFolder); + } + } + } + + private async Task> RunEncodingAsync( + string inputFolder, + int expectedCount, + CancellationToken cancellationToken) + { + return await RunWithTimeoutAsync(async token => + { + var pythonPath = await EnsureEnvironmentAsync(token); + var scriptPath = Path.Combine(VaultPath.Resolve(options.RuntimeFolder), "encode.py"); + await File.WriteAllTextAsync(scriptPath, BuildEncodingScript(), token); + var result = await RunRequiredAsync( + pythonPath, + [scriptPath, Path.GetFullPath(inputFolder)], + "encoding", + token); + return ParseAndValidateVectors(result.StandardOutput, expectedCount); + }, cancellationToken); + } + + public async Task WarmUpAsync(CancellationToken cancellationToken) + { + await commandLock.WaitAsync(cancellationToken); + try + { + await RunWithTimeoutAsync(async token => + { + await EnsureEnvironmentAsync(token); + return true; + }, cancellationToken); + } + finally + { + commandLock.Release(); + } + } + + private async Task EnsureEnvironmentAsync(CancellationToken cancellationToken) + { + ValidateDependencySettings(); + var runtimeFolder = VaultPath.Resolve(options.RuntimeFolder); + var environmentFolder = Path.Combine(runtimeFolder, "venv", BuildEnvironmentKey()); + var pythonPath = GetEnvironmentPythonPath(environmentFolder); + var readyPath = Path.Combine(environmentFolder, ".ready"); + if (string.Equals(verifiedEnvironmentPythonPath, pythonPath, StringComparison.OrdinalIgnoreCase) + && File.Exists(pythonPath) + && File.Exists(readyPath)) + { + return pythonPath; + } + + if (File.Exists(pythonPath) && File.Exists(readyPath)) + { + var verification = await commandRunner.RunAsync( + pythonPath, + WarmupArguments, + cancellationToken); + if (verification.ExitCode == 0) + { + verifiedEnvironmentPythonPath = pythonPath; + return pythonPath; + } + + logger.LogWarning( + "Existing Resemblyzer environment verification failed with exit code {ExitCode}: {Error}. Reprovisioning it.", + verification.ExitCode, + verification.StandardError); + File.Delete(readyPath); + } + + Directory.CreateDirectory(environmentFolder); + await RunRequiredAsync( + options.PythonCommand, + ["-m", "venv", "--clear", environmentFolder], + "virtual environment creation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + ["-m", "pip", "install", "--upgrade", "pip"], + "pip upgrade", + cancellationToken); + await RunRequiredAsync( + pythonPath, + [ + "-m", "pip", "install", + "--index-url", options.TorchIndexUrl, + TorchRequirement + ], + "CPU PyTorch installation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + [ + "-m", "pip", "install", + .. RuntimeDependencyRequirements + ], + "Resemblyzer dependency installation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + [ + "-m", "pip", "install", "--no-deps", + ResemblyzerRequirement + ], + "Resemblyzer installation", + cancellationToken); + await RunRequiredAsync( + pythonPath, + WarmupArguments, + "environment verification", + cancellationToken); + await File.WriteAllTextAsync(readyPath, BuildEnvironmentKey(), cancellationToken); + verifiedEnvironmentPythonPath = pythonPath; + return pythonPath; + } + + private async Task RunRequiredAsync( + string fileName, + IReadOnlyList arguments, + string operation, + CancellationToken cancellationToken) + { + var result = await commandRunner.RunAsync(fileName, arguments, cancellationToken); + ThrowIfFailed(result, operation); + return result; + } + + private static string BuildEncodingScript() + { + return + "import json\n" + + "import sys\n" + + "import numpy as np\n" + + "from pathlib import Path\n" + + "from resemblyzer import VoiceEncoder, preprocess_wav\n" + + "encoder = VoiceEncoder('cpu', verbose=False)\n" + + "vectors = []\n" + + "for path in sorted(Path(sys.argv[1]).glob('*.wav')):\n" + + " wav = preprocess_wav(path)\n" + + " vector = encoder.embed_utterance(wav)\n" + + " vectors.append(np.asarray(vector, dtype=np.float32).tolist())\n" + + $"print('{JsonStart}')\n" + + "print(json.dumps(vectors, allow_nan=False))\n" + + $"print('{JsonEnd}')\n"; + } + + private static IReadOnlyList ParseAndValidateVectors(string output, int expectedCount) + { + var json = ExtractJson(output); + float[][]? vectors; + try + { + vectors = JsonSerializer.Deserialize(json); + } + catch (JsonException exception) + { + throw new InvalidDataException("Resemblyzer returned malformed vector JSON.", exception); + } + + if (vectors is null || vectors.Length != expectedCount) + { + throw new InvalidDataException( + $"Resemblyzer returned {vectors?.Length ?? 0} vectors for {expectedCount} WAV samples."); + } + + return vectors + .Select((vector, index) => SpeakerVoiceVectors.Normalize(vector, $"Resemblyzer vector {index}")) + .ToList(); + } + + private static string ExtractJson(string output) + { + var start = output.IndexOf(JsonStart, StringComparison.Ordinal); + if (start < 0) + { + throw new InvalidDataException("Resemblyzer output did not contain the JSON start marker."); + } + + start += JsonStart.Length; + var end = output.IndexOf(JsonEnd, start, StringComparison.Ordinal); + if (end < 0) + { + throw new InvalidDataException("Resemblyzer output did not contain the JSON end marker."); + } + + return output[start..end].Trim(); + } + + private string BuildEnvironmentKey() + { + var settings = string.Join('\n', EnvironmentManifest); + var hash = SHA256.HashData(Encoding.UTF8.GetBytes(settings)); + return Convert.ToHexString(hash)[..16].ToLowerInvariant(); + } + + private string TorchRequirement => $"torch=={options.TorchVersion}"; + + private string ResemblyzerRequirement => $"Resemblyzer=={options.PackageVersion}"; + + private static string[] WarmupArguments => + [ + "-c", + "from resemblyzer import VoiceEncoder; VoiceEncoder('cpu', verbose=False); print('Resemblyzer warm-up complete')" + ]; + + private string[] RuntimeDependencyRequirements => + [ + NumpyBeforePython313Requirement, + NumpyFromPython313Requirement, + LibrosaRequirement, + ScipyRequirement, + $"webrtcvad-wheels=={options.WebRtcVadVersion}" + ]; + + private IEnumerable EnvironmentManifest => + [ + EnvironmentSchemaVersion, + options.PythonCommand, + options.TorchIndexUrl, + TorchRequirement, + .. RuntimeDependencyRequirements, + ResemblyzerRequirement + ]; + + private static string GetEnvironmentPythonPath(string environmentFolder) + { + return OperatingSystem.IsWindows() + ? Path.Combine(environmentFolder, "Scripts", "python.exe") + : Path.Combine(environmentFolder, "bin", "python"); + } + + private void ValidateDependencySettings() + { + if (string.IsNullOrWhiteSpace(options.PythonCommand)) + { + throw new InvalidOperationException("The Resemblyzer Python command cannot be empty."); + } + + ValidateVersion(options.PackageVersion, "Resemblyzer package"); + ValidateVersion(options.TorchVersion, "PyTorch"); + ValidateVersion(options.WebRtcVadVersion, "webrtcvad-wheels"); + if (!Uri.TryCreate(options.TorchIndexUrl, UriKind.Absolute, out var indexUri) + || indexUri.Scheme != Uri.UriSchemeHttps) + { + throw new InvalidOperationException( + $"Invalid Resemblyzer PyTorch index URL '{options.TorchIndexUrl}'."); + } + } + + private static void ValidateVersion(string version, string dependency) + { + if (!PackageVersionPattern().IsMatch(version)) + { + throw new InvalidOperationException($"Invalid {dependency} version '{version}'."); + } + } + + private async Task RunWithTimeoutAsync( + Func> operation, + CancellationToken cancellationToken) + { + using var timeoutSource = options.CommandTimeout > TimeSpan.Zero + ? new CancellationTokenSource(options.CommandTimeout) + : null; + using var linkedSource = timeoutSource is null + ? null + : CancellationTokenSource.CreateLinkedTokenSource(cancellationToken, timeoutSource.Token); + try + { + return await operation(linkedSource?.Token ?? cancellationToken); + } + catch (OperationCanceledException) when ( + !cancellationToken.IsCancellationRequested && timeoutSource?.IsCancellationRequested == true) + { + throw new TimeoutException( + $"Resemblyzer command timed out after {options.CommandTimeout.ToString(null, CultureInfo.InvariantCulture)}."); + } + } + + private static void ThrowIfFailed(CommandResult result, string operation) + { + if (result.ExitCode != 0) + { + throw new InvalidOperationException( + $"Resemblyzer {operation} failed with exit code {result.ExitCode}: {result.StandardError}"); + } + } + + [GeneratedRegex("^[0-9A-Za-z.+-]+$", RegexOptions.CultureInvariant)] + private static partial Regex PackageVersionPattern(); +} diff --git a/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs b/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs index a49660c..8e28d40 100644 --- a/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs +++ b/MeetingAssistant/Transcription/PyannoteDiarizationWarmupHostedService.cs @@ -120,7 +120,8 @@ public sealed class PyannoteDiarizationWarmupHostedService : IHostedService private static IEnumerable GetEnabledSpeakerValidationDiarizationOptions( MeetingAssistantOptions options) { - if (options.SpeakerIdentification.PyannoteValidation.Enabled) + if (!options.SpeakerIdentification.Resemblyzer.Enabled && + options.SpeakerIdentification.PyannoteValidation.Enabled) { yield return options.SpeakerIdentification.PyannoteValidation.Diarization; } diff --git a/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs b/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs index 8167ffe..dd32d95 100644 --- a/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs +++ b/MeetingAssistant/Workflow/WorkflowRulesEditorChatPipeline.cs @@ -24,6 +24,7 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi private readonly IMeetingMetadataProvider meetingMetadataProvider; private readonly ILaunchProfileOptionsProvider launchProfiles; private readonly ISpeakerIdentityMergeService identityMergeService; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; private readonly AsrDiagnosticService asrDiagnosticService; private readonly IConfiguration configuration; @@ -38,6 +39,7 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi IMeetingMetadataProvider meetingMetadataProvider, ILaunchProfileOptionsProvider launchProfiles, ISpeakerIdentityMergeService identityMergeService, + ResemblyzerVoiceVectorOutlierPruner outlierPruner, AsrDiagnosticService asrDiagnosticService, IConfiguration configuration) { @@ -51,6 +53,7 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi this.meetingMetadataProvider = meetingMetadataProvider; this.launchProfiles = launchProfiles; this.identityMergeService = identityMergeService; + this.outlierPruner = outlierPruner; this.asrDiagnosticService = asrDiagnosticService; this.configuration = configuration; } @@ -78,7 +81,8 @@ public sealed class WorkflowRulesEditorChatPipeline : IWorkflowRulesEditorChatPi identityMergeService, asrDiagnosticService, configuration, - logger: logger); + logger: logger, + outlierPruner: outlierPruner); var messages = conversation .Select(ToChatMessage) .Append(new ChatMessage(ChatRole.User, userMessage.Trim())) diff --git a/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs b/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs index 70405e3..f772973 100644 --- a/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs +++ b/MeetingAssistant/Workflow/WorkflowRulesEditorTools.cs @@ -10,6 +10,7 @@ using MeetingAssistant.Transcription; using Microsoft.EntityFrameworkCore; using Microsoft.Extensions.Configuration; using Microsoft.Extensions.Logging; +using Microsoft.Extensions.Logging.Abstractions; using YamlDotNet.Core; using YamlDotNet.Serialization; @@ -42,6 +43,7 @@ public sealed class WorkflowRulesEditorTools private readonly IMeetingMetadataProvider? meetingMetadataProvider; private readonly ILaunchProfileOptionsProvider? launchProfiles; private readonly ISpeakerIdentityMergeService? identityMergeService; + private readonly ResemblyzerVoiceVectorOutlierPruner outlierPruner; private readonly AsrDiagnosticService? asrDiagnosticService; private readonly IConfiguration? configuration; private readonly ILogger? logger; @@ -61,7 +63,8 @@ public sealed class WorkflowRulesEditorTools string? logDirectory = null, string? specRootPath = null, string? projectAgentsTemplatePath = null, - ILogger? logger = null) + ILogger? logger = null, + ResemblyzerVoiceVectorOutlierPruner? outlierPruner = null) { this.options = options; rulesPath = WorkflowRulesPathResolver.Resolve(options.Automation.RulesPath); @@ -82,6 +85,9 @@ public sealed class WorkflowRulesEditorTools this.meetingMetadataProvider = meetingMetadataProvider; this.launchProfiles = launchProfiles; this.identityMergeService = identityMergeService; + this.outlierPruner = outlierPruner ?? new ResemblyzerVoiceVectorOutlierPruner( + speakerOptions.Resemblyzer, + NullLogger.Instance); this.asrDiagnosticService = asrDiagnosticService; this.configuration = configuration; this.logger = logger; @@ -743,8 +749,8 @@ public sealed class WorkflowRulesEditorTools string[]? candidateNames = null) { return Task.FromResult( - "Refused: speaker identities require at least one audio sample. " + - "Use a speaker override from an existing transcript speaker sample, or update/merge an existing sampled identity."); + "Refused: speaker identities require audio evidence (a WAV sample or voice vector). " + + "Use a speaker override from an existing transcript speaker, or update/merge an existing identity."); } public async Task UpdateIdentity( @@ -826,10 +832,12 @@ public sealed class WorkflowRulesEditorTools return $"Could not find target {targetIdentityId} or source {sourceIdentityId}."; } - SpeakerIdentityMerger.MergeInto( + SpeakerIdentityMerger.MergeIntoAndPrune( target, source, - speakerOptions.MaxSnippetsPerSpeaker); + speakerOptions.MaxSnippetsPerSpeaker, + speakerOptions.Resemblyzer.MaxVectorsPerIdentity, + outlierPruner); context.SpeakerIdentities.Remove(source); await context.SaveChangesAsync(); return ToJson(ToIdentityDetail(target)); @@ -1567,9 +1575,11 @@ public sealed class WorkflowRulesEditorTools private static IQueryable LoadIdentities(SpeakerIdentityDbContext context) { return context.SpeakerIdentities + .AsSplitQuery() .Include(identity => identity.Aliases) .Include(identity => identity.CandidateNames) .Include(identity => identity.Snippets) + .Include(identity => identity.VoiceVectors) .Include(identity => identity.References); } @@ -1591,6 +1601,7 @@ public sealed class WorkflowRulesEditorTools identity.Aliases.Select(alias => alias.Name).Order(StringComparer.OrdinalIgnoreCase).ToArray(), identity.CandidateNames.Select(candidate => candidate.Name).Order(StringComparer.OrdinalIgnoreCase).ToArray(), identity.Snippets.Count, + identity.VoiceVectors.Count, identity.References.Count, identity.UpdatedAt); } @@ -1650,6 +1661,7 @@ public sealed class WorkflowRulesEditorTools IReadOnlyList Aliases, IReadOnlyList CandidateNames, int SampleCount, + int VoiceVectorCount, int ReferenceCount, DateTimeOffset UpdatedAt); diff --git a/MeetingAssistant/appsettings.json b/MeetingAssistant/appsettings.json index c75892a..585e142 100644 --- a/MeetingAssistant/appsettings.json +++ b/MeetingAssistant/appsettings.json @@ -120,8 +120,8 @@ "MaxMatchCandidates": 100, "MatchIdentityActiveAge": "365.00:00:00", "MaxSnippetsPerSpeaker": 3, - "MinimumSampleSpeechDuration": "00:00:30", - "MaximumSampleSegmentGap": "00:00:01", + "MinimumSampleSpeechDuration": "00:00:10", + "MaximumSampleDuration": "00:01:00", "SilenceBetweenSnippetsSeconds": 1, "LiveSampleBufferDuration": "00:10:00", "MergeRecentIdentityAge": "14.00:00:00", @@ -142,6 +142,26 @@ "TokenEnv": "HF_TOKEN", "CommandTimeout": "01:00:00" } + }, + "Resemblyzer": { + "Enabled": false, + "RequiredVectorsPerSpeaker": 5, + "MaxVectorsPerIdentity": 1000, + "OutlierPruningMinimumVectors": 20, + "OutlierPruningNeighborSimilarity": 0.75, + "OutlierPruningMinimumNeighbors": 3, + "OutlierPruningMinimumClusterRatio": 0.60, + "MinimumClusterCohesion": 0.75, + "MinimumIdentitySimilarity": 0.75, + "MinimumSimilarityMargin": 0.05, + "ModelId": "resemblyzer-0.1.4-pretrained", + "PackageVersion": "0.1.4", + "PythonCommand": "python", + "TorchVersion": "2.14.0+cpu", + "TorchIndexUrl": "https://download.pytorch.org/whl/cpu", + "WebRtcVadVersion": "2.0.14", + "RuntimeFolder": "%LOCALAPPDATA%\\MeetingAssistant\\Resemblyzer", + "CommandTimeout": "00:15:00" } }, "Automation": { diff --git a/README.md b/README.md index e62c13b..e5c0573 100644 --- a/README.md +++ b/README.md @@ -71,7 +71,8 @@ Agents are intentionally stateful. Depending on the invoked tools, they can chan Local runtime state outside the vault includes: - `%LOCALAPPDATA%\MeetingAssistant\Recordings`: mixed WAV files are normally deleted after completion, and unqueued stale files are deleted at startup. If an Azure stop cannot drain within `Recording:StopProcessingTimeout`, the WAV plus a JSON item under `offline-transcription-backlog` are retained and retried every minute. Both are removed only after successful replay, transcript finalization, and summary processing. -- `%LOCALAPPDATA%\MeetingAssistant\SpeakerIdentity\speaker-identities.db`: SQLite identities, aliases, meeting references, and bounded voice snippets. +- `%LOCALAPPDATA%\MeetingAssistant\SpeakerIdentity\speaker-identities.db`: SQLite identities, aliases, meeting references, bounded WAV snippets for the existing matcher, and separate versioned voice vectors for the optional Resemblyzer matcher. +- `%LOCALAPPDATA%\MeetingAssistant\Resemblyzer`: content-versioned managed Python environments, the local encoder script, and temporary encoder inputs. Per-meeting WAV inputs are deleted after each encoding command. - `%LOCALAPPDATA%\MeetingAssistant\FunASR\models` and `%LOCALAPPDATA%\MeetingAssistant\Pyannote\models`: persistent model, hotword, Hugging Face, and torch caches for optional local backends. - `%TEMP%\MeetingAssistant\Logs\meeting-assistant.log`: application log with four rotated predecessors. Paths, transcript text, agent diagnostics, and provider errors can make these logs sensitive. @@ -84,7 +85,7 @@ Abort is destructive: it removes the active run's note, transcript, context, sum - The default `azure-speech` provider sends mixed meeting audio and dictation phrase hints to Azure AI Speech. Azure-backed speaker matching also sends selected voice audio. - Summary, screenshot OCR, and interactive-agent requests go to the configured OpenAI-compatible Responses endpoint. They can include meeting/transcript/project text, screenshots, configuration, logs, and speaker samples when corresponding tools are used. The checked-in endpoint is a loopback proxy; its ultimate provider, data path, and retention policy are outside this repository. - Outlook Classic access is local COM and reads appointment metadata; it does not provide the primary capture path. -- A managed FunASR run pulls its configured image, removes any same-named container, starts a disposable container privileged by default, publishes the configured host port, and mounts the model/hotword cache. Pyannote may build a local image and starts disposable containers with the input WAV mounted read-only and its model cache read/write. These paths require Docker Desktop or a compatible Docker CLI and may download images/models from external registries. +- A managed FunASR run pulls its configured image, removes any same-named container, starts a disposable container privileged by default, publishes the configured host port, and mounts the model/hotword cache. Pyannote may build a local image and start disposable containers with audio inputs mounted read-only; those paths require Docker Desktop or a compatible Docker CLI. The opt-in Resemblyzer speaker matcher instead provisions an isolated local Python venv with CPU-only dependencies. First use may download images, Python packages, or models from external registries. ## Configuration @@ -97,6 +98,7 @@ The settings with the largest operational effect are: - `Recording:MicrophoneDeviceId`, mix gains, stop timeout, minimum duration, and temporary folder: control capture selection, audio, cleanup, and Azure backlog behavior. - `Recording:InactivitySafeguard`: prompts and can auto-finish a run after no new transcript text; it is not an audio-silence detector. - `LaunchProfiles`: overlay named recording/ASR/agent settings and require distinct hotkeys. +- `SpeakerIdentification:Resemblyzer:Enabled`: selects the local, managed-Python-venv vector matcher for the whole application; when disabled, the existing WAV/Azure path stays active. Five vectors unlock matching by default without capping retained evidence, and mature profiles use configurable fail-safe density clustering to remove likely mixed-speaker outliers. - `Automation:RulesPath`: points to the local YAML rules file, normally ignored `meeting-rules.local.yaml`. - `CalendarRecordingPrompts` and `Screenshots`: control Outlook prompts, capture, attachments, and configured OCR. - `Agent` and `WorkflowRulesEditor`: select the Responses endpoint/model, streaming or non-streaming transport, reasoning, retry, output, and compaction behavior; the available tools are defined by the application. diff --git a/docs/meeting-assistant-configuration.md b/docs/meeting-assistant-configuration.md index b42c3f8..3d9d4ce 100644 --- a/docs/meeting-assistant-configuration.md +++ b/docs/meeting-assistant-configuration.md @@ -268,7 +268,9 @@ Azure returns generic speaker IDs such as `Guest-1` and `Guest-2`, which Meeting ## Speaker Identification -Speaker identity matching keeps candidate samples only after a diarized speaker has at least `SpeakerIdentification:MinimumSampleSpeechDuration` of continuous speech, defaulting to 30 seconds. Adjacent same-speaker segments may be combined when the gap is no larger than `MaximumSampleSegmentGap`, but a different speaker resets the pending span. +Speaker identity matching keeps candidate samples only after a diarized speaker has at least `SpeakerIdentification:MinimumSampleSpeechDuration` of same-speaker audio, defaulting to 10 seconds. Consecutive lines with the same diarized speaker are combined even when any configured STT backend splits them around pauses. This applies to both live and final-only diarization. A different speaker ends the pending span, and `MaximumSampleDuration` caps every recognition WAV at 60 seconds by default. + +Configuration is rejected when the minimum duration is negative, the maximum is not positive, or the minimum exceeds the maximum. The same validation applies to launch-profile overrides. | Setting | Purpose | | --- | --- | @@ -276,12 +278,12 @@ Speaker identity matching keeps candidate samples only after a diarized speaker | `DatabasePath` | SQLite database path for identities, aliases, references, and samples. | | `InitialDelay` | Delay after recording starts before the first live identity pass. | | `Interval` | Interval between live identity passes. | -| `MatchBatchSize` | Number of identities processed per model matching batch. | +| `MatchBatchSize` | Number of identities processed per Azure model matching batch. Resemblyzer scores the full capped candidate set together so its ambiguity margin includes the global runner-up. | | `MaxMatchCandidates` | Maximum known identities considered during one match pass. | | `MatchIdentityActiveAge` | Age window for identities considered active enough for automatic matching. | -| `MaxSnippetsPerSpeaker` | Maximum stored voice snippets retained per speaker identity. | -| `MinimumSampleSpeechDuration` | Minimum uninterrupted same-speaker speech span needed before storing a sample. | -| `MaximumSampleSegmentGap` | Maximum gap allowed when combining adjacent same-speaker segments into one sample. | +| `MaxSnippetsPerSpeaker` | Maximum stored WAV snippets retained per speaker identity by the existing backend. | +| `MinimumSampleSpeechDuration` | Minimum total diarized speaker audio needed in a same-speaker sample; provider-created pauses do not count toward it. | +| `MaximumSampleDuration` | Maximum duration of an extracted speaker-recognition WAV. Defaults to 60 seconds. | | `SilenceBetweenSnippetsSeconds` | Silence padding inserted between snippets during Azure identity matching. | | `LiveSampleBufferDuration` | How long live transcript/audio material is retained for extracting samples. | | `MergeRecentIdentityAge` | Age window used by diagnostics that merge recent duplicate identities. | @@ -298,6 +300,37 @@ Speaker identity matching keeps candidate samples only after a diarized speaker | `MinimumMatchingKnownSnippetRatio` | Required pyannote agreement ratio between the new sample and known snippets for an accepted identity. | | `Diarization` | Nested pyannote settings used for this validation pass. | +`SpeakerIdentification:Resemblyzer` is a separate, application-level speaker-recognition backend and defaults to disabled. Setting its single `Enabled` flag to `true` selects the Resemblyzer identification and merge services for the process lifetime; Azure Speech and pyannote are then not used for identity matching or identity-match validation. Launch profiles do not override this choice. The normal `SpeakerIdentification:Enabled` setting remains the master switch for speaker identification as a whole. + +The selected service reuses temporary WAV samples from diarized same-speaker runs, but starts a fresh span after every accepted sample so one speaker's vector inputs do not overlap. It waits for five samples by default before automatic matching; five is an eligibility threshold, not a storage cap. Additional samples for an unresolved speaker trigger another attempt at the configured interval, even when attendees and speaker labels have not changed. All distinct qualifying vectors from the run are retained up to the per-identity limit, including later evidence collected after a live match has already renamed the transcript. Providers that only diarize during finalization can extract missing non-overlapping samples from the completed mixed WAV. Resemblyzer runs locally in an application-managed Python virtual environment with CPU-only PyTorch, produces 256-value embeddings, and stores only versioned float32 vectors in the identity database; it does not persist the temporary WAV inputs as identity evidence. Explicit speaker assignments from the summary agent may save fewer than five available vectors. + +For a query, the matcher first requires the mean pairwise cosine similarity of its vectors to meet `MinimumClusterCohesion`. It represents each compatible known identity by a normalized centroid, scores it with the median query-to-centroid cosine similarity, and accepts only if the best score meets `MinimumIdentitySimilarity` and beats the runner-up by `MinimumSimilarityMargin`. The runner-up margin is waived when only one identity is scoreable. Logs include measured scores and rejection reasons so these initial thresholds can be calibrated. + +Once an identity has at least `OutlierPruningMinimumVectors` compatible vectors, a separate DBSCAN-style cosine-density pass protects the profile from mixed-speaker diarization errors. A vector is a neighbor when its cosine similarity meets `OutlierPruningNeighborSimilarity`, and a dense point requires `OutlierPruningMinimumNeighbors` neighbors including itself. Vectors outside the uniquely largest dense cluster are removed only when that cluster contains at least `OutlierPruningMinimumClusterRatio` of all compatible vectors. If there is no dense cluster, the largest clusters tie, or the ratio is too low, pruning fails safe and retains every vector. Invalid vectors and vectors produced by other model versions are preserved. + +| Resemblyzer setting | Purpose | +| --- | --- | +| `Enabled` | Selects the complete local Resemblyzer identity backend. Defaults to `false`. | +| `RequiredVectorsPerSpeaker` | Independent vectors required for automatic matching and per-cluster merge validation. Defaults to `5`; this does not cap collection or persistence. | +| `MaxVectorsPerIdentity` | Maximum distinct vectors retained per identity. Defaults to `1000`; merges prefer the newest vectors. | +| `OutlierPruningMinimumVectors` | Compatible vectors required before density-cluster pruning runs. Defaults to `20`. | +| `OutlierPruningNeighborSimilarity` | Minimum cosine similarity for two vectors to count as density neighbors. Defaults to `0.75`. | +| `OutlierPruningMinimumNeighbors` | Neighbors required for a dense point, including the point itself. Defaults to `3`. | +| `OutlierPruningMinimumClusterRatio` | Minimum fraction of compatible vectors that the uniquely largest cluster must contain before other vectors are removed. Defaults to `0.60`. | +| `MinimumClusterCohesion` | Minimum mean pairwise cosine similarity within a query cluster. | +| `MinimumIdentitySimilarity` | Minimum median similarity from query vectors to a known identity centroid. | +| `MinimumSimilarityMargin` | Required difference between the best and runner-up identity scores. | +| `ModelId` | Version identifier persisted with vectors and required for compatible comparisons. | +| `PackageVersion` | Resemblyzer package version installed in the managed virtual environment. | +| `PythonCommand` | Python executable used to create the managed virtual environment. | +| `TorchVersion` | CPU-only PyTorch wheel version installed in the managed environment. | +| `TorchIndexUrl` | HTTPS package index used for the CPU-only PyTorch wheel. | +| `WebRtcVadVersion` | Version of the prebuilt Windows-compatible `webrtcvad-wheels` package. | +| `RuntimeFolder` | Local folder for versioned virtual environments, the encoder script, and short-lived WAV input batches. | +| `CommandTimeout` | Bound for first-time environment provisioning, warm-up, and encoder commands. | + +The first enabled startup creates a content-versioned virtual environment under `RuntimeFolder`. Dependency-version changes select a new environment automatically. The workstation's global Python packages are not modified, and Docker is not required for Resemblyzer speaker recognition. + ## Automation `Automation:RulesPath` points to an optional local YAML rules file. The default `meeting-rules.local.yaml` is ignored by git. Rules can trigger on meeting creation, assistant-context state transitions, identified speakers, or transcript line writes; conditions are evaluated with NCalc-style expressions and step values can use Razor syntax against `Model.Meeting`, `Model.Event`, `Model.Speaker`, and `Model.Transcript`. diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/.openspec.yaml b/openspec/changes/add-resemblyzer-speaker-recognition/.openspec.yaml new file mode 100644 index 0000000..032461f --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-02 diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/design.md b/openspec/changes/add-resemblyzer-speaker-recognition/design.md new file mode 100644 index 0000000..c23e794 --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/design.md @@ -0,0 +1,102 @@ +## Context + +The existing speaker-identification implementation stores bounded WAV snippets and sends composite audio to a dedicated Azure Speech diarization verifier, optionally followed by pyannote validation. Recording already maintains timestamped mixed audio and creates candidate WAV samples for diarized speaker labels. Identity names, aliases, candidate names, meeting references, transcript relabeling, summarizer overrides, deletion, and merges are stored locally in SQLite. + +Resemblyzer 0.1.4 exposes a local `VoiceEncoder` that produces L2-normalized 256-value embeddings. Its upstream examples compare embeddings with dot products, which are cosine similarities for normalized vectors. The package includes its pretrained model but has Python, PyTorch, audio, and native VAD dependencies, so the application isolates them in a managed virtual environment rather than mutating the workstation Python installation. + +The feature must remain opt-in and must not silently mix Resemblyzer vectors with evidence from the WAV/Azure backend. Existing identities remain shared because their names, aliases, references, and downstream behavior are backend-independent, but each backend reads and writes only its own voice evidence. + +## Goals / Non-Goals + +**Goals:** + +- Select a separate local Resemblyzer recognition path with one application-level feature flag that defaults off. +- Convert temporary per-speaker WAV samples to versioned 256-float embeddings and persist only the embeddings for this path. +- Require five independent, coherent query embeddings before automatic matching. +- Make cohesion, acceptance similarity, ambiguity margin, required-vector count, runtime, and per-identity limit configurable. +- Preserve existing naming, attendee, relabeling, override, deletion, reference, and merge outcomes. +- Bound persisted embeddings to 1,000 per identity by default. + +**Non-Goals:** + +- Convert existing WAV snippets to Resemblyzer vectors automatically. +- Use Resemblyzer for ASR speaker diarization; diarized labels still come from the configured transcription backend. +- Run the Azure or pyannote speaker verifier as a second opinion when the Resemblyzer path is selected. +- Guarantee calibrated production thresholds before real meeting data has been observed. + +## Decisions + +### Select a complete backend at the application boundary + +Add `SpeakerIdentification:Resemblyzer:Enabled`, defaulting to `false`. Dependency injection selects either the existing `SpeakerIdentityService`/`SpeakerIdentityMergeService` pair or a separate Resemblyzer identification/merge pair for the process lifetime. Resemblyzer configuration is application-level and launch profiles do not override it because the identity database and selected singleton backend are application-wide. + +The alternative of adding conditional vector branches throughout the existing WAV service was rejected because it would make it easy to mix evidence types or accidentally invoke Azure/pyannote while the local backend is enabled. + +### Reuse temporary sample capture but make vector samples independent + +The recording run retains the configured number of best WAV samples in memory. Consecutive transcript lines with the same diarized speaker are treated as one same-speaker run regardless of which STT backend emitted them, even when that backend splits them around a pause; only an intervening different speaker or the configured maximum sample duration ends the run. Provider-created pauses remain inside the extracted time range but do not count toward its minimum speaker-audio duration. Samples are capped at 60 seconds by default. With Resemblyzer enabled, the collector starts a fresh span after each accepted sample so the five embeddings are based on non-overlapping speech. The WAV bytes are temporary inputs only and are not written to the identity database by the Resemblyzer service. + +For providers that only yield speaker labels during finalization, the service extracts disjoint qualifying spans from the completed mixed WAV. Explicit summarizer assignments may learn from fewer than five valid vectors, but automatic matching and automatic unnamed-candidate learning wait for the configured required count. + +`RequiredVectorsPerSpeaker` is an eligibility threshold, not a collection or persistence cap. The collector retains qualifying non-overlapping samples up to `MaxVectorsPerIdentity`, matching scores the configured minimum high-quality vectors, and an accepted live or final assignment persists every distinct compatible vector available for that meeting. A speaker matched during live transcription receives a final evidence-accumulation pass so samples collected after the initial match are not lost. + +### Persist versioned float32 embeddings in a separate table + +Add a `SpeakerVoiceVectors` table related to `SpeakerIdentities` with cascade deletion. Each row stores a little-endian float32 blob, dimension count, model identifier, SHA-256 fingerprint, and creation timestamp. The model identifier prevents comparisons across incompatible encoder versions. A unique identity/fingerprint index makes retrying the same evidence idempotent. + +Vector rows are capped by `MaxVectorsPerIdentity`, default 1,000. Normal additions stop at the cap. Merges combine distinct rows and keep the most recently created vectors when the combined set exceeds the cap. WAV snippets and voice vectors remain independent collections. + +### Run Resemblyzer in an application-managed Python virtual environment + +`VenvResemblyzerVoiceEncoder` batches WAV files into one invocation of the managed virtual environment's Python executable, preprocesses each file with `preprocess_wav`, calls `VoiceEncoder("cpu").embed_utterance`, and returns marked JSON. The environment is content-versioned from its dependency settings, pins Resemblyzer and a CPU-only PyTorch wheel, and uses `webrtcvad-wheels` on Windows to avoid requiring Visual C++ build tooling. NumPy stays below 2 on Python versions where a compatible NumPy 1.x wheel exists and uses NumPy 2 on Python 3.13 or later. A non-blocking startup warm-up provisions and verifies the environment only when the feature is enabled. + +The encoder validates result count, dimension, finite values, and nonzero magnitude before returning normalized vectors. Temporary input directories are deleted after each bounded invocation. + +Installing packages into the workstation Python environment was rejected because it creates dependency conflicts and upstream `webrtcvad` requires native build tooling on clean Windows systems. The managed venv avoids both issues, while the compatible VAD wheel removes the compiler requirement. Docker was rejected because it adds an unnecessary VM/runtime dependency and caused CPU inference to pull multi-gigabyte CUDA packages from the default Linux PyTorch distribution. A long-lived inference service was deferred until measured process-start overhead warrants the extra lifecycle complexity. + +### Use a coherent-query centroid heuristic with ambiguity rejection + +All input vectors are normalized before scoring. + +1. Query cohesion is the mean pairwise cosine similarity among the required query vectors. A query below `MinimumClusterCohesion` is rejected before identity comparison. +2. Each identity is represented by the normalized centroid of all stored vectors having the configured model identifier. +3. Candidate similarity is the median cosine similarity from the query vectors to that identity centroid. The median limits the effect of one noisy query sample. +4. The best candidate must meet `MinimumIdentitySimilarity` and exceed the runner-up by `MinimumSimilarityMargin`. The margin is waived when there is no runner-up. + +Defaults are five vectors, `0.75` minimum cohesion, `0.75` minimum identity similarity, and `0.05` minimum margin. These are initial conservative values between the same-speaker and different-speaker similarities shown in Resemblyzer's upstream demonstrations; every threshold is configurable for calibration from local logs. + +### Prune mature identity evidence with fail-safe density clustering + +Once an identity has at least `OutlierPruningMinimumVectors` valid vectors for the configured model, defaulting to 20, run a separate DBSCAN-style clustering pass using cosine similarity. Two vectors are neighbors when their similarity meets `OutlierPruningNeighborSimilarity`; a dense point requires `OutlierPruningMinimumNeighbors`, including itself. This distinguishes isolated or small foreign-speaker groups without forcing every vector toward the matching centroid. + +Pruning keeps a uniquely largest dense cluster only when it contains at least `OutlierPruningMinimumClusterRatio` of the compatible evidence, defaulting to 60%. Vectors outside that dominant cluster are removed, while incompatible-model and malformed rows are left untouched. If no dense cluster dominates, retain all evidence and log the ambiguity rather than arbitrarily selecting one voice. Run pruning after vector additions and identity merges; also prune before additions so a full identity can recover capacity previously occupied by outliers. + +### Persist accepted evidence while keeping downstream identity behavior + +When live or finished matching accepts a known identity, the query vectors and meeting reference are added immediately, bounded and deduplicated, and the existing canonical name is used for transcript relabeling and attendee updates. Final processing still performs candidate-name intersection/promotion and creates unmatched candidates using summary-refined attendees. + +Summarizer overrides attach all available valid current-run vectors to the named identity, merge a current-run unnamed candidate when present, and create a named identity only when evidence or such a candidate exists. Identity deletion cascades to both evidence types. + +Diagnostic automatic merge uses two disjoint query clusters and requires both to select the same target, preserving the existing two-pass confirmation rule. Manual merges always move bounded vector evidence along with aliases, candidates, references, and WAV snippets. + +## Risks / Trade-offs + +- [Initial thresholds may be too strict or permissive for mixed microphone/system audio] → Log cohesion, best similarity, runner-up similarity, margin, sample count, and rejection reason; expose every decision threshold in configuration. +- [Five independent 10-second samples can delay recognition] → Keep required count and minimum speech duration configurable; explicit summarizer assignments can seed an identity with fewer vectors. +- [Provider-created pauses can add silence to a same-speaker sample] → Bound every sample to 60 seconds and rely on Resemblyzer preprocessing to remove non-speech before embedding. +- [Existing identities have no vector evidence] → Do not cross-use WAV evidence automatically; identities become matchable after an explicit assignment or new vector-backed learning. +- [First-time virtual-environment provisioning and model startup add latency] → Pin CPU-only dependencies, content-version and reuse the environment, batch samples, warm non-blockingly, bound commands, and serialize encoder invocations to avoid concurrent model memory spikes. +- [A false positive can contaminate an identity with five vectors] → Require query cohesion, an absolute similarity threshold, an ambiguity margin, and two independent clusters for automatic merges. +- [Density clustering could discard a legitimate secondary acoustic mode] → Do not prune below 20 vectors or without a uniquely dominant 60% cluster; expose the neighborhood and dominance settings and log every decision. +- [Changing the encoder model invalidates comparisons] → Store and filter by model identifier; require an explicit configuration/migration decision for future model upgrades. + +## Migration Plan + +1. Apply the additive SQLite table/index migration while the flag remains disabled. +2. Provision and warm the configured local virtual environment, then enable Resemblyzer explicitly. +3. Calibrate thresholds from decision logs and corrected summarizer assignments. +4. Roll back by disabling the feature flag; the existing WAV/Azure backend and its stored snippets remain intact, while vector rows stay dormant. + +## Open Questions + +None. diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/proposal.md b/openspec/changes/add-resemblyzer-speaker-recognition/proposal.md new file mode 100644 index 0000000..0370aae --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/proposal.md @@ -0,0 +1,31 @@ +## Why + +The current speaker-recognition path persists WAV snippets and depends on Azure Speech plus optional pyannote validation. Meeting Assistant needs an opt-in, fully local alternative that persists compact voice embeddings and can accumulate stronger identity evidence over time without replacing the existing backend. + +## What Changes + +- Add an application-level feature flag that selects a separate Resemblyzer speaker-recognition backend while leaving the current WAV/Azure backend unchanged when disabled. +- Create temporary WAV samples during recording, encode each retained sample locally into a 256-value Resemblyzer voice vector, and persist vectors rather than WAV data for this backend. +- Require a configurable minimum of five coherent vectors for automatic recognition, compare their cluster with known identity vector clusters using configurable cosine-similarity, cohesion, and ambiguity thresholds, and learn the accepted vectors. +- Store at most a configurable 1,000 vectors per identity and retain vectors through identity naming, summarizer overrides, deletion, and merge operations. +- Keep transcript relabeling, attendee updates, candidate-name learning, meeting references, and identity-management behavior consistent with the existing speaker-identification flow. +- Add a managed local Python virtual environment for Resemblyzer with CPU-only PyTorch and document its configuration and tuning parameters. +- Merge consecutive transcript lines from any STT backend for the same diarized speaker into recognition samples despite provider-created pauses, while capping every sample at 60 seconds by default. +- Treat five vectors only as the default automatic-decision threshold, retain all qualifying current-run vectors up to the identity limit, and prune accumulated outliers with a configurable density-clustering pass once an identity has at least 20 compatible vectors. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `meeting-transcription`: Add an opt-in local voice-vector speaker-recognition backend and define its collection, matching, persistence, and lifecycle behavior. + +## Impact + +- Speaker identity options, dependency registration, recording sample retention, and live/final identification orchestration. +- SQLite schema and identity merge/management tools gain a separate voice-vector collection. +- A local Python installation is required only when the feature is enabled; Resemblyzer and CPU-only PyTorch are isolated in an application-managed virtual environment. +- Canonical configuration and speaker-identification documentation gain the feature flag and tunable matching thresholds. diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/specs/meeting-transcription/spec.md b/openspec/changes/add-resemblyzer-speaker-recognition/specs/meeting-transcription/spec.md new file mode 100644 index 0000000..3207201 --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/specs/meeting-transcription/spec.md @@ -0,0 +1,361 @@ +## ADDED Requirements + +### Requirement: Speaker recognition can use local Resemblyzer voice vectors +Meeting Assistant SHALL expose `SpeakerIdentification:Resemblyzer:Enabled` as an application-level feature flag that defaults to disabled. + +When the feature is disabled, Meeting Assistant SHALL use the existing WAV-snippet, Azure Speech, and optional pyannote speaker-identification backend without reading or writing Resemblyzer voice vectors. + +When the feature is enabled, Meeting Assistant SHALL use a separate local Resemblyzer speaker-identification backend and SHALL NOT invoke the Azure Speech or pyannote speaker-identity matchers. + +The Resemblyzer backend SHALL create temporary WAV samples from diarized same-speaker runs during recording, SHALL encode each retained sample locally as a versioned 256-value voice vector, and SHALL NOT persist those temporary WAV samples as identity evidence. + +Resemblyzer sample spans retained for one speaker SHALL not overlap. For transcription providers that only produce diarized speakers during finalization, Meeting Assistant SHALL extract qualifying non-overlapping samples from the completed mixed recording. + +Automatic matching SHALL wait until the configured required number of valid vectors is available for a diarized speaker. The default required count SHALL be five. + +The required vector count SHALL be an automatic-decision threshold and SHALL NOT cap collection, encoding, or persistence. After the threshold is met, Meeting Assistant SHALL retain every distinct qualifying current-run vector up to the configured per-identity limit. When a speaker was assigned during live transcription, final processing SHALL attach qualifying vectors collected after that assignment to the same identity. + +The matcher SHALL reject a query cluster whose mean pairwise cosine similarity is below the configured minimum cluster cohesion. For a coherent query, it SHALL represent each known identity by the normalized centroid of compatible stored vectors, SHALL score that identity using the median cosine similarity from query vectors to the centroid, and SHALL select an identity only when the best score meets the configured minimum identity similarity and exceeds the runner-up by the configured minimum similarity margin. The runner-up margin SHALL be waived when only one candidate can be scored. + +The required vector count, minimum cluster cohesion, minimum identity similarity, minimum runner-up margin, encoder model identifier, local runtime settings, and command timeout SHALL be configurable. + +When an identity has at least the configured outlier-pruning minimum number of valid vectors for the active model, defaulting to 20, Meeting Assistant SHALL run a separate cosine-density clustering pass. Neighbor similarity, minimum neighbors, and minimum dominant-cluster ratio SHALL be configurable. + +Meeting Assistant SHALL remove vectors outside the uniquely largest dense cluster only when that cluster meets the configured minimum ratio of compatible evidence, defaulting to 60%. When no cluster qualifies or the largest cluster is tied, Meeting Assistant SHALL retain the evidence and log that pruning was skipped. Vectors for other model identifiers SHALL NOT be removed by this pass. + +When enabled, the local encoder SHALL provision and reuse an application-managed Python virtual environment under the configured runtime folder. It SHALL install a pinned CPU-only PyTorch distribution and Windows-compatible VAD wheel without requiring Docker or a system-wide Python package installation. + +The local encoder SHALL reject missing, malformed, non-finite, zero-magnitude, wrong-count, and wrong-dimension results without persisting them or falling back to the existing remote matcher. + +#### Scenario: Disabled feature preserves existing backend +- **GIVEN** Resemblyzer speaker recognition is disabled +- **WHEN** Meeting Assistant tries to identify a diarized speaker +- **THEN** it uses the existing WAV-snippet speaker-identification backend +- **AND** does not create or compare Resemblyzer voice vectors + +#### Scenario: Automatic matching waits for five vectors +- **GIVEN** Resemblyzer speaker recognition requires five vectors +- **AND** an unresolved diarized speaker has four valid samples +- **WHEN** live speaker identification runs +- **THEN** Meeting Assistant does not compare that speaker with known identities +- **WHEN** a fifth valid sample becomes available +- **THEN** Meeting Assistant can encode and compare the coherent five-vector cluster + +#### Scenario: Five vectors do not cap retained evidence +- **GIVEN** Resemblyzer automatic matching requires five vectors +- **AND** a meeting yields eight distinct qualifying vectors for one speaker +- **WHEN** Meeting Assistant accepts or creates that speaker identity +- **THEN** it stores all eight vectors within the configured identity limit + +#### Scenario: Final processing retains evidence collected after a live match +- **GIVEN** a diarized speaker was matched after five vectors during live transcription +- **AND** three more qualifying vectors were collected later in the meeting +- **AND** the finished transcript already uses the matched speaker's name while retained samples use the original diarized label +- **WHEN** final speaker processing runs with the existing mapping +- **THEN** the three later vectors are attached to the matched identity + +#### Scenario: Mature identity outliers are pruned +- **GIVEN** an identity has at least 20 compatible vectors +- **AND** a uniquely largest cosine-density cluster contains at least 60% of them +- **WHEN** vector evidence is added or identities are merged +- **THEN** vectors outside the dominant cluster are removed +- **AND** the pruning decision and removed count are logged + +#### Scenario: Ambiguous clusters are retained +- **GIVEN** an identity has at least 20 compatible vectors split between equally large or non-dominant dense clusters +- **WHEN** outlier pruning runs +- **THEN** Meeting Assistant removes no vectors +- **AND** logs that no uniquely dominant cluster qualified + +#### Scenario: Incoherent query cluster is rejected +- **GIVEN** five query vectors have mean pairwise cosine similarity below the configured cohesion threshold +- **WHEN** Resemblyzer speaker identification runs +- **THEN** Meeting Assistant does not assign the speaker to a known identity +- **AND** logs the measured cohesion and rejection reason + +#### Scenario: Similar and unambiguous cluster is accepted +- **GIVEN** a coherent five-vector query cluster +- **AND** its median similarity to Chris's vector centroid meets the configured identity threshold +- **AND** its score exceeds every other scored identity by the configured margin +- **WHEN** Resemblyzer speaker identification runs +- **THEN** Meeting Assistant identifies the diarized speaker as Chris +- **AND** adds the five query vectors to Chris's identity within the configured limit + +#### Scenario: Ambiguous best cluster is rejected +- **GIVEN** a coherent five-vector query cluster meets the identity similarity threshold for Chris +- **AND** another identity's score is within the configured runner-up margin +- **WHEN** Resemblyzer speaker identification runs +- **THEN** Meeting Assistant leaves the diarized speaker unresolved +- **AND** logs both candidate scores and the insufficient margin + +#### Scenario: Encoder failure preserves diarized labels +- **GIVEN** Resemblyzer speaker recognition is enabled +- **WHEN** the local encoder fails or returns invalid vectors +- **THEN** Meeting Assistant does not invoke the existing Azure or pyannote identity matcher as a fallback +- **AND** keeps the available diarized speaker labels + +#### Scenario: Encoder provisions an isolated CPU environment +- **GIVEN** Resemblyzer speaker recognition is enabled +- **AND** its versioned virtual environment is not ready +- **WHEN** encoder warm-up runs +- **THEN** Meeting Assistant creates the virtual environment with the configured Python command +- **AND** installs the configured CPU-only PyTorch, Windows-compatible VAD, and Resemblyzer versions inside that environment +- **AND** does not invoke Docker + +## MODIFIED Requirements + +### Requirement: Speaker identity samples require uninterrupted speech +Meeting Assistant SHALL only retain speaker identity samples after a diarized speaker has produced a same-speaker sample span meeting the configured minimum duration. + +The default minimum sample duration SHALL be 10 seconds. + +Meeting Assistant SHALL combine consecutive transcript segments for the same diarized speaker into one sample span even when the transcription provider splits those segments around pauses. Provider-created pauses SHALL remain in the bounded extracted WAV but SHALL NOT count toward the configured minimum speaker-audio duration. + +This aggregation behavior SHALL apply uniformly to diarized segments from every configured STT backend, whether segments arrive during live transcription or become available during finalization. + +Meeting Assistant SHALL end the pending span when a different diarized speaker interrupts it or when the configured maximum sample duration is reached. The default maximum sample duration SHALL be 60 seconds, and no extracted recognition WAV SHALL exceed it. + +When Resemblyzer recognition is enabled, Meeting Assistant SHALL start a new non-overlapping sample after accepting the previous sample from the same speaker. + +#### Scenario: Short speaker span is not retained +- **GIVEN** the configured minimum sample duration is 10 seconds +- **WHEN** a diarized speaker produces only 8 seconds of uninterrupted speech +- **THEN** Meeting Assistant does not retain a speaker identity sample for that span + +#### Scenario: Adjacent same-speaker segments form a sample +- **GIVEN** the configured minimum sample duration is 10 seconds +- **WHEN** a diarized speaker produces consecutive provider segments containing at least 10 seconds of speaker audio without another speaker interrupting +- **THEN** Meeting Assistant retains one speaker identity sample covering the continuous span + +#### Scenario: Provider pause does not split a same-speaker sample +- **GIVEN** any configured STT backend emits consecutive lines for `Guest01` with a pause longer than the former segment-gap threshold +- **WHEN** no differently labeled speaker appears between those lines +- **THEN** Meeting Assistant combines the lines into one speaker-recognition sample span +- **AND** counts only their diarized speaker-audio durations toward the minimum + +#### Scenario: Speaker sample is capped at 60 seconds +- **GIVEN** the maximum sample duration is 60 seconds +- **WHEN** consecutive transcript lines for one speaker span more than 60 seconds +- **THEN** every extracted speaker-recognition WAV is at most 60 seconds long + +#### Scenario: Different speaker interrupts pending span +- **GIVEN** the configured minimum sample duration is 10 seconds +- **WHEN** `Guest01` speaks for 8 seconds and then `Guest02` speaks +- **THEN** Meeting Assistant discards the pending `Guest01` span instead of retaining or later extending it + +### Requirement: Meeting Assistant learns speaker identities locally +Meeting Assistant SHALL maintain a local SQLite speaker identity database in the user's application data folder. + +The speaker identity database SHALL store speaker identities, optional canonical names, aliases, candidate names, meeting file references, a bounded set of WAV snippets per identity for the existing backend, and a separate bounded set of versioned voice vectors per identity for the Resemblyzer backend. + +Each persisted voice vector SHALL store its model identifier, dimension, creation time, and a fingerprint that makes adding the same vector to the same identity idempotent. + +Meeting file references SHALL include the meeting note file address and the transcript file address. + +Meeting Assistant SHALL calculate speaker identity participation counts from meeting file references when needed instead of persisting a denormalized transcript count. + +Each speaker identity SHALL store a last-modified timestamp used by active-age filtering, and Meeting Assistant SHALL update it whenever the identity is created or modified by identification, candidate updates, snippet changes, voice-vector changes, reference changes, or merge operations. + +The configured maximum snippet count and maximum voice-vector count per identity SHALL prevent unbounded growth. The default maximum voice-vector count SHALL be 1,000. + +Except for adding newly accepted Resemblyzer match evidence and its meeting reference, final candidate elimination, canonical promotion, and new unmatched identity creation SHALL happen only after transcription is finished and after automatic summary generation has completed, using the latest meeting note frontmatter. + +When the summary agent records a speaker override from a diarized transcript label to a named speaker, final speaker identity processing SHALL attach the current run's evidence to an existing identity with that name when one exists, or create a new canonical speaker identity with that name when none exists. For the existing backend that evidence SHALL be the resolved WAV snippet; for the Resemblyzer backend it SHALL be all available valid current-run voice vectors up to the configured per-run count. Meeting Assistant SHALL NOT create a new speaker identity for an override when no current run evidence or current run candidate can be resolved for the source speaker label. + +When a speaker override maps a current-run unnamed candidate to an existing named identity, Meeting Assistant SHALL merge the candidate's meeting reference and useful backend-specific evidence into the named identity instead of leaving a duplicate candidate. + +When the summary agent records that a speaker identity was wrongfully matched, final speaker identity processing SHALL delete the matching identity and all of its WAV and voice-vector evidence from the local speaker identity database so it cannot be matched again unless it is newly created in the future. + +#### Scenario: Unknown speaker is learned from meeting attendees +- **WHEN** a finished transcript contains an unmatched diarized speaker and the meeting note has attendees +- **THEN** Meeting Assistant stores a new unnamed speaker identity with candidate names from the attendees that were not already matched in that meeting +- **AND** stores a meeting file reference for that identity + +#### Scenario: Speaker snippets are bounded +- **WHEN** Meeting Assistant adds a snippet for an identity that already has the configured maximum number of snippets +- **THEN** Meeting Assistant does not store more snippets for that identity + +#### Scenario: Speaker voice vectors are bounded +- **GIVEN** the Resemblyzer vector limit is 1,000 +- **WHEN** Meeting Assistant adds vectors to an identity that already has 1,000 stored vectors +- **THEN** Meeting Assistant does not store more than 1,000 vectors for that identity + +#### Scenario: Retried vector evidence is idempotent +- **GIVEN** an identity already contains a voice vector +- **WHEN** Meeting Assistant retries adding the same vector to that identity +- **THEN** it stores only one copy of that vector + +#### Scenario: Identity modification updates active-age timestamp +- **WHEN** Meeting Assistant creates, identifies, updates candidates for, stores snippets or voice vectors for, stores references for, or merges a speaker identity +- **THEN** Meeting Assistant updates that identity's last-modified timestamp + +#### Scenario: Final speaker identity learning uses summary-refined attendees +- **GIVEN** the summary agent changes meeting note attendees during automatic summary generation +- **WHEN** Meeting Assistant performs final speaker identity learning and candidate creation +- **THEN** it uses the attendee list from the meeting note after the summary agent changes + +#### Scenario: Speaker override attaches to existing identity +- **GIVEN** the speaker identity database contains canonical speaker `Sabrina` +- **AND** the summary agent records that transcript speaker `Guest-01` is `Sabrina` +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant stores the meeting reference and current backend-specific speaker evidence on Sabrina's identity +- **AND** does not create a separate unnamed candidate for `Guest-01` + +#### Scenario: Resemblyzer override stores available vectors +- **GIVEN** Resemblyzer speaker recognition is enabled +- **AND** the current run has three valid vectors for `Guest-01` +- **WHEN** the summary agent assigns `Guest-01` to `Sabrina` +- **THEN** Meeting Assistant attaches those three vectors to Sabrina's identity +- **AND** does not require five vectors for the explicit assignment + +#### Scenario: Speaker override creates named identity +- **GIVEN** the speaker identity database has no accepted name `Sabrina` +- **AND** the summary agent records that transcript speaker `Guest-01` is `Sabrina` +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant creates a canonical speaker identity named `Sabrina` +- **AND** stores the meeting reference and current backend-specific speaker evidence on that identity + +#### Scenario: Speaker override with missing source sample is skipped +- **GIVEN** the speaker identity database has no accepted name `Sabrina` +- **AND** the summary agent records that transcript speaker `Guest-5` is `Sabrina` +- **AND** final speaker identity processing has no sample, vector, or segment for `Guest-5` +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant does not create a speaker identity for `Sabrina` + +#### Scenario: Speaker identity deletion removes a wrong match +- **GIVEN** the speaker identity database contains canonical speaker `Sabrina` +- **AND** the summary agent records that `Sabrina` was wrongfully matched +- **WHEN** final speaker identity processing runs +- **THEN** Meeting Assistant removes Sabrina's identity and backend-specific evidence from the speaker identity database +- **AND** the relabeled transcript uses `Removed-1` instead of `Sabrina` + +### Requirement: Speaker identities can be merged diagnostically +Meeting Assistant SHALL expose a diagnostic endpoint that merges duplicate speaker identities. + +The merge process SHALL compare recently-created identities, using a configurable recent age that defaults to two weeks, against all other identities using the selected backend's candidate-scoring strategy. + +For the existing WAV backend, the merge process SHALL require a match and a second validation match using a different source sample. For the Resemblyzer backend, it SHALL require two disjoint coherent source-vector clusters to select the same target identity. + +When identities are merged, Meeting Assistant SHALL retain one identity, move useful names from the merged identity into aliases, combine meeting file references, retain bounded sets of snippets and voice vectors from both identities, and append an audit line to each referenced transcript in the form ` and were merged`. + +When combined Resemblyzer evidence exceeds the configured vector limit, Meeting Assistant SHALL keep no more than that limit, preferring the most recently created distinct vectors. + +After combining Resemblyzer evidence, Meeting Assistant SHALL apply the configured mature-identity outlier-pruning policy. + +#### Scenario: Recently-created duplicate identity is merged +- **GIVEN** a recently-created identity and an older identity have matching backend-specific speaker evidence +- **WHEN** the diagnostic merge endpoint is triggered +- **THEN** Meeting Assistant validates the match twice with different source evidence +- **AND** merges the recent identity into the older identity +- **AND** stores the recent identity name as an alias on the retained identity +- **AND** keeps meeting file references and bounded backend-specific evidence from both identities +- **AND** appends the merge audit line to the referenced transcripts + +#### Scenario: Resemblyzer merge needs two clusters +- **GIVEN** Resemblyzer speaker recognition requires five vectors per cluster +- **AND** a recent identity has ten vectors split into two coherent clusters +- **WHEN** both clusters independently match the same target identity +- **THEN** Meeting Assistant merges the recent identity into that target + +#### Scenario: Old identities are not used as merge sources +- **GIVEN** two identities older than the configured recent age +- **WHEN** the diagnostic merge endpoint is triggered +- **THEN** Meeting Assistant does not compare them as source identities + +### Requirement: Speaker identity matches relabel transcripts +Meeting Assistant SHALL attempt to match unknown diarized speaker evidence against known speaker identities ordered by calculated meeting reference count. + +When Resemblyzer speaker recognition is disabled, matching SHALL use the existing dedicated Azure Speech diarization verifier and optional pyannote validator with WAV snippets. When Resemblyzer speaker recognition is enabled, matching SHALL instead use only compatible locally calculated Resemblyzer voice-vector clusters. + +For the existing WAV backend, the matcher SHALL test at most the configured batch size of known people per matching round and continue with later batches until a match is found or no candidates remain. The Resemblyzer backend SHALL score the capped candidate set together so ambiguity is measured against the global runner-up. + +The matcher SHALL prioritize identities whose canonical name or aliases match current meeting attendees. + +After attendee-matched identities, the matcher SHALL order identities by calculated meeting reference count, filter out non-attendee identities whose last update is older than the configured active age, and cap the candidate set at the configured maximum match candidate count. + +When a match is confirmed, Meeting Assistant SHALL store a meeting file reference and the accepted backend-specific evidence for that identity within its configured limit. + +When a match is confirmed and the identity has a canonical name, Meeting Assistant SHALL rewrite finished transcript segments for that diarized speaker with the canonical name. + +When a match is confirmed and the matched speaker is not already listed in meeting note attendees by display name or alias, Meeting Assistant SHALL add the speaker display name to the attendee list. + +When a match is confirmed and the meeting note attendees contain both the speaker display name and one or more accepted aliases for that same speaker, Meeting Assistant SHALL remove the alias attendee entries and keep the display name entry. + +When Meeting Assistant writes attendees from calendar metadata, it SHALL match attendee display names exactly against known identity canonical names and aliases, replace matches with the identity display name, and deduplicate attendees that map to the same identity. + +#### Scenario: Finished transcript is relabeled after a confirmed match +- **GIVEN** the speaker identity database contains canonical speaker `Chris` +- **WHEN** a finished transcript has diarized speaker `Guest03` and the selected matching backend confirms it is `Chris` +- **THEN** Meeting Assistant rewrites `Guest03` segments in the transcript as `Chris` + +#### Scenario: Confirmed match stores meeting reference +- **GIVEN** the speaker identity database contains canonical speaker `Chris` +- **WHEN** a finished transcript has diarized speaker `Guest03` and the selected matching backend confirms it is `Chris` +- **THEN** Meeting Assistant stores the meeting note and transcript file addresses as a reference for `Chris` +- **AND** stores the accepted backend-specific evidence within its configured limit + +#### Scenario: Confirmed match removes duplicate aliases +- **GIVEN** the speaker identity database contains canonical speaker `Christopher` with alias `Chris` +- **AND** the meeting note attendees contain both `Christopher` and `Chris ` +- **WHEN** live or final speaker matching confirms a diarized speaker is `Christopher` +- **THEN** Meeting Assistant keeps `Christopher` in the meeting note attendees +- **AND** removes `Chris ` from the meeting note attendees + +### Requirement: Speaker matching runs during active transcription +Meeting Assistant SHALL start speaker identity matching only after the configured initial transcription duration has elapsed. + +For backends that emit live diarized transcript segments, Meeting Assistant SHALL keep a bounded in-memory sliding audio buffer with chunk timestamps and extract candidate WAV samples from that buffer when live diarized segments arrive. + +Meeting Assistant SHALL keep only the configured best candidate samples per diarized speaker in memory. Better samples SHALL be preferred when the segment looks like a continuous medium-length sentence. When Resemblyzer is enabled, accepted samples for one speaker SHALL be non-overlapping and the retained count SHALL be at least the configured required vector count. + +Meeting Assistant SHALL periodically match unresolved diarized speaker evidence while transcription is active and attempt to match it against the local identity database. + +Meeting Assistant SHALL run live matching incrementally at the configured interval only when at least one new unmapped diarized speaker sample appears or the meeting note attendee frontmatter changes while unmapped speaker samples still exist. For Resemblyzer, additional samples for an existing unresolved speaker SHALL also trigger another attempt so an earlier insufficient-vector result does not suppress matching when the required count becomes available. + +When a speaker is matched during transcription, Meeting Assistant SHALL rewrite already-written live transcript segments for that diarized speaker and write future transcript segments using the canonical name. + +For the existing WAV backend, live speaker matching SHALL be read-only with respect to the speaker identity database. For the Resemblyzer backend, a confirmed live match SHALL persist the accepted deduplicated voice vectors and meeting reference immediately so an assignment made during transcription is learned. Candidate elimination, canonical promotion, and new unmatched identity creation SHALL still happen only after transcription is finished and after automatic summary generation has completed, using the latest meeting note frontmatter. + +For backends that only provide diarization after finalization, Meeting Assistant SHALL defer speaker identity matching until finished diarization is available, extract candidate samples from the completed temporary recording, complete identity matching, and only then allow summary generation to start. + +#### Scenario: Matching waits for useful speech duration +- **WHEN** transcription has been active for less than the configured speaker identification initial delay +- **THEN** Meeting Assistant does not run speaker identity matching yet + +#### Scenario: Live matching uses in-memory speaker samples +- **WHEN** a live diarized transcript segment identifies an unresolved speaker +- **THEN** Meeting Assistant extracts a temporary WAV sample for that segment from the in-memory sliding audio buffer +- **AND** uses retained speaker evidence for live identity matching without reading the temporary recording file + +#### Scenario: Live match rewrites current and future transcript writes +- **WHEN** periodic matching confirms that diarized speaker `Guest03` is canonical speaker `Chris` +- **THEN** already-written live transcript segments for `Guest03` are rewritten as `Chris` +- **AND** later live transcript segments for `Guest03` are written as `Chris` + +#### Scenario: Resemblyzer live match persists vectors +- **GIVEN** Resemblyzer speaker recognition is enabled +- **WHEN** periodic matching confirms a coherent five-vector cluster for `Guest03` as canonical speaker `Chris` +- **THEN** Meeting Assistant stores those vectors and the current meeting reference on Chris's identity +- **AND** does not persist the temporary WAV samples + +#### Scenario: New live speaker triggers another identification round +- **GIVEN** live matching already checked the current unresolved speaker samples +- **WHEN** a new unmapped diarized speaker sample appears +- **THEN** Meeting Assistant runs another live matching round at the next configured interval + +#### Scenario: Additional samples unlock live Resemblyzer matching +- **GIVEN** an earlier live attempt had fewer than five samples for an unresolved speaker +- **AND** no new speaker or attendee change occurs +- **WHEN** that speaker accumulates five qualifying samples +- **THEN** Meeting Assistant attempts matching again at the next configured interval +- **AND** does not repeatedly match unchanged evidence + +#### Scenario: Attendee changes trigger another identification round +- **GIVEN** live matching already checked unresolved speaker samples +- **WHEN** the meeting note attendee frontmatter changes +- **THEN** Meeting Assistant runs another live matching round at the next configured interval using the latest attendees + +#### Scenario: Final decisions use summary-refined attendees +- **WHEN** live matching finds or does not find a possible speaker identity during transcription +- **THEN** Meeting Assistant does not eliminate candidate names, promote canonical names, or create unmatched identities during that live pass +- **AND** the final speaker identity pass uses the latest meeting note attendees after transcription finishes diff --git a/openspec/changes/add-resemblyzer-speaker-recognition/tasks.md b/openspec/changes/add-resemblyzer-speaker-recognition/tasks.md new file mode 100644 index 0000000..967a772 --- /dev/null +++ b/openspec/changes/add-resemblyzer-speaker-recognition/tasks.md @@ -0,0 +1,73 @@ +## 1. Voice-vector persistence + +- [x] 1.1 Add a failing database behavior test for versioned, deduplicated voice vectors and cascade deletion +- [x] 1.2 Add the voice-vector entity, EF mapping, and additive SQLite schema migration + +## 2. Local Resemblyzer encoding + +- [x] 2.1 Add a failing encoder behavior test for batching WAV samples into validated 256-value vectors +- [x] 2.2 Implement the bounded local Resemblyzer encoder and non-blocking feature-gated warm-up +- [x] 2.3 Add behavior coverage for malformed, wrong-dimension, non-finite, and failed encoder results + +## 3. Tunable cluster matching + +- [x] 3.1 Add a failing behavior test for accepting a coherent, similar, unambiguous five-vector cluster +- [x] 3.2 Implement normalized-centroid, median-cosine, cohesion, threshold, and runner-up-margin scoring +- [x] 3.3 Add behavior coverage for insufficient, incoherent, below-threshold, and ambiguous clusters + +## 4. Resemblyzer identity lifecycle + +- [x] 4.1 Add a failing service behavior test proving a live vector match relabels the speaker and persists five vectors without WAV snippets +- [x] 4.2 Implement the separate Resemblyzer identification service with existing candidate ordering, naming, attendee, reference, and transcript outcomes +- [x] 4.3 Add and pass behavior tests for summary overrides with fewer than five vectors, unmatched learning, deduplication, and the 1,000-vector cap + +## 5. Recording and merge integration + +- [x] 5.1 Add behavior tests and implement non-overlapping Resemblyzer sample collection with at least the configured required count +- [x] 5.2 Add behavior tests and implement application-level feature selection without Azure/pyannote identity fallback +- [x] 5.3 Add behavior tests and implement two-cluster Resemblyzer diagnostic merging plus bounded vector retention in manual merges +- [x] 5.4 Expose vector counts in identity-management tools while keeping WAV playback operations separate + +## 6. Configuration and verification + +- [x] 6.1 Add the disabled-by-default canonical configuration and document runtime, persistence, and tuning behavior +- [x] 6.2 Run focused speaker, schema, encoder, matching, merge, recording, and workflow-tool tests +- [x] 6.3 Run the full solution test suite and validate the OpenSpec change strictly + +## 7. Local virtual-environment correction + +- [x] 7.1 Add a failing behavior test for provisioning a versioned venv with CPU-only PyTorch and no Docker command +- [x] 7.2 Replace the Docker encoder with managed-venv provisioning and direct venv Python batch encoding +- [x] 7.3 Replace Docker-specific Resemblyzer configuration and documentation with Python/venv settings +- [x] 7.4 Run focused and full tests, validate OpenSpec strictly, and verify enabled application warm-up through logs + +## 8. Sample-duration tuning + +- [x] 8.1 Add a failing configuration-default test for a 10-second minimum speaker sample +- [x] 8.2 Change the sample-duration default and canonical configuration to 10 seconds and update documentation +- [x] 8.3 Run focused tests, validate OpenSpec strictly, restart the enabled application, and verify health + +## 9. STT segment aggregation and sample cap + +- [x] 9.1 Add a failing live-collector behavior test proving consecutive same-speaker STT lines survive provider-created pauses +- [x] 9.2 Implement same-speaker aggregation for live and finalized samples while excluding provider pauses from the minimum speech duration +- [x] 9.3 Add a failing behavior test proving recognition WAVs never exceed the configurable 60-second default +- [x] 9.4 Implement and document the maximum sample duration across live and finalized collection +- [x] 9.5 Run focused and full tests, refactor, validate OpenSpec strictly, and verify operational readiness without interrupting active work + +## 10. Complete evidence retention and outlier pruning + +- [x] 10.1 Add a failing service behavior test proving five vectors unlock a decision without capping all qualifying current-run evidence +- [x] 10.2 Retain, encode, and persist all qualifying current-run vectors up to the configured identity limit, including evidence collected after a live match +- [x] 10.3 Add failing behavior tests for dominant density-cluster pruning and fail-safe ambiguous-cluster retention at the 20-vector floor +- [x] 10.4 Implement configurable cosine-density outlier pruning after vector additions and identity merges +- [x] 10.5 Document tuning settings, run focused/full tests and sequential refactor passes, validate OpenSpec strictly, and verify the enabled application without interrupting active work + +## 11. Release verification corrections + +- [x] 11.1 Reproduce and fix final evidence retention after live transcript relabeling +- [x] 11.2 Reproduce and fix live matching when an existing speaker reaches the required sample count +- [x] 11.3 Run focused/full tests, strictly validate OpenSpec, and verify operational readiness +- [x] 11.4 Split evidence-loading queries to avoid multiplying stored WAV blobs across vector, reference, and name rows + +Release verification (2026-09-11): 128 initial focused tests passed. Both lifecycle regressions were reproduced and fixed; the full suite passed all 519 tests with compilation complete after two timing-sensitive audio tests failed during the concurrent Windows build and passed in isolation. The Windows target built successfully and strict OpenSpec validation passed. Local `/health` returned `ok`, recording status was idle, and application logs showed successful Resemblyzer warm-up, local encoding, and vector pruning. The release corrections were verified through behavior tests; the running workstation process was not restarted.