forked from Manuel/meeting-assistant
feat: add local Resemblyzer speaker recognition
This commit is contained in:
@@ -34,6 +34,7 @@ public sealed class MeetingRecordingCoordinator
|
||||
private readonly IMeetingInactivityClock inactivityClock;
|
||||
private readonly IOfflineTranscriptionBacklog offlineTranscriptionBacklog;
|
||||
private readonly MeetingAssistantOptions options;
|
||||
private readonly SpeakerSampleCollectionPolicy speakerSampleCollectionPolicy;
|
||||
private readonly ILogger<MeetingRecordingCoordinator> logger;
|
||||
private readonly SemaphoreSlim gate = new(1, 1);
|
||||
private RecordingRun? currentRun;
|
||||
@@ -63,7 +64,8 @@ public sealed class MeetingRecordingCoordinator
|
||||
IMeetingRunArtifactCleaner? artifactCleaner = null,
|
||||
IMeetingInactivityPromptService? inactivityPromptService = null,
|
||||
IMeetingInactivityClock? inactivityClock = null,
|
||||
IOfflineTranscriptionBacklog? offlineTranscriptionBacklog = null)
|
||||
IOfflineTranscriptionBacklog? offlineTranscriptionBacklog = null,
|
||||
SpeakerSampleCollectionPolicy? speakerSampleCollectionPolicy = null)
|
||||
{
|
||||
this.audioSource = audioSource;
|
||||
this.speechRecognitionPipelineFactory = speechRecognitionPipelineFactory;
|
||||
@@ -86,6 +88,8 @@ public sealed class MeetingRecordingCoordinator
|
||||
this.inactivityClock = inactivityClock ?? new SystemMeetingInactivityClock();
|
||||
this.offlineTranscriptionBacklog = offlineTranscriptionBacklog ?? NoopOfflineTranscriptionBacklog.Instance;
|
||||
this.options = options.Value;
|
||||
this.speakerSampleCollectionPolicy = speakerSampleCollectionPolicy
|
||||
?? SpeakerSampleCollectionPolicy.ExistingBackend;
|
||||
this.logger = logger;
|
||||
}
|
||||
|
||||
@@ -290,7 +294,9 @@ public sealed class MeetingRecordingCoordinator
|
||||
startedAt,
|
||||
launchProfile.Name,
|
||||
runOptions.SpeakerIdentification.LiveSampleBufferDuration,
|
||||
runOptions.SpeakerIdentification.MaxSnippetsPerSpeaker,
|
||||
speakerSampleCollectionPolicy.ResolveRetainedSampleLimit(
|
||||
runOptions.SpeakerIdentification.MaxSnippetsPerSpeaker),
|
||||
speakerSampleCollectionPolicy.RequireNonOverlappingSamples,
|
||||
logger);
|
||||
run.Task = Task.Run(() => RecordAsync(run), CancellationToken.None);
|
||||
if (ShouldRunInactivitySafeguard(runOptions.Recording.InactivitySafeguard))
|
||||
@@ -1169,11 +1175,11 @@ public sealed class MeetingRecordingCoordinator
|
||||
try
|
||||
{
|
||||
var meetingNote = await meetingNoteStore.ReadAsync(run.MeetingNotePath, cancellationToken);
|
||||
var checkpoint = run.CreateLiveIdentificationCheckpoint(meetingNote);
|
||||
var checkpoint = run.CreateLiveIdentificationCheckpoint(meetingNote, samples);
|
||||
if (checkpoint is null)
|
||||
{
|
||||
logger.LogInformation(
|
||||
"Skipping live speaker identity matching because no new unmapped sample speakers or attendee changes were found");
|
||||
"Skipping live speaker identity matching because no new eligible speaker evidence or attendee changes were found");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2079,6 +2085,7 @@ public sealed class MeetingRecordingCoordinator
|
||||
string launchProfileName,
|
||||
TimeSpan liveSampleBufferDuration,
|
||||
int maxSpeakerSamples,
|
||||
bool requireNonOverlappingSpeakerSamples,
|
||||
ILogger logger)
|
||||
{
|
||||
CaptureCancellationSource = captureCancellation;
|
||||
@@ -2099,8 +2106,9 @@ public sealed class MeetingRecordingCoordinator
|
||||
liveSampleBufferDuration,
|
||||
maxSpeakerSamples,
|
||||
options.SpeakerIdentification.MinimumSampleSpeechDuration,
|
||||
options.SpeakerIdentification.MaximumSampleSegmentGap,
|
||||
logger);
|
||||
options.SpeakerIdentification.MaximumSampleDuration,
|
||||
logger,
|
||||
requireNonOverlappingSpeakerSamples);
|
||||
}
|
||||
|
||||
public CancellationTokenSource CaptureCancellationSource { get; }
|
||||
@@ -2508,9 +2516,11 @@ public sealed class MeetingRecordingCoordinator
|
||||
return speakerSampleCollector.Snapshot();
|
||||
}
|
||||
|
||||
public LiveIdentificationCheckpoint? CreateLiveIdentificationCheckpoint(MeetingNote meetingNote)
|
||||
public LiveIdentificationCheckpoint? CreateLiveIdentificationCheckpoint(
|
||||
MeetingNote meetingNote,
|
||||
IReadOnlyList<SpeakerAudioSample> samples)
|
||||
{
|
||||
var speakers = GetUnmappedSampleSpeakers();
|
||||
var speakers = GetUnmappedSampleSpeakers(samples);
|
||||
if (speakers.Count == 0)
|
||||
{
|
||||
return null;
|
||||
@@ -2518,7 +2528,11 @@ public sealed class MeetingRecordingCoordinator
|
||||
|
||||
var checkpoint = new LiveIdentificationCheckpoint(
|
||||
speakers,
|
||||
BuildAttendeeSignature(meetingNote));
|
||||
BuildAttendeeSignature(meetingNote),
|
||||
Options.SpeakerIdentification.Resemblyzer.Enabled
|
||||
? speakers.Select(speaker => samples.Count(sample =>
|
||||
string.Equals(sample.Speaker, speaker, StringComparison.OrdinalIgnoreCase))).ToArray()
|
||||
: []);
|
||||
lock (liveIdentificationGate)
|
||||
{
|
||||
return lastLiveIdentificationCheckpoint?.Matches(checkpoint) == true
|
||||
@@ -2576,9 +2590,9 @@ public sealed class MeetingRecordingCoordinator
|
||||
StringComparer.OrdinalIgnoreCase);
|
||||
}
|
||||
|
||||
private IReadOnlyList<string> GetUnmappedSampleSpeakers()
|
||||
private IReadOnlyList<string> GetUnmappedSampleSpeakers(IReadOnlyList<SpeakerAudioSample> samples)
|
||||
{
|
||||
return GetSpeakerSamplesSnapshot()
|
||||
return samples
|
||||
.Select(sample => sample.Speaker)
|
||||
.Where(speaker => !string.IsNullOrWhiteSpace(speaker))
|
||||
.Where(speaker => !string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase))
|
||||
@@ -2634,11 +2648,13 @@ public sealed class MeetingRecordingCoordinator
|
||||
|
||||
public sealed record LiveIdentificationCheckpoint(
|
||||
IReadOnlyList<string> Speakers,
|
||||
string AttendeeSignature)
|
||||
string AttendeeSignature,
|
||||
IReadOnlyList<int> SampleCounts)
|
||||
{
|
||||
public bool Matches(LiveIdentificationCheckpoint other)
|
||||
{
|
||||
return string.Equals(AttendeeSignature, other.AttendeeSignature, StringComparison.Ordinal) &&
|
||||
SampleCounts.SequenceEqual(other.SampleCounts) &&
|
||||
Speakers.Count == other.Speakers.Count &&
|
||||
Speakers.SequenceEqual(other.Speakers, StringComparer.OrdinalIgnoreCase);
|
||||
}
|
||||
|
||||
@@ -9,37 +9,31 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
private readonly object gate = new();
|
||||
private readonly RollingAudioBuffer audioBuffer;
|
||||
private readonly Dictionary<string, List<SpeakerAudioSample>> samplesBySpeaker = new(StringComparer.OrdinalIgnoreCase);
|
||||
private readonly Dictionary<string, TimeSpan> lastAcceptedEndBySpeaker = new(StringComparer.OrdinalIgnoreCase);
|
||||
private readonly int maxSamplesPerSpeaker;
|
||||
private readonly TimeSpan minimumUninterruptedSpeechDuration;
|
||||
private readonly TimeSpan maximumSegmentGap;
|
||||
private readonly TimeSpan minimumSampleSpeechDuration;
|
||||
private readonly TimeSpan maximumSampleDuration;
|
||||
private readonly bool requireNonOverlappingSamples;
|
||||
private readonly ILogger? logger;
|
||||
private PendingSpeakerSpan? pendingSpan;
|
||||
|
||||
public SpeakerAudioSampleCollector(TimeSpan bufferDuration, int maxSamplesPerSpeaker)
|
||||
: this(
|
||||
bufferDuration,
|
||||
maxSamplesPerSpeaker,
|
||||
TimeSpan.FromSeconds(30),
|
||||
TimeSpan.FromSeconds(1),
|
||||
logger: null)
|
||||
{
|
||||
}
|
||||
private SpeakerSampleSpan? pendingSpan;
|
||||
|
||||
public SpeakerAudioSampleCollector(
|
||||
TimeSpan bufferDuration,
|
||||
int maxSamplesPerSpeaker,
|
||||
TimeSpan minimumUninterruptedSpeechDuration,
|
||||
TimeSpan maximumSegmentGap,
|
||||
ILogger? logger = null)
|
||||
TimeSpan minimumSampleSpeechDuration,
|
||||
TimeSpan maximumSampleDuration,
|
||||
ILogger? logger = null,
|
||||
bool requireNonOverlappingSamples = false)
|
||||
{
|
||||
SpeakerSampleDurationConfiguration.ValidateOrThrow(
|
||||
minimumSampleSpeechDuration,
|
||||
maximumSampleDuration,
|
||||
"Speaker sample collector configuration is invalid");
|
||||
audioBuffer = new RollingAudioBuffer(bufferDuration);
|
||||
this.maxSamplesPerSpeaker = Math.Max(1, maxSamplesPerSpeaker);
|
||||
this.minimumUninterruptedSpeechDuration = minimumUninterruptedSpeechDuration > TimeSpan.Zero
|
||||
? minimumUninterruptedSpeechDuration
|
||||
: TimeSpan.Zero;
|
||||
this.maximumSegmentGap = maximumSegmentGap >= TimeSpan.Zero
|
||||
? maximumSegmentGap
|
||||
: TimeSpan.Zero;
|
||||
this.minimumSampleSpeechDuration = minimumSampleSpeechDuration;
|
||||
this.maximumSampleDuration = maximumSampleDuration;
|
||||
this.requireNonOverlappingSamples = requireNonOverlappingSamples;
|
||||
this.logger = logger;
|
||||
}
|
||||
|
||||
@@ -53,6 +47,7 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
lock (gate)
|
||||
{
|
||||
samplesBySpeaker.Clear();
|
||||
lastAcceptedEndBySpeaker.Clear();
|
||||
pendingSpan = null;
|
||||
audioBuffer.Reset();
|
||||
}
|
||||
@@ -68,34 +63,54 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
return null;
|
||||
}
|
||||
|
||||
TranscriptionSegment sampleSegment;
|
||||
PendingSpanReset? reset;
|
||||
SpeakerSampleSpan? sampleSpan;
|
||||
SpeakerSampleSpan? previousSpan;
|
||||
lock (gate)
|
||||
{
|
||||
(sampleSegment, reset) = ExtendPendingSpan(segment);
|
||||
if (requireNonOverlappingSamples &&
|
||||
lastAcceptedEndBySpeaker.TryGetValue(segment.Speaker, out var lastAcceptedEnd) &&
|
||||
segment.Start < lastAcceptedEnd)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Discarding speaker identity sample for {Speaker} because it overlaps an accepted sample ending at {AcceptedSampleEnd}",
|
||||
segment.Speaker,
|
||||
lastAcceptedEnd);
|
||||
return null;
|
||||
}
|
||||
|
||||
(sampleSpan, previousSpan) = ExtendPendingSpan(segment);
|
||||
}
|
||||
|
||||
if (reset is not null)
|
||||
if (sampleSpan is null)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}, gap {Gap}, maximum gap {MaximumGap}",
|
||||
reset.PreviousSpeaker,
|
||||
segment.Speaker,
|
||||
reset.PreviousEnd,
|
||||
segment.Start,
|
||||
reset.Gap,
|
||||
maximumSegmentGap);
|
||||
"Discarding speaker identity sample for {Speaker} because the segment duration is not positive",
|
||||
segment.Speaker);
|
||||
return null;
|
||||
}
|
||||
|
||||
var score = Score(sampleSegment, minimumUninterruptedSpeechDuration);
|
||||
var sampleSegment = sampleSpan.ToSegment();
|
||||
|
||||
if (previousSpan is not null)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}",
|
||||
previousSpan.Speaker,
|
||||
segment.Speaker,
|
||||
previousSpan.End,
|
||||
segment.Start);
|
||||
}
|
||||
|
||||
var score = Score(sampleSegment, sampleSpan.SpeechDuration, minimumSampleSpeechDuration);
|
||||
if (!score.Accepted)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Discarding speaker identity sample for {Speaker} because {Reason}: duration {Duration}, minimum duration {MinimumDuration}, word count {WordCount}",
|
||||
"Discarding speaker identity sample for {Speaker} because {Reason}: speaker audio {SpeakerAudioDuration}, clip duration {ClipDuration}, minimum duration {MinimumDuration}, word count {WordCount}",
|
||||
sampleSegment.Speaker,
|
||||
score.Reason,
|
||||
sampleSpan.SpeechDuration,
|
||||
sampleSegment.End - sampleSegment.Start,
|
||||
minimumUninterruptedSpeechDuration,
|
||||
minimumSampleSpeechDuration,
|
||||
score.WordCount);
|
||||
return null;
|
||||
}
|
||||
@@ -115,6 +130,13 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
var sample = new SpeakerAudioSample(sampleSegment.Speaker, sampleSegment, wavBytes, score.Value);
|
||||
lock (gate)
|
||||
{
|
||||
if (requireNonOverlappingSamples &&
|
||||
ReferenceEquals(pendingSpan, sampleSpan))
|
||||
{
|
||||
pendingSpan = null;
|
||||
lastAcceptedEndBySpeaker[sampleSegment.Speaker] = sampleSegment.End;
|
||||
}
|
||||
|
||||
if (!samplesBySpeaker.TryGetValue(sampleSegment.Speaker, out var samples))
|
||||
{
|
||||
samples = [];
|
||||
@@ -163,35 +185,26 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
!string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
|
||||
private (TranscriptionSegment Segment, PendingSpanReset? Reset) ExtendPendingSpan(TranscriptionSegment segment)
|
||||
private (SpeakerSampleSpan? Span, SpeakerSampleSpan? PreviousSpan) ExtendPendingSpan(TranscriptionSegment segment)
|
||||
{
|
||||
if (pendingSpan is null ||
|
||||
!SpeakerSampleSpanSelector.CanExtend(pendingSpan.Speaker, pendingSpan.End, segment, maximumSegmentGap))
|
||||
if (pendingSpan is not null && pendingSpan.TryExtend(segment, out var extended))
|
||||
{
|
||||
var reset = pendingSpan is null
|
||||
? null
|
||||
: new PendingSpanReset(
|
||||
pendingSpan.Speaker,
|
||||
pendingSpan.End,
|
||||
segment.Start - pendingSpan.End);
|
||||
pendingSpan = new PendingSpeakerSpan(
|
||||
segment.Speaker,
|
||||
segment.Start,
|
||||
segment.End,
|
||||
[segment.Text]);
|
||||
return (pendingSpan.ToSegment(), reset);
|
||||
pendingSpan = extended;
|
||||
return (pendingSpan, null);
|
||||
}
|
||||
|
||||
pendingSpan = pendingSpan.Extend(segment);
|
||||
return (pendingSpan.ToSegment(), null);
|
||||
var previousSpan = pendingSpan;
|
||||
pendingSpan = SpeakerSampleSpan.Create(segment, maximumSampleDuration);
|
||||
return (pendingSpan, previousSpan);
|
||||
}
|
||||
|
||||
private static SampleScore Score(
|
||||
TranscriptionSegment segment,
|
||||
TimeSpan minimumUninterruptedSpeechDuration)
|
||||
TimeSpan speechDuration,
|
||||
TimeSpan minimumSampleSpeechDuration)
|
||||
{
|
||||
var durationSeconds = (segment.End - segment.Start).TotalSeconds;
|
||||
if (durationSeconds < minimumUninterruptedSpeechDuration.TotalSeconds)
|
||||
var durationSeconds = speechDuration.TotalSeconds;
|
||||
if (durationSeconds < minimumSampleSpeechDuration.TotalSeconds)
|
||||
{
|
||||
return new SampleScore(false, 0, "speech duration is below the configured minimum", WordCount(segment.Text));
|
||||
}
|
||||
@@ -202,7 +215,7 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
return new SampleScore(false, 0, "word count is below the minimum useful sample length", words);
|
||||
}
|
||||
|
||||
var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumUninterruptedSpeechDuration.TotalSeconds), 2);
|
||||
var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumSampleSpeechDuration.TotalSeconds), 2);
|
||||
var wordScore = Math.Min(words / 60.0, 1);
|
||||
var sentenceBonus = segment.Text.TrimEnd().EndsWith('.') ||
|
||||
segment.Text.TrimEnd().EndsWith('?') ||
|
||||
@@ -221,30 +234,4 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
|
||||
private sealed record SampleScore(bool Accepted, double Value, string? Reason, int WordCount);
|
||||
|
||||
private sealed record PendingSpanReset(string PreviousSpeaker, TimeSpan PreviousEnd, TimeSpan Gap);
|
||||
|
||||
private sealed record PendingSpeakerSpan(
|
||||
string Speaker,
|
||||
TimeSpan Start,
|
||||
TimeSpan End,
|
||||
IReadOnlyList<string> TextParts)
|
||||
{
|
||||
public PendingSpeakerSpan Extend(TranscriptionSegment segment)
|
||||
{
|
||||
return this with
|
||||
{
|
||||
End = segment.End > End ? segment.End : End,
|
||||
TextParts = TextParts.Append(segment.Text).ToList()
|
||||
};
|
||||
}
|
||||
|
||||
public TranscriptionSegment ToSegment()
|
||||
{
|
||||
return new TranscriptionSegment(
|
||||
Start,
|
||||
End,
|
||||
Speaker,
|
||||
string.Join(' ', TextParts.Where(part => !string.IsNullOrWhiteSpace(part))));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user