forked from Manuel/meeting-assistant
feat: add local Resemblyzer speaker recognition
This commit is contained in:
@@ -9,37 +9,31 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
private readonly object gate = new();
|
||||
private readonly RollingAudioBuffer audioBuffer;
|
||||
private readonly Dictionary<string, List<SpeakerAudioSample>> samplesBySpeaker = new(StringComparer.OrdinalIgnoreCase);
|
||||
private readonly Dictionary<string, TimeSpan> lastAcceptedEndBySpeaker = new(StringComparer.OrdinalIgnoreCase);
|
||||
private readonly int maxSamplesPerSpeaker;
|
||||
private readonly TimeSpan minimumUninterruptedSpeechDuration;
|
||||
private readonly TimeSpan maximumSegmentGap;
|
||||
private readonly TimeSpan minimumSampleSpeechDuration;
|
||||
private readonly TimeSpan maximumSampleDuration;
|
||||
private readonly bool requireNonOverlappingSamples;
|
||||
private readonly ILogger? logger;
|
||||
private PendingSpeakerSpan? pendingSpan;
|
||||
|
||||
public SpeakerAudioSampleCollector(TimeSpan bufferDuration, int maxSamplesPerSpeaker)
|
||||
: this(
|
||||
bufferDuration,
|
||||
maxSamplesPerSpeaker,
|
||||
TimeSpan.FromSeconds(30),
|
||||
TimeSpan.FromSeconds(1),
|
||||
logger: null)
|
||||
{
|
||||
}
|
||||
private SpeakerSampleSpan? pendingSpan;
|
||||
|
||||
public SpeakerAudioSampleCollector(
|
||||
TimeSpan bufferDuration,
|
||||
int maxSamplesPerSpeaker,
|
||||
TimeSpan minimumUninterruptedSpeechDuration,
|
||||
TimeSpan maximumSegmentGap,
|
||||
ILogger? logger = null)
|
||||
TimeSpan minimumSampleSpeechDuration,
|
||||
TimeSpan maximumSampleDuration,
|
||||
ILogger? logger = null,
|
||||
bool requireNonOverlappingSamples = false)
|
||||
{
|
||||
SpeakerSampleDurationConfiguration.ValidateOrThrow(
|
||||
minimumSampleSpeechDuration,
|
||||
maximumSampleDuration,
|
||||
"Speaker sample collector configuration is invalid");
|
||||
audioBuffer = new RollingAudioBuffer(bufferDuration);
|
||||
this.maxSamplesPerSpeaker = Math.Max(1, maxSamplesPerSpeaker);
|
||||
this.minimumUninterruptedSpeechDuration = minimumUninterruptedSpeechDuration > TimeSpan.Zero
|
||||
? minimumUninterruptedSpeechDuration
|
||||
: TimeSpan.Zero;
|
||||
this.maximumSegmentGap = maximumSegmentGap >= TimeSpan.Zero
|
||||
? maximumSegmentGap
|
||||
: TimeSpan.Zero;
|
||||
this.minimumSampleSpeechDuration = minimumSampleSpeechDuration;
|
||||
this.maximumSampleDuration = maximumSampleDuration;
|
||||
this.requireNonOverlappingSamples = requireNonOverlappingSamples;
|
||||
this.logger = logger;
|
||||
}
|
||||
|
||||
@@ -53,6 +47,7 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
lock (gate)
|
||||
{
|
||||
samplesBySpeaker.Clear();
|
||||
lastAcceptedEndBySpeaker.Clear();
|
||||
pendingSpan = null;
|
||||
audioBuffer.Reset();
|
||||
}
|
||||
@@ -68,34 +63,54 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
return null;
|
||||
}
|
||||
|
||||
TranscriptionSegment sampleSegment;
|
||||
PendingSpanReset? reset;
|
||||
SpeakerSampleSpan? sampleSpan;
|
||||
SpeakerSampleSpan? previousSpan;
|
||||
lock (gate)
|
||||
{
|
||||
(sampleSegment, reset) = ExtendPendingSpan(segment);
|
||||
if (requireNonOverlappingSamples &&
|
||||
lastAcceptedEndBySpeaker.TryGetValue(segment.Speaker, out var lastAcceptedEnd) &&
|
||||
segment.Start < lastAcceptedEnd)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Discarding speaker identity sample for {Speaker} because it overlaps an accepted sample ending at {AcceptedSampleEnd}",
|
||||
segment.Speaker,
|
||||
lastAcceptedEnd);
|
||||
return null;
|
||||
}
|
||||
|
||||
(sampleSpan, previousSpan) = ExtendPendingSpan(segment);
|
||||
}
|
||||
|
||||
if (reset is not null)
|
||||
if (sampleSpan is null)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}, gap {Gap}, maximum gap {MaximumGap}",
|
||||
reset.PreviousSpeaker,
|
||||
segment.Speaker,
|
||||
reset.PreviousEnd,
|
||||
segment.Start,
|
||||
reset.Gap,
|
||||
maximumSegmentGap);
|
||||
"Discarding speaker identity sample for {Speaker} because the segment duration is not positive",
|
||||
segment.Speaker);
|
||||
return null;
|
||||
}
|
||||
|
||||
var score = Score(sampleSegment, minimumUninterruptedSpeechDuration);
|
||||
var sampleSegment = sampleSpan.ToSegment();
|
||||
|
||||
if (previousSpan is not null)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}",
|
||||
previousSpan.Speaker,
|
||||
segment.Speaker,
|
||||
previousSpan.End,
|
||||
segment.Start);
|
||||
}
|
||||
|
||||
var score = Score(sampleSegment, sampleSpan.SpeechDuration, minimumSampleSpeechDuration);
|
||||
if (!score.Accepted)
|
||||
{
|
||||
logger?.LogInformation(
|
||||
"Discarding speaker identity sample for {Speaker} because {Reason}: duration {Duration}, minimum duration {MinimumDuration}, word count {WordCount}",
|
||||
"Discarding speaker identity sample for {Speaker} because {Reason}: speaker audio {SpeakerAudioDuration}, clip duration {ClipDuration}, minimum duration {MinimumDuration}, word count {WordCount}",
|
||||
sampleSegment.Speaker,
|
||||
score.Reason,
|
||||
sampleSpan.SpeechDuration,
|
||||
sampleSegment.End - sampleSegment.Start,
|
||||
minimumUninterruptedSpeechDuration,
|
||||
minimumSampleSpeechDuration,
|
||||
score.WordCount);
|
||||
return null;
|
||||
}
|
||||
@@ -115,6 +130,13 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
var sample = new SpeakerAudioSample(sampleSegment.Speaker, sampleSegment, wavBytes, score.Value);
|
||||
lock (gate)
|
||||
{
|
||||
if (requireNonOverlappingSamples &&
|
||||
ReferenceEquals(pendingSpan, sampleSpan))
|
||||
{
|
||||
pendingSpan = null;
|
||||
lastAcceptedEndBySpeaker[sampleSegment.Speaker] = sampleSegment.End;
|
||||
}
|
||||
|
||||
if (!samplesBySpeaker.TryGetValue(sampleSegment.Speaker, out var samples))
|
||||
{
|
||||
samples = [];
|
||||
@@ -163,35 +185,26 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
!string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase);
|
||||
}
|
||||
|
||||
private (TranscriptionSegment Segment, PendingSpanReset? Reset) ExtendPendingSpan(TranscriptionSegment segment)
|
||||
private (SpeakerSampleSpan? Span, SpeakerSampleSpan? PreviousSpan) ExtendPendingSpan(TranscriptionSegment segment)
|
||||
{
|
||||
if (pendingSpan is null ||
|
||||
!SpeakerSampleSpanSelector.CanExtend(pendingSpan.Speaker, pendingSpan.End, segment, maximumSegmentGap))
|
||||
if (pendingSpan is not null && pendingSpan.TryExtend(segment, out var extended))
|
||||
{
|
||||
var reset = pendingSpan is null
|
||||
? null
|
||||
: new PendingSpanReset(
|
||||
pendingSpan.Speaker,
|
||||
pendingSpan.End,
|
||||
segment.Start - pendingSpan.End);
|
||||
pendingSpan = new PendingSpeakerSpan(
|
||||
segment.Speaker,
|
||||
segment.Start,
|
||||
segment.End,
|
||||
[segment.Text]);
|
||||
return (pendingSpan.ToSegment(), reset);
|
||||
pendingSpan = extended;
|
||||
return (pendingSpan, null);
|
||||
}
|
||||
|
||||
pendingSpan = pendingSpan.Extend(segment);
|
||||
return (pendingSpan.ToSegment(), null);
|
||||
var previousSpan = pendingSpan;
|
||||
pendingSpan = SpeakerSampleSpan.Create(segment, maximumSampleDuration);
|
||||
return (pendingSpan, previousSpan);
|
||||
}
|
||||
|
||||
private static SampleScore Score(
|
||||
TranscriptionSegment segment,
|
||||
TimeSpan minimumUninterruptedSpeechDuration)
|
||||
TimeSpan speechDuration,
|
||||
TimeSpan minimumSampleSpeechDuration)
|
||||
{
|
||||
var durationSeconds = (segment.End - segment.Start).TotalSeconds;
|
||||
if (durationSeconds < minimumUninterruptedSpeechDuration.TotalSeconds)
|
||||
var durationSeconds = speechDuration.TotalSeconds;
|
||||
if (durationSeconds < minimumSampleSpeechDuration.TotalSeconds)
|
||||
{
|
||||
return new SampleScore(false, 0, "speech duration is below the configured minimum", WordCount(segment.Text));
|
||||
}
|
||||
@@ -202,7 +215,7 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
return new SampleScore(false, 0, "word count is below the minimum useful sample length", words);
|
||||
}
|
||||
|
||||
var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumUninterruptedSpeechDuration.TotalSeconds), 2);
|
||||
var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumSampleSpeechDuration.TotalSeconds), 2);
|
||||
var wordScore = Math.Min(words / 60.0, 1);
|
||||
var sentenceBonus = segment.Text.TrimEnd().EndsWith('.') ||
|
||||
segment.Text.TrimEnd().EndsWith('?') ||
|
||||
@@ -221,30 +234,4 @@ internal sealed class SpeakerAudioSampleCollector
|
||||
|
||||
private sealed record SampleScore(bool Accepted, double Value, string? Reason, int WordCount);
|
||||
|
||||
private sealed record PendingSpanReset(string PreviousSpeaker, TimeSpan PreviousEnd, TimeSpan Gap);
|
||||
|
||||
private sealed record PendingSpeakerSpan(
|
||||
string Speaker,
|
||||
TimeSpan Start,
|
||||
TimeSpan End,
|
||||
IReadOnlyList<string> TextParts)
|
||||
{
|
||||
public PendingSpeakerSpan Extend(TranscriptionSegment segment)
|
||||
{
|
||||
return this with
|
||||
{
|
||||
End = segment.End > End ? segment.End : End,
|
||||
TextParts = TextParts.Append(segment.Text).ToList()
|
||||
};
|
||||
}
|
||||
|
||||
public TranscriptionSegment ToSegment()
|
||||
{
|
||||
return new TranscriptionSegment(
|
||||
Start,
|
||||
End,
|
||||
Speaker,
|
||||
string.Join(' ', TextParts.Where(part => !string.IsNullOrWhiteSpace(part))));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user