Public Access
238 lines
8.7 KiB
C#
238 lines
8.7 KiB
C#
using MeetingAssistant.Speakers;
|
|
using MeetingAssistant.Transcription;
|
|
using Microsoft.Extensions.Logging;
|
|
|
|
namespace MeetingAssistant.Recording;
|
|
|
|
internal sealed class SpeakerAudioSampleCollector
|
|
{
|
|
private readonly object gate = new();
|
|
private readonly RollingAudioBuffer audioBuffer;
|
|
private readonly Dictionary<string, List<SpeakerAudioSample>> samplesBySpeaker = new(StringComparer.OrdinalIgnoreCase);
|
|
private readonly Dictionary<string, TimeSpan> lastAcceptedEndBySpeaker = new(StringComparer.OrdinalIgnoreCase);
|
|
private readonly int maxSamplesPerSpeaker;
|
|
private readonly TimeSpan minimumSampleSpeechDuration;
|
|
private readonly TimeSpan maximumSampleDuration;
|
|
private readonly bool requireNonOverlappingSamples;
|
|
private readonly ILogger? logger;
|
|
private SpeakerSampleSpan? pendingSpan;
|
|
|
|
public SpeakerAudioSampleCollector(
|
|
TimeSpan bufferDuration,
|
|
int maxSamplesPerSpeaker,
|
|
TimeSpan minimumSampleSpeechDuration,
|
|
TimeSpan maximumSampleDuration,
|
|
ILogger? logger = null,
|
|
bool requireNonOverlappingSamples = false)
|
|
{
|
|
SpeakerSampleDurationConfiguration.ValidateOrThrow(
|
|
minimumSampleSpeechDuration,
|
|
maximumSampleDuration,
|
|
"Speaker sample collector configuration is invalid");
|
|
audioBuffer = new RollingAudioBuffer(bufferDuration);
|
|
this.maxSamplesPerSpeaker = Math.Max(1, maxSamplesPerSpeaker);
|
|
this.minimumSampleSpeechDuration = minimumSampleSpeechDuration;
|
|
this.maximumSampleDuration = maximumSampleDuration;
|
|
this.requireNonOverlappingSamples = requireNonOverlappingSamples;
|
|
this.logger = logger;
|
|
}
|
|
|
|
public void AppendAudio(AudioChunk chunk)
|
|
{
|
|
audioBuffer.Append(chunk);
|
|
}
|
|
|
|
public void Reset()
|
|
{
|
|
lock (gate)
|
|
{
|
|
samplesBySpeaker.Clear();
|
|
lastAcceptedEndBySpeaker.Clear();
|
|
pendingSpan = null;
|
|
audioBuffer.Reset();
|
|
}
|
|
}
|
|
|
|
public SpeakerAudioSample? TryAdd(TranscriptionSegment segment)
|
|
{
|
|
if (!IsDiarizedSpeaker(segment.Speaker))
|
|
{
|
|
logger?.LogInformation(
|
|
"Discarding speaker identity sample for {Speaker} because the segment has no diarized speaker label",
|
|
segment.Speaker);
|
|
return null;
|
|
}
|
|
|
|
SpeakerSampleSpan? sampleSpan;
|
|
SpeakerSampleSpan? previousSpan;
|
|
lock (gate)
|
|
{
|
|
if (requireNonOverlappingSamples &&
|
|
lastAcceptedEndBySpeaker.TryGetValue(segment.Speaker, out var lastAcceptedEnd) &&
|
|
segment.Start < lastAcceptedEnd)
|
|
{
|
|
logger?.LogInformation(
|
|
"Discarding speaker identity sample for {Speaker} because it overlaps an accepted sample ending at {AcceptedSampleEnd}",
|
|
segment.Speaker,
|
|
lastAcceptedEnd);
|
|
return null;
|
|
}
|
|
|
|
(sampleSpan, previousSpan) = ExtendPendingSpan(segment);
|
|
}
|
|
|
|
if (sampleSpan is null)
|
|
{
|
|
logger?.LogInformation(
|
|
"Discarding speaker identity sample for {Speaker} because the segment duration is not positive",
|
|
segment.Speaker);
|
|
return null;
|
|
}
|
|
|
|
var sampleSegment = sampleSpan.ToSegment();
|
|
|
|
if (previousSpan is not null)
|
|
{
|
|
logger?.LogInformation(
|
|
"Reset speaker identity sample span from {PreviousSpeaker} to {Speaker}: previous end {PreviousEnd}, next start {NextStart}",
|
|
previousSpan.Speaker,
|
|
segment.Speaker,
|
|
previousSpan.End,
|
|
segment.Start);
|
|
}
|
|
|
|
var score = Score(sampleSegment, sampleSpan.SpeechDuration, minimumSampleSpeechDuration);
|
|
if (!score.Accepted)
|
|
{
|
|
logger?.LogInformation(
|
|
"Discarding speaker identity sample for {Speaker} because {Reason}: speaker audio {SpeakerAudioDuration}, clip duration {ClipDuration}, minimum duration {MinimumDuration}, word count {WordCount}",
|
|
sampleSegment.Speaker,
|
|
score.Reason,
|
|
sampleSpan.SpeechDuration,
|
|
sampleSegment.End - sampleSegment.Start,
|
|
minimumSampleSpeechDuration,
|
|
score.WordCount);
|
|
return null;
|
|
}
|
|
|
|
var wavBytes = audioBuffer.TryExtractWav(sampleSegment.Start, sampleSegment.End);
|
|
if (wavBytes.Length == 0)
|
|
{
|
|
logger?.LogInformation(
|
|
"Discarding speaker identity sample for {Speaker} because no audio could be extracted from rolling buffer: start {Start}, end {End}, duration {Duration}",
|
|
sampleSegment.Speaker,
|
|
sampleSegment.Start,
|
|
sampleSegment.End,
|
|
sampleSegment.End - sampleSegment.Start);
|
|
return null;
|
|
}
|
|
|
|
var sample = new SpeakerAudioSample(sampleSegment.Speaker, sampleSegment, wavBytes, score.Value);
|
|
lock (gate)
|
|
{
|
|
if (requireNonOverlappingSamples &&
|
|
ReferenceEquals(pendingSpan, sampleSpan))
|
|
{
|
|
pendingSpan = null;
|
|
lastAcceptedEndBySpeaker[sampleSegment.Speaker] = sampleSegment.End;
|
|
}
|
|
|
|
if (!samplesBySpeaker.TryGetValue(sampleSegment.Speaker, out var samples))
|
|
{
|
|
samples = [];
|
|
samplesBySpeaker[sampleSegment.Speaker] = samples;
|
|
}
|
|
|
|
samples.Add(sample);
|
|
var beforeCount = samples.Count;
|
|
var bestSamples = samples
|
|
.OrderByDescending(candidate => candidate.Score)
|
|
.Take(maxSamplesPerSpeaker)
|
|
.ToList();
|
|
var retained = bestSamples.Contains(sample);
|
|
samples.Clear();
|
|
samples.AddRange(bestSamples);
|
|
if (!retained)
|
|
{
|
|
logger?.LogInformation(
|
|
"Discarding speaker identity sample for {Speaker} because it was not among the best {MaxSamplesPerSpeaker} retained sample(s): score {Score}, candidate count {CandidateCount}",
|
|
sample.Speaker,
|
|
maxSamplesPerSpeaker,
|
|
sample.Score,
|
|
beforeCount);
|
|
return null;
|
|
}
|
|
}
|
|
|
|
return sample;
|
|
}
|
|
|
|
public IReadOnlyList<SpeakerAudioSample> Snapshot()
|
|
{
|
|
lock (gate)
|
|
{
|
|
return samplesBySpeaker.Values
|
|
.SelectMany(samples => samples)
|
|
.OrderBy(sample => sample.Speaker, StringComparer.OrdinalIgnoreCase)
|
|
.ThenByDescending(sample => sample.Score)
|
|
.ToList();
|
|
}
|
|
}
|
|
|
|
private static bool IsDiarizedSpeaker(string speaker)
|
|
{
|
|
return !string.IsNullOrWhiteSpace(speaker) &&
|
|
!string.Equals(speaker, "Unknown", StringComparison.OrdinalIgnoreCase);
|
|
}
|
|
|
|
private (SpeakerSampleSpan? Span, SpeakerSampleSpan? PreviousSpan) ExtendPendingSpan(TranscriptionSegment segment)
|
|
{
|
|
if (pendingSpan is not null && pendingSpan.TryExtend(segment, out var extended))
|
|
{
|
|
pendingSpan = extended;
|
|
return (pendingSpan, null);
|
|
}
|
|
|
|
var previousSpan = pendingSpan;
|
|
pendingSpan = SpeakerSampleSpan.Create(segment, maximumSampleDuration);
|
|
return (pendingSpan, previousSpan);
|
|
}
|
|
|
|
private static SampleScore Score(
|
|
TranscriptionSegment segment,
|
|
TimeSpan speechDuration,
|
|
TimeSpan minimumSampleSpeechDuration)
|
|
{
|
|
var durationSeconds = speechDuration.TotalSeconds;
|
|
if (durationSeconds < minimumSampleSpeechDuration.TotalSeconds)
|
|
{
|
|
return new SampleScore(false, 0, "speech duration is below the configured minimum", WordCount(segment.Text));
|
|
}
|
|
|
|
var words = WordCount(segment.Text);
|
|
if (words < 3)
|
|
{
|
|
return new SampleScore(false, 0, "word count is below the minimum useful sample length", words);
|
|
}
|
|
|
|
var durationScore = Math.Min(durationSeconds / Math.Max(1, minimumSampleSpeechDuration.TotalSeconds), 2);
|
|
var wordScore = Math.Min(words / 60.0, 1);
|
|
var sentenceBonus = segment.Text.TrimEnd().EndsWith('.') ||
|
|
segment.Text.TrimEnd().EndsWith('?') ||
|
|
segment.Text.TrimEnd().EndsWith('!')
|
|
? 5
|
|
: 0;
|
|
return new SampleScore(true, durationScore * 70 + wordScore * 30 + sentenceBonus, null, words);
|
|
}
|
|
|
|
private static int WordCount(string text)
|
|
{
|
|
return text
|
|
.Split(' ', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries)
|
|
.Length;
|
|
}
|
|
|
|
private sealed record SampleScore(bool Accepted, double Value, string? Reason, int WordCount);
|
|
|
|
}
|