fix: unify pyannote validation toggle
PR and Push Build/Test / build-and-test (push) Successful in 12m8s

This commit is contained in:
2026-09-02 13:46:51 +02:00
parent 70ffaa7357
commit b1365b202d
14 changed files with 329 additions and 81 deletions
@@ -18,31 +18,8 @@ public sealed class PyannoteDiarizationWarmupHostedServiceTests
NullLogger<PyannoteTranscriptFinalizer>.Instance);
var service = new PyannoteDiarizationWarmupHostedService(
finalizer,
new FakeLaunchProfileOptionsProvider(new MeetingAssistantOptions
{
Recording = { TranscriptionProvider = "azure-speech" },
SpeakerIdentification =
{
PyannoteValidation =
{
Enabled = true,
Diarization =
{
Enabled = true,
DockerCommand = "docker",
Image = "meeting-assistant-pyannote-validation:local",
ModelsFolder = Path.Combine(
Path.GetTempPath(),
"meeting-assistant-tests",
Guid.NewGuid().ToString("N"),
"models"),
Token = "hf_test",
TokenEnv = "",
CommandTimeout = TimeSpan.FromMinutes(1)
}
}
}
}),
new FakeLaunchProfileOptionsProvider(
CreateValidationOptions("meeting-assistant-pyannote-validation:local")),
NullLogger<PyannoteDiarizationWarmupHostedService>.Instance);
await service.StartAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(1));
@@ -54,23 +31,87 @@ public sealed class PyannoteDiarizationWarmupHostedServiceTests
Assert.True(commandRunner.RunCancellationWasObserved);
}
[Fact]
public async Task HostedServiceUsesApplicationValidationRuntimeInsteadOfNamedProfileOverride()
{
var commandRunner = new BlockingCommandRunner();
var finalizer = new PyannoteTranscriptFinalizer(
commandRunner,
Options.Create(new MeetingAssistantOptions()),
NullLogger<PyannoteTranscriptFinalizer>.Instance);
var defaultOptions = CreateValidationOptions("meeting-assistant-pyannote-default:local");
var namedOptions = CreateValidationOptions("meeting-assistant-pyannote-named:local");
var service = new PyannoteDiarizationWarmupHostedService(
finalizer,
new FakeLaunchProfileOptionsProvider(
[
new LaunchProfile("english", namedOptions),
new LaunchProfile(ConfigurationLaunchProfileOptionsProvider.DefaultProfileName, defaultOptions)
]),
NullLogger<PyannoteDiarizationWarmupHostedService>.Instance);
await service.StartAsync(CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(1));
await commandRunner.WaitForRunAsync();
await service.StopAsync(CancellationToken.None);
Assert.Contains(commandRunner.Commands, command => command.Arguments.Contains("meeting-assistant-pyannote-default:local"));
Assert.DoesNotContain(commandRunner.Commands, command => command.Arguments.Contains("meeting-assistant-pyannote-named:local"));
}
private static MeetingAssistantOptions CreateValidationOptions(string image)
{
return new MeetingAssistantOptions
{
Recording = { TranscriptionProvider = "azure-speech" },
SpeakerIdentification =
{
PyannoteValidation =
{
Enabled = true,
Diarization =
{
DockerCommand = "docker",
Image = image,
ModelsFolder = Path.Combine(
Path.GetTempPath(),
"meeting-assistant-tests",
Guid.NewGuid().ToString("N"),
"models"),
Token = "hf_test",
TokenEnv = "",
CommandTimeout = TimeSpan.FromMinutes(1)
}
}
}
};
}
private sealed class FakeLaunchProfileOptionsProvider : ILaunchProfileOptionsProvider
{
private readonly MeetingAssistantOptions options;
private readonly IReadOnlyList<LaunchProfile> profiles;
public FakeLaunchProfileOptionsProvider(MeetingAssistantOptions options)
: this([new LaunchProfile(ConfigurationLaunchProfileOptionsProvider.DefaultProfileName, options)])
{
this.options = options;
}
public FakeLaunchProfileOptionsProvider(IReadOnlyList<LaunchProfile> profiles)
{
this.profiles = profiles;
}
public LaunchProfile GetRequiredProfile(string? name)
{
return new LaunchProfile(ConfigurationLaunchProfileOptionsProvider.DefaultProfileName, options);
var profileName = string.IsNullOrWhiteSpace(name)
? ConfigurationLaunchProfileOptionsProvider.DefaultProfileName
: name;
return profiles.Single(profile => profile.Name.Equals(profileName, StringComparison.OrdinalIgnoreCase));
}
public IReadOnlyList<LaunchProfile> GetProfiles()
{
return [GetRequiredProfile(null)];
return profiles;
}
public IReadOnlyList<LaunchProfileHotkey> GetHotkeys()
@@ -2,6 +2,7 @@ using MeetingAssistant;
using MeetingAssistant.Recording;
using MeetingAssistant.Speakers;
using MeetingAssistant.Transcription;
using Microsoft.Extensions.Configuration;
using Microsoft.Extensions.Logging.Abstractions;
using Microsoft.Extensions.Options;
using NAudio.Wave;
@@ -57,17 +58,49 @@ public sealed class PyannoteSpeakerIdentityMatchValidatorTests
Assert.Empty(commandRunner.Commands);
}
[Fact]
public async Task OuterToggleControlsValidationWhenLegacyNestedToggleIsFalse()
{
var commandRunner = new CapturingCommandRunner(
"""
__MEETING_ASSISTANT_PYANNOTE_JSON_START__
[{"start":0.0,"end":20.0,"speaker":"SPEAKER_00"}]
__MEETING_ASSISTANT_PYANNOTE_JSON_END__
""");
var modelsFolder = Path.Combine(
Path.GetTempPath(),
"meeting-assistant-tests",
Guid.NewGuid().ToString("N"),
"models");
var configuration = new ConfigurationBuilder()
.AddInMemoryCollection(new Dictionary<string, string?>
{
["SpeakerIdentification:PyannoteValidation:Enabled"] = "true",
["SpeakerIdentification:PyannoteValidation:Diarization:Enabled"] = "false",
["SpeakerIdentification:PyannoteValidation:Diarization:BuildImage"] = "false",
["SpeakerIdentification:PyannoteValidation:Diarization:ModelsFolder"] = modelsFolder,
["SpeakerIdentification:PyannoteValidation:Diarization:Token"] = "hf_test",
["SpeakerIdentification:PyannoteValidation:Diarization:TokenEnv"] = ""
})
.Build();
var configuredOptions = configuration.Get<MeetingAssistantOptions>()!;
var validator = CreateValidator(commandRunner, configuredOptions);
var valid = await validator.ValidateSampleAsync(
CreateWav(TimeSpan.FromSeconds(20)),
CancellationToken.None);
Assert.True(valid);
Assert.Contains(commandRunner.Commands, command => command.Arguments.Contains("run"));
}
private static PyannoteSpeakerIdentityMatchValidator CreateValidator(
CapturingCommandRunner commandRunner,
bool enabled = true)
{
var finalizer = new PyannoteTranscriptFinalizer(
return CreateValidator(
commandRunner,
Options.Create(new MeetingAssistantOptions()),
NullLogger<PyannoteTranscriptFinalizer>.Instance);
return new PyannoteSpeakerIdentityMatchValidator(
finalizer,
Options.Create(new MeetingAssistantOptions
new MeetingAssistantOptions
{
SpeakerIdentification = new SpeakerIdentificationOptions
{
@@ -76,9 +109,8 @@ public sealed class PyannoteSpeakerIdentityMatchValidatorTests
Enabled = enabled,
MinimumSingleSpeakerCoverage = 0.90,
MinimumMatchingKnownSnippetRatio = 1,
Diarization = new PyannoteDiarizationOptions
Diarization = new PyannoteRuntimeOptions
{
Enabled = true,
BuildImage = false,
DockerCommand = "docker",
Image = "meeting-assistant-pyannote:local",
@@ -94,7 +126,20 @@ public sealed class PyannoteSpeakerIdentityMatchValidatorTests
}
}
}
}),
});
}
private static PyannoteSpeakerIdentityMatchValidator CreateValidator(
CapturingCommandRunner commandRunner,
MeetingAssistantOptions configuredOptions)
{
var finalizer = new PyannoteTranscriptFinalizer(
commandRunner,
Options.Create(new MeetingAssistantOptions()),
NullLogger<PyannoteTranscriptFinalizer>.Instance);
return new PyannoteSpeakerIdentityMatchValidator(
finalizer,
Options.Create(configuredOptions),
NullLogger<PyannoteSpeakerIdentityMatchValidator>.Instance);
}
@@ -156,9 +156,8 @@ public sealed class PyannoteTranscriptFinalizerTests
commandRunner,
Options.Create(new MeetingAssistantOptions()),
NullLogger<PyannoteTranscriptFinalizer>.Instance);
var explicitDiarization = new PyannoteDiarizationOptions
var explicitDiarization = new PyannoteRuntimeOptions
{
Enabled = true,
DockerCommand = "docker",
BaseImage = "python:3.11-slim",
Image = "meeting-assistant-pyannote-azure:local",
@@ -169,7 +168,7 @@ public sealed class PyannoteTranscriptFinalizerTests
CommandTimeout = TimeSpan.FromMinutes(1)
};
await finalizer.FinalizeAsync(
await finalizer.FinalizeEnabledAsync(
audioPath,
[new TranscriptionSegment(TimeSpan.Zero, TimeSpan.FromSeconds(1), "Unknown", "hello")],
explicitDiarization,
@@ -186,9 +185,8 @@ public sealed class PyannoteTranscriptFinalizerTests
{
var commandRunner = new CapturingCommandRunner("");
var finalizer = CreateFinalizer(commandRunner, token: "hf_test");
var diarization = new PyannoteDiarizationOptions
var diarization = new PyannoteRuntimeOptions
{
Enabled = true,
DockerCommand = "docker",
BaseImage = "python:3.11-slim",
Image = "meeting-assistant-pyannote-warmup:local",
@@ -218,9 +216,8 @@ public sealed class PyannoteTranscriptFinalizerTests
var commandRunner = new CapturingCommandRunner("");
var finalizer = CreateFinalizer(commandRunner, token: null);
await finalizer.WarmUpAsync(new PyannoteDiarizationOptions
await finalizer.WarmUpAsync(new PyannoteRuntimeOptions
{
Enabled = true,
Token = "",
TokenEnv = "",
ModelsFolder = Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N"), "models")
+7 -5
View File
@@ -183,10 +183,8 @@ public sealed class WhisperLocalOptions
public PyannoteDiarizationOptions Diarization { get; set; } = new();
}
public sealed class PyannoteDiarizationOptions
public class PyannoteRuntimeOptions
{
public bool Enabled { get; set; } = true;
public string DockerCommand { get; set; } = "docker";
public string BaseImage { get; set; } = "python:3.11-slim";
@@ -210,6 +208,11 @@ public sealed class PyannoteDiarizationOptions
public TimeSpan CommandTimeout { get; set; } = TimeSpan.FromMinutes(30);
}
public sealed class PyannoteDiarizationOptions : PyannoteRuntimeOptions
{
public bool Enabled { get; set; } = true;
}
public sealed class AzureSpeechOptions
{
public string Endpoint { get; set; } = "";
@@ -367,9 +370,8 @@ public sealed class SpeakerIdentityPyannoteValidationOptions
public double MinimumMatchingKnownSnippetRatio { get; set; } = 0.50;
public PyannoteDiarizationOptions Diarization { get; set; } = new()
public PyannoteRuntimeOptions Diarization { get; set; } = new()
{
Enabled = true,
CommandTimeout = TimeSpan.FromHours(1),
AlignmentMode = PyannoteAlignmentMode.PyannoteTurns
};
@@ -190,7 +190,7 @@ public sealed class PyannoteSpeakerIdentityMatchValidator : ISpeakerIdentityMatc
return [];
}
return await finalizer.FinalizeAsync(
return await finalizer.FinalizeEnabledAsync(
wavPath,
[new TranscriptionSegment(TimeSpan.Zero, duration, "Unknown", "speaker identity validation sample")],
options.Diarization,
@@ -96,14 +96,18 @@ public sealed class PyannoteDiarizationWarmupHostedService : IHostedService
}
}
private IEnumerable<PyannoteDiarizationOptions> GetEnabledDiarizationOptions()
private IEnumerable<PyannoteRuntimeOptions> GetEnabledDiarizationOptions()
{
return launchProfiles.GetProfiles()
.SelectMany(profile => GetEnabledDiarizationOptions(profile.Options))
var transcriptionRuntimes = launchProfiles.GetProfiles()
.SelectMany(profile => GetEnabledTranscriptionDiarizationOptions(profile.Options));
var speakerValidationRuntimes = GetEnabledSpeakerValidationDiarizationOptions(
launchProfiles.GetRequiredProfile(null).Options);
return transcriptionRuntimes
.Concat(speakerValidationRuntimes)
.DistinctBy(CreateWarmUpKey);
}
private static IEnumerable<PyannoteDiarizationOptions> GetEnabledDiarizationOptions(
private static IEnumerable<PyannoteRuntimeOptions> GetEnabledTranscriptionDiarizationOptions(
MeetingAssistantOptions options)
{
if (options.Recording.TranscriptionProvider.Equals("whisper-local", StringComparison.OrdinalIgnoreCase) &&
@@ -111,15 +115,18 @@ public sealed class PyannoteDiarizationWarmupHostedService : IHostedService
{
yield return options.WhisperLocal.Diarization;
}
}
if (options.SpeakerIdentification.PyannoteValidation.Enabled &&
options.SpeakerIdentification.PyannoteValidation.Diarization.Enabled)
private static IEnumerable<PyannoteRuntimeOptions> GetEnabledSpeakerValidationDiarizationOptions(
MeetingAssistantOptions options)
{
if (options.SpeakerIdentification.PyannoteValidation.Enabled)
{
yield return options.SpeakerIdentification.PyannoteValidation.Diarization;
}
}
private static string CreateWarmUpKey(PyannoteDiarizationOptions diarization)
private static string CreateWarmUpKey(PyannoteRuntimeOptions diarization)
{
return string.Join(
'\u001f',
@@ -28,7 +28,12 @@ public sealed class PyannoteTranscriptFinalizer
SpeechRecognitionPipelineOptions pipelineOptions,
CancellationToken cancellationToken)
{
return await FinalizeAsync(
if (!options.WhisperLocal.Diarization.Enabled)
{
return [];
}
return await FinalizeEnabledAsync(
audioPath,
liveSegments,
options.WhisperLocal.Diarization,
@@ -36,14 +41,14 @@ public sealed class PyannoteTranscriptFinalizer
cancellationToken);
}
public async Task<IReadOnlyList<TranscriptionSegment>> FinalizeAsync(
internal async Task<IReadOnlyList<TranscriptionSegment>> FinalizeEnabledAsync(
string audioPath,
IReadOnlyList<TranscriptionSegment> liveSegments,
PyannoteDiarizationOptions diarization,
PyannoteRuntimeOptions diarization,
SpeechRecognitionPipelineOptions pipelineOptions,
CancellationToken cancellationToken)
{
if (!diarization.Enabled || liveSegments.Count == 0)
if (liveSegments.Count == 0)
{
return [];
}
@@ -83,14 +88,9 @@ public sealed class PyannoteTranscriptFinalizer
}
public async Task WarmUpAsync(
PyannoteDiarizationOptions diarization,
PyannoteRuntimeOptions diarization,
CancellationToken cancellationToken)
{
if (!diarization.Enabled)
{
return;
}
var token = ResolveToken(diarization);
if (string.IsNullOrWhiteSpace(token))
{
@@ -137,7 +137,7 @@ public sealed class PyannoteTranscriptFinalizer
private async Task<CommandResult> RunDiarizationAsync(
string fullAudioPath,
string token,
PyannoteDiarizationOptions diarization,
PyannoteRuntimeOptions diarization,
SpeechRecognitionPipelineOptions pipelineOptions,
CancellationToken cancellationToken)
{
@@ -171,7 +171,7 @@ public sealed class PyannoteTranscriptFinalizer
}
private async Task EnsureDockerImageAsync(
PyannoteDiarizationOptions diarization,
PyannoteRuntimeOptions diarization,
string modelsFolder,
CancellationToken cancellationToken)
{
@@ -198,7 +198,7 @@ public sealed class PyannoteTranscriptFinalizer
}
}
private static string BuildDockerfile(PyannoteDiarizationOptions diarization)
private static string BuildDockerfile(PyannoteRuntimeOptions diarization)
{
return
$"FROM {diarization.BaseImage}\n"
@@ -211,7 +211,7 @@ public sealed class PyannoteTranscriptFinalizer
}
private string[] BuildDockerArguments(
PyannoteDiarizationOptions diarization,
PyannoteRuntimeOptions diarization,
string fullAudioPath,
string modelsFolder,
SpeechRecognitionPipelineOptions pipelineOptions)
@@ -242,7 +242,7 @@ public sealed class PyannoteTranscriptFinalizer
}
private string[] BuildWarmUpDockerArguments(
PyannoteDiarizationOptions diarization,
PyannoteRuntimeOptions diarization,
string modelsFolder)
{
return
@@ -269,7 +269,7 @@ public sealed class PyannoteTranscriptFinalizer
}
private static string BuildPythonCommand(
PyannoteDiarizationOptions diarization,
PyannoteRuntimeOptions diarization,
SpeechRecognitionPipelineOptions pipelineOptions)
{
var model = diarization.Model;
@@ -296,7 +296,7 @@ public sealed class PyannoteTranscriptFinalizer
+ "python /tmp/meeting_assistant_pyannote.py";
}
private static string BuildWarmUpPythonCommand(PyannoteDiarizationOptions diarization)
private static string BuildWarmUpPythonCommand(PyannoteRuntimeOptions diarization)
{
var model = diarization.Model;
return
@@ -475,7 +475,7 @@ public sealed class PyannoteTranscriptFinalizer
return turns;
}
private static string ResolveToken(PyannoteDiarizationOptions options)
private static string ResolveToken(PyannoteRuntimeOptions options)
{
if (!string.IsNullOrWhiteSpace(options.Token))
{
+1 -2
View File
@@ -127,11 +127,10 @@
"MergeRecentIdentityAge": "14.00:00:00",
"MatchTimeout": "00:03:00",
"PyannoteValidation": {
"Enabled": true,
"Enabled": false,
"MinimumSingleSpeakerCoverage": 0.9,
"MinimumMatchingKnownSnippetRatio": 0.5,
"Diarization": {
"Enabled": false,
"DockerCommand": "docker",
"BaseImage": "python:3.11-slim",
"Image": "meeting-assistant-pyannote:local",
+4 -3
View File
@@ -223,11 +223,11 @@ When `WhisperLocal:Diarization:Enabled` is true, the final post-processing pass
Active pyannote runtimes are warmed up on application start so image setup and model download do not wait for the first diarization request.
Pyannote diarization settings are shared by local Whisper finalization and speaker-identification validation:
Pyannote runtime settings are shared by local Whisper finalization and speaker-identification validation. `WhisperLocal:Diarization:Enabled` controls the optional Whisper finalization pass; speaker-identification validation instead uses its single outer `SpeakerIdentification:PyannoteValidation:Enabled` switch.
| Setting | Purpose |
| --- | --- |
| `Enabled` | Enables the pyannote-backed pass. |
| `Enabled` | Available under `WhisperLocal:Diarization` to enable the pyannote-backed Whisper finalization pass. It is not part of the nested speaker-validation runtime settings. |
| `DockerCommand` | Docker executable name or path. |
| `BaseImage` | Python base image used when building the local pyannote image. |
| `Image` | Local pyannote Docker image tag. |
@@ -289,10 +289,11 @@ Speaker identity matching keeps candidate samples only after a diarized speaker
`SpeakerIdentification:AzureSpeech` is an advanced nested override for speaker identity matching. It uses the same shape as `AzureSpeech` and lets identity matching use different Azure language, endpoint, or key settings than live transcription when needed. If it is left unset, the normal Azure Speech settings remain the practical default.
`SpeakerIdentification:PyannoteValidation` is an optional secondary confidence layer. When enabled, pyannote rejects multi-speaker samples and must confirm an Azure-confirmed identity match before Meeting Assistant accepts it. It uses the same Docker-based pyannote runtime shape as local Whisper finalization and defaults the validation command timeout to 1 hour because local model setup can take substantial time.
`SpeakerIdentification:PyannoteValidation` is an optional application-level secondary confidence layer and defaults to disabled. Its outer `Enabled` setting is the only validation toggle; the nested `Diarization` block contains runtime settings but no second enable switch. Launch profiles do not override speaker validation or its runtime settings. When enabled, pyannote rejects multi-speaker samples and must confirm an Azure-confirmed identity match before Meeting Assistant accepts it. It uses the same Docker-based pyannote runtime shape as local Whisper finalization and defaults the validation command timeout to 1 hour because local model setup can take substantial time.
| Setting | Purpose |
| --- | --- |
| `Enabled` | Sole switch for speaker-identity pyannote validation and its startup warm-up. |
| `MinimumSingleSpeakerCoverage` | Required fraction of the tested sample that pyannote must attribute to a single speaker. |
| `MinimumMatchingKnownSnippetRatio` | Required pyannote agreement ratio between the new sample and known snippets for an accepted identity. |
| `Diarization` | Nested pyannote settings used for this validation pass. |
@@ -0,0 +1,2 @@
schema: spec-driven
created: 2026-09-02
@@ -0,0 +1,52 @@
## Context
Speaker identity validation reuses the general pyannote diarization option type. That type includes an `Enabled` property for the optional Whisper finalization feature, while speaker validation already has its own outer `Enabled` property. The checked-in and deployed configuration currently sets those properties to opposite values. The validator observes the outer value, calls the finalizer, and receives an empty result because the finalizer observes the nested value, causing every sample to be rejected before Azure matching.
## Goals / Non-Goals
**Goals:**
- Expose one authoritative enable switch for speaker identity pyannote validation.
- Keep Whisper diarization independently configurable.
- Make validation, matching, and startup warm-up observe the same speaker-validation state.
- Preserve the existing pyannote runtime settings and validation thresholds.
**Non-Goals:**
- Change speaker sample duration or gap thresholds.
- Change Azure Speech matching semantics.
- Enable pyannote validation without the required Docker runtime and Hugging Face token.
## Decisions
### Give speaker validation a runtime-options type without an enable flag
Extract the shared pyannote runtime properties into `PyannoteRuntimeOptions`. Keep `PyannoteDiarizationOptions` as the Whisper-facing derived type that adds `Enabled`, and type `SpeakerIdentityPyannoteValidationOptions.Diarization` as `PyannoteRuntimeOptions`.
This makes contradictory speaker-validation state unrepresentable through the typed configuration model. The alternative—retaining the nested flag and overriding it at runtime—would leave a misleading configuration surface and permit the same mistake to recur.
### Gate once at the feature boundary
`PyannoteSpeakerIdentityMatchValidator` bypasses pyannote when the application-level outer validation switch is off. When it is on, the validator calls an enabled-runtime finalization path that does not evaluate another toggle. The warm-up service selects the same application-level speaker-validation runtime instead of resolving speaker validation independently for each launch profile.
Whisper finalization continues checking `WhisperLocal:Diarization:Enabled` before it invokes the shared runtime path.
### Migrate configuration by removing the nested key
Remove `SpeakerIdentification:PyannoteValidation:Diarization:Enabled` from the canonical configuration and document `SpeakerIdentification:PyannoteValidation:Enabled` as the sole switch. Existing unknown nested keys are ignored by .NET configuration binding after the typed property is removed; deployments should republish from the canonical configuration to remove the stale key.
## Risks / Trade-offs
- [Enabling the outer switch now really invokes Docker/pyannote] → Preserve the existing token/runtime error reporting and document that disabling the outer switch is the supported bypass.
- [The shared options refactor touches Whisper code] → Preserve its independent `Enabled` property on the derived type and run focused Whisper/pyannote tests plus the full suite.
- [Old deployed appsettings may retain the removed nested key] → The binder ignores it, so runtime behavior remains controlled by the outer switch; republishing removes it from the canonical deployed file.
## Migration Plan
1. Publish the updated application configuration with the nested key removed.
2. Keep `SpeakerIdentification:PyannoteValidation:Enabled` on only where Docker, the pyannote model, and `HF_TOKEN` are available.
3. Roll back by deploying the prior build and configuration together if necessary.
## Open Questions
None.
@@ -0,0 +1,27 @@
## Why
Speaker identity matching currently exposes two independent pyannote validation switches. Enabling the outer validation switch while disabling the nested diarization switch silently rejects every otherwise usable speaker sample, so the default configuration can prevent all automatic identity matches.
## What Changes
- Make `SpeakerIdentification:PyannoteValidation:Enabled` the only switch controlling secondary pyannote validation.
- Remove the nested `SpeakerIdentification:PyannoteValidation:Diarization:Enabled` configuration setting.
- Ensure enabling validation also enables its pyannote runtime and startup warm-up, while disabling validation preserves primary Azure identity matching without invoking pyannote.
- Add regression coverage and configuration documentation for both toggle states.
## Capabilities
### New Capabilities
None.
### Modified Capabilities
- `meeting-transcription`: Clarify that secondary pyannote validation has one authoritative enable setting and cannot be partially enabled.
## Impact
- Speaker identification options and pyannote runtime invocation.
- Pyannote startup warm-up selection.
- Checked-in application configuration and configuration reference.
- Speaker validation and warm-up behavior tests.
@@ -0,0 +1,60 @@
## MODIFIED Requirements
### Requirement: Speaker identity matching can use pyannote secondary validation
Meeting Assistant SHALL support an optional configurable pyannote secondary validation layer for speaker identity matching.
`SpeakerIdentification:PyannoteValidation:Enabled` SHALL be the only enable setting for speaker identity pyannote validation. The nested pyannote runtime settings SHALL NOT expose or honor a second enable setting.
Speaker identity pyannote validation SHALL use the application-level setting consistently for matching and startup warm-up. Launch profiles SHALL NOT override this validation setting or its runtime configuration.
When pyannote secondary validation is enabled, Meeting Assistant SHALL verify candidate speaker samples before retaining them for identity matching. Samples that pyannote reports as containing multiple speakers SHALL be rejected.
When pyannote secondary validation is enabled, Meeting Assistant SHALL verify speaker-override samples before retaining them on speaker identities. Speaker overrides whose source samples are rejected SHALL NOT create a new speaker identity from that rejected sample.
When pyannote secondary validation is enabled and the primary identity matcher confirms a speaker, Meeting Assistant SHALL run a second validation pass through pyannote before accepting the match.
If pyannote secondary validation cannot confirm that the unknown live sample and matched identity samples belong to one speaker, Meeting Assistant SHALL reject the match.
When pyannote secondary validation is disabled, Meeting Assistant SHALL preserve the primary identity matching behavior.
When pyannote secondary validation is enabled, Meeting Assistant SHALL start a non-blocking startup warm-up that builds or verifies the configured pyannote runtime image and downloads the configured model into the persistent model cache before the first validation request when possible.
#### Scenario: Multi-speaker sample is rejected
- **GIVEN** pyannote secondary validation is enabled
- **WHEN** pyannote reports multiple speakers in a candidate sample
- **THEN** Meeting Assistant does not retain that sample for identity matching
#### Scenario: Multi-speaker speaker-override sample is rejected
- **GIVEN** pyannote secondary validation is enabled
- **WHEN** a summary speaker override resolves a source sample that pyannote reports as containing multiple speakers
- **THEN** Meeting Assistant does not retain that sample on a speaker identity
- **AND** does not create a new speaker identity from that rejected sample
#### Scenario: Pyannote rejects primary match
- **GIVEN** pyannote secondary validation is enabled
- **AND** the primary identity matcher confirms `Guest03` as `Chris`
- **WHEN** pyannote reports that the unknown `Guest03` sample and known `Chris` samples contain different speakers
- **THEN** Meeting Assistant rejects the match
#### Scenario: Enabled pyannote validation invokes its runtime
- **GIVEN** speaker identity pyannote validation is enabled
- **WHEN** Meeting Assistant validates a readable speaker sample
- **THEN** it invokes the configured pyannote runtime without requiring another enable setting
#### Scenario: Launch profile cannot override speaker validation
- **GIVEN** application-level speaker identity pyannote validation is disabled
- **AND** a named launch profile contains different speaker-validation settings
- **WHEN** Meeting Assistant starts or matches a speaker for that profile
- **THEN** it keeps application-level validation disabled
- **AND** does not warm or invoke the named profile's speaker-validation runtime
#### Scenario: Disabled pyannote validation preserves primary match
- **GIVEN** pyannote secondary validation is disabled
- **WHEN** the primary identity matcher confirms `Guest03` as `Chris`
- **THEN** Meeting Assistant accepts the match without running pyannote secondary validation
#### Scenario: Pyannote validation warms up on startup
- **GIVEN** pyannote secondary validation is enabled
- **WHEN** Meeting Assistant starts
- **THEN** it begins preparing the configured pyannote runtime image and model cache without waiting for the first validation request
- **AND** application startup is not blocked by the warm-up task
@@ -0,0 +1,15 @@
## 1. Single-toggle behavior
- [x] 1.1 Add a failing behavior test proving enabled speaker validation invokes pyannote without a nested enable setting
- [x] 1.2 Introduce toggle-free speaker-validation runtime options and make the validator use them
- [x] 1.3 Add or update behavior coverage proving disabled validation bypasses pyannote and enabled validation warms the runtime
- [x] 1.4 Keep application-level validation and warm-up consistent when named launch profiles contain speaker-validation overrides
## 2. Configuration migration
- [x] 2.1 Remove the nested speaker-validation diarization toggle from canonical configuration and document the outer toggle as authoritative
## 3. Verification
- [x] 3.1 Run focused speaker validation, warm-up, and pyannote finalizer tests
- [x] 3.2 Run the full solution test suite and validate the OpenSpec change strictly