Public Access
Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e35ccf661f | ||
|
|
aa42e8edda | ||
|
|
2f12a96688 | ||
|
|
5d0ae84426 | ||
|
|
b9547ae4c4 |
@@ -102,6 +102,23 @@ public sealed class AudioMixingTests
|
||||
Assert.Equal(2_000, BitConverter.ToInt16(chunks[0].Pcm));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task CompositeAudioSourceKeepsSystemAudioWhileMicrophoneIsRecovering()
|
||||
{
|
||||
var microphone = new WaitingAudioSource();
|
||||
var system = new FixedAudioSource(Pcm16(10_000));
|
||||
var source = CreateSource(microphone, system);
|
||||
using var cancellation = new CancellationTokenSource(TimeSpan.FromSeconds(5));
|
||||
await using var chunks = source
|
||||
.CaptureAsync(new MeetingAssistantOptions(), cancellation.Token)
|
||||
.GetAsyncEnumerator(cancellation.Token);
|
||||
|
||||
Assert.True(await chunks.MoveNextAsync());
|
||||
Assert.Equal(10_000, BitConverter.ToInt16(chunks.Current.Pcm));
|
||||
|
||||
await cancellation.CancelAsync();
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void AdaptiveEchoCancellerReducesEchoFromMicrophoneSignal()
|
||||
{
|
||||
@@ -221,6 +238,16 @@ public sealed class AudioMixingTests
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class WaitingAudioSource : IMeetingAudioSource
|
||||
{
|
||||
public async IAsyncEnumerable<AudioChunk> CaptureAsync(
|
||||
[System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken cancellationToken)
|
||||
{
|
||||
await Task.Delay(Timeout.InfiniteTimeSpan, cancellationToken);
|
||||
yield break;
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class OptionsCapturingAudioSource : IMeetingAudioSource
|
||||
{
|
||||
private readonly AudioChunk chunk;
|
||||
|
||||
@@ -13,21 +13,50 @@ namespace MeetingAssistant.Tests;
|
||||
|
||||
public sealed class LiteLlmScreenshotOcrClientTests
|
||||
{
|
||||
[Fact]
|
||||
public async Task ExtractUsesAgentStreamingTransportByDefault()
|
||||
{
|
||||
var screenshotPath = await CreateScreenshotAsync([1, 2, 3]);
|
||||
var handler = new RecordingHandler(
|
||||
CreateStreamedTextResponse("Streamed OCR text"),
|
||||
"text/event-stream");
|
||||
var client = new LiteLlmScreenshotOcrClient(
|
||||
() => handler,
|
||||
NullLogger<LiteLlmScreenshotOcrClient>.Instance);
|
||||
var options = new MeetingAssistantOptions
|
||||
{
|
||||
Agent =
|
||||
{
|
||||
Endpoint = "https://summary.local",
|
||||
Model = "vision-model",
|
||||
Key = "agent-key",
|
||||
UseStreaming = true
|
||||
}
|
||||
};
|
||||
|
||||
var result = await client.ExtractAsync(
|
||||
screenshotPath,
|
||||
"Extract screenshot.",
|
||||
options,
|
||||
CancellationToken.None);
|
||||
|
||||
Assert.Equal("Streamed OCR text", result.Text);
|
||||
using var payload = JsonDocument.Parse(handler.RequestBody!);
|
||||
Assert.True(payload.RootElement.GetProperty("stream").GetBoolean());
|
||||
var message = Assert.Single(payload.RootElement.GetProperty("input").EnumerateArray());
|
||||
Assert.Equal("message", message.GetProperty("type").GetString());
|
||||
var content = message.GetProperty("content").EnumerateArray().ToArray();
|
||||
Assert.Equal("input_text", content[0].GetProperty("type").GetString());
|
||||
Assert.Equal("Extract screenshot.", content[0].GetProperty("text").GetString());
|
||||
Assert.Equal("input_image", content[1].GetProperty("type").GetString());
|
||||
Assert.Equal("data:image/png;base64,AQID", content[1].GetProperty("image_url").GetString());
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task ExtractUsesAgentEndpointAndModelWhenOcrEndpointAndModelAreBlank()
|
||||
{
|
||||
var screenshotPath = await CreateScreenshotAsync([1, 2, 3]);
|
||||
var handler = new RecordingHandler("""
|
||||
{
|
||||
"output": [
|
||||
{
|
||||
"content": [
|
||||
{ "type": "output_text", "text": "Visible slide text" }
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
""");
|
||||
var handler = new RecordingHandler(CreateNonStreamingTextResponse("Visible slide text"));
|
||||
var client = new LiteLlmScreenshotOcrClient(
|
||||
() => handler,
|
||||
NullLogger<LiteLlmScreenshotOcrClient>.Instance);
|
||||
@@ -37,7 +66,8 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
{
|
||||
Endpoint = "https://summary.local",
|
||||
Model = "summary-model",
|
||||
Key = "agent-key"
|
||||
Key = "agent-key",
|
||||
UseStreaming = false
|
||||
},
|
||||
Screenshots =
|
||||
{
|
||||
@@ -62,6 +92,7 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
Assert.Equal("Bearer", handler.Authorization?.Scheme);
|
||||
Assert.Equal("ocr-key", handler.Authorization?.Parameter);
|
||||
using var payload = JsonDocument.Parse(handler.RequestBody!);
|
||||
Assert.False(payload.RootElement.GetProperty("stream").GetBoolean());
|
||||
Assert.Equal("summary-model", payload.RootElement.GetProperty("model").GetString());
|
||||
var content = payload.RootElement
|
||||
.GetProperty("input")[0]
|
||||
@@ -74,7 +105,7 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
public async Task ExtractUsesScreenshotOcrEndpointAndModelWhenConfigured()
|
||||
{
|
||||
var screenshotPath = await CreateScreenshotAsync([4, 5, 6]);
|
||||
var handler = new RecordingHandler("""{ "output_text": "OCR result" }""");
|
||||
var handler = new RecordingHandler(CreateNonStreamingTextResponse("OCR result"));
|
||||
var client = new LiteLlmScreenshotOcrClient(
|
||||
() => handler,
|
||||
NullLogger<LiteLlmScreenshotOcrClient>.Instance);
|
||||
@@ -84,7 +115,8 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
{
|
||||
Endpoint = "https://summary.local",
|
||||
Model = "summary-model",
|
||||
Key = "agent-key"
|
||||
Key = "agent-key",
|
||||
UseStreaming = false
|
||||
},
|
||||
Screenshots =
|
||||
{
|
||||
@@ -113,11 +145,14 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
public async Task ExtractParsesCropMetadataAndOmitsMetadataFromReturnedText()
|
||||
{
|
||||
var screenshotPath = await CreateScreenshotAsync(CreatePngBytes(8, 6));
|
||||
var handler = new RecordingHandler("""
|
||||
{
|
||||
"output_text": "Slide text\n\n```json\n{ \"crop\": { \"x\": 1, \"y\": 2, \"width\": 3, \"height\": 4 } }\n```"
|
||||
}
|
||||
""");
|
||||
var handler = new RecordingHandler(CreateNonStreamingTextResponse(
|
||||
"""
|
||||
Slide text
|
||||
|
||||
```json
|
||||
{ "crop": { "x": 1, "y": 2, "width": 3, "height": 4 } }
|
||||
```
|
||||
"""));
|
||||
var client = new LiteLlmScreenshotOcrClient(
|
||||
() => handler,
|
||||
NullLogger<LiteLlmScreenshotOcrClient>.Instance);
|
||||
@@ -125,7 +160,8 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
{
|
||||
Agent =
|
||||
{
|
||||
Key = "agent-key"
|
||||
Key = "agent-key",
|
||||
UseStreaming = false
|
||||
}
|
||||
};
|
||||
|
||||
@@ -148,11 +184,14 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
public async Task ExtractParsesAttendeeMetadataAndOmitsMetadataFromReturnedText()
|
||||
{
|
||||
var screenshotPath = await CreateScreenshotAsync([1, 2, 3]);
|
||||
var handler = new RecordingHandler("""
|
||||
{
|
||||
"output_text": "Visible participant tiles: Ada and Grace.\n\n```json\n{ \"crop\": null, \"attendees\": [\"Ada Lovelace\", \"Grace Hopper\"] }\n```"
|
||||
}
|
||||
""");
|
||||
var handler = new RecordingHandler(CreateNonStreamingTextResponse(
|
||||
"""
|
||||
Visible participant tiles: Ada and Grace.
|
||||
|
||||
```json
|
||||
{ "crop": null, "attendees": ["Ada Lovelace", "Grace Hopper"] }
|
||||
```
|
||||
"""));
|
||||
var client = new LiteLlmScreenshotOcrClient(
|
||||
() => handler,
|
||||
NullLogger<LiteLlmScreenshotOcrClient>.Instance);
|
||||
@@ -160,7 +199,8 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
{
|
||||
Agent =
|
||||
{
|
||||
Key = "agent-key"
|
||||
Key = "agent-key",
|
||||
UseStreaming = false
|
||||
}
|
||||
};
|
||||
|
||||
@@ -178,11 +218,14 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
public async Task ExtractIgnoresMalformedAttendeesMetadataAndStillParsesCrop()
|
||||
{
|
||||
var screenshotPath = await CreateScreenshotAsync(CreatePngBytes(8, 6));
|
||||
var handler = new RecordingHandler("""
|
||||
{
|
||||
"output_text": "Slide text\n\n```json\n{ \"crop\": { \"x\": 1, \"y\": 2, \"width\": 3, \"height\": 4 }, \"attendees\": \"Ada\" }\n```"
|
||||
}
|
||||
""");
|
||||
var handler = new RecordingHandler(CreateNonStreamingTextResponse(
|
||||
"""
|
||||
Slide text
|
||||
|
||||
```json
|
||||
{ "crop": { "x": 1, "y": 2, "width": 3, "height": 4 }, "attendees": "Ada" }
|
||||
```
|
||||
"""));
|
||||
var client = new LiteLlmScreenshotOcrClient(
|
||||
() => handler,
|
||||
NullLogger<LiteLlmScreenshotOcrClient>.Instance);
|
||||
@@ -190,7 +233,8 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
{
|
||||
Agent =
|
||||
{
|
||||
Key = "agent-key"
|
||||
Key = "agent-key",
|
||||
UseStreaming = false
|
||||
}
|
||||
};
|
||||
|
||||
@@ -208,10 +252,12 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
private sealed class RecordingHandler : HttpMessageHandler
|
||||
{
|
||||
private readonly string responseBody;
|
||||
private readonly string mediaType;
|
||||
|
||||
public RecordingHandler(string responseBody)
|
||||
public RecordingHandler(string responseBody, string mediaType = "application/json")
|
||||
{
|
||||
this.responseBody = responseBody;
|
||||
this.mediaType = mediaType;
|
||||
}
|
||||
|
||||
public Uri? RequestUri { get; private set; }
|
||||
@@ -231,7 +277,7 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
: await request.Content.ReadAsStringAsync(cancellationToken);
|
||||
return new HttpResponseMessage(HttpStatusCode.OK)
|
||||
{
|
||||
Content = new StringContent(responseBody, Encoding.UTF8, "application/json")
|
||||
Content = new StringContent(responseBody, Encoding.UTF8, mediaType)
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -249,6 +295,55 @@ public sealed class LiteLlmScreenshotOcrClientTests
|
||||
return stream.ToArray();
|
||||
}
|
||||
|
||||
private static string CreateNonStreamingTextResponse(string text)
|
||||
{
|
||||
return JsonSerializer.Serialize(new
|
||||
{
|
||||
id = "resp_ocr",
|
||||
created_at = 1779147100,
|
||||
model = "vision-model",
|
||||
@object = "response",
|
||||
output = new[]
|
||||
{
|
||||
new
|
||||
{
|
||||
id = "msg_ocr",
|
||||
type = "message",
|
||||
status = "completed",
|
||||
content = new[]
|
||||
{
|
||||
new
|
||||
{
|
||||
type = "output_text",
|
||||
annotations = Array.Empty<object>(),
|
||||
text
|
||||
}
|
||||
},
|
||||
role = "assistant"
|
||||
}
|
||||
},
|
||||
parallel_tool_calls = true,
|
||||
status = "completed",
|
||||
store = false
|
||||
});
|
||||
}
|
||||
|
||||
private static string CreateStreamedTextResponse(string text)
|
||||
{
|
||||
var delta = JsonSerializer.Serialize(new
|
||||
{
|
||||
type = "response.output_text.delta",
|
||||
item_id = "msg_ocr",
|
||||
output_index = 0,
|
||||
content_index = 0,
|
||||
delta = text
|
||||
});
|
||||
return
|
||||
$"data: {delta}{Environment.NewLine}{Environment.NewLine}" +
|
||||
"""data: {"type":"response.completed","response":{"id":"resp_ocr","created_at":1779147100,"model":"vision-model","object":"response","output":[],"parallel_tool_calls":true,"status":"completed","store":false}}""" +
|
||||
$"{Environment.NewLine}{Environment.NewLine}data: [DONE]{Environment.NewLine}{Environment.NewLine}";
|
||||
}
|
||||
|
||||
private static async Task<string> CreateScreenshotAsync(byte[] bytes)
|
||||
{
|
||||
var screenshotPath = Path.Combine(Path.GetTempPath(), "meeting-assistant-tests", Guid.NewGuid().ToString("N") + ".png");
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
using MeetingAssistant.Recording;
|
||||
using Microsoft.Extensions.Logging.Abstractions;
|
||||
|
||||
namespace MeetingAssistant.Tests;
|
||||
|
||||
public sealed class MicrophoneAudioSourceTests
|
||||
{
|
||||
[Fact]
|
||||
public async Task CaptureMovesToNewlyResolvedMicrophoneWhenCurrentCaptureFails()
|
||||
{
|
||||
var captureSources = new SequenceMicrophoneCaptureSourceFactory(
|
||||
new FailingAfterChunkAudioSource(Pcm16(1_000)),
|
||||
new ActiveAudioSource(Pcm16(2_000)));
|
||||
var source = new MicrophoneAudioSource(
|
||||
captureSources,
|
||||
NullLogger<MicrophoneAudioSource>.Instance,
|
||||
TimeSpan.Zero);
|
||||
using var cancellation = new CancellationTokenSource(TimeSpan.FromSeconds(5));
|
||||
await using var chunks = source
|
||||
.CaptureAsync(new MeetingAssistantOptions(), cancellation.Token)
|
||||
.GetAsyncEnumerator(cancellation.Token);
|
||||
|
||||
Assert.True(await chunks.MoveNextAsync());
|
||||
Assert.Equal(1_000, BitConverter.ToInt16(chunks.Current.Pcm));
|
||||
Assert.True(await chunks.MoveNextAsync());
|
||||
Assert.Equal(2_000, BitConverter.ToInt16(chunks.Current.Pcm));
|
||||
Assert.Equal(2, captureSources.CaptureCreationCount);
|
||||
|
||||
await cancellation.CancelAsync();
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public async Task CaptureWaitsForMicrophoneToBecomeAvailable()
|
||||
{
|
||||
var captureSources = new InitiallyUnavailableMicrophoneCaptureSourceFactory(
|
||||
new ActiveAudioSource(Pcm16(3_000)));
|
||||
var source = new MicrophoneAudioSource(
|
||||
captureSources,
|
||||
NullLogger<MicrophoneAudioSource>.Instance,
|
||||
TimeSpan.Zero);
|
||||
using var cancellation = new CancellationTokenSource(TimeSpan.FromSeconds(5));
|
||||
await using var chunks = source
|
||||
.CaptureAsync(new MeetingAssistantOptions(), cancellation.Token)
|
||||
.GetAsyncEnumerator(cancellation.Token);
|
||||
|
||||
Assert.True(await chunks.MoveNextAsync());
|
||||
Assert.Equal(3_000, BitConverter.ToInt16(chunks.Current.Pcm));
|
||||
Assert.Equal(2, captureSources.CaptureCreationCount);
|
||||
|
||||
await cancellation.CancelAsync();
|
||||
}
|
||||
|
||||
private static byte[] Pcm16(short sample)
|
||||
{
|
||||
return BitConverter.GetBytes(sample);
|
||||
}
|
||||
|
||||
private sealed class SequenceMicrophoneCaptureSourceFactory(params IMeetingAudioSource[] sources)
|
||||
: IMicrophoneCaptureSourceFactory
|
||||
{
|
||||
private readonly Queue<IMeetingAudioSource> sources = new(sources);
|
||||
|
||||
public int CaptureCreationCount { get; private set; }
|
||||
|
||||
public IMeetingAudioSource CreateCapture(MeetingAssistantOptions options)
|
||||
{
|
||||
CaptureCreationCount++;
|
||||
return sources.Dequeue();
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class InitiallyUnavailableMicrophoneCaptureSourceFactory(IMeetingAudioSource availableSource)
|
||||
: IMicrophoneCaptureSourceFactory
|
||||
{
|
||||
public int CaptureCreationCount { get; private set; }
|
||||
|
||||
public IMeetingAudioSource CreateCapture(MeetingAssistantOptions options)
|
||||
{
|
||||
CaptureCreationCount++;
|
||||
if (CaptureCreationCount == 1)
|
||||
{
|
||||
throw new InvalidOperationException("No microphone is currently available.");
|
||||
}
|
||||
|
||||
return availableSource;
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class FailingAfterChunkAudioSource(byte[] pcm) : IMeetingAudioSource
|
||||
{
|
||||
public async IAsyncEnumerable<AudioChunk> CaptureAsync(
|
||||
[System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken cancellationToken)
|
||||
{
|
||||
await Task.Yield();
|
||||
yield return new AudioChunk(pcm, 16000, 1);
|
||||
throw new InvalidOperationException("The active microphone was disconnected.");
|
||||
}
|
||||
}
|
||||
|
||||
private sealed class ActiveAudioSource(byte[] pcm) : IMeetingAudioSource
|
||||
{
|
||||
public async IAsyncEnumerable<AudioChunk> CaptureAsync(
|
||||
[System.Runtime.CompilerServices.EnumeratorCancellation] CancellationToken cancellationToken)
|
||||
{
|
||||
yield return new AudioChunk(pcm, 16000, 1);
|
||||
await Task.Delay(Timeout.InfiniteTimeSpan, cancellationToken);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -51,4 +51,17 @@ public sealed class MicrophoneSelectionTests
|
||||
|
||||
Assert.Equal("runtime-id", selected?.Id);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void UnavailableDefaultMicrophoneFallsBackToAnotherActiveDevice()
|
||||
{
|
||||
var selection = new MicrophoneDeviceSelection();
|
||||
|
||||
var selected = selection.Resolve(
|
||||
configuredDeviceId: null,
|
||||
new MicrophoneDevice("disconnected-id", "disconnected microphone"),
|
||||
[new MicrophoneDevice("backup-id", "backup microphone")]);
|
||||
|
||||
Assert.Equal("backup-id", selected?.Id);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,22 +13,23 @@ public sealed class TaskbarIconTests
|
||||
{
|
||||
var menu = MeetingTaskbarMenuBuilder.Build(
|
||||
Status(),
|
||||
[Profile("default", "Ctrl+Alt+M"), Profile("english", "Ctrl+Alt+L")]);
|
||||
[Profile("default", "Ctrl+Alt+M"), Profile("english", "Ctrl+Alt+L")],
|
||||
[new MicrophoneDevice("integrated", "integrated microphone")],
|
||||
"integrated");
|
||||
|
||||
Assert.Equal(RecordingProcessState.Idle, menu.State);
|
||||
Assert.Contains(menu.Items, item =>
|
||||
item.Action == MeetingTaskbarAction.EditRules &&
|
||||
item.Text == "Open agent");
|
||||
Assert.Contains(menu.Items, item =>
|
||||
item.Action == MeetingTaskbarAction.StartRecording &&
|
||||
item.ProfileName == "default" &&
|
||||
item.Text == "Start meeting recording (default)\tCtrl+Alt+M");
|
||||
Assert.Contains(menu.Items, item =>
|
||||
item.Action == MeetingTaskbarAction.StartRecording &&
|
||||
item.ProfileName == "english" &&
|
||||
item.Text == "Start meeting recording (english)\tCtrl+Alt+L");
|
||||
Assert.DoesNotContain(menu.Items, item => item.Action == MeetingTaskbarAction.StopRecording);
|
||||
Assert.DoesNotContain(menu.Items, item => item.Action == MeetingTaskbarAction.AbortRecording);
|
||||
AssertMenuLayout(
|
||||
menu,
|
||||
("Open agent", MeetingTaskbarAction.EditRules, false),
|
||||
("Microphone", MeetingTaskbarAction.OpenSubmenu, true),
|
||||
("Start meeting recording (default)\tCtrl+Alt+M", MeetingTaskbarAction.StartRecording, false),
|
||||
("Start meeting recording (english)\tCtrl+Alt+L", MeetingTaskbarAction.StartRecording, false),
|
||||
("Exit", MeetingTaskbarAction.Exit, true));
|
||||
Assert.Equal(
|
||||
["default", "english"],
|
||||
menu.Items
|
||||
.Where(item => item.Action == MeetingTaskbarAction.StartRecording)
|
||||
.Select(item => item.ProfileName));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
@@ -60,27 +61,41 @@ public sealed class TaskbarIconTests
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void RecordingMenuOffersStopAbortAndOtherProfileSwitches()
|
||||
public void RecordingMenuPrioritizesFinishMeetingInDedicatedSection()
|
||||
{
|
||||
var menu = MeetingTaskbarMenuBuilder.Build(
|
||||
Status(isRecording: true, state: RecordingProcessState.Recording, profile: "default"),
|
||||
[Profile("default", "Ctrl+Alt+M"), Profile("english", "Ctrl+Alt+L"), Profile("french", "Ctrl+Alt+F")]);
|
||||
[Profile("default", "Ctrl+Alt+M"), Profile("english", "Ctrl+Alt+L")],
|
||||
[new MicrophoneDevice("integrated", "integrated microphone")],
|
||||
"integrated");
|
||||
|
||||
Assert.Equal(RecordingProcessState.Recording, menu.State);
|
||||
Assert.Contains(menu.Items, item => item.Action == MeetingTaskbarAction.StopRecording);
|
||||
Assert.Contains(menu.Items, item => item.Action == MeetingTaskbarAction.AbortRecording);
|
||||
Assert.Contains(menu.Items, item =>
|
||||
item.Action == MeetingTaskbarAction.SwitchProfile &&
|
||||
item.ProfileName == "english" &&
|
||||
item.Text == "Switch to english\tCtrl+Alt+L");
|
||||
Assert.Contains(menu.Items, item =>
|
||||
item.Action == MeetingTaskbarAction.SwitchProfile &&
|
||||
item.ProfileName == "french" &&
|
||||
item.Text == "Switch to french\tCtrl+Alt+F");
|
||||
Assert.DoesNotContain(menu.Items, item =>
|
||||
item.Action == MeetingTaskbarAction.SwitchProfile &&
|
||||
item.ProfileName == "default");
|
||||
Assert.DoesNotContain(menu.Items, item => item.Action == MeetingTaskbarAction.StartRecording);
|
||||
AssertMenuLayout(
|
||||
menu,
|
||||
("Open agent", MeetingTaskbarAction.EditRules, false),
|
||||
("Finish meeting", MeetingTaskbarAction.StopRecording, true),
|
||||
("Microphone", MeetingTaskbarAction.OpenSubmenu, true),
|
||||
("Cancel meeting recording and discard", MeetingTaskbarAction.AbortRecording, false),
|
||||
("Switch to english\tCtrl+Alt+L", MeetingTaskbarAction.SwitchProfile, false),
|
||||
("Exit", MeetingTaskbarAction.Exit, true));
|
||||
Assert.Equal(
|
||||
"english",
|
||||
Assert.Single(menu.Items, item => item.Action == MeetingTaskbarAction.SwitchProfile).ProfileName);
|
||||
}
|
||||
|
||||
[Fact]
|
||||
public void RecordingMenuKeepsFinishMeetingIsolatedWithoutMicrophones()
|
||||
{
|
||||
var menu = MeetingTaskbarMenuBuilder.Build(
|
||||
Status(isRecording: true, state: RecordingProcessState.Recording, profile: "default"),
|
||||
[Profile("default")]);
|
||||
|
||||
AssertMenuLayout(
|
||||
menu,
|
||||
("Open agent", MeetingTaskbarAction.EditRules, false),
|
||||
("Finish meeting", MeetingTaskbarAction.StopRecording, true),
|
||||
("Cancel meeting recording and discard", MeetingTaskbarAction.AbortRecording, true),
|
||||
("Exit", MeetingTaskbarAction.Exit, true));
|
||||
}
|
||||
|
||||
[Fact]
|
||||
@@ -173,6 +188,15 @@ public sealed class TaskbarIconTests
|
||||
});
|
||||
}
|
||||
|
||||
private static void AssertMenuLayout(
|
||||
MeetingTaskbarMenu menu,
|
||||
params (string Text, MeetingTaskbarAction Action, bool StartsSection)[] expected)
|
||||
{
|
||||
Assert.Equal(
|
||||
expected,
|
||||
menu.Items.Select(item => (item.Text, item.Action, item.StartsSection)));
|
||||
}
|
||||
|
||||
private static RecordingStatus Status(
|
||||
bool isRecording = false,
|
||||
RecordingProcessState state = RecordingProcessState.Idle,
|
||||
|
||||
@@ -20,13 +20,13 @@
|
||||
</ItemGroup>
|
||||
|
||||
<ItemGroup>
|
||||
<PackageReference Include="Microsoft.Agents.AI.OpenAI" Version="1.13.0" />
|
||||
<PackageReference Include="Microsoft.Agents.AI.OpenAI" Version="1.17.0" />
|
||||
<PackageReference Include="DiffPlex" Version="1.9.0" />
|
||||
<PackageReference Include="Microsoft.CognitiveServices.Speech" Version="$(MicrosoftSpeechVersion)" />
|
||||
<PackageReference Include="Microsoft.CognitiveServices.Speech.Extension.MAS" Version="$(MicrosoftSpeechVersion)" ExcludeAssets="build" />
|
||||
<PackageReference Include="Microsoft.EntityFrameworkCore.Sqlite" Version="10.0.10" />
|
||||
<PackageReference Include="NAudio" Version="2.3.0" />
|
||||
<PackageReference Include="NCalcSync" Version="7.0.0" />
|
||||
<PackageReference Include="NCalcSync" Version="6.4.0" />
|
||||
<PackageReference Include="RazorLight" Version="2.3.1" />
|
||||
<PackageReference Include="SQLitePCLRaw.bundle_e_sqlite3" Version="3.0.5" />
|
||||
<PackageReference Include="System.Drawing.Common" Version="10.0.10" />
|
||||
@@ -65,6 +65,7 @@
|
||||
<ItemGroup Condition="$([MSBuild]::GetTargetPlatformIdentifier('$(TargetFramework)')) != 'windows'">
|
||||
<Compile Remove="Hotkeys\GlobalHotkeyService.cs" />
|
||||
<Compile Remove="Recording\NaudioCaptureSource.cs" />
|
||||
<Compile Remove="Recording\WindowsMicrophoneDeviceProvider.cs" />
|
||||
<Compile Remove="MeetingNotes\OutlookClassicMeetingMetadataProvider.Windows.cs" />
|
||||
<Compile Remove="Screenshots\ActiveWindowScreenshotCapture.Windows.cs" />
|
||||
<Compile Remove="Taskbar\UnoTaskbarIconService.Windows.cs" />
|
||||
|
||||
@@ -20,7 +20,11 @@ builder.Services.Configure<MeetingAssistantOptions>(builder.Configuration.GetSec
|
||||
builder.Services.AddSingleton<ILaunchProfileOptionsProvider, ConfigurationLaunchProfileOptionsProvider>();
|
||||
#if WINDOWS
|
||||
builder.Services.AddSingleton<MicrophoneDeviceSelection>();
|
||||
builder.Services.AddSingleton<IMicrophoneDeviceProvider, WindowsMicrophoneDeviceProvider>();
|
||||
builder.Services.AddSingleton<WindowsMicrophoneDeviceProvider>();
|
||||
builder.Services.AddSingleton<IMicrophoneDeviceProvider>(services =>
|
||||
services.GetRequiredService<WindowsMicrophoneDeviceProvider>());
|
||||
builder.Services.AddSingleton<IMicrophoneCaptureSourceFactory>(services =>
|
||||
services.GetRequiredService<WindowsMicrophoneDeviceProvider>());
|
||||
builder.Services.AddSingleton<MicrophoneAudioSource>();
|
||||
builder.Services.AddSingleton<SystemAudioSource>();
|
||||
builder.Services.AddSingleton<IAcousticEchoCancellerFactory, AdaptiveFilterAcousticEchoCancellerFactory>();
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
namespace MeetingAssistant.Recording;
|
||||
|
||||
public interface IMicrophoneCaptureSourceFactory
|
||||
{
|
||||
IMeetingAudioSource CreateCapture(MeetingAssistantOptions options);
|
||||
}
|
||||
@@ -1,5 +1,3 @@
|
||||
using NAudio.Wave;
|
||||
|
||||
namespace MeetingAssistant.Recording;
|
||||
|
||||
public interface IMicrophoneDeviceProvider
|
||||
@@ -7,6 +5,4 @@ public interface IMicrophoneDeviceProvider
|
||||
IReadOnlyList<MicrophoneDevice> GetAvailableMicrophones();
|
||||
|
||||
MicrophoneDeviceSnapshot GetMicrophoneSnapshot(MeetingAssistantOptions options);
|
||||
|
||||
IWaveIn CreateCapture(MeetingAssistantOptions options);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
using System.Runtime.CompilerServices;
|
||||
|
||||
namespace MeetingAssistant.Recording;
|
||||
|
||||
public sealed class MicrophoneAudioSource : IMeetingAudioSource
|
||||
{
|
||||
private static readonly TimeSpan DefaultRecoveryDelay = TimeSpan.FromSeconds(1);
|
||||
private readonly IMicrophoneCaptureSourceFactory captureSources;
|
||||
private readonly ILogger<MicrophoneAudioSource> logger;
|
||||
private readonly TimeSpan recoveryDelay;
|
||||
|
||||
public MicrophoneAudioSource(
|
||||
IMicrophoneCaptureSourceFactory captureSources,
|
||||
ILogger<MicrophoneAudioSource> logger)
|
||||
: this(captureSources, logger, DefaultRecoveryDelay)
|
||||
{
|
||||
}
|
||||
|
||||
internal MicrophoneAudioSource(
|
||||
IMicrophoneCaptureSourceFactory captureSources,
|
||||
ILogger<MicrophoneAudioSource> logger,
|
||||
TimeSpan recoveryDelay)
|
||||
{
|
||||
this.captureSources = captureSources;
|
||||
this.logger = logger;
|
||||
this.recoveryDelay = recoveryDelay;
|
||||
}
|
||||
|
||||
public IAsyncEnumerable<AudioChunk> CaptureAsync(CancellationToken cancellationToken)
|
||||
{
|
||||
return CaptureAsync(new MeetingAssistantOptions(), cancellationToken);
|
||||
}
|
||||
|
||||
public async IAsyncEnumerable<AudioChunk> CaptureAsync(
|
||||
MeetingAssistantOptions options,
|
||||
[EnumeratorCancellation] CancellationToken cancellationToken)
|
||||
{
|
||||
var failedAttempts = 0;
|
||||
while (!cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
IAsyncEnumerator<AudioChunk>? capture = null;
|
||||
Exception? failure = null;
|
||||
|
||||
try
|
||||
{
|
||||
capture = captureSources
|
||||
.CreateCapture(options)
|
||||
.CaptureAsync(options, cancellationToken)
|
||||
.GetAsyncEnumerator(cancellationToken);
|
||||
}
|
||||
catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
}
|
||||
catch (Exception exception)
|
||||
{
|
||||
failure = exception;
|
||||
}
|
||||
|
||||
if (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
yield break;
|
||||
}
|
||||
|
||||
if (capture is not null)
|
||||
{
|
||||
try
|
||||
{
|
||||
while (!cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
var hasNext = false;
|
||||
try
|
||||
{
|
||||
hasNext = await capture.MoveNextAsync();
|
||||
}
|
||||
catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
}
|
||||
catch (Exception exception)
|
||||
{
|
||||
failure = exception;
|
||||
}
|
||||
|
||||
if (cancellationToken.IsCancellationRequested || failure is not null || !hasNext)
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
if (failedAttempts > 0)
|
||||
{
|
||||
logger.LogInformation(
|
||||
"Microphone capture recovered after {FailedAttemptCount} failed attempt(s)",
|
||||
failedAttempts);
|
||||
failedAttempts = 0;
|
||||
}
|
||||
|
||||
yield return capture.Current;
|
||||
}
|
||||
}
|
||||
finally
|
||||
{
|
||||
try
|
||||
{
|
||||
await capture.DisposeAsync();
|
||||
}
|
||||
catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
}
|
||||
catch (Exception exception)
|
||||
{
|
||||
failure ??= exception;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
yield break;
|
||||
}
|
||||
|
||||
failedAttempts++;
|
||||
logger.LogWarning(
|
||||
failure,
|
||||
"Microphone capture stopped unexpectedly; re-resolving an available microphone in {RecoveryDelay}",
|
||||
recoveryDelay);
|
||||
|
||||
if (!await WaitForRecoveryAsync(cancellationToken))
|
||||
{
|
||||
yield break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private async Task<bool> WaitForRecoveryAsync(CancellationToken cancellationToken)
|
||||
{
|
||||
try
|
||||
{
|
||||
await Task.Delay(recoveryDelay, cancellationToken);
|
||||
return true;
|
||||
}
|
||||
catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -34,6 +34,8 @@ public sealed class MicrophoneDeviceSelection
|
||||
var selected = SelectedDeviceId;
|
||||
return FindById(availableDevices, selected) ??
|
||||
FindById(availableDevices, configuredDeviceId) ??
|
||||
FindById(availableDevices, defaultDevice?.Id) ??
|
||||
availableDevices.FirstOrDefault() ??
|
||||
defaultDevice;
|
||||
}
|
||||
|
||||
|
||||
@@ -4,41 +4,28 @@ using NAudio.Wave;
|
||||
|
||||
namespace MeetingAssistant.Recording;
|
||||
|
||||
public sealed class MicrophoneAudioSource : IMeetingAudioSource
|
||||
internal sealed class NaudioCaptureAudioSource : IMeetingAudioSource
|
||||
{
|
||||
private readonly IMicrophoneDeviceProvider microphones;
|
||||
private readonly ILogger<MicrophoneAudioSource> logger;
|
||||
private readonly IWaveIn capture;
|
||||
private readonly string sourceName;
|
||||
private readonly ILogger logger;
|
||||
|
||||
public MicrophoneAudioSource(
|
||||
IMicrophoneDeviceProvider microphones,
|
||||
ILogger<MicrophoneAudioSource> logger)
|
||||
public NaudioCaptureAudioSource(
|
||||
IWaveIn capture,
|
||||
string sourceName,
|
||||
ILogger logger)
|
||||
{
|
||||
this.microphones = microphones;
|
||||
this.capture = capture;
|
||||
this.sourceName = sourceName;
|
||||
this.logger = logger;
|
||||
}
|
||||
|
||||
public IAsyncEnumerable<AudioChunk> CaptureAsync(CancellationToken cancellationToken)
|
||||
{
|
||||
return CaptureAsync(new MeetingAssistantOptions(), cancellationToken);
|
||||
return CaptureWith(capture, sourceName, logger, cancellationToken);
|
||||
}
|
||||
|
||||
public IAsyncEnumerable<AudioChunk> CaptureAsync(
|
||||
MeetingAssistantOptions options,
|
||||
CancellationToken cancellationToken)
|
||||
{
|
||||
return CaptureAsync(microphones.CreateCapture(options), options, cancellationToken);
|
||||
}
|
||||
|
||||
private IAsyncEnumerable<AudioChunk> CaptureAsync(
|
||||
IWaveIn capture,
|
||||
MeetingAssistantOptions options,
|
||||
CancellationToken cancellationToken)
|
||||
{
|
||||
capture.WaveFormat = new WaveFormat(options.Recording.SampleRate, 16, options.Recording.Channels);
|
||||
return CaptureWith(capture, "microphone", logger, cancellationToken);
|
||||
}
|
||||
|
||||
internal static async IAsyncEnumerable<AudioChunk> CaptureWith(
|
||||
private static async IAsyncEnumerable<AudioChunk> CaptureWith(
|
||||
IWaveIn capture,
|
||||
string sourceName,
|
||||
ILogger logger,
|
||||
@@ -124,6 +111,6 @@ public sealed class SystemAudioSource : IMeetingAudioSource
|
||||
WaveFormat = new WaveFormat(options.Recording.SampleRate, 16, options.Recording.Channels)
|
||||
};
|
||||
|
||||
return MicrophoneAudioSource.CaptureWith(capture, "system", logger, cancellationToken);
|
||||
return new NaudioCaptureAudioSource(capture, "system", logger).CaptureAsync(cancellationToken);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,7 +3,7 @@ using NAudio.Wave;
|
||||
|
||||
namespace MeetingAssistant.Recording;
|
||||
|
||||
public sealed class WindowsMicrophoneDeviceProvider : IMicrophoneDeviceProvider
|
||||
public sealed class WindowsMicrophoneDeviceProvider : IMicrophoneDeviceProvider, IMicrophoneCaptureSourceFactory
|
||||
{
|
||||
private readonly MicrophoneDeviceSelection selection;
|
||||
private readonly ILogger<WindowsMicrophoneDeviceProvider> logger;
|
||||
@@ -42,21 +42,38 @@ public sealed class WindowsMicrophoneDeviceProvider : IMicrophoneDeviceProvider
|
||||
selection.Resolve(options.Recording.MicrophoneDeviceId, GetDefaultMicrophone(), devices));
|
||||
}
|
||||
|
||||
public IWaveIn CreateCapture(MeetingAssistantOptions options)
|
||||
public IMeetingAudioSource CreateCapture(MeetingAssistantOptions options)
|
||||
{
|
||||
var current = GetMicrophoneSnapshot(options).Current;
|
||||
IWaveIn capture;
|
||||
if (current is null)
|
||||
{
|
||||
logger.LogInformation("Starting microphone capture from Windows default capture endpoint");
|
||||
return new WasapiCapture();
|
||||
capture = new WasapiCapture();
|
||||
}
|
||||
else
|
||||
{
|
||||
logger.LogInformation(
|
||||
"Starting microphone capture from {MicrophoneName} ({MicrophoneDeviceId})",
|
||||
current.Name,
|
||||
current.Id);
|
||||
using var enumerator = new MMDeviceEnumerator();
|
||||
capture = new WasapiCapture(enumerator.GetDevice(current.Id));
|
||||
}
|
||||
|
||||
logger.LogInformation(
|
||||
"Starting microphone capture from {MicrophoneName} ({MicrophoneDeviceId})",
|
||||
current.Name,
|
||||
current.Id);
|
||||
using var enumerator = new MMDeviceEnumerator();
|
||||
return new WasapiCapture(enumerator.GetDevice(current.Id));
|
||||
try
|
||||
{
|
||||
capture.WaveFormat = new WaveFormat(
|
||||
options.Recording.SampleRate,
|
||||
16,
|
||||
options.Recording.Channels);
|
||||
return new NaudioCaptureAudioSource(capture, "microphone", logger);
|
||||
}
|
||||
catch
|
||||
{
|
||||
capture.Dispose();
|
||||
throw;
|
||||
}
|
||||
}
|
||||
|
||||
private static MicrophoneDevice? GetDefaultMicrophone()
|
||||
|
||||
@@ -1,15 +1,13 @@
|
||||
using System.Net.Http.Headers;
|
||||
using System.Text;
|
||||
using System.Text.Json;
|
||||
using System.Text.Json.Nodes;
|
||||
using System.Text.RegularExpressions;
|
||||
using MeetingAssistant.MeetingNotes;
|
||||
using MeetingAssistant.Summary;
|
||||
using Microsoft.Extensions.AI;
|
||||
|
||||
namespace MeetingAssistant.Screenshots;
|
||||
|
||||
public sealed partial class LiteLlmScreenshotOcrClient : IScreenshotOcrClient
|
||||
{
|
||||
private static readonly JsonSerializerOptions JsonOptions = new(JsonSerializerDefaults.Web);
|
||||
private readonly ILogger<LiteLlmScreenshotOcrClient> logger;
|
||||
private readonly Func<HttpMessageHandler>? httpMessageHandlerFactory;
|
||||
|
||||
@@ -40,20 +38,30 @@ public sealed partial class LiteLlmScreenshotOcrClient : IScreenshotOcrClient
|
||||
: options.Agent.Model;
|
||||
var key = ResolveApiKey(options);
|
||||
var imageBytes = await File.ReadAllBytesAsync(screenshotPath, cancellationToken);
|
||||
using var httpClient = CreateHttpClient();
|
||||
httpClient.BaseAddress = NormalizeEndpoint(new Uri(endpoint));
|
||||
httpClient.DefaultRequestHeaders.Authorization = new AuthenticationHeaderValue("Bearer", key);
|
||||
var payload = CreatePayload(model, CreatePrompt(prompt, imageBytes), imageBytes);
|
||||
using var content = new StringContent(payload.ToJsonString(JsonOptions), Encoding.UTF8, "application/json");
|
||||
using var response = await httpClient.PostAsync("responses", content, cancellationToken);
|
||||
var responseJson = await response.Content.ReadAsStringAsync(cancellationToken);
|
||||
if (!response.IsSuccessStatusCode)
|
||||
{
|
||||
throw new InvalidOperationException(
|
||||
$"Screenshot OCR request failed with {(int)response.StatusCode} {response.ReasonPhrase}: {responseJson}");
|
||||
}
|
||||
|
||||
var text = ParseOutputText(responseJson);
|
||||
var httpClient = CreateHttpClient();
|
||||
httpClient.BaseAddress = LiteLlmResponsesChatClient.NormalizeEndpoint(new Uri(endpoint));
|
||||
using var chatClient = new LiteLlmResponsesChatClient(
|
||||
httpClient,
|
||||
key,
|
||||
model,
|
||||
enableThinking: false,
|
||||
reasoningEffort: "none",
|
||||
reconnectionAttempts: options.Agent.ReconnectionAttempts,
|
||||
reconnectionDelay: options.Agent.ReconnectionDelay,
|
||||
logger: logger,
|
||||
firstRequestIsUser: false,
|
||||
useStreaming: options.Agent.UseStreaming);
|
||||
var response = await chatClient.GetResponseAsync(
|
||||
[
|
||||
new ChatMessage(
|
||||
ChatRole.User,
|
||||
[
|
||||
new TextContent(CreatePrompt(prompt, imageBytes)),
|
||||
new DataContent(imageBytes, "image/png")
|
||||
])
|
||||
],
|
||||
cancellationToken: cancellationToken);
|
||||
var text = response.Text ?? string.Empty;
|
||||
logger.LogInformation("Screenshot OCR completed for {ScreenshotPath}", screenshotPath);
|
||||
return ParseOcrResult(text);
|
||||
}
|
||||
@@ -65,35 +73,6 @@ public sealed partial class LiteLlmScreenshotOcrClient : IScreenshotOcrClient
|
||||
: new HttpClient(httpMessageHandlerFactory());
|
||||
}
|
||||
|
||||
private static JsonObject CreatePayload(string model, string prompt, byte[] imageBytes)
|
||||
{
|
||||
return new JsonObject
|
||||
{
|
||||
["model"] = model,
|
||||
["store"] = false,
|
||||
["input"] = new JsonArray
|
||||
{
|
||||
new JsonObject
|
||||
{
|
||||
["role"] = "user",
|
||||
["content"] = new JsonArray
|
||||
{
|
||||
new JsonObject
|
||||
{
|
||||
["type"] = "input_text",
|
||||
["text"] = prompt
|
||||
},
|
||||
new JsonObject
|
||||
{
|
||||
["type"] = "input_image",
|
||||
["image_url"] = "data:image/png;base64," + Convert.ToBase64String(imageBytes)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
private static string CreatePrompt(string prompt, byte[] imageBytes)
|
||||
{
|
||||
return TryReadPngDimensions(imageBytes, out var width, out var height)
|
||||
@@ -210,46 +189,6 @@ public sealed partial class LiteLlmScreenshotOcrClient : IScreenshotOcrClient
|
||||
bytes[offset + 3];
|
||||
}
|
||||
|
||||
private static string ParseOutputText(string responseJson)
|
||||
{
|
||||
using var document = JsonDocument.Parse(responseJson);
|
||||
var root = document.RootElement;
|
||||
var parts = new List<string>();
|
||||
if (root.TryGetProperty("output_text", out var outputText) &&
|
||||
outputText.ValueKind == JsonValueKind.String &&
|
||||
!string.IsNullOrWhiteSpace(outputText.GetString()))
|
||||
{
|
||||
parts.Add(outputText.GetString()!);
|
||||
}
|
||||
|
||||
if (root.TryGetProperty("output", out var output) && output.ValueKind == JsonValueKind.Array)
|
||||
{
|
||||
foreach (var item in output.EnumerateArray())
|
||||
{
|
||||
if (!item.TryGetProperty("content", out var content) || content.ValueKind != JsonValueKind.Array)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
foreach (var block in content.EnumerateArray())
|
||||
{
|
||||
if (block.TryGetProperty("type", out var type) &&
|
||||
type.GetString() == "output_text" &&
|
||||
block.TryGetProperty("text", out var text) &&
|
||||
text.ValueKind == JsonValueKind.String &&
|
||||
!string.IsNullOrWhiteSpace(text.GetString()))
|
||||
{
|
||||
parts.Add(text.GetString()!);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return parts.Count == 0
|
||||
? ""
|
||||
: string.Join(Environment.NewLine + Environment.NewLine, parts);
|
||||
}
|
||||
|
||||
private static string ResolveApiKey(MeetingAssistantOptions options)
|
||||
{
|
||||
if (!string.IsNullOrWhiteSpace(options.Screenshots.Ocr.Key))
|
||||
@@ -278,17 +217,6 @@ public sealed partial class LiteLlmScreenshotOcrClient : IScreenshotOcrClient
|
||||
$"No screenshot OCR API key configured. Set MeetingAssistant:Screenshots:Ocr:Key or environment variable '{options.Screenshots.Ocr.KeyEnv}'.");
|
||||
}
|
||||
|
||||
private static Uri NormalizeEndpoint(Uri endpoint)
|
||||
{
|
||||
var value = endpoint.ToString().TrimEnd('/');
|
||||
if (!value.EndsWith("/v1", StringComparison.OrdinalIgnoreCase))
|
||||
{
|
||||
value += "/v1";
|
||||
}
|
||||
|
||||
return new Uri(value + "/");
|
||||
}
|
||||
|
||||
[GeneratedRegex("```json\\s*(?<json>.*?)\\s*```", RegexOptions.Singleline | RegexOptions.IgnoreCase)]
|
||||
private static partial Regex JsonCodeBlockRegex();
|
||||
}
|
||||
|
||||
@@ -659,9 +659,9 @@ public sealed class LiteLlmResponsesChatClient : IChatClient
|
||||
private static void AddInputItem(JsonArray input, StringBuilder instructions, ChatMessage message)
|
||||
{
|
||||
var role = message.Role.Value;
|
||||
var text = message.Text;
|
||||
if (role == ChatRole.System.Value)
|
||||
{
|
||||
var text = message.Text;
|
||||
if (!string.IsNullOrWhiteSpace(text))
|
||||
{
|
||||
instructions.AppendLine(text);
|
||||
@@ -670,21 +670,37 @@ public sealed class LiteLlmResponsesChatClient : IChatClient
|
||||
return;
|
||||
}
|
||||
|
||||
if (!string.IsNullOrWhiteSpace(text))
|
||||
var isAssistant = role == ChatRole.Assistant.Value;
|
||||
var messageContent = new JsonArray();
|
||||
foreach (var content in message.Contents)
|
||||
{
|
||||
if (content is TextContent textContent && !string.IsNullOrWhiteSpace(textContent.Text))
|
||||
{
|
||||
messageContent.Add(new JsonObject
|
||||
{
|
||||
["type"] = isAssistant ? "output_text" : "input_text",
|
||||
["text"] = textContent.Text
|
||||
});
|
||||
}
|
||||
else if (!isAssistant &&
|
||||
content is DataContent dataContent &&
|
||||
dataContent.HasTopLevelMediaType("image"))
|
||||
{
|
||||
messageContent.Add(new JsonObject
|
||||
{
|
||||
["type"] = "input_image",
|
||||
["image_url"] = dataContent.Uri.ToString()
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (messageContent.Count > 0)
|
||||
{
|
||||
var isAssistant = role == ChatRole.Assistant.Value;
|
||||
input.Add(new JsonObject
|
||||
{
|
||||
["type"] = "message",
|
||||
["role"] = isAssistant ? "assistant" : "user",
|
||||
["content"] = new JsonArray
|
||||
{
|
||||
new JsonObject
|
||||
{
|
||||
["type"] = isAssistant ? "output_text" : "input_text",
|
||||
["text"] = text
|
||||
}
|
||||
}
|
||||
["content"] = messageContent
|
||||
});
|
||||
}
|
||||
|
||||
@@ -761,7 +777,7 @@ public sealed class LiteLlmResponsesChatClient : IChatClient
|
||||
return Math.Max(1, (int)Math.Ceiling(json.Length / 4.0));
|
||||
}
|
||||
|
||||
private static Uri NormalizeEndpoint(Uri endpoint)
|
||||
internal static Uri NormalizeEndpoint(Uri endpoint)
|
||||
{
|
||||
var value = endpoint.ToString().TrimEnd('/');
|
||||
if (!value.EndsWith("/v1", StringComparison.OrdinalIgnoreCase))
|
||||
|
||||
@@ -26,7 +26,8 @@ public sealed record MeetingTaskbarMenuItem(
|
||||
string? ProfileName = null,
|
||||
string? MicrophoneDeviceId = null,
|
||||
bool IsChecked = false,
|
||||
IReadOnlyList<MeetingTaskbarMenuItem>? Items = null);
|
||||
IReadOnlyList<MeetingTaskbarMenuItem>? Items = null,
|
||||
bool StartsSection = false);
|
||||
|
||||
public static class MeetingTaskbarMenuBuilder
|
||||
{
|
||||
@@ -41,23 +42,29 @@ public static class MeetingTaskbarMenuBuilder
|
||||
new("Open agent", MeetingTaskbarAction.EditRules)
|
||||
};
|
||||
|
||||
if (status.IsRecording)
|
||||
{
|
||||
items.Add(new MeetingTaskbarMenuItem(
|
||||
"Finish meeting",
|
||||
MeetingTaskbarAction.StopRecording,
|
||||
StartsSection: true));
|
||||
}
|
||||
|
||||
var secondaryControls = new List<MeetingTaskbarMenuItem>();
|
||||
if (microphones is { Count: > 0 })
|
||||
{
|
||||
items.Add(BuildMicrophoneMenu(microphones, currentMicrophoneDeviceId));
|
||||
secondaryControls.Add(BuildMicrophoneMenu(microphones, currentMicrophoneDeviceId));
|
||||
}
|
||||
|
||||
if (status.IsRecording)
|
||||
{
|
||||
items.Add(new MeetingTaskbarMenuItem(
|
||||
"Stop meeting recording and transcribe",
|
||||
MeetingTaskbarAction.StopRecording));
|
||||
items.Add(new MeetingTaskbarMenuItem(
|
||||
secondaryControls.Add(new MeetingTaskbarMenuItem(
|
||||
"Cancel meeting recording and discard",
|
||||
MeetingTaskbarAction.AbortRecording));
|
||||
|
||||
foreach (var profile in launchProfiles.Where(profile => !IsActiveProfile(profile, status)))
|
||||
{
|
||||
items.Add(new MeetingTaskbarMenuItem(
|
||||
secondaryControls.Add(new MeetingTaskbarMenuItem(
|
||||
AppendHotkey($"Switch to {profile.Name}", profile.Options.Hotkey.Toggle),
|
||||
MeetingTaskbarAction.SwitchProfile,
|
||||
profile.Name));
|
||||
@@ -67,16 +74,18 @@ public static class MeetingTaskbarMenuBuilder
|
||||
{
|
||||
foreach (var profile in launchProfiles)
|
||||
{
|
||||
items.Add(new MeetingTaskbarMenuItem(
|
||||
secondaryControls.Add(new MeetingTaskbarMenuItem(
|
||||
AppendHotkey($"Start meeting recording ({profile.Name})", profile.Options.Hotkey.Toggle),
|
||||
MeetingTaskbarAction.StartRecording,
|
||||
profile.Name));
|
||||
}
|
||||
}
|
||||
|
||||
AddSection(items, secondaryControls);
|
||||
items.Add(new MeetingTaskbarMenuItem(
|
||||
"Exit",
|
||||
MeetingTaskbarAction.Exit));
|
||||
MeetingTaskbarAction.Exit,
|
||||
StartsSection: true));
|
||||
|
||||
return new MeetingTaskbarMenu(
|
||||
status.State,
|
||||
@@ -102,6 +111,19 @@ public static class MeetingTaskbarMenuBuilder
|
||||
Items: microphoneItems);
|
||||
}
|
||||
|
||||
private static void AddSection(
|
||||
List<MeetingTaskbarMenuItem> items,
|
||||
IReadOnlyList<MeetingTaskbarMenuItem> section)
|
||||
{
|
||||
if (section.Count == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
items.Add(section[0] with { StartsSection = true });
|
||||
items.AddRange(section.Skip(1));
|
||||
}
|
||||
|
||||
private static string BuildTooltip(RecordingStatus status)
|
||||
{
|
||||
return status.State switch
|
||||
|
||||
@@ -196,14 +196,12 @@ public sealed class UnoTaskbarIconService : IHostedService, IDisposable
|
||||
var popupMenu = new PopupMenu();
|
||||
for (var index = 0; index < menu.Items.Count; index++)
|
||||
{
|
||||
if (index == 1 ||
|
||||
(menu.Items[index].Action == MeetingTaskbarAction.Exit &&
|
||||
menu.Items[index - 1].Action != MeetingTaskbarAction.EditRules))
|
||||
var menuItem = menu.Items[index];
|
||||
if (index > 0 && menuItem.StartsSection)
|
||||
{
|
||||
popupMenu.Items.Add(new PopupMenuSeparator());
|
||||
}
|
||||
|
||||
var menuItem = menu.Items[index];
|
||||
popupMenu.Items.Add(BuildPopupItem(menuItem));
|
||||
}
|
||||
|
||||
@@ -290,7 +288,7 @@ public sealed class UnoTaskbarIconService : IHostedService, IDisposable
|
||||
return string.Join(
|
||||
"|",
|
||||
FlattenMenuItems(menu.Items).Select(item =>
|
||||
$"{item.Action}:{item.ProfileName}:{item.MicrophoneDeviceId}:{item.IsChecked}:{item.Text}"));
|
||||
$"{item.Action}:{item.ProfileName}:{item.MicrophoneDeviceId}:{item.IsChecked}:{item.StartsSection}:{item.Text}"));
|
||||
}
|
||||
|
||||
private static IEnumerable<MeetingTaskbarMenuItem> FlattenMenuItems(
|
||||
|
||||
@@ -317,7 +317,7 @@ When enabled on Windows, Meeting Assistant periodically syncs today's Outlook Cl
|
||||
|
||||
`Screenshots:Hotkey` configures a global hotkey that captures the currently active window during an active meeting. Screenshots are written under `Screenshots:AttachmentsFolder`, which defaults to an `Attachments` folder beside the assistant context note, and each capture appends a timestamped markdown image link to the assistant context.
|
||||
|
||||
`Screenshots:Ocr` optionally enables vision extraction for screenshots. Blank `Endpoint` or `Model` values fall back to the summary `Agent` endpoint and model. `Key` or `KeyEnv` can be set specifically for OCR; otherwise the summary agent key configuration is used. Automatic summarization scans the meeting note for user-added Obsidian image embeds such as `![[whiteboard.png]]` and Markdown image embeds such as ``, adds resolvable local images to the assistant context without copying them or changing the meeting note, runs OCR without crop or attendee updates, and waits for all pending OCR work to complete or hit `Timeout` before the assistant context moves to `summarizing`. Failed or timed-out screenshot OCR writes a retry link that targets `/meetings/screenshot-ocr/retry` with the exact saved screenshot and OCR block id.
|
||||
`Screenshots:Ocr` optionally enables vision extraction for screenshots. Blank `Endpoint` or `Model` values fall back to the summary `Agent` endpoint and model. `Key` or `KeyEnv` can be set specifically for OCR; otherwise the summary agent key configuration is used. Screenshot OCR also inherits `Agent:UseStreaming`, using the Responses SSE transport when it is `true` and the non-streaming Responses transport when it is `false`. Automatic summarization scans the meeting note for user-added Obsidian image embeds such as `![[whiteboard.png]]` and Markdown image embeds such as ``, adds resolvable local images to the assistant context without copying them or changing the meeting note, runs OCR without crop or attendee updates, and waits for all pending OCR work to complete or hit `Timeout` before the assistant context moves to `summarizing`. Failed or timed-out screenshot OCR writes a retry link that targets `/meetings/screenshot-ocr/retry` with the exact saved screenshot and OCR block id.
|
||||
|
||||
| Setting | Purpose |
|
||||
| --- | --- |
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-29
|
||||
@@ -0,0 +1,34 @@
|
||||
## Context
|
||||
|
||||
`LiteLlmScreenshotOcrClient` currently builds and posts a raw Responses JSON payload, then parses the successful body as one JSON document. It inherits endpoint, model, and key values from `AgentOptions`, but never reads `AgentOptions.UseStreaming`. The summary and workflow agents already use `LiteLlmResponsesChatClient`, which selects the OpenAI SDK streaming or non-streaming Responses method and maps both through the Microsoft.Extensions.AI adapter.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Make screenshot OCR use `Agent:UseStreaming` without adding another setting.
|
||||
- Reuse the supported Responses SDK transport and response adapter.
|
||||
- Preserve screenshot-specific endpoint, model, key, prompt, image, crop, attendee, and timeout behavior.
|
||||
- Keep non-streaming screenshot OCR working when streaming is disabled.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Expose OCR token deltas to the UI or assistant context.
|
||||
- Change screenshot OCR retry, crop, attendee, or note-block semantics.
|
||||
- Add file upload or remote image URL support.
|
||||
|
||||
## Decisions
|
||||
|
||||
1. Route screenshot OCR through `LiteLlmResponsesChatClient` instead of maintaining a second Responses parser. The screenshot client will construct one user chat message containing prompt text and PNG `DataContent`, then consume the buffered `ChatResponse.Text`. This keeps transport selection, SDK request creation, SSE assembly, and non-streaming mapping in one client.
|
||||
|
||||
2. Extend the shared Responses message translator to map image `DataContent` to an `input_image` content block using its data URI. Text and image blocks remain in one `type: message` input item, matching the existing OCR payload shape.
|
||||
|
||||
3. Use `AgentOptions.UseStreaming` for screenshot OCR even when the screenshot-specific endpoint or model overrides are set. Endpoint/model/key remain independently overrideable; transport is an agent-wide behavior setting.
|
||||
|
||||
4. Disable reasoning and compaction for the one-turn OCR request, preserving the existing screenshot client behavior while still using the agent reconnection settings and selected transport.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- **[Shared-client diagnostics mention summary context]** Some low-level logs are named for the summary pipeline. → Avoid passing summary compaction state and keep screenshot-specific completion/failure logs at the screenshot client boundary.
|
||||
- **[Multimodal translation expands shared client scope]** Incorrect content mapping could affect summary requests. → Add request-body behavior coverage proving prompt and PNG data URI are preserved, while existing summary message tests protect text translation.
|
||||
- **[Provider image support varies]** A configured model may reject image input. → Preserve the provider error and existing screenshot OCR failure/retry behavior.
|
||||
@@ -0,0 +1,24 @@
|
||||
## Why
|
||||
|
||||
Screenshot OCR inherits its endpoint, model, and key from the summary agent, but it bypasses the shared Responses client and ignores `Agent:UseStreaming`. With streaming enabled, the screenshot client still expects one JSON document and cannot consume the configured LiteLLM Responses event stream.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Make screenshot OCR inherit the summary agent's streaming transport selection.
|
||||
- Send screenshot image input through the same OpenAI Responses SDK and Agent Framework adapter used by the summary client.
|
||||
- Preserve the existing non-streaming screenshot OCR path when streaming is disabled.
|
||||
- Add behavior coverage for streamed and non-streamed screenshot OCR responses.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
- None.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- `meeting-summary`: Require screenshot OCR to honor the configured agent streaming transport while preserving image input and OCR metadata parsing.
|
||||
|
||||
## Impact
|
||||
|
||||
The change affects the screenshot OCR client, the shared LiteLLM Responses message translation, focused tests, and agent configuration documentation. It does not change the local HTTP API or screenshot note format.
|
||||
@@ -0,0 +1,103 @@
|
||||
## MODIFIED Requirements
|
||||
|
||||
### Requirement: Meeting screenshots are captured into assistant context
|
||||
Meeting Assistant SHALL expose a configurable screenshot hotkey.
|
||||
|
||||
When a meeting is active and the screenshot hotkey is pressed, Meeting Assistant SHALL capture the currently active window.
|
||||
|
||||
The screenshot image SHALL be saved into a configurable attachments folder for the assistant context note. By default, the folder SHALL be `Attachments` beside the assistant context note.
|
||||
|
||||
After the image is saved, Meeting Assistant SHALL append a markdown image link to the assistant context note with a meeting-relative timestamp that correlates to transcript timestamps.
|
||||
|
||||
Meeting Assistant SHALL allow optional screenshot OCR configuration with endpoint URL, API key or key environment variable, model, prompt, and timeout.
|
||||
|
||||
When screenshot OCR is configured, Meeting Assistant SHALL send the screenshot and prompt to the configured OpenAI-compatible Responses endpoint and append the model result after the screenshot link in the assistant context note.
|
||||
|
||||
Screenshot OCR SHALL honor the configured `MeetingAssistant:Agent:UseStreaming` transport selection.
|
||||
|
||||
When streaming is enabled, screenshot OCR SHALL consume the Responses result through the supported OpenAI Responses and Agent Framework Server-Sent Events adapter.
|
||||
|
||||
When streaming is disabled, screenshot OCR SHALL consume the result through the supported non-streaming OpenAI Responses client and adapter.
|
||||
|
||||
The screenshot OCR prompt SHALL ask the model to return pixel crop coordinates when it can confidently isolate only the presentation, shared screen, or similarly relevant meeting content.
|
||||
|
||||
When OCR returns valid crop coordinates within the original image bounds, Meeting Assistant SHALL save a cropped PNG beside the original screenshot and SHALL link the cropped image before the OCR result in the assistant context note.
|
||||
|
||||
When OCR returns no crop coordinates or invalid crop coordinates, Meeting Assistant SHALL keep the original screenshot link and OCR result without writing a cropped image.
|
||||
|
||||
After transcription finishes and before summarization starts, Meeting Assistant SHALL scan the meeting note for user-authored Obsidian image embeds and Markdown image embeds.
|
||||
|
||||
When configured screenshot OCR is enabled and the meeting note contains image embeds, Meeting Assistant SHALL append each resolvable image to the assistant context note, state that the image came from the meeting note, preserve the original embed text for cross-reference, and run OCR for the linked image.
|
||||
|
||||
Meeting-note image OCR SHALL NOT copy the image file, SHALL NOT write crop images, SHALL NOT add attendees from OCR metadata, and SHALL NOT modify the meeting note.
|
||||
|
||||
Meeting Assistant SHALL wait for meeting-note image OCR to finish or time out before transitioning the assistant context to summarizing.
|
||||
|
||||
When screenshot OCR fails or times out, Meeting Assistant SHALL write the failure status into the assistant context note with a retry link for that exact screenshot.
|
||||
|
||||
When the screenshot OCR retry link is activated, Meeting Assistant SHALL rerun OCR for the saved screenshot and replace that screenshot's OCR block in the assistant context note.
|
||||
|
||||
When screenshot OCR is not configured, Meeting Assistant SHALL skip OCR and keep the screenshot link.
|
||||
|
||||
The default OCR prompt SHALL explain that the image is from a meeting and ask the model to identify who is talking, who is presenting, what is presented, capture slide text in markdown, convert diagrams to Mermaid when possible, indicate whether visible people are clearly the exact meeting participants or only a partial result, return crop coordinates only for confidently isolated presentation/shared-screen content, and otherwise describe the scene.
|
||||
|
||||
#### Scenario: Screenshot is linked with meeting timestamp
|
||||
- **GIVEN** a meeting started at `10:00:00`
|
||||
- **WHEN** the user captures a screenshot at `10:03:05`
|
||||
- **THEN** Meeting Assistant saves the screenshot under the configured attachments folder
|
||||
- **AND** appends a markdown image link to assistant context with timestamp `[00:03:05]`
|
||||
|
||||
#### Scenario: OCR result is appended after screenshot
|
||||
- **GIVEN** screenshot OCR is configured
|
||||
- **WHEN** the user captures a screenshot
|
||||
- **THEN** Meeting Assistant appends the screenshot link to assistant context
|
||||
- **AND** appends the OCR result for that screenshot after the link when processing completes
|
||||
|
||||
#### Scenario: Streaming screenshot OCR is assembled
|
||||
- **GIVEN** screenshot OCR is configured
|
||||
- **AND** `MeetingAssistant:Agent:UseStreaming` is `true`
|
||||
- **WHEN** the Responses endpoint returns screenshot OCR output as Server-Sent Events
|
||||
- **THEN** Meeting Assistant sends the prompt and screenshot as one multimodal Responses message
|
||||
- **AND** appends the assembled OCR text without a JSON document parse failure
|
||||
|
||||
#### Scenario: Screenshot OCR streaming can be disabled
|
||||
- **GIVEN** screenshot OCR is configured
|
||||
- **AND** `MeetingAssistant:Agent:UseStreaming` is `false`
|
||||
- **WHEN** Meeting Assistant requests screenshot OCR
|
||||
- **THEN** it uses the supported non-streaming Responses client and adapter
|
||||
- **AND** preserves the prompt and screenshot image input
|
||||
|
||||
#### Scenario: OCR crop is saved and linked before OCR text
|
||||
- **GIVEN** screenshot OCR is configured
|
||||
- **AND** OCR returns valid crop coordinates for a shared screen
|
||||
- **WHEN** OCR processing completes
|
||||
- **THEN** Meeting Assistant saves a cropped screenshot beside the original image
|
||||
- **AND** links the cropped screenshot before the OCR text in assistant context
|
||||
|
||||
#### Scenario: Meeting note image embeds are OCRed before summarization
|
||||
- **GIVEN** screenshot OCR is configured
|
||||
- **AND** the meeting note contains `![[whiteboard.png]]`
|
||||
- **AND** the meeting note contains ``
|
||||
- **WHEN** transcription finishes
|
||||
- **THEN** Meeting Assistant appends both images to the assistant context as images from the meeting note
|
||||
- **AND** includes the original embed text for each image
|
||||
- **AND** runs OCR for each image without copying files, writing crop images, adding attendees, or modifying the meeting note
|
||||
- **AND** waits for this OCR to finish or time out before transitioning to summarizing
|
||||
|
||||
#### Scenario: OCR failure can be retried for the same screenshot
|
||||
- **GIVEN** screenshot OCR is configured
|
||||
- **AND** OCR fails or times out for a captured screenshot
|
||||
- **WHEN** Meeting Assistant writes the OCR failure status
|
||||
- **THEN** the assistant context includes a retry link for that exact screenshot
|
||||
- **WHEN** the retry link is activated
|
||||
- **THEN** Meeting Assistant reruns OCR against the saved screenshot
|
||||
- **AND** replaces that screenshot's OCR block with the retry result
|
||||
|
||||
#### Scenario: OCR is skipped when not configured
|
||||
- **GIVEN** screenshot OCR is not configured
|
||||
- **WHEN** the user captures a screenshot
|
||||
- **THEN** Meeting Assistant saves and links the screenshot without calling a model endpoint
|
||||
|
||||
#### Scenario: OCR reports whether visible people are complete or partial
|
||||
- **WHEN** Meeting Assistant uses the built-in screenshot OCR prompt
|
||||
- **THEN** the prompt asks the model to state whether the screenshot clearly shows exactly who is in the meeting or only a partial participant result
|
||||
@@ -0,0 +1,15 @@
|
||||
## 1. Streaming screenshot OCR
|
||||
|
||||
- [x] 1.1 Add a failing screenshot-client behavior test for an SSE response with prompt and image input.
|
||||
- [x] 1.2 Route screenshot OCR through the shared Responses client and add multimodal message translation.
|
||||
|
||||
## 2. Non-streaming compatibility
|
||||
|
||||
- [x] 2.1 Add behavior coverage proving `Agent:UseStreaming=false` preserves non-streaming screenshot OCR and image input.
|
||||
- [x] 2.2 Document that screenshot OCR inherits the agent streaming setting.
|
||||
|
||||
## 3. Verification
|
||||
|
||||
- [x] 3.1 Refactor the touched screenshot and shared client paths for DRYness, SOLID boundaries, and simplicity while preserving behavior.
|
||||
- [x] 3.2 Run focused tests, the full solution tests, and strict OpenSpec validation.
|
||||
- [x] 3.3 Restart Meeting Assistant only while idle and verify screenshot OCR against the deployed LiteLLM endpoint.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-08-04
|
||||
@@ -0,0 +1,45 @@
|
||||
## Context
|
||||
|
||||
The tray-menu builder currently returns a flat list of semantic actions, while the Windows renderer infers separators from item indexes and the Exit action. During an active recording, the normal stop action is added after the microphone submenu and uses a long implementation-oriented label. This makes the primary meeting-completion action look equivalent to cancel, profile switching, and device selection.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Give normal meeting completion the concise label `Finish meeting`.
|
||||
- Make that action the only item in the section immediately below `Open agent` while recording.
|
||||
- Keep fine-grained recording controls in a distinct following section.
|
||||
- Make section boundaries observable in platform-independent menu behavior tests.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Change what normal stop, abort, profile switching, or microphone selection does.
|
||||
- Change idle-menu actions, hotkeys, endpoints, or recording state transitions.
|
||||
- Add icons, confirmation prompts, or nested submenus.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Represent section starts in the menu model
|
||||
|
||||
Add a section-start flag to `MeetingTaskbarMenuItem`. The Windows renderer will insert a separator before items carrying the flag instead of deriving layout from array indexes and action types.
|
||||
|
||||
This keeps layout intent in the platform-independent builder where behavior tests can observe it. Keeping another renderer-only special case was rejected because it would leave the requested prominence untestable without Windows UI automation.
|
||||
|
||||
### Build prioritized and fine-grained controls as separate groups
|
||||
|
||||
While recording, the builder will add `Open agent`, then `Finish meeting` as a new section, then collect microphone, cancel/discard, and profile-switch actions into a fine-grained group whose first item starts another section. Exit remains the final section.
|
||||
|
||||
The action continues to use the existing normal-stop command so transcription, speaker processing, OCR, and summarization semantics do not change.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- **A section flag could produce adjacent separators if assigned carelessly** → The builder marks only the first item of each non-empty group, and the renderer follows those explicit starts.
|
||||
- **Menu ordering changes while recording** → Limit reordering to the active-recording state; idle and processing actions retain their existing relative order.
|
||||
|
||||
## Migration Plan
|
||||
|
||||
No configuration or data migration is required. Deploying the updated executable changes only tray-menu presentation. Rollback restores the previous label and grouping.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None.
|
||||
@@ -0,0 +1,25 @@
|
||||
## Why
|
||||
|
||||
The active-recording tray menu labels its most important completion action as the verbose `Stop meeting recording and transcribe` and groups it with rarely used controls. Finishing a meeting should be immediately recognizable and visually prioritized during normal use.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Rename the active-recording stop action to `Finish meeting` without changing its normal stop, transcription, or summary behavior.
|
||||
- Place `Finish meeting` by itself in the section immediately below `Open agent`.
|
||||
- Place microphone selection, cancel/discard, and profile-switch controls in a separate lower-priority section.
|
||||
- Represent tray-menu section boundaries explicitly so ordering and prominence are behavior-testable.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
None.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- `meeting-recording`: Prioritize the normal meeting completion action in the Windows tray menu with a concise label and dedicated section.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects the platform-independent tray-menu model/builder, Windows tray-menu rendering, and taskbar behavior tests.
|
||||
- Does not change recording lifecycle semantics, hotkeys, endpoints, or generated meeting artifacts.
|
||||
+62
@@ -0,0 +1,62 @@
|
||||
## MODIFIED Requirements
|
||||
|
||||
### Requirement: Windows taskbar icon controls recording
|
||||
Meeting Assistant SHALL show a Windows taskbar notification icon when running on Windows.
|
||||
|
||||
The taskbar icon SHALL indicate whether the newest meeting process is idle, actively recording, or post-recording processing/summarizing.
|
||||
|
||||
When a new meeting is actively recording while an older stopped meeting is still transcribing, recognizing speakers, or summarizing, the taskbar icon SHALL show the new active recording state.
|
||||
|
||||
The taskbar icon right-click menu SHALL expose recording controls based on the current state and configured launch profiles.
|
||||
|
||||
The taskbar icon right-click menu SHALL expose an Exit action in every recording state.
|
||||
|
||||
When Meeting Assistant is idle or only processing older stopped meetings, the menu SHALL allow starting a meeting recording for each configured launch profile.
|
||||
|
||||
When a meeting is actively recording, the menu SHALL allow stopping the recording and continuing transcription/summary generation.
|
||||
|
||||
During an active recording, the normal stop action SHALL be labeled `Finish meeting` and SHALL be the only action in a dedicated menu section immediately below the `Open agent` section.
|
||||
|
||||
During an active recording, microphone selection, cancel/discard, and profile-switch actions SHALL appear in a separate fine-grained controls section below `Finish meeting`.
|
||||
|
||||
When a meeting is actively recording, the menu SHALL allow canceling the recording and discarding that run's artifacts.
|
||||
|
||||
When a meeting is actively recording, the menu SHALL allow switching to each configured launch profile other than the current active profile.
|
||||
|
||||
Selecting Exit while Meeting Assistant is idle SHALL stop the application without an additional confirmation prompt.
|
||||
|
||||
Selecting Exit while Meeting Assistant is recording, transcribing, recognizing speakers, or summarizing SHALL show a confirmation dialog before stopping the application.
|
||||
|
||||
#### Scenario: Idle tray menu can start configured profiles
|
||||
- **GIVEN** launch profiles `default` and `english` are configured
|
||||
- **AND** no meeting recording is active
|
||||
- **WHEN** the taskbar menu is opened
|
||||
- **THEN** it offers start recording actions for `default` and `english`
|
||||
|
||||
#### Scenario: Recording tray menu prioritizes finishing the meeting
|
||||
- **GIVEN** launch profiles `default` and `english` are configured
|
||||
- **AND** a meeting is actively recording with profile `default`
|
||||
- **WHEN** the taskbar menu is opened
|
||||
- **THEN** `Finish meeting` is the only action in the section immediately below `Open agent`
|
||||
- **AND** microphone selection, cancel/discard, and switching to `english` appear in a separate following section
|
||||
- **AND** the menu does not offer switching to `default`
|
||||
|
||||
#### Scenario: Active recording has priority over older summarizing runs
|
||||
- **GIVEN** an older meeting is still summarizing
|
||||
- **WHEN** a newer meeting is actively recording
|
||||
- **THEN** the taskbar icon indicates recording
|
||||
|
||||
#### Scenario: Tray menu always exposes Exit
|
||||
- **GIVEN** Meeting Assistant is running
|
||||
- **WHEN** the taskbar menu is opened
|
||||
- **THEN** it offers an Exit action
|
||||
|
||||
#### Scenario: Idle Exit stops immediately
|
||||
- **GIVEN** no recording, transcription, speaker recognition, or summary work is running
|
||||
- **WHEN** the user selects Exit from the taskbar menu
|
||||
- **THEN** Meeting Assistant stops the application without an additional confirmation prompt
|
||||
|
||||
#### Scenario: In-progress Exit asks for confirmation
|
||||
- **GIVEN** Meeting Assistant is recording, transcribing, recognizing speakers, or summarizing
|
||||
- **WHEN** the user selects Exit from the taskbar menu
|
||||
- **THEN** Meeting Assistant asks for confirmation before stopping the application
|
||||
@@ -0,0 +1,15 @@
|
||||
## 1. Tray Menu Behavior
|
||||
|
||||
- [x] 1.1 Add a failing behavior test proving that an active recording labels the normal stop action `Finish meeting`, places it alone immediately below `Open agent`, and keeps fine-grained controls in the following section.
|
||||
- [x] 1.2 Add explicit section metadata to the tray-menu model, reorder the active-recording actions, and render separators from that metadata.
|
||||
|
||||
## 2. Verification
|
||||
|
||||
- [x] 2.1 Review the touched menu builder and renderer for DRYness, SOLID design, and simplicity while preserving behavior.
|
||||
- [x] 2.2 Run focused taskbar-menu tests, the Windows application build, the full solution tests, and strict OpenSpec validation.
|
||||
|
||||
## 3. Refactor Follow-up
|
||||
|
||||
- [x] 3.1 Lock down idle section boundaries and active-recording layout when no microphone is available.
|
||||
- [x] 3.2 Remove the tray-menu section helper's hidden input mutation without changing rendered behavior.
|
||||
- [x] 3.3 Run focused and full verification, then validate the OpenSpec change strictly.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-08-03
|
||||
@@ -0,0 +1,60 @@
|
||||
## Context
|
||||
|
||||
The Windows microphone source currently creates one NAudio `IWaveIn` for the lifetime of a recording. When the endpoint is unplugged, NAudio reports a WASAPI exception through `RecordingStopped`; the source completes exceptionally, the composite source treats that as fatal, and `MeetingRecordingCoordinator` ends the run. The composite source already tolerates a temporarily quiet microphone by mixing system audio with synthetic silence after its alignment timeout, so recovery can be isolated to the microphone side.
|
||||
|
||||
The microphone selection provider already re-enumerates active endpoints whenever it creates a capture. Its selection rules ignore an unavailable runtime/configured device and fall back to the current Windows default. The missing behavior is retrying that resolution after an active capture fails.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Keep the active meeting run alive when microphone capture fails or stops unexpectedly.
|
||||
- Re-resolve the effective microphone on every recovery attempt so another active endpoint can take over.
|
||||
- Keep system-loopback audio flowing while microphone recovery is pending.
|
||||
- Verify recovery deterministically through the public audio-source contract without physical audio devices.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Recover system-loopback capture failures.
|
||||
- Persist or change the user's runtime microphone selection.
|
||||
- Add UI, endpoint, or configuration controls for recovery.
|
||||
- Splice or manufacture microphone audio for the disconnected interval.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Keep retry orchestration outside the NAudio adapter
|
||||
|
||||
`MicrophoneAudioSource` will own a recovery loop and ask `IMicrophoneDeviceProvider` for a new capture source on each attempt. The Windows provider will continue to own endpoint enumeration and selection, while an NAudio-specific adapter will own one `IWaveIn` lifetime.
|
||||
|
||||
This keeps device selection and WASAPI details behind a narrow boundary and makes the observable recovery behavior testable with deterministic capture sources. Retrying the same `IWaveIn` instance was rejected because a disconnected WASAPI client is not a reliable basis for endpoint failover.
|
||||
|
||||
### Treat unexpected completion and capture exceptions as recoverable
|
||||
|
||||
While the recording cancellation token remains active, microphone-source creation failures, capture exceptions, and clean-but-unexpected capture completion will all trigger another attempt. Cancellation remains the only normal terminal condition for the microphone stream.
|
||||
|
||||
This deliberately contains microphone failures without changing the composite source's handling of system-audio failures.
|
||||
|
||||
### Re-resolve after a bounded delay
|
||||
|
||||
Each recovery attempt will call the provider again after a short fixed delay. Recreating through the provider re-enumerates active devices and applies the existing runtime selection, configured selection, and Windows-default fallback rules. The delay prevents a busy loop while Windows is still updating endpoint state.
|
||||
|
||||
No new setting is introduced because recovery timing is an internal reliability detail and does not need user tuning for the current scope.
|
||||
|
||||
### Reuse the composite source's missing-stream behavior
|
||||
|
||||
The recovering microphone enumerable remains active between attempts instead of completing. The independently pumped system source therefore continues writing chunks, and the composite source's existing alignment timeout mixes those chunks with silent microphone samples until real microphone chunks resume.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- **Windows endpoint enumeration can lag behind physical disconnects** → Retry through fresh provider calls until the device list and default endpoint stabilize.
|
||||
- **A persistent microphone or driver failure can retry indefinitely** → Use a delay, log each failed attempt, and stop immediately when the recording is canceled.
|
||||
- **The replacement endpoint can have different native capabilities** → Continue requesting the run's configured PCM format through the same NAudio adapter; failed formats remain recoverable and retryable.
|
||||
- **There is an unavoidable microphone gap during failover** → Preserve the meeting and system audio rather than inventing microphone samples; the mixed stream contains silence for the missing microphone interval.
|
||||
|
||||
## Migration Plan
|
||||
|
||||
No data or configuration migration is required. Deploy the updated executable normally. Rollback consists of restoring the previous executable; existing meeting artifacts are unaffected.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None for this change.
|
||||
@@ -0,0 +1,26 @@
|
||||
## Why
|
||||
|
||||
Unplugging the active microphone currently propagates a WASAPI capture error through the recording pipeline and terminates the active meeting recording. Recording must remain available through transient device changes so that already-captured meeting work and continued system audio are not lost.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Recover microphone capture when the active Windows capture endpoint disappears or otherwise stops unexpectedly.
|
||||
- Re-resolve the effective microphone for each recovery attempt so an available configured, runtime-selected, default, or fallback endpoint can take over.
|
||||
- Keep the active recording and its independent system-audio capture alive while no microphone is temporarily available.
|
||||
- Log microphone recovery failures and successful capture restarts without terminating the meeting run.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
None.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- `meeting-recording`: Active recording becomes resilient to microphone endpoint disconnection and automatically resumes microphone capture from an available endpoint.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects the Windows microphone capture source and device-provider boundary.
|
||||
- Adds behavior tests around the public meeting audio-source contract.
|
||||
- Does not change recording endpoints, tray controls, system-loopback capture, or transcription-provider APIs.
|
||||
@@ -0,0 +1,125 @@
|
||||
## MODIFIED Requirements
|
||||
|
||||
### Requirement: Recording mode captures microphone and computer output
|
||||
Meeting Assistant SHALL capture microphone input and computer output and combine them into one audio stream for transcription.
|
||||
|
||||
Meeting Assistant SHALL capture audio as 16 kHz mono PCM chunks for the existing recording and transcription pipeline.
|
||||
|
||||
Meeting Assistant SHALL capture microphone and system loopback as separate input streams before producing the final mono chunks.
|
||||
|
||||
Meeting Assistant SHALL clean the microphone stream with a local acoustic echo cancellation stage that uses system loopback as the far-end reference.
|
||||
|
||||
Meeting Assistant SHALL produce final mono chunks by adding the cleaned microphone samples and system samples.
|
||||
|
||||
Meeting Assistant SHALL align microphone and system samples through per-source buffers before mixing and SHALL NOT emit normal live audio chunks that contain only one source while the other source is merely delayed.
|
||||
|
||||
When one source stays quiet beyond the alignment timeout, Meeting Assistant SHALL mix the available source with synthetic silence for the missing source instead of blocking transcription.
|
||||
|
||||
Meeting Assistant SHALL allow the final microphone/system mono mix to apply configurable microphone and system gain before combining samples.
|
||||
|
||||
Meeting Assistant SHALL use the active run or launch profile recording options when configuring capture format and final microphone/system gains.
|
||||
|
||||
Meeting Assistant SHALL clamp mixed samples after gain is applied.
|
||||
|
||||
Meeting Assistant SHALL write only the mixed stream to the temporary WAV used by transcription and finalization.
|
||||
|
||||
Meeting Assistant SHALL allow `Recording:MicrophoneDeviceId` to select a Windows microphone capture endpoint.
|
||||
|
||||
When `Recording:MicrophoneDeviceId` is blank or absent, Meeting Assistant SHALL use the Windows default capture endpoint.
|
||||
|
||||
When a microphone is selected from the tray icon menu, Meeting Assistant SHALL use that selected microphone for later recording starts until another microphone is selected or the process exits.
|
||||
|
||||
The tray icon right-click menu SHALL expose a `Microphone` submenu listing active microphone capture endpoints.
|
||||
|
||||
The `Microphone` submenu SHALL mark exactly one effective microphone as checked.
|
||||
|
||||
When no runtime microphone override is selected, the checked microphone SHALL be the configured microphone when it is available, otherwise the Windows default capture endpoint.
|
||||
|
||||
When the active microphone endpoint disappears, microphone capture fails, or microphone capture stops unexpectedly while a meeting recording is active, Meeting Assistant SHALL keep the meeting recording active and SHALL repeatedly re-resolve and restart microphone capture until capture succeeds or the recording is stopped.
|
||||
|
||||
Each microphone recovery attempt SHALL re-enumerate active microphone endpoints and apply the existing runtime-selected, configured, and Windows-default selection rules so an available endpoint can take over.
|
||||
|
||||
While microphone recovery is pending, Meeting Assistant SHALL keep system-loopback capture active and SHALL continue producing mixed audio with synthetic silence for the missing microphone stream.
|
||||
|
||||
#### Scenario: Both sources produce audio
|
||||
- **WHEN** microphone and computer output audio chunks are available
|
||||
- **THEN** Meeting Assistant mixes them into one PCM stream before transcription
|
||||
|
||||
#### Scenario: Mixed audio uses cleaned microphone and system audio
|
||||
- **GIVEN** the echo canceller cleans a microphone chunk to sample `2000`
|
||||
- **AND** the matching system chunk has sample `10000`
|
||||
- **WHEN** microphone and system chunks are mixed with gains `1` and `1`
|
||||
- **THEN** the mixed sample is `12000`
|
||||
|
||||
#### Scenario: Temporary recording stores only the mixed stream
|
||||
- **GIVEN** Meeting Assistant has mixed microphone and system audio into one PCM chunk
|
||||
- **WHEN** Meeting Assistant appends the chunk to the temporary recording
|
||||
- **THEN** the main temporary WAV contains the mixed PCM
|
||||
- **AND** no microphone or system sidecar WAV is written
|
||||
|
||||
#### Scenario: Launch profile recording options configure capture and gains
|
||||
- **GIVEN** an active launch profile configures sample format and microphone/system mix gains
|
||||
- **WHEN** Meeting Assistant captures and mixes audio for that run
|
||||
- **THEN** the microphone and system capture sources receive that launch profile recording configuration
|
||||
- **AND** the mixed output uses that launch profile's microphone/system gains
|
||||
|
||||
#### Scenario: Delayed sources are buffered before mixing
|
||||
- **GIVEN** microphone audio arrives before matching system audio
|
||||
- **WHEN** matching system audio arrives after a short delay
|
||||
- **THEN** Meeting Assistant emits one mixed chunk for the aligned samples
|
||||
- **AND** it does not emit separate microphone-only and system-only chunks for that delayed pair
|
||||
|
||||
#### Scenario: Quiet system audio does not block microphone transcription
|
||||
- **GIVEN** microphone audio arrives
|
||||
- **AND** system loopback audio does not arrive within the alignment timeout
|
||||
- **WHEN** Meeting Assistant mixes the available audio
|
||||
- **THEN** it emits the microphone audio mixed with silent system audio
|
||||
- **AND** live transcription can continue while system loopback is quiet
|
||||
|
||||
#### Scenario: Continuous microphone audio does not suppress the alignment timeout
|
||||
- **GIVEN** microphone audio keeps arriving
|
||||
- **AND** system loopback audio stays unavailable past the alignment timeout
|
||||
- **WHEN** Meeting Assistant checks the buffered microphone audio
|
||||
- **THEN** it emits the buffered microphone audio mixed with silent system audio
|
||||
- **AND** it does not wait indefinitely for a loopback chunk
|
||||
|
||||
#### Scenario: Device-level capture cannot be verified in tests
|
||||
- **WHEN** automated tests run without live audio devices
|
||||
- **THEN** Meeting Assistant verifies the audio mixer through deterministic source abstractions rather than depending on physical microphone or speaker devices
|
||||
|
||||
#### Scenario: Configured microphone is used for capture
|
||||
- **GIVEN** `Recording:MicrophoneDeviceId` identifies an active microphone endpoint
|
||||
- **WHEN** Meeting Assistant starts microphone capture
|
||||
- **THEN** it captures from that endpoint
|
||||
|
||||
#### Scenario: Blank microphone setting uses Windows default
|
||||
- **GIVEN** `Recording:MicrophoneDeviceId` is blank
|
||||
- **WHEN** Meeting Assistant starts microphone capture
|
||||
- **THEN** it captures from the Windows default capture endpoint
|
||||
|
||||
#### Scenario: Tray menu lists microphones with current selection checked
|
||||
- **GIVEN** active microphone endpoints `integrated microphone` and `other microphone`
|
||||
- **AND** `integrated microphone` is the effective microphone
|
||||
- **WHEN** the taskbar menu is opened
|
||||
- **THEN** it shows a `Microphone` submenu
|
||||
- **AND** the `integrated microphone` item is checked
|
||||
- **AND** the `other microphone` item is unchecked
|
||||
|
||||
#### Scenario: Tray microphone selection changes later capture
|
||||
- **GIVEN** active microphone endpoints `integrated microphone` and `other microphone`
|
||||
- **WHEN** the user selects `other microphone` from the taskbar microphone submenu
|
||||
- **THEN** later recording starts capture from `other microphone`
|
||||
|
||||
#### Scenario: Disconnected microphone fails over during recording
|
||||
- **GIVEN** a meeting is actively recording from one microphone and another microphone is available
|
||||
- **WHEN** the active microphone is disconnected and its capture fails
|
||||
- **THEN** the meeting recording remains active
|
||||
- **AND** Meeting Assistant re-resolves the effective microphone and resumes capture from the available microphone
|
||||
|
||||
#### Scenario: Recording continues while no microphone is available
|
||||
- **GIVEN** a meeting is actively recording
|
||||
- **WHEN** the active microphone disconnects and no microphone is temporarily available
|
||||
- **THEN** Meeting Assistant keeps the meeting recording and system-loopback capture active
|
||||
- **AND** emits system audio mixed with synthetic microphone silence
|
||||
- **WHEN** a microphone becomes available
|
||||
- **THEN** Meeting Assistant resumes microphone capture for the same meeting run
|
||||
@@ -0,0 +1,14 @@
|
||||
## 1. Microphone recovery behavior
|
||||
|
||||
- [x] 1.1 Add a failing behavior test proving active microphone capture moves to a newly resolved capture source after the current source fails.
|
||||
- [x] 1.2 Refactor the microphone device-provider boundary so recovery orchestration is platform-independent and individual NAudio capture lifetimes remain Windows-specific.
|
||||
- [x] 1.3 Implement bounded-delay microphone recovery that re-resolves devices after creation failures, capture failures, and unexpected capture completion until recording cancellation.
|
||||
- [x] 1.4 Add coverage proving capture recovers when no microphone is initially available and a later resolution succeeds.
|
||||
- [x] 1.5 Add a failing selection test and fall back to an active endpoint when the selected and Windows-default endpoints are unavailable.
|
||||
|
||||
## 2. Verification
|
||||
|
||||
- [x] 2.1 Refactor the touched capture path for DRYness, SOLID boundaries, and KISS while preserving behavior.
|
||||
- [x] 2.2 Run the focused microphone-selection and audio-source behavior tests plus the Windows application build.
|
||||
- [x] 2.3 Run the full solution test suite and `openspec validate recover-microphone-disconnect --strict`.
|
||||
- [x] 2.4 Verify the local health and recording-status surfaces without interrupting an active meeting run.
|
||||
Reference in New Issue
Block a user