From 6f4eb2a2b76bc2ac11d3d6e6059cea81cb829483 Mon Sep 17 00:00:00 2001 From: David Bond Date: Mon, 5 Oct 2026 15:18:37 +0100 Subject: [PATCH 1/6] PDChat: Voice Mode, so a user can ask out loud and hear the answer A host offers it by implementing IChatService.VoiceEndpoints, a new default interface member that returns null, so existing services compile unchanged and show nothing. - A "Voice" switch in the header, off by default, whose title says what it does; a status line says what is happening. - Microphone capture resampled to 24 kHz in an AudioWorklet, sent to the host's listen endpoint. The host decides when the question has ended. - A spoken question is sent through the same path as a typed one, so the question and the written answer are both in the transcript. - Only the first finished reply to a spoken question is read aloud (never a Typing placeholder, never a later notification), through the host's speak endpoint, with playback starting as the first audio arrives. - Half duplex: the microphone sends nothing while the answer is awaited or spoken, so the assistant never hears itself. - Refused microphones and closed connections are explained, not silent. For Magic Suite MS-26473; the host side is panoramicdata/MagicSuite#451. Co-Authored-By: Claude Opus 5.5 --- .../Components/PDChatTests.Fakes.cs | 1 + .../Components/PDChatTests.Voice.cs | 169 +++++++++++++++ .../Interfaces/IChatService.cs | 6 + .../Models/PDChatVoiceEndpoints.cs | 16 ++ PanoramicData.Blazor/PDChat.Voice.cs | 202 +++++++++++++++++ PanoramicData.Blazor/PDChat.razor | 16 ++ PanoramicData.Blazor/PDChat.razor.cs | 4 + PanoramicData.Blazor/PDChat.razor.css | 28 +++ .../wwwroot/js/pdchat-voice.js | 203 ++++++++++++++++++ 9 files changed, 645 insertions(+) create mode 100644 PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs create mode 100644 PanoramicData.Blazor/Models/PDChatVoiceEndpoints.cs create mode 100644 PanoramicData.Blazor/PDChat.Voice.cs create mode 100644 PanoramicData.Blazor/wwwroot/js/pdchat-voice.js diff --git a/PanoramicData.Blazor.Test/Components/PDChatTests.Fakes.cs b/PanoramicData.Blazor.Test/Components/PDChatTests.Fakes.cs index 62ce0bfeb..34a56d816 100644 --- a/PanoramicData.Blazor.Test/Components/PDChatTests.Fakes.cs +++ b/PanoramicData.Blazor.Test/Components/PDChatTests.Fakes.cs @@ -54,6 +54,7 @@ private sealed partial class FakeChatService : IChatService public string ToastMaxHeight { get; set; } = string.Empty; public int ToastMaxVisible { get; set; } = 3; public PDChatButtonPosition ToastAnchor { get; set; } = PDChatButtonPosition.BottomRight; + public PDChatVoiceEndpoints? VoiceEndpoints { get; set; } public IReadOnlyList Messages => Store; public bool SupportsConversations { get; init; } diff --git a/PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs b/PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs new file mode 100644 index 000000000..11a9b126c --- /dev/null +++ b/PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs @@ -0,0 +1,169 @@ +using AwesomeAssertions; +using Bunit; +using Microsoft.JSInterop; +using PanoramicData.Blazor.Interfaces; +using PanoramicData.Blazor.Models; + +namespace PanoramicData.Blazor.Test; + +/// +/// Voice Mode: offered only when the service supplies endpoints, off by default, and when on, a spoken question is +/// sent like a typed one and the first finished reply is read aloud. +/// +public partial class PDChatTests +{ + private const string VoiceModulePath = "./_content/PanoramicData.Blazor/js/pdchat-voice.js"; + private static readonly PDChatVoiceEndpoints _voiceEndpoints = new("/voice/listen", "/voice/speak"); + + /// Verifies that a service without voice endpoints offers no Voice Mode at all. + [Fact] + public void Without_voice_endpoints_there_is_no_Voice_Mode_control() + => RenderChat(new FakeChatService()).FindAll(".pdchat-voice-toggle").Should().BeEmpty(); + + /// Verifies that Voice Mode is off until the user turns it on, and its control says what it does. + [Fact] + public void Voice_Mode_is_offered_but_off_by_default() + { + var component = RenderChat(new FakeChatService { VoiceEndpoints = _voiceEndpoints }); + + var toggle = component.Find(".pdchat-voice-toggle"); + toggle.TextContent.Should().Contain("Voice"); + toggle.GetAttribute("title").Should().Be("Turn Voice Mode on: ask out loud and hear the answer"); + toggle.GetAttribute("aria-pressed").Should().Be("false"); + component.FindAll(".pdchat-voice-status").Should().BeEmpty(); + component.Instance.VoiceState.Should().Be(PDChatVoiceState.Off); + } + + /// Verifies that a chat whose input is not permitted offers no Voice Mode either. + [Fact] + public void Voice_Mode_is_not_offered_where_typing_is_not_permitted() + { + var service = new FakeChatService { VoiceEndpoints = _voiceEndpoints }; + ((IChatService)service).IsInputPermitted = false; + + RenderChat(service).FindAll(".pdchat-voice-toggle").Should().BeEmpty(); + } + + /// Verifies that turning Voice Mode on opens the host's listening endpoint and says how to use it. + [Fact] + public async Task Turning_Voice_Mode_on_starts_listening() + { + var module = SetUpVoiceModule(); + var component = RenderChat(new FakeChatService { VoiceEndpoints = _voiceEndpoints }); + + await component.Find(".pdchat-voice-toggle").ClickAsync(new()); + + module.Invocations["start"].Should().ContainSingle().Which.Arguments[0].Should().Be("/voice/listen"); + component.Instance.VoiceState.Should().Be(PDChatVoiceState.Listening); + component.Find(".pdchat-voice-toggle").GetAttribute("aria-pressed").Should().Be("true"); + component.Find(".pdchat-voice-status").TextContent.Trim().Should().Be("Listening. Ask your question, then pause."); + } + + /// Verifies that a spoken question is sent as the user, as typed text is, and the microphone pauses. + [Fact] + public async Task A_spoken_question_is_sent_like_a_typed_one() + { + var module = SetUpVoiceModule(); + var service = new FakeChatService { VoiceEndpoints = _voiceEndpoints }; + var component = RenderChat(service); + await component.Find(".pdchat-voice-toggle").ClickAsync(new()); + + await component.InvokeAsync(() => component.Instance.OnVoiceTurn("Is MS-26473 done?")); + + var sent = service.Sent.Should().ContainSingle().Subject; + sent.Message.Should().Be("Is MS-26473 done?"); + sent.Sender.Should().BeSameAs(_user); + module.Invocations["pause"].Should().ContainSingle().Which.Arguments[0].Should().Be(true); + component.Instance.VoiceState.Should().Be(PDChatVoiceState.Thinking); + } + + /// Verifies that only the finished answer is read aloud, as plain text, and listening then resumes. + [Fact] + public async Task The_finished_answer_is_spoken_and_listening_resumes() + { + var module = SetUpVoiceModule(); + var service = new FakeChatService { VoiceEndpoints = _voiceEndpoints }; + var component = RenderChat(service); + await component.Find(".pdchat-voice-toggle").ClickAsync(new()); + await component.InvokeAsync(() => component.Instance.OnVoiceTurn("Is it done?")); + + var typing = Message("Looking it up", MessageType.Typing); + await component.InvokeAsync(() => service.Receive(typing)); + module.Invocations["speak"].Should().BeEmpty("a typing placeholder is not the answer"); + + var answer = Message("

Yes, it is done.

"); + answer.IsMessageHtml = true; + await component.InvokeAsync(() => service.Receive(answer)); + + var speak = module.Invocations["speak"].Should().ContainSingle().Subject; + speak.Arguments[0].Should().Be("/voice/speak"); + ((string)speak.Arguments[1]!).Trim().Should().Be("Yes, it is done ."); + module.Invocations["pause"].Select(call => call.Arguments[0]).Should().Equal(true, false); + component.Instance.VoiceState.Should().Be(PDChatVoiceState.Listening); + } + + /// Verifies that only the first finished reply to a spoken question is spoken, never later ones. + [Fact] + public async Task Only_the_first_reply_to_a_spoken_question_is_spoken() + { + var module = SetUpVoiceModule(); + var service = new FakeChatService { VoiceEndpoints = _voiceEndpoints }; + var component = RenderChat(service); + await component.Find(".pdchat-voice-toggle").ClickAsync(new()); + await component.InvokeAsync(() => component.Instance.OnVoiceTurn("Is it done?")); + + await component.InvokeAsync(() => service.Receive(Message("Yes."))); + await component.InvokeAsync(() => service.Receive(Message("An unrelated notification"))); + + module.Invocations["speak"].Should().ContainSingle(); + } + + /// Verifies that with Voice Mode off, answers are never spoken. + [Fact] + public async Task With_Voice_Mode_off_nothing_is_spoken() + { + var module = SetUpVoiceModule(); + var service = new FakeChatService { VoiceEndpoints = _voiceEndpoints }; + var component = RenderChat(service); + + await component.InvokeAsync(() => service.Receive(Message("An answer"))); + + module.Invocations["speak"].Should().BeEmpty(); + } + + /// Verifies that turning Voice Mode off releases the microphone and removes the status line. + [Fact] + public async Task Turning_Voice_Mode_off_stops_it() + { + var module = SetUpVoiceModule(); + var component = RenderChat(new FakeChatService { VoiceEndpoints = _voiceEndpoints }); + await component.Find(".pdchat-voice-toggle").ClickAsync(new()); + + await component.Find(".pdchat-voice-toggle").ClickAsync(new()); + + module.Invocations["stop"].Should().ContainSingle(); + component.Instance.VoiceState.Should().Be(PDChatVoiceState.Off); + component.FindAll(".pdchat-voice-status").Should().BeEmpty(); + } + + /// Verifies that a refused microphone leaves Voice Mode off and says what to check. + [Fact] + public async Task A_refused_microphone_is_explained() + { + var module = SetUpVoiceModule(); + _ = module.SetupVoid("start", _ => true).SetException(new JSException("NotAllowedError")); + var component = RenderChat(new FakeChatService { VoiceEndpoints = _voiceEndpoints }); + + await component.Find(".pdchat-voice-toggle").ClickAsync(new()); + + component.Instance.VoiceState.Should().Be(PDChatVoiceState.Off); + component.Find(".pdchat-voice-status").TextContent.Should().Contain("microphone could not be opened"); + } + + private BunitJSModuleInterop SetUpVoiceModule() + { + var module = JSInterop.SetupModule(VoiceModulePath); + module.Mode = JSRuntimeMode.Loose; + return module; + } +} diff --git a/PanoramicData.Blazor/Interfaces/IChatService.cs b/PanoramicData.Blazor/Interfaces/IChatService.cs index fd001eccd..e108f280a 100644 --- a/PanoramicData.Blazor/Interfaces/IChatService.cs +++ b/PanoramicData.Blazor/Interfaces/IChatService.cs @@ -88,6 +88,12 @@ bool IsInputPermitted set => ChatServiceDefaultState.For(this).InputDisabledMessage = value; } + /// + /// Gets where Voice Mode sends speech and fetches spoken answers, or null (the default) when the host + /// offers no Voice Mode, in which case no Voice Mode control is shown. + /// + PDChatVoiceEndpoints? VoiceEndpoints => null; + /// /// Gets or sets whether the chat should auto-restore when new messages arrive. /// diff --git a/PanoramicData.Blazor/Models/PDChatVoiceEndpoints.cs b/PanoramicData.Blazor/Models/PDChatVoiceEndpoints.cs new file mode 100644 index 000000000..3b1e36fcd --- /dev/null +++ b/PanoramicData.Blazor/Models/PDChatVoiceEndpoints.cs @@ -0,0 +1,16 @@ +namespace PanoramicData.Blazor.Models; + +/// +/// The host's two websocket endpoints behind 's Voice Mode. +/// +/// +/// Receives 24 kHz mono float32 audio as binary messages, and answers with JSON text messages: +/// {"type":"word","text":...} as words are recognised, {"type":"turn","text":...} once the speaker has +/// finished, and {"type":"error","text":...}. +/// +/// +/// Receives one JSON message, {"text":...}, and answers with 24 kHz mono float32 audio as binary messages, +/// then {"type":"done"}. +/// +/// Relative URLs are resolved against the page. Speech is processed wherever the host's endpoints send it. +public sealed record PDChatVoiceEndpoints(string ListenUrl, string SpeakUrl); diff --git a/PanoramicData.Blazor/PDChat.Voice.cs b/PanoramicData.Blazor/PDChat.Voice.cs new file mode 100644 index 000000000..918c38fe3 --- /dev/null +++ b/PanoramicData.Blazor/PDChat.Voice.cs @@ -0,0 +1,202 @@ +using System.Net; + +namespace PanoramicData.Blazor; + +/// Where 's Voice Mode is in a spoken exchange. +public enum PDChatVoiceState +{ + /// Voice Mode is off. + Off, + + /// The microphone is being opened. + Starting, + + /// Waiting for the user to speak. + Listening, + + /// The question has been sent and the answer is awaited. + Thinking, + + /// The answer is being spoken. + Speaking, +} + +/// +/// Voice Mode for : the user speaks a question, pauses, and hears the answer. Off by default, and +/// offered only when the chat service supplies . +/// +/// +/// A spoken question goes through the same path as a typed one, so the question and the written answer are both in +/// the transcript. Only the first finished reply to a spoken question is read aloud. +/// +public partial class PDChat +{ + private const string _voiceModulePath = "./_content/PanoramicData.Blazor/js/pdchat-voice.js"; + + private IJSObjectReference? _voiceModule; + private DotNetObjectReference? _voiceReference; + private bool _isAwaitingSpokenAnswer; + private string _voiceHeard = string.Empty; + + /// Gets where Voice Mode is in a spoken exchange. + public PDChatVoiceState VoiceState { get; private set; } = PDChatVoiceState.Off; + + /// Gets the last problem Voice Mode reported, shown in place of its status. + public string? VoiceError { get; private set; } + + private bool IsVoiceModeOffered => ChatService.VoiceEndpoints is not null && ChatService.IsInputPermitted; + + private bool IsVoiceModeOn => VoiceState != PDChatVoiceState.Off; + + private string VoiceButtonTitle => IsVoiceModeOn + ? "Turn Voice Mode off" + : "Turn Voice Mode on: ask out loud and hear the answer"; + + private string VoiceStatusCssClass => $"pdchat-voice-status pdchat-voice-{VoiceState.ToString().ToLowerInvariant()}" + (VoiceError is null ? string.Empty : " pdchat-voice-error"); + + private string VoiceStatusText => VoiceError ?? VoiceState switch + { + PDChatVoiceState.Starting => "Opening the microphone…", + PDChatVoiceState.Listening => _voiceHeard.Length > 0 ? _voiceHeard : "Listening. Ask your question, then pause.", + PDChatVoiceState.Thinking => "Thinking…", + PDChatVoiceState.Speaking => "Speaking. Turn Voice Mode off to stop.", + _ => string.Empty, + }; + + private async Task ToggleVoiceModeAsync() + { + if (IsVoiceModeOn) + { + await StopVoiceModeAsync(); + return; + } + + if (ChatService.VoiceEndpoints is not { } endpoints) + { + return; + } + + VoiceError = null; + VoiceState = PDChatVoiceState.Starting; + try + { + _voiceModule ??= await JSRuntime.InvokeAsync("import", _voiceModulePath); + _voiceReference ??= DotNetObjectReference.Create(this); + await _voiceModule.InvokeVoidAsync("start", endpoints.ListenUrl, _voiceReference); + VoiceState = PDChatVoiceState.Listening; + } + catch (JSException) + { + VoiceState = PDChatVoiceState.Off; + VoiceError = "The microphone could not be opened. Check that this site may use it."; + } + } + + private async Task StopVoiceModeAsync() + { + VoiceState = PDChatVoiceState.Off; + _isAwaitingSpokenAnswer = false; + _voiceHeard = string.Empty; + if (_voiceModule is not null) + { + await _voiceModule.InvokeVoidAsync("stop"); + } + } + + /// Called by the voice module as words are recognised. + /// The word just heard. + [JSInvokable] + public Task OnVoiceWord(string text) + { + _voiceHeard = $"{_voiceHeard} {text}".Trim(); + return InvokeAsync(StateHasChanged); + } + + /// Called by the voice module when the speaker has finished; sends what they said as a question. + /// The whole question. + [JSInvokable] + public async Task OnVoiceTurn(string text) + { + if (VoiceState != PDChatVoiceState.Listening || string.IsNullOrWhiteSpace(text)) + { + return; + } + + _voiceHeard = string.Empty; + _currentInput = text; + VoiceState = PDChatVoiceState.Thinking; + _isAwaitingSpokenAnswer = true; + + // Half duplex: while Merlin thinks and speaks, the microphone sends nothing, so it never hears itself. + await (_voiceModule?.InvokeVoidAsync("pause", true) ?? ValueTask.CompletedTask); + await SendCurrentMessageAsync(); + await InvokeAsync(StateHasChanged); + } + + /// Called by the voice module when speech recognition reports a problem. + /// What went wrong, for the user. + [JSInvokable] + public Task OnVoiceError(string text) + { + VoiceError = text; + return InvokeAsync(StateHasChanged); + } + + /// Called by the voice module when the listening connection has closed. + [JSInvokable] + public Task OnVoiceClosed() + { + VoiceState = PDChatVoiceState.Off; + _isAwaitingSpokenAnswer = false; + VoiceError ??= "Voice Mode stopped: the connection closed."; + return InvokeAsync(StateHasChanged); + } + + /// Reads the first finished reply to a spoken question aloud, then listens again. + private async Task SpeakAnswerIfAwaitedAsync(ChatMessage message) + { + if (!_isAwaitingSpokenAnswer || message.Sender.IsUser || message.Type == MessageType.Typing + || ChatService.VoiceEndpoints is not { } endpoints || _voiceModule is null) + { + return; + } + + _isAwaitingSpokenAnswer = false; + VoiceState = PDChatVoiceState.Speaking; + await InvokeAsync(StateHasChanged); + + await _voiceModule.InvokeVoidAsync("speak", endpoints.SpeakUrl, ToSpeakableText(message)); + + if (VoiceState == PDChatVoiceState.Speaking) + { + VoiceState = PDChatVoiceState.Listening; + await _voiceModule.InvokeVoidAsync("pause", false); + await InvokeAsync(StateHasChanged); + } + } + + // The host turns the text into speech; this only removes HTML markup, which would otherwise be read out. + private static string ToSpeakableText(ChatMessage message) + => message.IsMessageHtml ? WebUtility.HtmlDecode(HtmlTag().Replace(message.Message, " ")) : message.Message; + + [GeneratedRegex("<[^>]+>")] + private static partial Regex HtmlTag(); + + private async ValueTask DisposeVoiceAsync() + { + if (_voiceModule is not null) + { + try + { + await _voiceModule.InvokeVoidAsync("stop"); + await _voiceModule.DisposeAsync(); + } + catch (Exception ex) when (ex is JSDisconnectedException or OperationCanceledException or ObjectDisposedException) + { + // The circuit is going away; the browser releases the microphone with the page. + } + } + + _voiceReference?.Dispose(); + } +} diff --git a/PanoramicData.Blazor/PDChat.razor b/PanoramicData.Blazor/PDChat.razor index 09a52bed8..0ab5fbde1 100644 --- a/PanoramicData.Blazor/PDChat.razor +++ b/PanoramicData.Blazor/PDChat.razor @@ -84,6 +84,15 @@ 🗑️ } + @if (IsVoiceModeOffered) + { + + } @@ -112,6 +121,13 @@ + @if (IsVoiceModeOn || VoiceError is not null) + { +
+ @VoiceStatusText +
+ } + @* Conversation history: full-screen only, and only when the host supplied a store to back it (issues #111 and #112). Takes precedence over the canvas layout below because both want the same space, and a conversation list with nowhere to open a conversation into is useless. *@ diff --git a/PanoramicData.Blazor/PDChat.razor.cs b/PanoramicData.Blazor/PDChat.razor.cs index 1d4876a6e..9af92c3d9 100644 --- a/PanoramicData.Blazor/PDChat.razor.cs +++ b/PanoramicData.Blazor/PDChat.razor.cs @@ -275,6 +275,8 @@ private async Task OnMessageReceivedAsync(ChatMessage message) { var isNewMessage = UpsertMessage(message); + await SpeakAnswerIfAwaitedAsync(message); + // Emit OnMessageReceived event for new messages if (isNewMessage && OnMessageReceivedEvent.HasDelegate) { @@ -585,6 +587,8 @@ private void OnTabAdded() /// public override async ValueTask DisposeAsync() { + await DisposeVoiceAsync(); + // Clean up event handlers ChatService.OnMessageReceived -= OnMessageReceived; diff --git a/PanoramicData.Blazor/PDChat.razor.css b/PanoramicData.Blazor/PDChat.razor.css index feef44cb1..71610bde4 100644 --- a/PanoramicData.Blazor/PDChat.razor.css +++ b/PanoramicData.Blazor/PDChat.razor.css @@ -543,6 +543,34 @@ body:has(.pdchat-container.dock-fullscreen) { outline-offset: 1px; } +/* Voice Mode: the switch carries a word as well as an icon, so it says what it does. */ +.pdchat-voice-toggle { + font-size: 14px; + min-width: auto; + white-space: nowrap; +} + +.pdchat-voice-toggle.active { + background: rgba(255,255,255,0.25); + box-shadow: inset 0 0 0 1px rgba(255,255,255,0.6); +} + +.pdchat-voice-status { + padding: 6px 12px; + font-size: 13px; + border-bottom: 1px solid var(--pd-chat-border, #dee2e6); + background: var(--pd-chat-info-bg, #e7f1ff); + color: var(--pd-chat-color, inherit); +} + +.pdchat-voice-status.pdchat-voice-listening::before { + content: "● "; + color: #dc3545; +} + +.pdchat-voice-status.pdchat-voice-error { + background: var(--pd-chat-warning-bg, #fff3cd); +} /* ============================================== PDMessages Layout - Ensure Full Height ============================================== */ diff --git a/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js new file mode 100644 index 000000000..99c152ca2 --- /dev/null +++ b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js @@ -0,0 +1,203 @@ +// PDChat Voice Mode: the microphone to a host websocket at 24 kHz, and spoken answers played as they arrive. + +const SAMPLE_RATE = 24000; +const FRAME_SAMPLES = 1920; + +// Resampled in the worklet from the device's own rate: a 24 kHz AudioContext cannot be connected to a +// microphone in every browser. +const CAPTURE_WORKLET = ` +class PdChatVoiceCapture extends AudioWorkletProcessor { + constructor() { + super(); + this.step = sampleRate / ${SAMPLE_RATE}; + this.position = 0; + this.frame = new Float32Array(${FRAME_SAMPLES}); + this.length = 0; + } + process(inputs) { + const input = inputs[0] && inputs[0][0]; + if (!input) { + return true; + } + while (this.position < input.length) { + const index = Math.floor(this.position); + const next = Math.min(index + 1, input.length - 1); + const fraction = this.position - index; + this.frame[this.length++] = input[index] + (input[next] - input[index]) * fraction; + if (this.length === ${FRAME_SAMPLES}) { + this.port.postMessage(this.frame.buffer, [this.frame.buffer]); + this.frame = new Float32Array(${FRAME_SAMPLES}); + this.length = 0; + } + this.position += this.step; + } + this.position -= input.length; + return true; + } +} +registerProcessor("pdchat-voice-capture", PdChatVoiceCapture); +`; + +let listening = null; +let speaking = null; + +function toWebSocketUrl(url) { + const absolute = new URL(url, document.baseURI); + absolute.protocol = absolute.protocol === "https:" ? "wss:" : "ws:"; + return absolute.toString(); +} + +function onListenMessage(event, dotNet) { + if (typeof event.data !== "string") { + return; + } + const message = JSON.parse(event.data); + if (message.type === "word") { + dotNet.invokeMethodAsync("OnVoiceWord", message.text); + } else if (message.type === "turn") { + dotNet.invokeMethodAsync("OnVoiceTurn", message.text); + } else if (message.type === "error") { + dotNet.invokeMethodAsync( + "OnVoiceError", + message.text || "Voice Mode is unavailable.", + ); + } +} + +/** Starts listening. Rejects when the microphone is refused or unavailable. */ +export async function start(listenUrl, dotNet) { + stop(); + const stream = await navigator.mediaDevices.getUserMedia({ + audio: { + channelCount: 1, + echoCancellation: true, + noiseSuppression: true, + autoGainControl: true, + }, + }); + const context = new AudioContext(); + const workletUrl = URL.createObjectURL( + new Blob([CAPTURE_WORKLET], { type: "text/javascript" }), + ); + await context.audioWorklet.addModule(workletUrl); + URL.revokeObjectURL(workletUrl); + + const capture = new AudioWorkletNode(context, "pdchat-voice-capture"); + const socket = new WebSocket(toWebSocketUrl(listenUrl)); + socket.binaryType = "arraybuffer"; + const state = { stream, context, socket, paused: false }; + + capture.port.onmessage = (event) => { + if (!state.paused && socket.readyState === WebSocket.OPEN) { + socket.send(event.data); + } + }; + socket.onmessage = (event) => onListenMessage(event, dotNet); + socket.onclose = () => { + if (listening === state) { + listening = null; + dotNet.invokeMethodAsync("OnVoiceClosed"); + } + }; + + context.createMediaStreamSource(stream).connect(capture); + listening = state; +} + +/** While paused the microphone is still open but nothing is sent, so Merlin does not hear itself. */ +export function pause(paused) { + if (listening) { + listening.paused = paused; + } +} + +/** Stops listening and speaking, and releases the microphone. */ +export function stop() { + stopSpeaking(); + if (!listening) { + return; + } + const state = listening; + listening = null; + state.socket.close(); + state.stream.getTracks().forEach((track) => track.stop()); + state.context.close(); +} + +/** Stops an answer part-way through. */ +export function stopSpeaking() { + if (!speaking) { + return; + } + const state = speaking; + speaking = null; + state.socket.close(); + state.context.close(); + state.finish(); +} + +/** Speaks an answer, resolving once the last of the audio has played. */ +export function speak(speakUrl, text) { + stopSpeaking(); + return new Promise((resolve) => { + const context = new AudioContext({ sampleRate: SAMPLE_RATE }); + const socket = new WebSocket(toWebSocketUrl(speakUrl)); + socket.binaryType = "arraybuffer"; + const state = { context, socket, startAt: 0, last: null, finished: false }; + state.finish = () => { + if (!state.finished) { + state.finished = true; + resolve(); + } + }; + + socket.onopen = () => socket.send(JSON.stringify({ text })); + socket.onmessage = (event) => { + if (typeof event.data === "string") { + if (JSON.parse(event.data).type === "done") { + finishAfterPlayback(state); + } + return; + } + queueAudio(state, new Float32Array(event.data)); + }; + socket.onerror = () => finishAfterPlayback(state); + socket.onclose = () => finishAfterPlayback(state); + speaking = state; + }); +} + +function queueAudio(state, samples) { + if (samples.length === 0) { + return; + } + const buffer = state.context.createBuffer(1, samples.length, SAMPLE_RATE); + buffer.copyToChannel(samples, 0); + const source = state.context.createBufferSource(); + source.buffer = buffer; + source.connect(state.context.destination); + state.startAt = Math.max(state.startAt, state.context.currentTime + 0.05); + source.start(state.startAt); + state.startAt += buffer.duration; + state.last = source; +} + +function finishAfterPlayback(state) { + const done = () => { + if (speaking === state) { + speaking = null; + state.context.close(); + } + state.finish(); + }; + // A short answer may already have finished playing, and then onended will not fire again. + if ( + state.last && + !state.finished && + state.context.currentTime < state.startAt + ) { + state.last.onended = done; + } else { + done(); + } +} From 046a4bc15d427b440d15eca38e0e352c42f62cb6 Mon Sep 17 00:00:00 2001 From: David Bond Date: Mon, 5 Oct 2026 15:37:39 +0100 Subject: [PATCH 2/6] Declare supported browsers: Opera Mini cannot run Blazor ESLint's compat rule checks against browserslist defaults, which include Opera Mini. It cannot run Blazor (no WebSockets), so its URL and Promise findings on pdchat-voice.js are not real gaps. Co-Authored-By: Claude Opus 5.5 --- .browserslistrc | 3 +++ 1 file changed, 3 insertions(+) create mode 100644 .browserslistrc diff --git a/.browserslistrc b/.browserslistrc new file mode 100644 index 000000000..402568060 --- /dev/null +++ b/.browserslistrc @@ -0,0 +1,3 @@ +# Browsers the library supports. Opera Mini cannot run Blazor at all: it does not support WebSockets. +defaults +not op_mini all From faebb3c086a6092b260cdf29083e6c5c69499dcb Mon Sep 17 00:00:00 2001 From: David Bond Date: Mon, 5 Oct 2026 18:34:55 +0100 Subject: [PATCH 3/6] Revert "Declare supported browsers: Opera Mini cannot run Blazor" This reverts commit 046a4bc15d427b440d15eca38e0e352c42f62cb6. --- .browserslistrc | 3 --- 1 file changed, 3 deletions(-) delete mode 100644 .browserslistrc diff --git a/.browserslistrc b/.browserslistrc deleted file mode 100644 index 402568060..000000000 --- a/.browserslistrc +++ /dev/null @@ -1,3 +0,0 @@ -# Browsers the library supports. Opera Mini cannot run Blazor at all: it does not support WebSockets. -defaults -not op_mini all From 362532603aa7f51e10891a8d2e70ac687018237e Mon Sep 17 00:00:00 2001 From: David Bond Date: Mon, 5 Oct 2026 18:37:28 +0100 Subject: [PATCH 4/6] PDChat Voice Mode: no Promise or URL in the module (Codacy compat) Codacy's ESLint ignores the repository's browserslist, so the Opera Mini findings are fixed in code instead: - speak() calls back OnVoiceSpoken rather than returning a Promise; the wait is on the C# side. - The websocket URL is built from location with string operations. - The capture worklet is its own static file rather than a Blob URL, which also means a Content Security Policy need not allow blob: scripts. Co-Authored-By: Claude Opus 5.5 --- .../Components/PDChatTests.Voice.cs | 4 + PanoramicData.Blazor/PDChat.Voice.cs | 18 ++- .../wwwroot/js/pdchat-voice-capture.js | 39 ++++++ .../wwwroot/js/pdchat-voice.js | 113 +++++++----------- 4 files changed, 97 insertions(+), 77 deletions(-) create mode 100644 PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js diff --git a/PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs b/PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs index 11a9b126c..45d527daa 100644 --- a/PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs +++ b/PanoramicData.Blazor.Test/Components/PDChatTests.Voice.cs @@ -98,6 +98,10 @@ public async Task The_finished_answer_is_spoken_and_listening_resumes() var speak = module.Invocations["speak"].Should().ContainSingle().Subject; speak.Arguments[0].Should().Be("/voice/speak"); ((string)speak.Arguments[1]!).Trim().Should().Be("Yes, it is done ."); + component.Instance.VoiceState.Should().Be(PDChatVoiceState.Speaking); + + await component.InvokeAsync(() => component.Instance.OnVoiceSpoken()); + module.Invocations["pause"].Select(call => call.Arguments[0]).Should().Equal(true, false); component.Instance.VoiceState.Should().Be(PDChatVoiceState.Listening); } diff --git a/PanoramicData.Blazor/PDChat.Voice.cs b/PanoramicData.Blazor/PDChat.Voice.cs index 918c38fe3..b87374d49 100644 --- a/PanoramicData.Blazor/PDChat.Voice.cs +++ b/PanoramicData.Blazor/PDChat.Voice.cs @@ -165,14 +165,22 @@ private async Task SpeakAnswerIfAwaitedAsync(ChatMessage message) VoiceState = PDChatVoiceState.Speaking; await InvokeAsync(StateHasChanged); - await _voiceModule.InvokeVoidAsync("speak", endpoints.SpeakUrl, ToSpeakableText(message)); + // The module calls OnVoiceSpoken when the last of the audio has played. + await _voiceModule.InvokeVoidAsync("speak", endpoints.SpeakUrl, ToSpeakableText(message), _voiceReference); + } - if (VoiceState == PDChatVoiceState.Speaking) + /// Called by the voice module when an answer has finished playing; listening resumes. + [JSInvokable] + public async Task OnVoiceSpoken() + { + if (VoiceState != PDChatVoiceState.Speaking) { - VoiceState = PDChatVoiceState.Listening; - await _voiceModule.InvokeVoidAsync("pause", false); - await InvokeAsync(StateHasChanged); + return; } + + VoiceState = PDChatVoiceState.Listening; + await (_voiceModule?.InvokeVoidAsync("pause", false) ?? ValueTask.CompletedTask); + await InvokeAsync(StateHasChanged); } // The host turns the text into speech; this only removes HTML markup, which would otherwise be read out. diff --git a/PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js b/PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js new file mode 100644 index 000000000..775f3f0dc --- /dev/null +++ b/PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js @@ -0,0 +1,39 @@ +// PDChat Voice Mode's capture worklet: the microphone, resampled from the device's own rate to 24 kHz, in 80 ms +// frames. A 24 kHz AudioContext cannot be connected to a microphone in every browser, so it is done here. + +const TARGET_RATE = 24000; +const FRAME_SAMPLES = 1920; + +class PdChatVoiceCapture extends AudioWorkletProcessor { + constructor() { + super(); + this.step = sampleRate / TARGET_RATE; + this.position = 0; + this.frame = new Float32Array(FRAME_SAMPLES); + this.length = 0; + } + + process(inputs) { + const input = inputs[0] && inputs[0][0]; + if (!input) { + return true; + } + while (this.position < input.length) { + const index = Math.floor(this.position); + const next = Math.min(index + 1, input.length - 1); + const fraction = this.position - index; + this.frame[this.length++] = + input[index] + (input[next] - input[index]) * fraction; + if (this.length === FRAME_SAMPLES) { + this.port.postMessage(this.frame.buffer, [this.frame.buffer]); + this.frame = new Float32Array(FRAME_SAMPLES); + this.length = 0; + } + this.position += this.step; + } + this.position -= input.length; + return true; + } +} + +registerProcessor("pdchat-voice-capture", PdChatVoiceCapture); diff --git a/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js index 99c152ca2..710abcacb 100644 --- a/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js +++ b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js @@ -1,50 +1,25 @@ // PDChat Voice Mode: the microphone to a host websocket at 24 kHz, and spoken answers played as they arrive. const SAMPLE_RATE = 24000; -const FRAME_SAMPLES = 1920; - -// Resampled in the worklet from the device's own rate: a 24 kHz AudioContext cannot be connected to a -// microphone in every browser. -const CAPTURE_WORKLET = ` -class PdChatVoiceCapture extends AudioWorkletProcessor { - constructor() { - super(); - this.step = sampleRate / ${SAMPLE_RATE}; - this.position = 0; - this.frame = new Float32Array(${FRAME_SAMPLES}); - this.length = 0; - } - process(inputs) { - const input = inputs[0] && inputs[0][0]; - if (!input) { - return true; - } - while (this.position < input.length) { - const index = Math.floor(this.position); - const next = Math.min(index + 1, input.length - 1); - const fraction = this.position - index; - this.frame[this.length++] = input[index] + (input[next] - input[index]) * fraction; - if (this.length === ${FRAME_SAMPLES}) { - this.port.postMessage(this.frame.buffer, [this.frame.buffer]); - this.frame = new Float32Array(${FRAME_SAMPLES}); - this.length = 0; - } - this.position += this.step; - } - this.position -= input.length; - return true; - } -} -registerProcessor("pdchat-voice-capture", PdChatVoiceCapture); -`; + +// Beside this module; a separate file rather than a Blob so a Content Security Policy need not allow blob: scripts. +const CAPTURE_MODULE = import.meta.url.replace( + /[^/]*$/, + "pdchat-voice-capture.js", +); let listening = null; let speaking = null; function toWebSocketUrl(url) { - const absolute = new URL(url, document.baseURI); - absolute.protocol = absolute.protocol === "https:" ? "wss:" : "ws:"; - return absolute.toString(); + if (/^wss?:\/\//i.test(url)) { + return url; + } + if (/^https?:\/\//i.test(url)) { + return url.replace(/^http/i, "ws"); + } + const scheme = location.protocol === "https:" ? "wss://" : "ws://"; + return scheme + location.host + (/^\//.test(url) ? url : "/" + url); } function onListenMessage(event, dotNet) { @@ -76,11 +51,7 @@ export async function start(listenUrl, dotNet) { }, }); const context = new AudioContext(); - const workletUrl = URL.createObjectURL( - new Blob([CAPTURE_WORKLET], { type: "text/javascript" }), - ); - await context.audioWorklet.addModule(workletUrl); - URL.revokeObjectURL(workletUrl); + await context.audioWorklet.addModule(CAPTURE_MODULE); const capture = new AudioWorkletNode(context, "pdchat-voice-capture"); const socket = new WebSocket(toWebSocketUrl(listenUrl)); @@ -104,7 +75,7 @@ export async function start(listenUrl, dotNet) { listening = state; } -/** While paused the microphone is still open but nothing is sent, so Merlin does not hear itself. */ +/** While paused the microphone is still open but nothing is sent, so the assistant does not hear itself. */ export function pause(paused) { if (listening) { listening.paused = paused; @@ -136,35 +107,33 @@ export function stopSpeaking() { state.finish(); } -/** Speaks an answer, resolving once the last of the audio has played. */ -export function speak(speakUrl, text) { +/** Speaks an answer, and calls OnVoiceSpoken once the last of the audio has played. */ +export function speak(speakUrl, text, dotNet) { stopSpeaking(); - return new Promise((resolve) => { - const context = new AudioContext({ sampleRate: SAMPLE_RATE }); - const socket = new WebSocket(toWebSocketUrl(speakUrl)); - socket.binaryType = "arraybuffer"; - const state = { context, socket, startAt: 0, last: null, finished: false }; - state.finish = () => { - if (!state.finished) { - state.finished = true; - resolve(); - } - }; - - socket.onopen = () => socket.send(JSON.stringify({ text })); - socket.onmessage = (event) => { - if (typeof event.data === "string") { - if (JSON.parse(event.data).type === "done") { - finishAfterPlayback(state); - } - return; + const context = new AudioContext({ sampleRate: SAMPLE_RATE }); + const socket = new WebSocket(toWebSocketUrl(speakUrl)); + socket.binaryType = "arraybuffer"; + const state = { context, socket, startAt: 0, last: null, finished: false }; + state.finish = () => { + if (!state.finished) { + state.finished = true; + dotNet.invokeMethodAsync("OnVoiceSpoken"); + } + }; + + socket.onopen = () => socket.send(JSON.stringify({ text })); + socket.onmessage = (event) => { + if (typeof event.data === "string") { + if (JSON.parse(event.data).type === "done") { + finishAfterPlayback(state); } - queueAudio(state, new Float32Array(event.data)); - }; - socket.onerror = () => finishAfterPlayback(state); - socket.onclose = () => finishAfterPlayback(state); - speaking = state; - }); + return; + } + queueAudio(state, new Float32Array(event.data)); + }; + socket.onerror = () => finishAfterPlayback(state); + socket.onclose = () => finishAfterPlayback(state); + speaking = state; } function queueAudio(state, samples) { From 0702ec35fdd049dda73c04d284d7a7318f854338 Mon Sep 17 00:00:00 2001 From: David Bond Date: Mon, 5 Oct 2026 18:47:44 +0100 Subject: [PATCH 5/6] PDChat Voice Mode: socket scheme from the page; worklet reads globalThis.sampleRate Codacy: no literal ws:// (an https page always gets wss), and sampleRate read as the AudioWorkletGlobalScope property it is. Co-Authored-By: Claude Opus 5.5 --- .../wwwroot/js/pdchat-voice-capture.js | 2 +- PanoramicData.Blazor/wwwroot/js/pdchat-voice.js | 10 ++++------ 2 files changed, 5 insertions(+), 7 deletions(-) diff --git a/PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js b/PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js index 775f3f0dc..03d179b74 100644 --- a/PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js +++ b/PanoramicData.Blazor/wwwroot/js/pdchat-voice-capture.js @@ -7,7 +7,7 @@ const FRAME_SAMPLES = 1920; class PdChatVoiceCapture extends AudioWorkletProcessor { constructor() { super(); - this.step = sampleRate / TARGET_RATE; + this.step = globalThis.sampleRate / TARGET_RATE; this.position = 0; this.frame = new Float32Array(FRAME_SAMPLES); this.length = 0; diff --git a/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js index 710abcacb..a36a6b8ca 100644 --- a/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js +++ b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js @@ -12,14 +12,12 @@ let listening = null; let speaking = null; function toWebSocketUrl(url) { - if (/^wss?:\/\//i.test(url)) { - return url; - } - if (/^https?:\/\//i.test(url)) { + // The page's own scheme decides: an https page gets wss, so a secure page never opens an insecure socket. + if (/^(wss?|https?):/i.test(url)) { return url.replace(/^http/i, "ws"); } - const scheme = location.protocol === "https:" ? "wss://" : "ws://"; - return scheme + location.host + (/^\//.test(url) ? url : "/" + url); + const scheme = location.protocol.replace(/^http/, "ws"); + return scheme + "//" + location.host + (/^\//.test(url) ? url : "/" + url); } function onListenMessage(event, dotNet) { From d14aa9b0a1a5ce11475c7afd57286fcee9cb4686 Mon Sep 17 00:00:00 2001 From: David Bond Date: Mon, 5 Oct 2026 18:56:35 +0100 Subject: [PATCH 6/6] PDChat Voice Mode: non-capturing group in the URL scheme test (Codacy) Co-Authored-By: Claude Opus 5.5 --- PanoramicData.Blazor/wwwroot/js/pdchat-voice.js | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js index a36a6b8ca..3e39a8e1d 100644 --- a/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js +++ b/PanoramicData.Blazor/wwwroot/js/pdchat-voice.js @@ -13,7 +13,7 @@ let speaking = null; function toWebSocketUrl(url) { // The page's own scheme decides: an https page gets wss, so a secure page never opens an insecure socket. - if (/^(wss?|https?):/i.test(url)) { + if (/^(?:wss?|https?):/i.test(url)) { return url.replace(/^http/i, "ws"); } const scheme = location.protocol.replace(/^http/, "ws");