From: Taran Nathan Date: Thu, 23 Jul 2026 22:52:13 +0000 (-0700) Subject: more document and presentation stuff X-Git-Url: https://git.taranathan.com/?a=commitdiff_plain;p=audio-stream.git more document and presentation stuff --- diff --git a/document/main.typ b/document/main.typ index 9bc9586..f96c817 100644 --- a/document/main.typ +++ b/document/main.typ @@ -1,12 +1,21 @@ #import "@preview/fletcher:0.5.8" as fletcher: diagram, edge, node #import fletcher.shapes: circle, diamond, ellipse, parallelogram, pill, rect + +#import "@local/codly:1.3.1": * +#import "@preview/codly-languages:0.1.10": * +#show: codly-init.with() + +#codly(languages: codly-languages, display-name: true, display-icon: true) +#show raw: set text(size: 8pt) + #set page(paper: "a4", numbering: "1", number-align: top + right, margin: (x: 2em, y: 3em)) #set document(author: "Taran Nathan", title: "Independent Audio Transcription PoC") #set text(14pt, weight: "medium", font: "Times New Roman") #show title: tootle => align(center)[#tootle] #show heading.where(depth: 1): x => align(center)[#x] #set heading(numbering: "1.a:") - +#show link: set text(blue) +#show link: underline //header @@ -20,18 +29,16 @@ Taran Nathan = Executive Product Brief == Problem Statement -Duke Uni Hospital is already integrated with someone else, and we want to tap into their meetings and such(consensually, of course) to be able to provide automated meeting notes for them? +AmplifyMD has built AI features that use transcription of the video encounter with the patient as context for helping physicians with documentation. Currently AmplifyMD has access to these transcripts if its native video platform is used. However, many customers use 3rd party video hardware/software solutions which may not allow AmplifyMD access to this data. == Value Propositions -Allows us to work within our competitor's system while still providing value and integrating them with our system. - -== Success Metrics -List target KPIs like user adoption rates, reduction in task time, or clinical compliance. +Allows customers to use any video solution while still having access to AmplifyMD’s AI documentation features. -Doctor usage and reported satisfaction with the tool in surveys. +== Proposed Solution +The Proof of Concept is an embeddable script, that records audio from both the microphone and the system audio source(pipewire, coreaudio, wasapi) via browser APIs. The system audio stream is obtained through the screen sharing API, so even though it will show a popup asking for screen share, it discards the video stream and only uses the audio stream. Once the audio streams are combined into one, it can be sent to a server that provides a speech-to-text translation model, preferably hosted by AmplifyMD. Currently, for demonstration purposes, we are using WhisperAI's API. The simple websocket backend I wrote was made to be discarded, as I don't think that the WhisperAI API would be a good final solution. It would be best to switch to a locally hosted instance of whisper or an alternative for control, cost, and privacy reasons. Currently the transcription and audio are only stored in the browser and is ephemeral(goes away on page reload), and is not sent anywhere yet. -== Scope Boundaries -Proof of Concept is basically a simple script (that can be easily embedded into another site), that records audio from both the microphone and the system audio source(pulseaudio, coreaudio, wasapi) via browser apis. The system audio stream is obtained through the screen sharing API, so even though it will show a popup asking for screen share, it discards the video stream and only uses the audio stream. Once the audio streams are combined into one, it is sent to little server that contains WhisperAI api credentials for speech transcription over a websocket. I deliberately did not include any sort of authorization yet because I'm not really sure how you would want to do that, and the primary use case I would see would be for you guys to embed the script into another dashboard and have it hooked up to some buttons and a different UI. The whisperAI docs are kinda murky, so the backend and some of the script code might be useful as reference, even though it would probably be better to rewrite it from scratch to integrate it better in whatever frontend framework you guys are using. The simple websocket backend I wrote was made to be discarded, as I don't think that the whisperAI api would be a final solution necessarily. It would be best to switch to a locally hosted instance of whisper for control, cost, and privacy reasons. +== Scope Limits +The PoC will probably only work on Chromium based browsers(Chrome, Edge, Opera, Vivaldi), as #link("https://developer.mozilla.org/en-US/docs/Web/API/MediaDevices/getDisplayMedia#browser_compatibility")[audio capture] is only supported on those. Though if the recording is done in a Chromium based browser, as long as the audio is being output through their device's speakers, it can be recorded(ie. Zoom or other video app can be run in the desktop app or in another browser, and it will still be able to record it). #pagebreak() = User Flow & Functional Specs @@ -54,7 +61,7 @@ Proof of Concept is basically a simple script (that can be easily embedded into node(start, [Start]) node(join_meeting, [Doctor Joins Meeting]) - node(record, [Press Record Button on Dashbard (ask for consent?)]) + node(record, [Press Record Button on Dashboard (ask for consent?)]) node(consult, [Doctor and Patient have Consult]) node(exit_meeting, [Doctor Exits Meeting]) node(stop_record, [Doctor Presses Stop Recording]) @@ -76,10 +83,6 @@ Proof of Concept is basically a simple script (that can be easily embedded into Doctor presses record at start of consult, and the system starts recording and transcribing. Once doctor presses stop, the recording and transcribing stops, and full transcription is uploaded to the database. - -== Compliance Framework -Outline how the user experience respects healthcare regulations like HIPAA or GDPR. - #pagebreak() = Technical Design Outline @@ -124,7 +127,178 @@ Outline how the user experience respects healthcare regulations like HIPAA or GD caption: [Current Data's Journey], ) -== Data Strategy -Define what Protected Health Information (PHI) is processed and where it is stored. +#pagebreak() + += Code Snippets + +== Frontend Script -The only Protected Health Info that is stored is whatever the doctor and patient talked about in their consult. Currently, the data is streamed to the whisperAI server, but I think that it would be best to switch to a locally hosted instance of whisper, as it is an open weight model. Currently the transcription is only stored in the browser and is ephemeral(goes away on page reload), and is not sent anywhere yet. +Audio Stream Combining: + +```javascript +microphoneStream = await navigator.mediaDevices.getUserMedia({ + audio: true, +}); +systemStream = await navigator.mediaDevices.getDisplayMedia({ + video: true, + audio: true, +}); + +if (systemStream.getAudioTracks().length === 0) { + console.error("No screen share audio track. Microphone only."); + systemStream.getTracks().forEach((track) => track.stop()); + systemStream = null; +} + +audioContext = new AudioContext(); +mixedDestination = audioContext.createMediaStreamDestination(); + +const micSource = audioContext.createMediaStreamSource(microphoneStream); +micSource.connect(mixedDestination); +if (systemStream) { + const systemSource = audioContext.createMediaStreamSource(systemStream); + systemSource.connect(mixedDestination); +} + +mixedAudioStream = mixedDestination.stream; +``` + + +=== Push audio on new audio available: +```javascript +audioChunks = []; + +mediaRecorder.addEventListener("dataavailable", (event) => { + if (event.data.size > 0) { + audioChunks.push(event.data); + } +}); +``` + +#pagebreak() +=== Handle WebSocket: +```javascript +const socket = new WebSocket(BACKEND_WS_URL); +socket.binaryType = "arraybuffer"; + +socket.addEventListener("open", () => { + transcriptionSocket = socket; + resolve(socket); +}); + +socket.addEventListener("message", (event) => { + if (typeof event.data !== "string") { + return; + } + + let parsed; + try { + parsed = JSON.parse(event.data); + } catch { + return; + } + + if (typeof parsed.error === "string") { + console.error("transcription error:", parsed.error); + return; + } + + if (parsed.type === "Turn") { + handleTurnEvent(parsed); + } else if (parsed.type === "SpeakerRevision") { + handleSpeakerRevisionEvent(parsed); + } +}); +``` + + +=== Sending Audio across WebSocket: +```javascript +streamProcessorNode.onaudioprocess = (audioProcessEvent) => { + if ( + !transcriptionSocket || + transcriptionSocket.readyState !== WebSocket.OPEN + ) { + return; + } + + const channelSamples = audioProcessEvent.inputBuffer.getChannelData(0); + const downsampled = downsampleToTargetSampleRate( + channelSamples, + audioContext.sampleRate, + TARGET_SAMPLE_RATE, + ); + const buffer = float32ToPcm16Buffer(downsampled); + transcriptionSocket.send(buffer); +}; +``` + +== Backend WebSocket Pipe + +=== Router: +```rust +let state = AppState { + whisper_api_key: Arc::new(whisper_api_key), + whisper_api_base: Arc::new(whisper_api_base), +}; + +let cors = CorsLayer::new() + .allow_origin(Any) + .allow_headers(Any) + .allow_methods(Any); + +let app = Router::new() + .route("/ws/transcribe", get(ws_transcribe)) + .with_state(state) + .layer(cors); +``` + +=== Websocket Handling (upstream to client is omitted because it is basically the same code, just flipped): +#codly(skips: ((7,56),(45,86))) +```rust +async fn ws_transcribe(ws: WebSocketUpgrade, State(state): State) -> impl IntoResponse { + ws.on_upgrade(move |socket| proxy_ws_to_whisper(socket, state)) +} + + +async fn proxy_ws_to_whisper(client_socket: WebSocket, state: AppState) { + let client_to_upstream = async { + while let Some(message_result) = client_rx.next().await { + let message = match message_result { + Ok(message) => message, + Err(error) => { + return Err(format!("Client socket read error: {error}")); + } + }; + + let upstream_message = match message { + Message::Binary(bytes) => UpstreamMessage::Binary(bytes), + Message::Text(text) => UpstreamMessage::Text(text.to_string().into()), + Message::Ping(bytes) => UpstreamMessage::Ping(bytes), + Message::Pong(bytes) => UpstreamMessage::Pong(bytes), + Message::Close(frame) => { + let close_frame = + frame.map(|f| tokio_tungstenite::tungstenite::protocol::CloseFrame { + code: f.code.into(), + reason: f.reason.to_string().into(), + }); + let _ = upstream_tx.send(UpstreamMessage::Close(close_frame)).await; + return Ok(()); + } + }; + + if let Err(error) = upstream_tx.send(upstream_message).await { + return Err(format!("Upstream socket write error: {error}")); + } + } + + let _ = upstream_tx + .send(UpstreamMessage::Text( + "{\"type\":\"Terminate\"}".to_string().into(), + )) + .await; + let _ = upstream_tx.send(UpstreamMessage::Close(None)).await; + Ok(()) + }; +} +``` diff --git a/script.js b/script.js index a4ce039..0d9fe4b 100644 --- a/script.js +++ b/script.js @@ -299,7 +299,7 @@ startButton.addEventListener("click", async () => { await connectTranscriptionSocket(); streamSourceNode = audioContext.createMediaStreamSource(mixedAudioStream); - streamProcessorNode = audioContext.createScriptProcessor(4096, 1, 1); + streamProcessorNode = audioContext.createScriptProcessor(4096, 1, 1); //should switch to audioworklet interface eventually streamMuteNode = audioContext.createGain(); streamMuteNode.gain.value = 0; @@ -307,7 +307,7 @@ startButton.addEventListener("click", async () => { streamProcessorNode.connect(streamMuteNode); streamMuteNode.connect(audioContext.destination); - streamProcessorNode.onaudioprocess = (audioProcessEvent) => { + streamProcessorNode.onaudioprocess = (audioProcessEvent) => {// see line 302 if ( !transcriptionSocket || transcriptionSocket.readyState !== WebSocket.OPEN @@ -315,7 +315,7 @@ startButton.addEventListener("click", async () => { return; } - const channelSamples = audioProcessEvent.inputBuffer.getChannelData(0); + const channelSamples = audioProcessEvent.inputBuffer.getChannelData(0);// see line 302 const downsampled = downsampleToTargetSampleRate( channelSamples, audioContext.sampleRate,