From c31bb898b1045b5a2b76039f8293856cc019f8f5 Mon Sep 17 00:00:00 2001 From: Taran Nathan Date: Wed, 22 Jul 2026 19:37:28 -0700 Subject: [PATCH] document --- document/.gitignore | 1 + document/justfile | 16 ++++++ document/main.typ | 130 ++++++++++++++++++++++++++++++++++++++++++++ slides/.gitignore | 1 + slides/justfile | 16 ++++++ slides/main.typ | 1 + 6 files changed, 165 insertions(+) create mode 100644 document/.gitignore create mode 100644 document/justfile create mode 100644 document/main.typ create mode 100644 slides/.gitignore create mode 100644 slides/justfile create mode 100644 slides/main.typ diff --git a/document/.gitignore b/document/.gitignore new file mode 100644 index 0000000..f0de8a3 --- /dev/null +++ b/document/.gitignore @@ -0,0 +1 @@ +main.pdf diff --git a/document/justfile b/document/justfile new file mode 100644 index 0000000..09b495e --- /dev/null +++ b/document/justfile @@ -0,0 +1,16 @@ +default: build + +build: + typst compile main.typ + zathura main.pdf & + +watch: noview-build + zathura main.pdf & + typst watch main.typ + +clean: + rm -rf *.pdf + +[private] +noview-build: + typst compile main.typ diff --git a/document/main.typ b/document/main.typ new file mode 100644 index 0000000..9bc9586 --- /dev/null +++ b/document/main.typ @@ -0,0 +1,130 @@ +#import "@preview/fletcher:0.5.8" as fletcher: diagram, edge, node +#import fletcher.shapes: circle, diamond, ellipse, parallelogram, pill, rect +#set page(paper: "a4", numbering: "1", number-align: top + right, margin: (x: 2em, y: 3em)) +#set document(author: "Taran Nathan", title: "Independent Audio Transcription PoC") +#set text(14pt, weight: "medium", font: "Times New Roman") +#show title: tootle => align(center)[#tootle] +#show heading.where(depth: 1): x => align(center)[#x] +#set heading(numbering: "1.a:") + + + +//header +Taran Nathan +#linebreak() +#datetime.today().display("[day] [month repr:long] [year]") +#linebreak() +#title() + + += Executive Product Brief + +== Problem Statement +Duke Uni Hospital is already integrated with someone else, and we want to tap into their meetings and such(consensually, of course) to be able to provide automated meeting notes for them? + +== Value Propositions +Allows us to work within our competitor's system while still providing value and integrating them with our system. + +== Success Metrics +List target KPIs like user adoption rates, reduction in task time, or clinical compliance. + +Doctor usage and reported satisfaction with the tool in surveys. + +== Scope Boundaries +Proof of Concept is basically a simple script (that can be easily embedded into another site), that records audio from both the microphone and the system audio source(pulseaudio, coreaudio, wasapi) via browser apis. The system audio stream is obtained through the screen sharing API, so even though it will show a popup asking for screen share, it discards the video stream and only uses the audio stream. Once the audio streams are combined into one, it is sent to little server that contains WhisperAI api credentials for speech transcription over a websocket. I deliberately did not include any sort of authorization yet because I'm not really sure how you would want to do that, and the primary use case I would see would be for you guys to embed the script into another dashboard and have it hooked up to some buttons and a different UI. The whisperAI docs are kinda murky, so the backend and some of the script code might be useful as reference, even though it would probably be better to rewrite it from scratch to integrate it better in whatever frontend framework you guys are using. The simple websocket backend I wrote was made to be discarded, as I don't think that the whisperAI api would be a final solution necessarily. It would be best to switch to a locally hosted instance of whisper for control, cost, and privacy reasons. + +#pagebreak() += User Flow & Functional Specs + +== Interaction Flow Diagram + +#figure( + { + set text(size: 8pt) + diagram(node-stroke: 0.6pt, node-fill: white, node-shape: rect, spacing: (6mm, 6mm), { + let (start, join_meeting, record, consult, exit_meeting, stop_record, recieve_transcript) = ( + (0, 0), + (0, 1), + (0, 2), + (0, 3), + (0, 4), + (0, 5), + (0, 6), + ) + + node(start, [Start]) + node(join_meeting, [Doctor Joins Meeting]) + node(record, [Press Record Button on Dashbard (ask for consent?)]) + node(consult, [Doctor and Patient have Consult]) + node(exit_meeting, [Doctor Exits Meeting]) + node(stop_record, [Doctor Presses Stop Recording]) + node(recieve_transcript, [Doctor Recieves Transcript, and is pushed to DB]) + + edge(start, join_meeting, "->") + edge(join_meeting, record, "->") + edge(record, consult, "->") + edge(consult, exit_meeting, "->") + edge(exit_meeting, stop_record, "->") + edge(stop_record, recieve_transcript, "->") + }) + }, + caption: [Clinician's Journey], +) + + +== Functional Requirements +Doctor presses record at start of consult, and the system starts recording and transcribing. Once doctor presses stop, the recording and transcribing stops, and full transcription is uploaded to the database. + + + +== Compliance Framework +Outline how the user experience respects healthcare regulations like HIPAA or GDPR. + +#pagebreak() += Technical Design Outline + +== Architecture Diagram + +#figure( + { + set text(size: 10pt) + diagram(node-stroke: 0.6pt, node-fill: white, node-shape: rect, spacing: (6mm, 12mm), cell-size: (70pt, 50pt), { + let (frontend, backend, whisperAI) = (((0, 0), (6, 0)), ((0, 1), (6, 1)), ((0, 2), (6, 2))) + + // websocket init, whisper ws init, confirmation, data stream front -> server -> whisper, transcription stream whisper -> server -> front + + node(enclose: frontend, [Frontend]) + node(enclose: backend, [Backend]) + node(enclose: whisperAI, [WhisperAI API]) + + edge((1, 0), (1, 1), "-->", [Backend Websocket Init]) + + // Whisper WebSocket initialization + edge((1.1, 1), (1.1, 2), "-->", [Whisper WS Init]) + + // Whisper connection confirmation + edge((2.6, 2), (2.6, 1), "-->", [Connection Confirmation], label-side: left) + + // Backend confirms to frontend + edge((2.7, 1), (2.7, 0), "-->", [Connection Confirmation], label-side: left) + + // Audio stream: Frontend -> Backend + edge((3.8, 0), (3.8, 1), "-->", [Audio Stream]) + + // Audio stream: Backend -> Whisper + edge((3.9, 1), (3.9, 2), "-->", [Repackaged Audio Stream]) + + // Transcription stream: Whisper -> Backend + edge((4.1, 2), (4.1, 1), "-->", [Transcription Stream]) + + // Transcription stream: Backend -> Frontend + edge((4.2, 1), (4.2, 0), "-->", [Repackaged Transcription Stream]) + }) + }, + caption: [Current Data's Journey], +) + +== Data Strategy +Define what Protected Health Information (PHI) is processed and where it is stored. + +The only Protected Health Info that is stored is whatever the doctor and patient talked about in their consult. Currently, the data is streamed to the whisperAI server, but I think that it would be best to switch to a locally hosted instance of whisper, as it is an open weight model. Currently the transcription is only stored in the browser and is ephemeral(goes away on page reload), and is not sent anywhere yet. diff --git a/slides/.gitignore b/slides/.gitignore new file mode 100644 index 0000000..f0de8a3 --- /dev/null +++ b/slides/.gitignore @@ -0,0 +1 @@ +main.pdf diff --git a/slides/justfile b/slides/justfile new file mode 100644 index 0000000..1d62db3 --- /dev/null +++ b/slides/justfile @@ -0,0 +1,16 @@ +default: watch + +build: + typst compile main.typ + zathura main.pdf & + +watch: noview-build + zathura main.pdf & + typst watch main.typ + +clean: + rm -rf *.pdf + +[private] +noview-build: + typst compile main.typ diff --git a/slides/main.typ b/slides/main.typ new file mode 100644 index 0000000..9cd5b6a --- /dev/null +++ b/slides/main.typ @@ -0,0 +1 @@ +start here! -- 2.47.3