diff --git a/.cursor/rules/overall-objective.mdc b/.cursor/rules/overall-objective.mdc
deleted file mode 100644
index 5b0413a..0000000
--- a/.cursor/rules/overall-objective.mdc
+++ /dev/null
@@ -1,20 +0,0 @@
----
-alwaysApply: true
----
-I want to build an open source version of Granola
-
-It will be a meeting notetaking macOS application
-
-Basically it will use your mic and system audio, create a live transcript, and then at the end of the meeting use the transcript + any notes you write in a notepad to generate the meeting notes, with a button to quickly copy the notes or transcript. Can also edit the resulting notes
-
-The user will provide their own deepgram and openai API keys (stored locally) for transcription and AI generation
-
-The user will also be able to modify the system prompt in addition to just being able to write a blurb about themself (that gets injected into the system prompt)
-
-All data will be stored locally on the device
-
-That will be great for the MVP to start. Later on I will add additional features such as:
-- connecting to your Google calendar
-- note templates
-- AI chat for asking questions about a meeting
-- Integrations for email, slack, etc.
\ No newline at end of file
diff --git a/README.md b/README.md
index 5cd5097..210bee1 100644
--- a/README.md
+++ b/README.md
@@ -1,7 +1,7 @@
-Introducing Notetaker, an open source Granola alternative for on-device AI meeting notes:
+Introducing Meetingnotes: the free, open-source AI notetaker for busy engineers.
-- 100% free (bring your own Deepgram & OpenAI API keys)
-- 100% data privacy (data stored on device)
+- 100% free (bring your own OpenAI API key)
+- 100% privacy (all data stored on device)
- 100% open source (please contribute)
## Features
@@ -9,7 +9,7 @@ Introducing Notetaker, an open source Granola alternative for on-device AI meeti
Implemented:
- Recording mic & system audio
-- Live transcript using Deepgram
+- Live transcript
- Ability to also write down additional notes
- AI generated enhanced notes
- Copy functionality
@@ -24,6 +24,7 @@ Todo:
- Markdown formatting
- Auto generate notes when recording is stopped
- Different note templates
+- Fix transcript UI alignment
Later:
diff --git a/landing-page/app/components/features.tsx b/landing-page/app/components/features.tsx
index 250af79..b3b2719 100644
--- a/landing-page/app/components/features.tsx
+++ b/landing-page/app/components/features.tsx
@@ -7,7 +7,7 @@ export default function Features() {
icon:
- Use your Deepgram and OpenAI API keys. Pay only for what you use, directly to the providers. + Use your OpenAI API key. Pay only for what you use, directly to the providers.
- ~$0.28/hour using Deepgram Nova-3 and GPT-4o-mini. Pay only for what you use, no markup or - subscriptions. + ~$0.20/hour using gpt-4o-mini-transcribe for transcription and gpt-4.1-mini for summarization.
- Your Deepgram and OpenAI API keys are stored securely on your device, never shared. + Your OpenAI API key is stored securely on your device, never shared.
diff --git a/notetaker/Managers/AudioManager.swift b/notetaker/Managers/AudioManager.swift index 96594e6..0fc9ccc 100644 --- a/notetaker/Managers/AudioManager.swift +++ b/notetaker/Managers/AudioManager.swift @@ -6,7 +6,7 @@ import Foundation import SwiftUI import ScreenCaptureKit -/// Manages audio capture from microphone and system audio and handles real-time transcription via Deepgram +/// Manages audio capture from microphone and system audio and handles real-time transcription via OpenAI class AudioManager: NSObject, ObservableObject { @Published var transcriptChunks: [TranscriptChunk] = [] @Published var isRecording = false @@ -14,7 +14,7 @@ class AudioManager: NSObject, ObservableObject { private var audioEngine = AVAudioEngine() private var micSocketTask: URLSessionWebSocketTask? private var systemSocketTask: URLSessionWebSocketTask? - private let deepgramURL = URL(string: "wss://api.deepgram.com/v1/listen?encoding=linear16&sample_rate=16000&channels=1&interim_results=true&model=nova-3")! + private let realtimeURL = URL(string: "wss://api.openai.com/v1/realtime?intent=transcription")! // ScreenCaptureKit properties private var stream: SCStream? @@ -22,6 +22,9 @@ class AudioManager: NSObject, ObservableObject { // Add properties near the top, after existing private vars private var micRetryCount = 0 private let maxMicRetries = 3 + + // Add current interim transcripts per source + private var currentInterim: [AudioSource: String] = [.mic: "", .system: ""] override init() { super.init() @@ -93,7 +96,7 @@ class AudioManager: NSObject, ObservableObject { } } - /// Starts a microphone tap without creating a new Deepgram connection (used when also capturing system audio) + /// Starts a microphone tap without creating a new OpenAI connection (used when also capturing system audio) private func startMicrophoneTap() { print("🎤 Starting microphone tap...") @@ -102,7 +105,7 @@ class AudioManager: NSObject, ObservableObject { let recordingFormat = inputNode.outputFormat(forBus: 0) guard let targetFormat = AVAudioFormat(commonFormat: .pcmFormatInt16, - sampleRate: 16000, + sampleRate: 24000, channels: 1, interleaved: false) else { print("❌ Failed to create target audio format for mic tap") @@ -142,7 +145,7 @@ class AudioManager: NSObject, ObservableObject { audioEngine.prepare() try audioEngine.start() - connectToDeepgram(source: .mic) + connectToOpenAIRealtime(source: .mic) print("✅ Microphone tap started successfully") micRetryCount = 0 // Reset on success @@ -222,7 +225,7 @@ class AudioManager: NSObject, ObservableObject { self.isRecording = true } - connectToDeepgram(source: .system) + connectToOpenAIRealtime(source: .system) print("✅ System audio capture started successfully") } catch { @@ -263,7 +266,7 @@ class AudioManager: NSObject, ObservableObject { private func processAudioBuffer(_ buffer: AVAudioPCMBuffer, converter: AVAudioConverter, targetFormat: AVAudioFormat, source: AudioSource) { let processBuffer = buffer - // Convert to target format (16kHz int16 mono) in a single step – AVAudioConverter will handle resampling and downmixing + // Convert to target format (24kHz int16 mono) in a single step – AVAudioConverter will handle resampling and downmixing let outputFrameCapacity = AVAudioFrameCount(Double(processBuffer.frameLength) * targetFormat.sampleRate / processBuffer.format.sampleRate) guard let outputBuffer = AVAudioPCMBuffer(pcmFormat: targetFormat, frameCapacity: outputFrameCapacity) else { return @@ -279,7 +282,7 @@ class AudioManager: NSObject, ObservableObject { return } - // Convert to Data for Deepgram + // Convert to Data for OpenAI guard let channelData = outputBuffer.int16ChannelData?[0] else { return } @@ -290,19 +293,51 @@ class AudioManager: NSObject, ObservableObject { sendAudioData(data, source: source) } - private func connectToDeepgram(source: AudioSource) { - guard let key = KeychainHelper.shared.get(forKey: "deepgramKey"), !key.isEmpty else { - print("❌ No Deepgram key found") + private func connectToOpenAIRealtime(source: AudioSource) { + guard let key = KeychainHelper.shared.get(forKey: "openAIKey"), !key.isEmpty else { + print("❌ No OpenAI key found") return } let session = URLSession(configuration: .default) - var request = URLRequest(url: deepgramURL) - request.addValue("Token \(key)", forHTTPHeaderField: "Authorization") + var request = URLRequest(url: realtimeURL) + request.addValue("Bearer \(key)", forHTTPHeaderField: "Authorization") + request.addValue("realtime=v1", forHTTPHeaderField: "OpenAI-Beta") let task = session.webSocketTask(with: request) task.resume() + // Send initial configuration + let config: [String: Any] = [ + "type": "transcription_session.update", + "session": [ + "input_audio_format": "pcm16", + "input_audio_transcription": [ + "model": "gpt-4o-mini-transcribe", + "language": "en" + ], + "turn_detection": [ + "type": "server_vad", + "threshold": 0.5, + "prefix_padding_ms": 300, + "silence_duration_ms": 200 + ] + ] + ] + + do { + let jsonData = try JSONSerialization.data(withJSONObject: config) + if let jsonStr = String(data: jsonData, encoding: .utf8) { + task.send(.string(jsonStr)) { error in + if let error = error { + print("❌ Config send error: \(error)") + } + } + } + } catch { + print("❌ Config JSON error: \(error)") + } + switch source { case .mic: micSocketTask = task @@ -311,7 +346,7 @@ class AudioManager: NSObject, ObservableObject { } receiveMessage(for: source) - print("🌐 Connected to Deepgram (\(source))") + print("🌐 Connected to OpenAI Realtime (\(source))") } private func receiveMessage(for source: AudioSource) { @@ -321,7 +356,7 @@ class AudioManager: NSObject, ObservableObject { case .success(let message): switch message { case .string(let text): - self?.parseTranscription(text, source: source) + self?.parseRealtimeEvent(text, source: source) case .data: break @unknown default: @@ -333,45 +368,54 @@ class AudioManager: NSObject, ObservableObject { // Attempt reconnect if still recording DispatchQueue.main.asyncAfter(deadline: .now() + 2) { if self?.isRecording == true { - self?.connectToDeepgram(source: source) + self?.connectToOpenAIRealtime(source: source) } } } } } - private func parseTranscription(_ text: String, source: AudioSource) { + private func parseRealtimeEvent(_ text: String, source: AudioSource) { guard let data = text.data(using: .utf8), let json = try? JSONSerialization.jsonObject(with: data) as? [String: Any], - let type = json["type"] as? String, type == "Results", - let channel = json["channel"] as? [String: Any], - let alternatives = channel["alternatives"] as? [[String: Any]], - let alt = alternatives.first, - let transcriptText = alt["transcript"] as? String, - !transcriptText.isEmpty else { return } + let type = json["type"] as? String else { return } - let isFinal = json["is_final"] as? Bool ?? false - - DispatchQueue.main.async { - let chunk = TranscriptChunk( - timestamp: Date(), - source: source, - text: transcriptText, - isFinal: isFinal - ) - - // For interim results, replace the last interim chunk from the same source - if !isFinal { - // Remove the last interim chunk from the same source - if let lastIndex = self.transcriptChunks.lastIndex(where: { !$0.isFinal && $0.source == source }) { - self.transcriptChunks.remove(at: lastIndex) + switch type { + case "conversation.item.input_audio_transcription.delta": + if let delta = json["delta"] as? String { + currentInterim[source]! += delta + + DispatchQueue.main.async { + // Remove previous interim chunk from the same source + if let lastIndex = self.transcriptChunks.lastIndex(where: { !$0.isFinal && $0.source == source }) { + self.transcriptChunks.remove(at: lastIndex) + } + let chunk = TranscriptChunk( + timestamp: Date(), + source: source, + text: self.currentInterim[source] ?? "", + isFinal: false + ) + self.transcriptChunks.append(chunk) } - self.transcriptChunks.append(chunk) - } else { - // For final results, remove any interim chunks from the same source and add the final chunk - self.transcriptChunks.removeAll { !$0.isFinal && $0.source == source } - self.transcriptChunks.append(chunk) } + case "conversation.item.input_audio_transcription.completed": + if let transcript = json["transcript"] as? String { + DispatchQueue.main.async { + // Remove any interim chunks from the same source + self.transcriptChunks.removeAll { !$0.isFinal && $0.source == source } + let chunk = TranscriptChunk( + timestamp: Date(), + source: source, + text: transcript, + isFinal: true + ) + self.transcriptChunks.append(chunk) + } + currentInterim[source] = "" + } + default: + break } } @@ -380,10 +424,20 @@ class AudioManager: NSObject, ObservableObject { guard let socket = task, socket.state == .running else { return } - socket.send(.data(data)) { error in - if let error = error { - print("❌ Send error (\(source)): \(error)") + let base64 = data.base64EncodedString() + let message: [String: Any] = ["type": "input_audio_buffer.append", "audio": base64] + + do { + let jsonData = try JSONSerialization.data(withJSONObject: message) + if let jsonStr = String(data: jsonData, encoding: .utf8) { + socket.send(.string(jsonStr)) { error in + if let error = error { + print("❌ Send error (\(source)): \(error)") + } + } } + } catch { + print("❌ JSON send error") } } @@ -402,9 +456,9 @@ extension AudioManager: SCStreamDelegate, SCStreamOutput { // Convert CMSampleBuffer to AVAudioPCMBuffer guard let pcmBuffer = sampleBuffer.asPCMBuffer else { return } - // Create converter for Deepgram format + // Create converter for OpenAI format let targetFormat = AVAudioFormat(commonFormat: .pcmFormatInt16, - sampleRate: 16000, + sampleRate: 24000, channels: 1, interleaved: false)! @@ -431,4 +485,4 @@ extension CMSampleBuffer { return AVAudioPCMBuffer(pcmFormat: format, bufferListNoCopy: audioBufferList.unsafePointer) } } -} \ No newline at end of file +} diff --git a/notetaker/Models/Settings.swift b/notetaker/Models/Settings.swift index 3cc637b..f1ded7a 100644 --- a/notetaker/Models/Settings.swift +++ b/notetaker/Models/Settings.swift @@ -1,72 +1,44 @@ import Foundation struct Settings: Codable { - var deepgramKey: String var openAIKey: String var userBlurb: String var systemPrompt: String - static let defaultSystemPrompt: String = { - guard let url = Bundle.main.url(forResource: "DefaultSystemPrompt", withExtension: "txt"), - let content = try? String(contentsOf: url) else { - assertionFailure("DefaultSystemPrompt.txt missing from bundle") - return "" + // System prompt default loading + static func defaultSystemPrompt() -> String { + guard let path = Bundle.main.path(forResource: "DefaultSystemPrompt", ofType: "txt"), + let content = try? String(contentsOfFile: path) else { + return "You are a helpful assistant that creates comprehensive meeting notes from transcript data." } return content - }() - - // Required variables for the template - static let requiredVariables: Set