Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3d093f1e1d | ||
|
|
579a91d357 |
@@ -276,7 +276,7 @@
|
|||||||
CODE_SIGN_IDENTITY = "Apple Development";
|
CODE_SIGN_IDENTITY = "Apple Development";
|
||||||
CODE_SIGN_STYLE = Automatic;
|
CODE_SIGN_STYLE = Automatic;
|
||||||
COMBINE_HIDPI_IMAGES = YES;
|
COMBINE_HIDPI_IMAGES = YES;
|
||||||
CURRENT_PROJECT_VERSION = 29;
|
CURRENT_PROJECT_VERSION = 31;
|
||||||
DEVELOPMENT_ASSET_PATHS = "\"meetingnotes/Preview Content\"";
|
DEVELOPMENT_ASSET_PATHS = "\"meetingnotes/Preview Content\"";
|
||||||
DEVELOPMENT_TEAM = G9LVHZAJNX;
|
DEVELOPMENT_TEAM = G9LVHZAJNX;
|
||||||
ENABLE_HARDENED_RUNTIME = YES;
|
ENABLE_HARDENED_RUNTIME = YES;
|
||||||
@@ -290,7 +290,7 @@
|
|||||||
"@executable_path/../Frameworks",
|
"@executable_path/../Frameworks",
|
||||||
);
|
);
|
||||||
MACOSX_DEPLOYMENT_TARGET = 15.0;
|
MACOSX_DEPLOYMENT_TARGET = 15.0;
|
||||||
MARKETING_VERSION = 1.1.17;
|
MARKETING_VERSION = 1.1.19;
|
||||||
ONLY_ACTIVE_ARCH = NO;
|
ONLY_ACTIVE_ARCH = NO;
|
||||||
OTHER_SWIFT_FLAGS = "$(inherited) -D ENABLE_TCC_SPI";
|
OTHER_SWIFT_FLAGS = "$(inherited) -D ENABLE_TCC_SPI";
|
||||||
PRODUCT_BUNDLE_IDENTIFIER = net.jamesbone.meetingnotes;
|
PRODUCT_BUNDLE_IDENTIFIER = net.jamesbone.meetingnotes;
|
||||||
@@ -312,7 +312,7 @@
|
|||||||
CODE_SIGN_IDENTITY = "Apple Development";
|
CODE_SIGN_IDENTITY = "Apple Development";
|
||||||
CODE_SIGN_STYLE = Automatic;
|
CODE_SIGN_STYLE = Automatic;
|
||||||
COMBINE_HIDPI_IMAGES = YES;
|
COMBINE_HIDPI_IMAGES = YES;
|
||||||
CURRENT_PROJECT_VERSION = 29;
|
CURRENT_PROJECT_VERSION = 31;
|
||||||
DEVELOPMENT_ASSET_PATHS = "\"meetingnotes/Preview Content\"";
|
DEVELOPMENT_ASSET_PATHS = "\"meetingnotes/Preview Content\"";
|
||||||
DEVELOPMENT_TEAM = G9LVHZAJNX;
|
DEVELOPMENT_TEAM = G9LVHZAJNX;
|
||||||
ENABLE_HARDENED_RUNTIME = YES;
|
ENABLE_HARDENED_RUNTIME = YES;
|
||||||
@@ -326,7 +326,7 @@
|
|||||||
"@executable_path/../Frameworks",
|
"@executable_path/../Frameworks",
|
||||||
);
|
);
|
||||||
MACOSX_DEPLOYMENT_TARGET = 15.0;
|
MACOSX_DEPLOYMENT_TARGET = 15.0;
|
||||||
MARKETING_VERSION = 1.1.17;
|
MARKETING_VERSION = 1.1.19;
|
||||||
ONLY_ACTIVE_ARCH = YES;
|
ONLY_ACTIVE_ARCH = YES;
|
||||||
OTHER_SWIFT_FLAGS = "$(inherited) -D ENABLE_TCC_SPI";
|
OTHER_SWIFT_FLAGS = "$(inherited) -D ENABLE_TCC_SPI";
|
||||||
PRODUCT_BUNDLE_IDENTIFIER = net.jamesbone.meetingnotes;
|
PRODUCT_BUNDLE_IDENTIFIER = net.jamesbone.meetingnotes;
|
||||||
|
|||||||
@@ -87,8 +87,8 @@ final class AudioManager: NSObject, ObservableObject {
|
|||||||
}
|
}
|
||||||
|
|
||||||
let model = UserDefaultsManager.shared.transcriptionModel
|
let model = UserDefaultsManager.shared.transcriptionModel
|
||||||
async let micResult = transcribe(files[0], model: model)
|
async let micResult = transcribe(files[0], model: model, diarization: false)
|
||||||
async let systemResult = transcribe(files[1], model: model)
|
async let systemResult = transcribe(files[1], model: model, diarization: true)
|
||||||
let (micTranscription, systemTranscription) = await (micResult, systemResult)
|
let (micTranscription, systemTranscription) = await (micResult, systemResult)
|
||||||
let results = [micTranscription, systemTranscription]
|
let results = [micTranscription, systemTranscription]
|
||||||
|
|
||||||
@@ -125,8 +125,8 @@ final class AudioManager: NSObject, ObservableObject {
|
|||||||
let model = UserDefaultsManager.shared.transcriptionModel
|
let model = UserDefaultsManager.shared.transcriptionModel
|
||||||
let micURL = recoveryFiles.first(where: { $0.source == .mic })?.url
|
let micURL = recoveryFiles.first(where: { $0.source == .mic })?.url
|
||||||
let systemURL = recoveryFiles.first(where: { $0.source == .system })?.url
|
let systemURL = recoveryFiles.first(where: { $0.source == .system })?.url
|
||||||
async let micResult = transcribe(micURL, model: model)
|
async let micResult = transcribe(micURL, model: model, diarization: false)
|
||||||
async let systemResult = transcribe(systemURL, model: model)
|
async let systemResult = transcribe(systemURL, model: model, diarization: true)
|
||||||
let (micTranscription, systemTranscription) = await (micResult, systemResult)
|
let (micTranscription, systemTranscription) = await (micResult, systemResult)
|
||||||
let results = [micTranscription, systemTranscription]
|
let results = [micTranscription, systemTranscription]
|
||||||
let (chunks, failures) = buildTranscriptChunks(
|
let (chunks, failures) = buildTranscriptChunks(
|
||||||
@@ -149,10 +149,19 @@ final class AudioManager: NSObject, ObservableObject {
|
|||||||
lastRecoveryAudioFolderName = nil
|
lastRecoveryAudioFolderName = nil
|
||||||
}
|
}
|
||||||
|
|
||||||
private func transcribe(_ fileURL: URL?, model: String) async -> Result<CoderAPIClient.Transcription, Error>? {
|
private func transcribe(
|
||||||
|
_ fileURL: URL?,
|
||||||
|
model: String,
|
||||||
|
diarization: Bool
|
||||||
|
) async -> Result<CoderAPIClient.Transcription, Error>? {
|
||||||
guard let fileURL else { return nil }
|
guard let fileURL else { return nil }
|
||||||
do {
|
do {
|
||||||
return .success(try await CoderAPIClient.shared.transcribe(fileURL: fileURL, model: model))
|
return .success(try await CoderAPIClient.shared.transcribe(
|
||||||
|
fileURL: fileURL,
|
||||||
|
model: model,
|
||||||
|
diarization: diarization,
|
||||||
|
maxSpeakerCount: 4
|
||||||
|
))
|
||||||
} catch {
|
} catch {
|
||||||
return .failure(error)
|
return .failure(error)
|
||||||
}
|
}
|
||||||
@@ -181,6 +190,7 @@ final class AudioManager: NSObject, ObservableObject {
|
|||||||
updated.append(TranscriptChunk(
|
updated.append(TranscriptChunk(
|
||||||
timestamp: captureStartedAt.addingTimeInterval(max(0, segment.start)),
|
timestamp: captureStartedAt.addingTimeInterval(max(0, segment.start)),
|
||||||
source: source,
|
source: source,
|
||||||
|
speaker: source == .system ? segment.speaker : nil,
|
||||||
text: text,
|
text: text,
|
||||||
isFinal: true
|
isFinal: true
|
||||||
))
|
))
|
||||||
|
|||||||
@@ -79,7 +79,13 @@ class UserDefaultsManager {
|
|||||||
}
|
}
|
||||||
|
|
||||||
var transcriptionModel: String {
|
var transcriptionModel: String {
|
||||||
get { userDefaults.string(forKey: Keys.transcriptionModel) ?? "groq/whisper-large-v3-turbo" }
|
get {
|
||||||
|
let stored = userDefaults.string(forKey: Keys.transcriptionModel)
|
||||||
|
if stored == "local-whisper/whisper-large-v3-turbo" {
|
||||||
|
return "local-parakeet/parakeet-tdt-0.6b-v3"
|
||||||
|
}
|
||||||
|
return stored ?? "local-parakeet/parakeet-tdt-0.6b-v3"
|
||||||
|
}
|
||||||
set { userDefaults.set(newValue, forKey: Keys.transcriptionModel) }
|
set { userDefaults.set(newValue, forKey: Keys.transcriptionModel) }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -48,30 +48,46 @@ struct TranscriptChunk: Codable, Identifiable, Hashable {
|
|||||||
let id: UUID
|
let id: UUID
|
||||||
let timestamp: Date
|
let timestamp: Date
|
||||||
let source: AudioSource
|
let source: AudioSource
|
||||||
|
let speaker: Int?
|
||||||
let text: String
|
let text: String
|
||||||
let isFinal: Bool
|
let isFinal: Bool
|
||||||
|
|
||||||
init(id: UUID = UUID(), timestamp: Date = Date(), source: AudioSource, text: String, isFinal: Bool = false) {
|
init(id: UUID = UUID(), timestamp: Date = Date(), source: AudioSource, speaker: Int? = nil, text: String, isFinal: Bool = false) {
|
||||||
self.id = id
|
self.id = id
|
||||||
self.timestamp = timestamp
|
self.timestamp = timestamp
|
||||||
self.source = source
|
self.source = source
|
||||||
|
self.speaker = speaker
|
||||||
self.text = text
|
self.text = text
|
||||||
self.isFinal = isFinal
|
self.isFinal = isFinal
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var displayName: String {
|
||||||
|
if source == .mic { return "Me" }
|
||||||
|
if let speaker { return "Speaker \(speaker)" }
|
||||||
|
return source.displayName
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
struct CollapsedTranscriptChunk: Identifiable {
|
struct CollapsedTranscriptChunk: Identifiable {
|
||||||
let id: UUID
|
let id: UUID
|
||||||
let timestamp: Date
|
let timestamp: Date
|
||||||
let source: AudioSource
|
let source: AudioSource
|
||||||
|
let speaker: Int?
|
||||||
let combinedText: String
|
let combinedText: String
|
||||||
|
|
||||||
init(id: UUID = UUID(), timestamp: Date, source: AudioSource, combinedText: String) {
|
init(id: UUID = UUID(), timestamp: Date, source: AudioSource, speaker: Int? = nil, combinedText: String) {
|
||||||
self.id = id
|
self.id = id
|
||||||
self.timestamp = timestamp
|
self.timestamp = timestamp
|
||||||
self.source = source
|
self.source = source
|
||||||
|
self.speaker = speaker
|
||||||
self.combinedText = combinedText
|
self.combinedText = combinedText
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var displayName: String {
|
||||||
|
if source == .mic { return "Me" }
|
||||||
|
if let speaker { return "Speaker \(speaker)" }
|
||||||
|
return source.displayName
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
struct Meeting: Codable, Identifiable, Hashable {
|
struct Meeting: Codable, Identifiable, Hashable {
|
||||||
@@ -115,7 +131,7 @@ struct Meeting: Codable, Identifiable, Hashable {
|
|||||||
var transcript: String {
|
var transcript: String {
|
||||||
return transcriptChunks
|
return transcriptChunks
|
||||||
.filter { $0.isFinal }
|
.filter { $0.isFinal }
|
||||||
.map { "[\($0.source.rawValue)] \($0.text)" }
|
.map { "[\($0.displayName)] \($0.text)" }
|
||||||
.joined(separator: " ")
|
.joined(separator: " ")
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -124,7 +140,7 @@ struct Meeting: Codable, Identifiable, Hashable {
|
|||||||
let finalChunks = transcriptChunks.filter { $0.isFinal }
|
let finalChunks = transcriptChunks.filter { $0.isFinal }
|
||||||
|
|
||||||
return finalChunks.map { chunk in
|
return finalChunks.map { chunk in
|
||||||
"[\(TranscriptTimestampFormatter.string(from: chunk.timestamp))] \(chunk.source.copyPrefix): \(chunk.text)"
|
"[\(TranscriptTimestampFormatter.string(from: chunk.timestamp))] \(chunk.displayName): \(chunk.text)"
|
||||||
}.joined(separator: "\n")
|
}.joined(separator: "\n")
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -135,6 +151,7 @@ struct Meeting: Codable, Identifiable, Hashable {
|
|||||||
id: chunk.id,
|
id: chunk.id,
|
||||||
timestamp: chunk.timestamp,
|
timestamp: chunk.timestamp,
|
||||||
source: chunk.source,
|
source: chunk.source,
|
||||||
|
speaker: chunk.speaker,
|
||||||
combinedText: chunk.text
|
combinedText: chunk.text
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -138,8 +138,11 @@ final class ProcessTap {
|
|||||||
tapDescription = CATapDescription(stereoMixdownOfProcesses: [process.objectID])
|
tapDescription = CATapDescription(stereoMixdownOfProcesses: [process.objectID])
|
||||||
logger.debug("Configuring tap for single process objectID: \(process.objectID)")
|
logger.debug("Configuring tap for single process objectID: \(process.objectID)")
|
||||||
case .systemAudio:
|
case .systemAudio:
|
||||||
tapDescription = CATapDescription(monoGlobalTapButExcludeProcesses: [])
|
// Keep the HAL tap's buffer layout consistent with the default
|
||||||
logger.debug("Configuring a global system audio tap.")
|
// output stream. AudioManager performs the stereo-to-mono mix when
|
||||||
|
// it converts the captured audio to the 16 kHz transcription file.
|
||||||
|
tapDescription = CATapDescription(stereoGlobalTapButExcludeProcesses: [])
|
||||||
|
logger.debug("Configuring a stereo global system audio tap.")
|
||||||
}
|
}
|
||||||
|
|
||||||
tapDescription.uuid = UUID()
|
tapDescription.uuid = UUID()
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ struct CoderModel: Codable, Identifiable, Hashable {
|
|||||||
|
|
||||||
var supportsChat: Bool { capabilities.isEmpty || capabilities.contains("chat") }
|
var supportsChat: Bool { capabilities.isEmpty || capabilities.contains("chat") }
|
||||||
var supportsTranscription: Bool { capabilities.contains("audio_transcription") }
|
var supportsTranscription: Bool { capabilities.contains("audio_transcription") }
|
||||||
|
var supportsSpeakerDiarization: Bool { capabilities.contains("speaker_diarization") }
|
||||||
}
|
}
|
||||||
|
|
||||||
enum CoderAPIError: LocalizedError {
|
enum CoderAPIError: LocalizedError {
|
||||||
@@ -54,6 +55,7 @@ final class CoderAPIClient {
|
|||||||
let start: TimeInterval
|
let start: TimeInterval
|
||||||
let end: TimeInterval
|
let end: TimeInterval
|
||||||
let text: String
|
let text: String
|
||||||
|
let speaker: Int?
|
||||||
}
|
}
|
||||||
|
|
||||||
let text: String
|
let text: String
|
||||||
@@ -70,8 +72,16 @@ final class CoderAPIClient {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private struct TranscriptionResponse: Decodable {
|
private struct TranscriptionResponse: Decodable {
|
||||||
|
struct Word: Decodable {
|
||||||
|
let word: String
|
||||||
|
let start: TimeInterval
|
||||||
|
let end: TimeInterval
|
||||||
|
let speaker: Int?
|
||||||
|
}
|
||||||
|
|
||||||
let text: String
|
let text: String
|
||||||
let segments: [Transcription.Segment]?
|
let segments: [Transcription.Segment]?
|
||||||
|
let words: [Word]?
|
||||||
}
|
}
|
||||||
|
|
||||||
private struct AudioChunk {
|
private struct AudioChunk {
|
||||||
@@ -159,11 +169,17 @@ final class CoderAPIClient {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func transcribe(fileURL: URL, model: String, language: String = "en") async throws -> Transcription {
|
func transcribe(
|
||||||
|
fileURL: URL,
|
||||||
|
model: String,
|
||||||
|
language: String = "en",
|
||||||
|
diarization: Bool = false,
|
||||||
|
maxSpeakerCount: Int = 4
|
||||||
|
) async throws -> Transcription {
|
||||||
let selectedModel = model.trimmingCharacters(in: .whitespacesAndNewlines)
|
let selectedModel = model.trimmingCharacters(in: .whitespacesAndNewlines)
|
||||||
guard !selectedModel.isEmpty else { throw CoderAPIError.missingModel("transcription") }
|
guard !selectedModel.isEmpty else { throw CoderAPIError.missingModel("transcription") }
|
||||||
let apiKey = try requiredAPIKey(KeychainHelper.shared.getCoderAPIKey() ?? "")
|
let apiKey = try requiredAPIKey(KeychainHelper.shared.getCoderAPIKey() ?? "")
|
||||||
let chunks = try makeAudioChunks(from: fileURL)
|
let chunks = try makeAudioChunks(from: fileURL, preserveSpeakerIdentity: diarization)
|
||||||
defer {
|
defer {
|
||||||
for chunk in chunks where chunk.isTemporary {
|
for chunk in chunks where chunk.isTemporary {
|
||||||
try? FileManager.default.removeItem(at: chunk.url)
|
try? FileManager.default.removeItem(at: chunk.url)
|
||||||
@@ -180,13 +196,15 @@ final class CoderAPIClient {
|
|||||||
chunk.url,
|
chunk.url,
|
||||||
model: selectedModel,
|
model: selectedModel,
|
||||||
language: language,
|
language: language,
|
||||||
apiKey: apiKey
|
apiKey: apiKey,
|
||||||
|
diarization: diarization,
|
||||||
|
maxSpeakerCount: maxSpeakerCount
|
||||||
)
|
)
|
||||||
if transcription.segments.isEmpty {
|
if transcription.segments.isEmpty {
|
||||||
let text = transcription.text.trimmingCharacters(in: .whitespacesAndNewlines)
|
let text = transcription.text.trimmingCharacters(in: .whitespacesAndNewlines)
|
||||||
if !text.isEmpty {
|
if !text.isEmpty {
|
||||||
textParts.append(text)
|
textParts.append(text)
|
||||||
segments.append(.init(start: chunk.offset, end: chunk.offset, text: text))
|
segments.append(.init(start: chunk.offset, end: chunk.offset, text: text, speaker: nil))
|
||||||
}
|
}
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
@@ -206,7 +224,8 @@ final class CoderAPIClient {
|
|||||||
segments.append(.init(
|
segments.append(.init(
|
||||||
start: segment.start + chunk.offset,
|
start: segment.start + chunk.offset,
|
||||||
end: segment.end + chunk.offset,
|
end: segment.end + chunk.offset,
|
||||||
text: text
|
text: text,
|
||||||
|
speaker: segment.speaker
|
||||||
))
|
))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -218,13 +237,17 @@ final class CoderAPIClient {
|
|||||||
_ fileURL: URL,
|
_ fileURL: URL,
|
||||||
model: String,
|
model: String,
|
||||||
language: String,
|
language: String,
|
||||||
apiKey: String
|
apiKey: String,
|
||||||
|
diarization: Bool,
|
||||||
|
maxSpeakerCount: Int
|
||||||
) async throws -> Transcription {
|
) async throws -> Transcription {
|
||||||
let boundary = "Meetingnotes-\(UUID().uuidString)"
|
let boundary = "Meetingnotes-\(UUID().uuidString)"
|
||||||
let bodyURL = try makeMultipartBody(
|
let bodyURL = try makeMultipartBody(
|
||||||
audioURL: fileURL,
|
audioURL: fileURL,
|
||||||
model: model,
|
model: model,
|
||||||
language: language,
|
language: language,
|
||||||
|
diarization: diarization,
|
||||||
|
maxSpeakerCount: maxSpeakerCount,
|
||||||
boundary: boundary
|
boundary: boundary
|
||||||
)
|
)
|
||||||
defer { try? FileManager.default.removeItem(at: bodyURL) }
|
defer { try? FileManager.default.removeItem(at: bodyURL) }
|
||||||
@@ -240,22 +263,18 @@ final class CoderAPIClient {
|
|||||||
let (data, response) = try await transcriptionSession.upload(for: request, fromFile: bodyURL)
|
let (data, response) = try await transcriptionSession.upload(for: request, fromFile: bodyURL)
|
||||||
try validate(response: response, data: data)
|
try validate(response: response, data: data)
|
||||||
let decoded = try JSONDecoder().decode(TranscriptionResponse.self, from: data)
|
let decoded = try JSONDecoder().decode(TranscriptionResponse.self, from: data)
|
||||||
return Transcription(text: decoded.text, segments: decoded.segments ?? [])
|
let segments = decoded.segments ?? segments(from: decoded.words ?? [])
|
||||||
|
return Transcription(text: decoded.text, segments: segments)
|
||||||
}
|
}
|
||||||
|
|
||||||
private func makeAudioChunks(from fileURL: URL) throws -> [AudioChunk] {
|
private func makeAudioChunks(from fileURL: URL, preserveSpeakerIdentity: Bool) throws -> [AudioChunk] {
|
||||||
let input = try AVAudioFile(forReading: fileURL)
|
let input = try AVAudioFile(forReading: fileURL)
|
||||||
let format = input.processingFormat
|
let format = input.processingFormat
|
||||||
guard format.sampleRate > 0 else {
|
guard format.sampleRate > 0 else { throw CoderAPIError.invalidResponse }
|
||||||
return [AudioChunk(url: fileURL, offset: 0, isTemporary: false)]
|
|
||||||
}
|
|
||||||
|
|
||||||
let duration = Double(input.length) / format.sampleRate
|
let framesPerChunk = preserveSpeakerIdentity
|
||||||
guard duration > transcriptionChunkDuration else {
|
? max(1, input.length)
|
||||||
return [AudioChunk(url: fileURL, offset: 0, isTemporary: false)]
|
: AVAudioFramePosition(format.sampleRate * transcriptionChunkDuration)
|
||||||
}
|
|
||||||
|
|
||||||
let framesPerChunk = AVAudioFramePosition(format.sampleRate * transcriptionChunkDuration)
|
|
||||||
var chunks: [AudioChunk] = []
|
var chunks: [AudioChunk] = []
|
||||||
var frameOffset: AVAudioFramePosition = 0
|
var frameOffset: AVAudioFramePosition = 0
|
||||||
|
|
||||||
@@ -263,7 +282,7 @@ final class CoderAPIClient {
|
|||||||
while frameOffset < input.length {
|
while frameOffset < input.length {
|
||||||
let frameCount = min(framesPerChunk, input.length - frameOffset)
|
let frameCount = min(framesPerChunk, input.length - frameOffset)
|
||||||
let chunkURL = FileManager.default.temporaryDirectory
|
let chunkURL = FileManager.default.temporaryDirectory
|
||||||
.appendingPathComponent("meetingnotes-transcription-\(UUID().uuidString).m4a")
|
.appendingPathComponent("meetingnotes-transcription-\(UUID().uuidString).wav")
|
||||||
try writeAudioChunk(
|
try writeAudioChunk(
|
||||||
from: input,
|
from: input,
|
||||||
frameCount: frameCount,
|
frameCount: frameCount,
|
||||||
@@ -293,10 +312,13 @@ final class CoderAPIClient {
|
|||||||
to outputURL: URL
|
to outputURL: URL
|
||||||
) throws {
|
) throws {
|
||||||
let settings: [String: Any] = [
|
let settings: [String: Any] = [
|
||||||
AVFormatIDKey: kAudioFormatMPEG4AAC,
|
AVFormatIDKey: kAudioFormatLinearPCM,
|
||||||
AVSampleRateKey: format.sampleRate,
|
AVSampleRateKey: format.sampleRate,
|
||||||
AVNumberOfChannelsKey: format.channelCount,
|
AVNumberOfChannelsKey: format.channelCount,
|
||||||
AVEncoderBitRateKey: 48_000 * max(1, Int(format.channelCount))
|
AVLinearPCMBitDepthKey: 16,
|
||||||
|
AVLinearPCMIsFloatKey: false,
|
||||||
|
AVLinearPCMIsBigEndianKey: false,
|
||||||
|
AVLinearPCMIsNonInterleaved: false
|
||||||
]
|
]
|
||||||
let output = try AVAudioFile(
|
let output = try AVAudioFile(
|
||||||
forWriting: outputURL,
|
forWriting: outputURL,
|
||||||
@@ -317,6 +339,51 @@ final class CoderAPIClient {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private func segments(from words: [TranscriptionResponse.Word]) -> [Transcription.Segment] {
|
||||||
|
var result: [Transcription.Segment] = []
|
||||||
|
var currentWords: [String] = []
|
||||||
|
var currentStart: TimeInterval?
|
||||||
|
var currentEnd: TimeInterval = 0
|
||||||
|
var currentSpeaker: Int?
|
||||||
|
|
||||||
|
func flush() {
|
||||||
|
guard let start = currentStart, !currentWords.isEmpty else { return }
|
||||||
|
result.append(.init(
|
||||||
|
start: start,
|
||||||
|
end: currentEnd,
|
||||||
|
text: currentWords.joined(separator: " "),
|
||||||
|
speaker: currentSpeaker
|
||||||
|
))
|
||||||
|
currentWords.removeAll(keepingCapacity: true)
|
||||||
|
currentStart = nil
|
||||||
|
currentEnd = 0
|
||||||
|
currentSpeaker = nil
|
||||||
|
}
|
||||||
|
|
||||||
|
for word in words {
|
||||||
|
let text = word.word.trimmingCharacters(in: .whitespacesAndNewlines)
|
||||||
|
guard !text.isEmpty else { continue }
|
||||||
|
let speakerChanged = currentStart != nil && word.speaker != currentSpeaker
|
||||||
|
let longPause = currentStart != nil && word.start - currentEnd > 1.5
|
||||||
|
if speakerChanged || longPause { flush() }
|
||||||
|
|
||||||
|
if currentStart == nil {
|
||||||
|
currentStart = word.start
|
||||||
|
currentSpeaker = word.speaker
|
||||||
|
}
|
||||||
|
currentWords.append(text)
|
||||||
|
currentEnd = word.end
|
||||||
|
|
||||||
|
let sentenceEnded = text.last.map { ".!?".contains($0) } ?? false
|
||||||
|
let duration = currentEnd - (currentStart ?? currentEnd)
|
||||||
|
if currentWords.count >= 40 || (sentenceEnded && (currentWords.count >= 12 || duration >= 8)) {
|
||||||
|
flush()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
flush()
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
private func endpoint(baseURL: String, path: String) throws -> URL {
|
private func endpoint(baseURL: String, path: String) throws -> URL {
|
||||||
guard var components = URLComponents(string: baseURL.trimmingCharacters(in: .whitespacesAndNewlines)),
|
guard var components = URLComponents(string: baseURL.trimmingCharacters(in: .whitespacesAndNewlines)),
|
||||||
let scheme = components.scheme?.lowercased(),
|
let scheme = components.scheme?.lowercased(),
|
||||||
@@ -346,7 +413,14 @@ final class CoderAPIClient {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private func makeMultipartBody(audioURL: URL, model: String, language: String, boundary: String) throws -> URL {
|
private func makeMultipartBody(
|
||||||
|
audioURL: URL,
|
||||||
|
model: String,
|
||||||
|
language: String,
|
||||||
|
diarization: Bool,
|
||||||
|
maxSpeakerCount: Int,
|
||||||
|
boundary: String
|
||||||
|
) throws -> URL {
|
||||||
let bodyURL = FileManager.default.temporaryDirectory.appendingPathComponent("meetingnotes-upload-\(UUID().uuidString).body")
|
let bodyURL = FileManager.default.temporaryDirectory.appendingPathComponent("meetingnotes-upload-\(UUID().uuidString).body")
|
||||||
_ = FileManager.default.createFile(atPath: bodyURL.path, contents: nil)
|
_ = FileManager.default.createFile(atPath: bodyURL.path, contents: nil)
|
||||||
let output = try FileHandle(forWritingTo: bodyURL)
|
let output = try FileHandle(forWritingTo: bodyURL)
|
||||||
@@ -358,7 +432,11 @@ final class CoderAPIClient {
|
|||||||
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"model\"\r\n\r\n\(model)\r\n")
|
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"model\"\r\n\r\n\(model)\r\n")
|
||||||
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n\(language)\r\n")
|
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n\(language)\r\n")
|
||||||
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"response_format\"\r\n\r\nverbose_json\r\n")
|
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"response_format\"\r\n\r\nverbose_json\r\n")
|
||||||
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"file\"; filename=\"\(audioURL.lastPathComponent)\"\r\nContent-Type: audio/mp4\r\n\r\n")
|
if diarization {
|
||||||
|
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"diarization\"\r\n\r\ntrue\r\n")
|
||||||
|
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"max_speaker_count\"\r\n\r\n\(maxSpeakerCount)\r\n")
|
||||||
|
}
|
||||||
|
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"file\"; filename=\"\(audioURL.lastPathComponent)\"\r\nContent-Type: audio/wav\r\n\r\n")
|
||||||
let input = try FileHandle(forReadingFrom: audioURL)
|
let input = try FileHandle(forReadingFrom: audioURL)
|
||||||
defer { try? input.close() }
|
defer { try? input.close() }
|
||||||
while let chunk = try input.read(upToCount: 1 << 20), !chunk.isEmpty {
|
while let chunk = try input.read(upToCount: 1 << 20), !chunk.isEmpty {
|
||||||
|
|||||||
@@ -208,12 +208,12 @@ struct CollapsedTranscriptChunkView: View {
|
|||||||
.font(.caption)
|
.font(.caption)
|
||||||
.foregroundColor(chunk.source == .mic ? .blue : .orange)
|
.foregroundColor(chunk.source == .mic ? .blue : .orange)
|
||||||
|
|
||||||
Text(chunk.source.displayName)
|
Text(chunk.displayName)
|
||||||
.font(.caption)
|
.font(.caption)
|
||||||
.fontWeight(.medium)
|
.fontWeight(.medium)
|
||||||
.foregroundColor(chunk.source == .mic ? .blue : .orange)
|
.foregroundColor(chunk.source == .mic ? .blue : .orange)
|
||||||
}
|
}
|
||||||
.frame(width: 50, alignment: .leading)
|
.frame(width: 78, alignment: .leading)
|
||||||
|
|
||||||
// Transcript text
|
// Transcript text
|
||||||
Text(chunk.combinedText)
|
Text(chunk.combinedText)
|
||||||
|
|||||||
@@ -58,6 +58,12 @@ struct SettingsView: View {
|
|||||||
Text(model.displayName).tag(model.id)
|
Text(model.displayName).tag(model.id)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if viewModel.coderModels.first(where: { $0.id == viewModel.settings.transcriptionModel })?.supportsSpeakerDiarization == true {
|
||||||
|
Label("Remote participants are labeled Speaker 1–4; your microphone is labeled Me.", systemImage: "person.2.wave.2")
|
||||||
|
.font(.caption)
|
||||||
|
.foregroundColor(.secondary)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Text("The token is stored locally in Keychain. Audio and note generation are sent only to this Coder service.")
|
Text("The token is stored locally in Keychain. Audio and note generation are sent only to this Coder service.")
|
||||||
|
|||||||
Reference in New Issue
Block a user