Compare commits

..
8 Commits
13 changed files with 498 additions and 91 deletions
+17 -11
View File
@@ -20,6 +20,9 @@ jobs:
runs-on: macos-15
env:
VERSION: ${{ inputs.version }}
APPLE_ID: ${{ secrets.APPLE_ID }}
APPLE_TEAM_ID: ${{ secrets.APPLE_TEAM_ID }}
APPLE_APP_PASSWORD: ${{ secrets.APPLE_APP_PASSWORD }}
SPARKLE_PRIVATE_KEY: ${{ secrets.SPARKLE_PRIVATE_KEY }}
steps:
- uses: actions/checkout@v7
@@ -31,7 +34,7 @@ jobs:
APPLE_CERTIFICATE_P12: ${{ secrets.APPLE_CERTIFICATE_P12 }}
APPLE_CERTIFICATE_PASSWORD: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }}
run: |
for variable in APPLE_CERTIFICATE_P12 APPLE_CERTIFICATE_PASSWORD SPARKLE_PRIVATE_KEY; do
for variable in APPLE_CERTIFICATE_P12 APPLE_CERTIFICATE_PASSWORD APPLE_ID APPLE_TEAM_ID APPLE_APP_PASSWORD SPARKLE_PRIVATE_KEY; do
if [[ -z "${!variable:-}" ]]; then
echo "Missing GitHub Actions secret: $variable" >&2
exit 1
@@ -53,7 +56,7 @@ jobs:
fi
echo "SIGNING_IDENTITY=$signing_identity" >> "$GITHUB_ENV"
- name: Build and sign release
- name: Build, sign, and notarize release
timeout-minutes: 30
run: scripts/package_release.sh
@@ -70,13 +73,16 @@ jobs:
GH_TOKEN: ${{ github.token }}
run: |
tag="v$VERSION"
if git rev-parse "$tag" >/dev/null 2>&1; then
echo "Tag already exists: $tag" >&2
exit 1
if gh release view "$tag" >/dev/null 2>&1; then
gh release upload "$tag" \
"$RUNNER_TEMP/meetingnotes-release/release/Meetingnotes-$VERSION.zip" \
"$RUNNER_TEMP/meetingnotes-release/release/appcast.xml" \
--clobber
else
gh release create "$tag" \
"$RUNNER_TEMP/meetingnotes-release/release/Meetingnotes-$VERSION.zip" \
"$RUNNER_TEMP/meetingnotes-release/release/appcast.xml" \
--target "$GITHUB_SHA" \
--title "Meetingnotes $VERSION" \
--generate-notes
fi
gh release create "$tag" \
"$RUNNER_TEMP/meetingnotes-release/release/Meetingnotes-$VERSION.zip" \
"$RUNNER_TEMP/meetingnotes-release/release/appcast.xml" \
--target "$GITHUB_SHA" \
--title "Meetingnotes $VERSION" \
--generate-notes
+4 -4
View File
@@ -276,7 +276,7 @@
CODE_SIGN_IDENTITY = "Apple Development";
CODE_SIGN_STYLE = Automatic;
COMBINE_HIDPI_IMAGES = YES;
CURRENT_PROJECT_VERSION = 28;
CURRENT_PROJECT_VERSION = 35;
DEVELOPMENT_ASSET_PATHS = "\"meetingnotes/Preview Content\"";
DEVELOPMENT_TEAM = G9LVHZAJNX;
ENABLE_HARDENED_RUNTIME = YES;
@@ -290,7 +290,7 @@
"@executable_path/../Frameworks",
);
MACOSX_DEPLOYMENT_TARGET = 15.0;
MARKETING_VERSION = 1.1.16;
MARKETING_VERSION = 1.1.23;
ONLY_ACTIVE_ARCH = NO;
OTHER_SWIFT_FLAGS = "$(inherited) -D ENABLE_TCC_SPI";
PRODUCT_BUNDLE_IDENTIFIER = net.jamesbone.meetingnotes;
@@ -312,7 +312,7 @@
CODE_SIGN_IDENTITY = "Apple Development";
CODE_SIGN_STYLE = Automatic;
COMBINE_HIDPI_IMAGES = YES;
CURRENT_PROJECT_VERSION = 28;
CURRENT_PROJECT_VERSION = 35;
DEVELOPMENT_ASSET_PATHS = "\"meetingnotes/Preview Content\"";
DEVELOPMENT_TEAM = G9LVHZAJNX;
ENABLE_HARDENED_RUNTIME = YES;
@@ -326,7 +326,7 @@
"@executable_path/../Frameworks",
);
MACOSX_DEPLOYMENT_TARGET = 15.0;
MARKETING_VERSION = 1.1.16;
MARKETING_VERSION = 1.1.23;
ONLY_ACTIVE_ARCH = YES;
OTHER_SWIFT_FLAGS = "$(inherited) -D ENABLE_TCC_SPI";
PRODUCT_BUNDLE_IDENTIFIER = net.jamesbone.meetingnotes;
+167 -15
View File
@@ -86,9 +86,22 @@ final class AudioManager: NSObject, ObservableObject {
isProcessing = false
}
repairHalfDurationSystemWAVIfNeeded(in: files)
let completedFiles = files.compactMap { $0 }
let audioFolder = preserveAudioFiles(completedFiles, meetingID: completedMeetingID)
let transcriptionFiles = preservedAudioFiles(files, in: audioFolder)
lastRecoveryAudioFolderName = audioFolder?.lastPathComponent
if let mismatch = captureDurationMismatch(in: transcriptionFiles) {
let recoveryMessage = audioFolder == nil
? " The audio remains in the app's temporary folder."
: " Audio was kept so it can be recovered."
errorMessage = "System audio timing was invalid (\(mismatch)). Transcription was stopped to avoid an out-of-order result." + recoveryMessage
return transcriptChunks.filter(\.isFinal)
}
let model = UserDefaultsManager.shared.transcriptionModel
async let micResult = transcribe(files[0], model: model)
async let systemResult = transcribe(files[1], model: model)
async let micResult = transcribe(transcriptionFiles[0], model: model, diarization: false)
async let systemResult = transcribe(transcriptionFiles[1], model: model, diarization: true)
let (micTranscription, systemTranscription) = await (micResult, systemResult)
let results = [micTranscription, systemTranscription]
@@ -98,9 +111,6 @@ final class AudioManager: NSObject, ObservableObject {
existingChunks: transcriptChunks.filter(\.isFinal)
)
transcriptChunks = updated
let completedFiles = files.compactMap { $0 }
let audioFolder = preserveAudioFiles(completedFiles, meetingID: completedMeetingID)
lastRecoveryAudioFolderName = audioFolder?.lastPathComponent
if !failures.isEmpty {
let retentionDays = UserDefaultsManager.shared.audioRetentionDays
let retentionUnit = retentionDays == 1 ? "day" : "days"
@@ -125,8 +135,8 @@ final class AudioManager: NSObject, ObservableObject {
let model = UserDefaultsManager.shared.transcriptionModel
let micURL = recoveryFiles.first(where: { $0.source == .mic })?.url
let systemURL = recoveryFiles.first(where: { $0.source == .system })?.url
async let micResult = transcribe(micURL, model: model)
async let systemResult = transcribe(systemURL, model: model)
async let micResult = transcribe(micURL, model: model, diarization: false)
async let systemResult = transcribe(systemURL, model: model, diarization: true)
let (micTranscription, systemTranscription) = await (micResult, systemResult)
let results = [micTranscription, systemTranscription]
let (chunks, failures) = buildTranscriptChunks(
@@ -149,10 +159,19 @@ final class AudioManager: NSObject, ObservableObject {
lastRecoveryAudioFolderName = nil
}
private func transcribe(_ fileURL: URL?, model: String) async -> Result<CoderAPIClient.Transcription, Error>? {
private func transcribe(
_ fileURL: URL?,
model: String,
diarization: Bool
) async -> Result<CoderAPIClient.Transcription, Error>? {
guard let fileURL else { return nil }
do {
return .success(try await CoderAPIClient.shared.transcribe(fileURL: fileURL, model: model))
return .success(try await CoderAPIClient.shared.transcribe(
fileURL: fileURL,
model: model,
diarization: diarization,
maxSpeakerCount: 4
))
} catch {
return .failure(error)
}
@@ -181,6 +200,7 @@ final class AudioManager: NSObject, ObservableObject {
updated.append(TranscriptChunk(
timestamp: captureStartedAt.addingTimeInterval(max(0, segment.start)),
source: source,
speaker: source == .system ? segment.speaker : nil,
text: text,
isFinal: true
))
@@ -199,15 +219,18 @@ final class AudioManager: NSObject, ObservableObject {
private func prepareAudioFiles() throws {
let settings: [String: Any] = [
AVFormatIDKey: kAudioFormatMPEG4AAC,
AVFormatIDKey: kAudioFormatLinearPCM,
AVSampleRateKey: 16_000,
AVNumberOfChannelsKey: 1,
AVEncoderBitRateKey: 48_000
AVLinearPCMBitDepthKey: 16,
AVLinearPCMIsFloatKey: false,
AVLinearPCMIsBigEndianKey: false,
AVLinearPCMIsNonInterleaved: false
]
let base = FileManager.default.temporaryDirectory
let id = sessionID.uuidString
let micURL = base.appendingPathComponent("meetingnotes-\(id)-mic.m4a")
let systemURL = base.appendingPathComponent("meetingnotes-\(id)-system.m4a")
let micURL = base.appendingPathComponent("meetingnotes-\(id)-mic.wav")
let systemURL = base.appendingPathComponent("meetingnotes-\(id)-system.wav")
let newMicAudioFile = try AVAudioFile(
forWriting: micURL,
settings: settings,
@@ -343,13 +366,25 @@ final class AudioManager: NSObject, ObservableObject {
private func startTapIO(_ tap: ProcessTap) throws {
guard var description = tap.tapStreamDescription,
let inputFormat = AVAudioFormat(streamDescription: &description),
let advertisedInputFormat = AVAudioFormat(streamDescription: &description),
let targetFormat = systemAudioFile?.processingFormat,
let converter = AVAudioConverter(from: inputFormat, to: targetFormat) else {
advertisedInputFormat.sampleRate > 0 else {
throw NSError(domain: "AudioManager", code: -1, userInfo: [NSLocalizedDescriptionKey: "Unsupported system audio format"])
}
var inputFormat: AVAudioFormat?
var converter: AVAudioConverter?
try tap.run(on: tapQueue) { [weak self] _, inputData, _, _, _ in
guard let self else { return }
if inputFormat == nil {
inputFormat = self.inputFormat(
for: inputData,
advertisedFormat: advertisedInputFormat
)
if let inputFormat {
converter = AVAudioConverter(from: inputFormat, to: targetFormat)
}
}
guard let inputFormat, let converter else { return }
// The tap queue is serial. Reusing the converter preserves its
// resampler state instead of discarding audio at every callback.
self.processAudioBuffer(
@@ -364,6 +399,30 @@ final class AudioManager: NSObject, ObservableObject {
}
}
private func inputFormat(
for inputData: UnsafePointer<AudioBufferList>,
advertisedFormat: AVAudioFormat
) -> AVAudioFormat? {
let buffers = UnsafeMutableAudioBufferListPointer(
UnsafeMutablePointer(mutating: inputData)
)
let channelCount = buffers.reduce(UInt32(0)) { $0 + $1.mNumberChannels }
guard channelCount > 0 else { return nil }
// HAL tap metadata can advertise interleaved stereo while the callback
// supplies one mono buffer per channel (or the reverse). Constructing a
// PCM buffer with that mismatched layout halves its frame count and
// produces 2x-speed system audio. The callback's AudioBufferList is the
// authoritative layout for the memory we are copying.
let isInterleaved = buffers.count == 1 && channelCount > 1
return AVAudioFormat(
commonFormat: advertisedFormat.commonFormat,
sampleRate: advertisedFormat.sampleRate,
channels: AVAudioChannelCount(channelCount),
interleaved: isInterleaved
)
}
private func copyAudioBuffer(
from inputData: UnsafePointer<AudioBufferList>,
format: AVAudioFormat
@@ -510,6 +569,99 @@ final class AudioManager: NSObject, ObservableObject {
private func preserveAudioFiles(_ urls: [URL], meetingID: UUID) -> URL? {
LocalStorageManager.shared.preserveAudioFiles(urls, for: meetingID)
}
private func preservedAudioFiles(_ urls: [URL?], in folder: URL?) -> [URL?] {
urls.map { sourceURL in
guard let sourceURL else { return nil }
guard let folder else { return sourceURL }
let preservedURL = folder.appendingPathComponent(sourceURL.lastPathComponent)
return FileManager.default.fileExists(atPath: preservedURL.path) ? preservedURL : sourceURL
}
}
private func captureDurationMismatch(in files: [URL?]) -> String? {
guard files.count >= 2,
let micDuration = audioDuration(at: files[0]),
let systemDuration = audioDuration(at: files[1]),
micDuration >= 60 else { return nil }
let ratio = systemDuration / micDuration
guard (0.45...0.55).contains(ratio) || (1.8...2.2).contains(ratio) else { return nil }
return String(format: "mic %.1fs, system %.1fs", micDuration, systemDuration)
}
private func repairHalfDurationSystemWAVIfNeeded(in files: [URL?]) {
guard files.count >= 2,
let micDuration = audioDuration(at: files[0]),
let systemURL = files[1],
let systemDuration = audioDuration(at: systemURL),
micDuration >= 60,
systemURL.pathExtension.caseInsensitiveCompare("wav") == .orderedSame,
(0.48...0.52).contains(systemDuration / micDuration) else { return }
try? halveWAVSampleRate(at: systemURL)
}
private func halveWAVSampleRate(at url: URL) throws {
let handle = try FileHandle(forUpdating: url)
defer { try? handle.close() }
try handle.seek(toOffset: 0)
guard let riffHeader = try handle.read(upToCount: 12),
riffHeader.count == 12,
String(data: riffHeader[0..<4], encoding: .ascii) == "RIFF",
String(data: riffHeader[8..<12], encoding: .ascii) == "WAVE" else {
throw NSError(domain: "AudioManager", code: -2, userInfo: [NSLocalizedDescriptionKey: "Invalid WAV header"])
}
var offset: UInt64 = 12
while true {
try handle.seek(toOffset: offset)
guard let chunkHeader = try handle.read(upToCount: 8), chunkHeader.count == 8 else { break }
let chunkID = String(data: chunkHeader[0..<4], encoding: .ascii)
let chunkSize = UInt32(chunkHeader[4])
| (UInt32(chunkHeader[5]) << 8)
| (UInt32(chunkHeader[6]) << 16)
| (UInt32(chunkHeader[7]) << 24)
let chunkDataOffset = offset + 8
if chunkID == "fmt ", chunkSize >= 16 {
try handle.seek(toOffset: chunkDataOffset)
guard let format = try handle.read(upToCount: 16), format.count == 16 else { break }
let audioFormat = UInt16(format[0]) | (UInt16(format[1]) << 8)
let blockAlign = UInt16(format[12]) | (UInt16(format[13]) << 8)
let sampleRate = UInt32(format[4])
| (UInt32(format[5]) << 8)
| (UInt32(format[6]) << 16)
| (UInt32(format[7]) << 24)
guard audioFormat == 1, sampleRate >= 16_000, sampleRate.isMultiple(of: 2) else {
throw NSError(domain: "AudioManager", code: -3, userInfo: [NSLocalizedDescriptionKey: "Unsupported WAV format"])
}
let correctedSampleRate = sampleRate / 2
let correctedByteRate = correctedSampleRate * UInt32(blockAlign)
try handle.seek(toOffset: chunkDataOffset + 4)
try handle.write(contentsOf: littleEndianData(correctedSampleRate))
try handle.write(contentsOf: littleEndianData(correctedByteRate))
try handle.synchronize()
return
}
offset = chunkDataOffset + UInt64(chunkSize) + UInt64(chunkSize % 2)
}
throw NSError(domain: "AudioManager", code: -4, userInfo: [NSLocalizedDescriptionKey: "WAV format chunk was not found"])
}
private func littleEndianData(_ value: UInt32) -> Data {
var littleEndianValue = value.littleEndian
return withUnsafeBytes(of: &littleEndianValue) { Data($0) }
}
private func audioDuration(at url: URL?) -> TimeInterval? {
guard let url,
let file = try? AVAudioFile(forReading: url),
file.processingFormat.sampleRate > 0 else { return nil }
return Double(file.length) / file.processingFormat.sampleRate
}
private func resetAudioLevels() {
micAudioLevel = 0
@@ -256,50 +256,12 @@ class LocalStorageManager {
}
func findRecoveryAudioFolder(for meeting: Meeting) -> URL? {
if let name = meeting.recoveryAudioFolderName,
let folder = recoveryAudioFolder(named: name) {
return folder
}
let claimedFolderNames = Set(
loadMeetings()
.filter { $0.id != meeting.id }
.compactMap(\.recoveryAudioFolderName)
)
guard let folders = try? FileManager.default.contentsOfDirectory(
at: recoveryDirectory,
includingPropertiesForKeys: [.isDirectoryKey, .creationDateKey, .contentModificationDateKey],
options: [.skipsHiddenFiles]
) else {
let canonicalName = meeting.id.uuidString
guard meeting.recoveryAudioFolderName == nil
|| meeting.recoveryAudioFolderName?.caseInsensitiveCompare(canonicalName) == .orderedSame else {
return nil
}
let candidates = folders.compactMap { folder -> (url: URL, distance: TimeInterval)? in
let folderValues = try? folder.resourceValues(
forKeys: [.isDirectoryKey, .creationDateKey, .contentModificationDateKey]
)
let files = recoveryAudioFiles(in: folder)
guard folderValues?.isDirectory == true,
!claimedFolderNames.contains(folder.lastPathComponent),
!files.isEmpty else {
return nil
}
let dates = files.compactMap { file -> Date? in
let values = try? file.url.resourceValues(forKeys: [.creationDateKey, .contentModificationDateKey])
return values?.creationDate ?? values?.contentModificationDate
}
let referenceDate = dates.min()
?? folderValues?.creationDate
?? folderValues?.contentModificationDate
guard let referenceDate else { return nil }
return (folder, abs(referenceDate.timeIntervalSince(meeting.date)))
}
// This fallback links recovery files created by older app versions.
return candidates
.filter { $0.distance <= 12 * 60 * 60 }
.min(by: { $0.distance < $1.distance })?
.url
return recoveryAudioFolder(named: canonicalName)
}
func deleteRecoveryAudioFolder(_ folder: URL) {
@@ -124,6 +124,9 @@ class RecordingSessionManager: ObservableObject {
// Find and update the active meeting
if let index = meetings.firstIndex(where: { $0.id == meetingId }) {
meetings[index].transcriptChunks = chunks
if let recoveryAudioFolderName = lastRecoveryAudioFolderName {
meetings[index].recoveryAudioFolderName = recoveryAudioFolderName
}
// Save the updated meeting
let success = LocalStorageManager.shared.saveMeeting(meetings[index])
@@ -79,7 +79,7 @@ class UserDefaultsManager {
}
var transcriptionModel: String {
get { userDefaults.string(forKey: Keys.transcriptionModel) ?? "groq/whisper-large-v3-turbo" }
get { userDefaults.string(forKey: Keys.transcriptionModel) ?? "local-parakeet/parakeet-tdt-0.6b-v3" }
set { userDefaults.set(newValue, forKey: Keys.transcriptionModel) }
}
+21 -4
View File
@@ -48,30 +48,46 @@ struct TranscriptChunk: Codable, Identifiable, Hashable {
let id: UUID
let timestamp: Date
let source: AudioSource
let speaker: Int?
let text: String
let isFinal: Bool
init(id: UUID = UUID(), timestamp: Date = Date(), source: AudioSource, text: String, isFinal: Bool = false) {
init(id: UUID = UUID(), timestamp: Date = Date(), source: AudioSource, speaker: Int? = nil, text: String, isFinal: Bool = false) {
self.id = id
self.timestamp = timestamp
self.source = source
self.speaker = speaker
self.text = text
self.isFinal = isFinal
}
var displayName: String {
if source == .mic { return "Me" }
if let speaker { return "Speaker \(speaker)" }
return source.displayName
}
}
struct CollapsedTranscriptChunk: Identifiable {
let id: UUID
let timestamp: Date
let source: AudioSource
let speaker: Int?
let combinedText: String
init(id: UUID = UUID(), timestamp: Date, source: AudioSource, combinedText: String) {
init(id: UUID = UUID(), timestamp: Date, source: AudioSource, speaker: Int? = nil, combinedText: String) {
self.id = id
self.timestamp = timestamp
self.source = source
self.speaker = speaker
self.combinedText = combinedText
}
var displayName: String {
if source == .mic { return "Me" }
if let speaker { return "Speaker \(speaker)" }
return source.displayName
}
}
struct Meeting: Codable, Identifiable, Hashable {
@@ -115,7 +131,7 @@ struct Meeting: Codable, Identifiable, Hashable {
var transcript: String {
return transcriptChunks
.filter { $0.isFinal }
.map { "[\($0.source.rawValue)] \($0.text)" }
.map { "[\($0.displayName)] \($0.text)" }
.joined(separator: " ")
}
@@ -124,7 +140,7 @@ struct Meeting: Codable, Identifiable, Hashable {
let finalChunks = transcriptChunks.filter { $0.isFinal }
return finalChunks.map { chunk in
"[\(TranscriptTimestampFormatter.string(from: chunk.timestamp))] \(chunk.source.copyPrefix): \(chunk.text)"
"[\(TranscriptTimestampFormatter.string(from: chunk.timestamp))] \(chunk.displayName): \(chunk.text)"
}.joined(separator: "\n")
}
@@ -135,6 +151,7 @@ struct Meeting: Codable, Identifiable, Hashable {
id: chunk.id,
timestamp: chunk.timestamp,
source: chunk.source,
speaker: chunk.speaker,
combinedText: chunk.text
)
}
+5 -2
View File
@@ -138,8 +138,11 @@ final class ProcessTap {
tapDescription = CATapDescription(stereoMixdownOfProcesses: [process.objectID])
logger.debug("Configuring tap for single process objectID: \(process.objectID)")
case .systemAudio:
tapDescription = CATapDescription(monoGlobalTapButExcludeProcesses: [])
logger.debug("Configuring a global system audio tap.")
// Keep the HAL tap's buffer layout consistent with the default
// output stream. AudioManager performs the stereo-to-mono mix when
// it converts the captured audio to the 16 kHz transcription file.
tapDescription = CATapDescription(stereoGlobalTapButExcludeProcesses: [])
logger.debug("Configuring a stereo global system audio tap.")
}
tapDescription.uuid = UUID()
+229 -5
View File
@@ -1,3 +1,4 @@
import AVFoundation
import Foundation
struct CoderModel: Codable, Identifiable, Hashable {
@@ -20,6 +21,7 @@ struct CoderModel: Codable, Identifiable, Hashable {
var supportsChat: Bool { capabilities.isEmpty || capabilities.contains("chat") }
var supportsTranscription: Bool { capabilities.contains("audio_transcription") }
var supportsSpeakerDiarization: Bool { capabilities.contains("speaker_diarization") }
}
enum CoderAPIError: LocalizedError {
@@ -53,6 +55,7 @@ final class CoderAPIClient {
let start: TimeInterval
let end: TimeInterval
let text: String
let speaker: Int?
}
let text: String
@@ -69,10 +72,25 @@ final class CoderAPIClient {
}
private struct TranscriptionResponse: Decodable {
struct Word: Decodable {
let word: String
let start: TimeInterval
let end: TimeInterval
let speaker: Int?
}
let text: String
let segments: [Transcription.Segment]?
let words: [Word]?
}
private struct AudioChunk {
let url: URL
let offset: TimeInterval
let isTemporary: Bool
}
private let transcriptionChunkDuration: TimeInterval = 3 * 60
private let transcriptionSession: URLSession
private init() {
@@ -151,15 +169,85 @@ final class CoderAPIClient {
}
}
func transcribe(fileURL: URL, model: String, language: String = "en") async throws -> Transcription {
func transcribe(
fileURL: URL,
model: String,
language: String = "en",
diarization: Bool = false,
maxSpeakerCount: Int = 4
) async throws -> Transcription {
let selectedModel = model.trimmingCharacters(in: .whitespacesAndNewlines)
guard !selectedModel.isEmpty else { throw CoderAPIError.missingModel("transcription") }
let apiKey = try requiredAPIKey(KeychainHelper.shared.getCoderAPIKey() ?? "")
let chunks = try makeAudioChunks(from: fileURL, preserveSpeakerIdentity: diarization)
defer {
for chunk in chunks where chunk.isTemporary {
try? FileManager.default.removeItem(at: chunk.url)
}
}
var textParts: [String] = []
var segments: [Transcription.Segment] = []
var lastNormalizedText = ""
var consecutiveDuplicateCount = 0
for chunk in chunks {
let transcription = try await transcribeChunk(
chunk.url,
model: selectedModel,
language: language,
apiKey: apiKey,
diarization: diarization,
maxSpeakerCount: maxSpeakerCount
)
if transcription.segments.isEmpty {
let text = transcription.text.trimmingCharacters(in: .whitespacesAndNewlines)
if !text.isEmpty {
textParts.append(text)
segments.append(.init(start: chunk.offset, end: chunk.offset, text: text, speaker: nil))
}
continue
}
for segment in transcription.segments {
let text = segment.text.trimmingCharacters(in: .whitespacesAndNewlines)
guard !text.isEmpty else { continue }
let normalized = text.lowercased()
if normalized == lastNormalizedText {
consecutiveDuplicateCount += 1
} else {
lastNormalizedText = normalized
consecutiveDuplicateCount = 1
}
guard consecutiveDuplicateCount <= 2 else { continue }
textParts.append(text)
segments.append(.init(
start: segment.start + chunk.offset,
end: segment.end + chunk.offset,
text: text,
speaker: segment.speaker
))
}
}
return Transcription(text: textParts.joined(separator: "\n"), segments: segments)
}
private func transcribeChunk(
_ fileURL: URL,
model: String,
language: String,
apiKey: String,
diarization: Bool,
maxSpeakerCount: Int
) async throws -> Transcription {
let boundary = "Meetingnotes-\(UUID().uuidString)"
let bodyURL = try makeMultipartBody(
audioURL: fileURL,
model: selectedModel,
model: model,
language: language,
diarization: diarization,
maxSpeakerCount: maxSpeakerCount,
boundary: boundary
)
defer { try? FileManager.default.removeItem(at: bodyURL) }
@@ -175,7 +263,132 @@ final class CoderAPIClient {
let (data, response) = try await transcriptionSession.upload(for: request, fromFile: bodyURL)
try validate(response: response, data: data)
let decoded = try JSONDecoder().decode(TranscriptionResponse.self, from: data)
return Transcription(text: decoded.text, segments: decoded.segments ?? [])
let segments = decoded.segments ?? segments(from: decoded.words ?? [])
return Transcription(text: decoded.text, segments: segments)
}
private func makeAudioChunks(from fileURL: URL, preserveSpeakerIdentity: Bool) throws -> [AudioChunk] {
let input = try AVAudioFile(forReading: fileURL)
let format = input.processingFormat
guard format.sampleRate > 0 else { throw CoderAPIError.invalidResponse }
let framesPerChunk = preserveSpeakerIdentity
? max(1, input.length)
: AVAudioFramePosition(format.sampleRate * transcriptionChunkDuration)
if fileURL.pathExtension.caseInsensitiveCompare("wav") == .orderedSame,
input.fileFormat.streamDescription.pointee.mFormatID == kAudioFormatLinearPCM,
input.length <= framesPerChunk {
return [AudioChunk(url: fileURL, offset: 0, isTemporary: false)]
}
var chunks: [AudioChunk] = []
var frameOffset: AVAudioFramePosition = 0
do {
while frameOffset < input.length {
let frameCount = min(framesPerChunk, input.length - frameOffset)
let chunkURL = FileManager.default.temporaryDirectory
.appendingPathComponent("meetingnotes-transcription-\(UUID().uuidString).wav")
try writeAudioChunk(
from: input,
frameCount: frameCount,
format: format,
to: chunkURL
)
chunks.append(AudioChunk(
url: chunkURL,
offset: Double(frameOffset) / format.sampleRate,
isTemporary: true
))
frameOffset += frameCount
}
return chunks
} catch {
for chunk in chunks {
try? FileManager.default.removeItem(at: chunk.url)
}
throw error
}
}
private func writeAudioChunk(
from input: AVAudioFile,
frameCount: AVAudioFramePosition,
format: AVAudioFormat,
to outputURL: URL
) throws {
let settings: [String: Any] = [
AVFormatIDKey: kAudioFormatLinearPCM,
AVSampleRateKey: format.sampleRate,
AVNumberOfChannelsKey: format.channelCount,
AVLinearPCMBitDepthKey: 16,
AVLinearPCMIsFloatKey: false,
AVLinearPCMIsBigEndianKey: false,
AVLinearPCMIsNonInterleaved: false
]
let output = try AVAudioFile(
forWriting: outputURL,
settings: settings,
commonFormat: format.commonFormat,
interleaved: format.isInterleaved
)
var remaining = frameCount
while remaining > 0 {
let requestedFrames = AVAudioFrameCount(min(remaining, 8_192))
guard let buffer = AVAudioPCMBuffer(pcmFormat: format, frameCapacity: requestedFrames) else {
throw CoderAPIError.invalidResponse
}
try input.read(into: buffer, frameCount: requestedFrames)
guard buffer.frameLength > 0 else { break }
try output.write(from: buffer)
remaining -= AVAudioFramePosition(buffer.frameLength)
}
}
private func segments(from words: [TranscriptionResponse.Word]) -> [Transcription.Segment] {
var result: [Transcription.Segment] = []
var currentWords: [String] = []
var currentStart: TimeInterval?
var currentEnd: TimeInterval = 0
var currentSpeaker: Int?
func flush() {
guard let start = currentStart, !currentWords.isEmpty else { return }
result.append(.init(
start: start,
end: currentEnd,
text: currentWords.joined(separator: " "),
speaker: currentSpeaker
))
currentWords.removeAll(keepingCapacity: true)
currentStart = nil
currentEnd = 0
currentSpeaker = nil
}
for word in words {
let text = word.word.trimmingCharacters(in: .whitespacesAndNewlines)
guard !text.isEmpty else { continue }
let speakerChanged = currentStart != nil && word.speaker != currentSpeaker
let longPause = currentStart != nil && word.start - currentEnd > 1.5
if speakerChanged || longPause { flush() }
if currentStart == nil {
currentStart = word.start
currentSpeaker = word.speaker
}
currentWords.append(text)
currentEnd = word.end
let sentenceEnded = text.last.map { ".!?".contains($0) } ?? false
let duration = currentEnd - (currentStart ?? currentEnd)
if currentWords.count >= 40 || (sentenceEnded && (currentWords.count >= 12 || duration >= 8)) {
flush()
}
}
flush()
return result
}
private func endpoint(baseURL: String, path: String) throws -> URL {
@@ -207,7 +420,14 @@ final class CoderAPIClient {
}
}
private func makeMultipartBody(audioURL: URL, model: String, language: String, boundary: String) throws -> URL {
private func makeMultipartBody(
audioURL: URL,
model: String,
language: String,
diarization: Bool,
maxSpeakerCount: Int,
boundary: String
) throws -> URL {
let bodyURL = FileManager.default.temporaryDirectory.appendingPathComponent("meetingnotes-upload-\(UUID().uuidString).body")
_ = FileManager.default.createFile(atPath: bodyURL.path, contents: nil)
let output = try FileHandle(forWritingTo: bodyURL)
@@ -219,7 +439,11 @@ final class CoderAPIClient {
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"model\"\r\n\r\n\(model)\r\n")
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n\(language)\r\n")
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"response_format\"\r\n\r\nverbose_json\r\n")
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"file\"; filename=\"\(audioURL.lastPathComponent)\"\r\nContent-Type: audio/mp4\r\n\r\n")
if diarization {
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"diarization\"\r\n\r\ntrue\r\n")
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"max_speaker_count\"\r\n\r\n\(maxSpeakerCount)\r\n")
}
try write("--\(boundary)\r\nContent-Disposition: form-data; name=\"file\"; filename=\"\(audioURL.lastPathComponent)\"\r\nContent-Type: audio/wav\r\n\r\n")
let input = try FileHandle(forReadingFrom: audioURL)
defer { try? input.close() }
while let chunk = try input.read(upToCount: 1 << 20), !chunk.isEmpty {
@@ -254,9 +254,7 @@ class MeetingViewModel: ObservableObject {
private func refreshRecoveryAudioFolder() {
recoveryAudioFolderURL = LocalStorageManager.shared.findRecoveryAudioFolder(for: meeting)
if let recoveryAudioFolderURL {
meeting.recoveryAudioFolderName = recoveryAudioFolderURL.lastPathComponent
}
meeting.recoveryAudioFolderName = recoveryAudioFolderURL?.lastPathComponent
}
func showAudioInFinder() {
+2 -2
View File
@@ -208,12 +208,12 @@ struct CollapsedTranscriptChunkView: View {
.font(.caption)
.foregroundColor(chunk.source == .mic ? .blue : .orange)
Text(chunk.source.displayName)
Text(chunk.displayName)
.font(.caption)
.fontWeight(.medium)
.foregroundColor(chunk.source == .mic ? .blue : .orange)
}
.frame(width: 50, alignment: .leading)
.frame(width: 78, alignment: .leading)
// Transcript text
Text(chunk.combinedText)
+12
View File
@@ -58,6 +58,18 @@ struct SettingsView: View {
Text(model.displayName).tag(model.id)
}
}
if let selectedModel = viewModel.coderModels.first(where: { $0.id == viewModel.settings.transcriptionModel }) {
if selectedModel.supportsSpeakerDiarization {
Label("Remote participants are labeled Speaker 14; your microphone is labeled Me.", systemImage: "person.2.wave.2")
.font(.caption)
.foregroundColor(.secondary)
} else {
Label("Remote participants are labeled Them; your microphone is labeled Me.", systemImage: "person.2")
.font(.caption)
.foregroundColor(.secondary)
}
}
}
Text("The token is stored locally in Keychain. Audio and note generation are sent only to this Coder service.")
+32 -2
View File
@@ -14,6 +14,9 @@ APP_PATH="$DERIVED_DATA/Build/Products/Release/$APP_NAME.app"
required_variables=(
VERSION
SIGNING_IDENTITY
APPLE_ID
APPLE_TEAM_ID
APPLE_APP_PASSWORD
SPARKLE_PRIVATE_KEY
GITHUB_REPOSITORY
)
@@ -80,6 +83,33 @@ ARCHIVE_NAME="$APP_NAME-$VERSION.zip"
ARCHIVE_PATH="$RELEASE_DIR/$ARCHIVE_NAME"
ditto -c -k --sequesterRsrc --keepParent "$APP_PATH" "$ARCHIVE_PATH"
NOTARY_RESULT="$BUILD_ROOT/notary-result.json"
xcrun notarytool submit "$ARCHIVE_PATH" \
--apple-id "$APPLE_ID" \
--team-id "$APPLE_TEAM_ID" \
--password "$APPLE_APP_PASSWORD" \
--wait \
--timeout 20m \
--output-format json > "$NOTARY_RESULT"
NOTARY_STATUS=$(plutil -extract status raw -o - "$NOTARY_RESULT")
if [[ "$NOTARY_STATUS" != "Accepted" ]]; then
submission_id=$(plutil -extract id raw -o - "$NOTARY_RESULT")
xcrun notarytool log "$submission_id" \
--apple-id "$APPLE_ID" \
--team-id "$APPLE_TEAM_ID" \
--password "$APPLE_APP_PASSWORD" || true
echo "Apple notarization failed with status: $NOTARY_STATUS" >&2
exit 1
fi
xcrun stapler staple "$APP_PATH"
xcrun stapler validate "$APP_PATH"
spctl --assess --type execute --verbose=2 "$APP_PATH"
rm -f "$ARCHIVE_PATH"
ditto -c -k --sequesterRsrc --keepParent "$APP_PATH" "$ARCHIVE_PATH"
GENERATE_APPCAST=$(find "$DERIVED_DATA/SourcePackages/artifacts" -type f -name generate_appcast -print -quit)
if [[ -z "$GENERATE_APPCAST" ]]; then
echo "Sparkle generate_appcast tool was not found" >&2
@@ -97,8 +127,8 @@ grep -q "$DOWNLOAD_URL$ARCHIVE_NAME" "$RELEASE_DIR/appcast.xml"
grep -q 'sparkle:edSignature=' "$RELEASE_DIR/appcast.xml"
if [[ -n "${GITHUB_STEP_SUMMARY:-}" ]]; then
printf 'Built and Developer ID-signed Meetingnotes %s. The GitHub release is ready to publish.\n' \
printf 'Built, Developer ID-signed, notarized, and stapled Meetingnotes %s. The GitHub release is ready to publish.\n' \
"$VERSION" >> "$GITHUB_STEP_SUMMARY"
fi
echo "Signed release artifacts are ready in $RELEASE_DIR"
echo "Signed and notarized release artifacts are ready in $RELEASE_DIR"