Initial commit of Edison Voice menu-bar dictation app

This commit is contained in:
2026-09-09 23:49:51 +02:00
commit ad02cfb638
40 changed files with 3305 additions and 0 deletions
@@ -0,0 +1,175 @@
import AVFoundation
import Foundation
import Speech
/// Streaming on-device transcription via macOS 26's `SpeechAnalyzer` / `SpeechTranscriber`.
///
/// No model ships with the app — the OS downloads and manages the assets, so the first
/// run for a given locale may block briefly while `AssetInstallationRequest` completes.
public actor AppleSpeechEngine: TranscriptionEngine {
private let locale: Locale
private var transcriber: SpeechTranscriber?
private var analyzer: SpeechAnalyzer?
private var inputContinuation: AsyncStream<AnalyzerInput>.Continuation?
private var resultsTask: Task<Void, Never>?
/// Text the engine has committed. Volatile results are appended on top for display
/// but discarded as soon as a final result covering the same range arrives.
private var finalizedText = ""
public init(locale: Locale = Locale.current) {
self.locale = locale
}
public func preferredInputFormat() async -> AVAudioFormat? {
let module = transcriber ?? Self.makeTranscriber(locale: locale)
return await SpeechAnalyzer.bestAvailableAudioFormat(compatibleWith: [module])
}
public func start() async throws -> AsyncThrowingStream<TranscriptionChunk, Error> {
guard SpeechTranscriber.isAvailable else {
throw TranscriptionError.localeUnsupported(locale)
}
let resolvedLocale = await SpeechTranscriber.supportedLocale(equivalentTo: locale)
?? Locale(identifier: "en-US")
let transcriber = Self.makeTranscriber(locale: resolvedLocale)
self.transcriber = transcriber
try await Self.ensureModelInstalled(for: transcriber)
let (inputStream, inputContinuation) = AsyncStream<AnalyzerInput>.makeStream()
self.inputContinuation = inputContinuation
// Bias the recognizer toward the dictionary's words before it hears anything. This
// is a nudge, not a guarantee — `DictionaryCorrector` is the pass that actually
// enforces spelling — but it's free and it catches things a post-hoc rewrite can't,
// like a name the engine would otherwise split into two ordinary words.
//
// The list is capped at `DictionaryCorrector.biasLimit`. A long context list makes
// these models drift: on quiet or ambiguous audio they start emitting the terms they
// were primed with, which is a far worse failure than the misspelling it prevents.
// Only the input-sequence initializers take a context up front, and this analyzer is
// fed by `analyzer.start(inputSequence:)` later — so the context is applied here
// instead. It must be set before any audio arrives to affect recognition.
let analyzer = SpeechAnalyzer(modules: [transcriber])
self.analyzer = analyzer
if let context = await Self.context() {
try? await analyzer.setContext(context)
}
finalizedText = ""
let (chunks, chunkContinuation) = AsyncThrowingStream<TranscriptionChunk, Error>.makeStream()
// Drain the transcriber's results into our simpler chunk stream.
resultsTask = Task { [weak self] in
do {
for try await result in transcriber.results {
guard let self else { break }
let snapshot = await self.absorb(result)
chunkContinuation.yield(TranscriptionChunk(text: snapshot, isFinal: false))
}
let final = await self?.finalizedText ?? ""
chunkContinuation.yield(TranscriptionChunk(text: final, isFinal: true))
chunkContinuation.finish()
} catch {
Log.speech.error("results stream failed: \(error.localizedDescription)")
chunkContinuation.finish(throwing: error)
}
}
try await analyzer.start(inputSequence: inputStream)
Log.speech.info("SpeechAnalyzer started for \(resolvedLocale.identifier)")
return chunks
}
public func feed(_ chunk: AudioChunk) async {
inputContinuation?.yield(AnalyzerInput(buffer: chunk.buffer))
}
public func finish() async {
inputContinuation?.finish()
inputContinuation = nil
do {
try await analyzer?.finalizeAndFinishThroughEndOfInput()
} catch {
Log.speech.error("finalize failed: \(error.localizedDescription)")
await analyzer?.cancelAndFinishNow()
}
analyzer = nil
transcriber = nil
resultsTask = nil
}
// MARK: - Result accumulation
/// Folds one result into the running transcript and returns the full text to display.
///
/// Final results are committed; a volatile result is shown appended to the committed
/// text but never stored, so the next revision replaces it cleanly.
private func absorb(_ result: SpeechTranscriber.Result) -> String {
let text = String(result.text.characters)
guard result.isFinal else {
return (finalizedText + text).trimmingCharacters(in: .whitespaces)
}
finalizedText += text
return finalizedText.trimmingCharacters(in: .whitespaces)
}
// MARK: - Setup helpers
/// The dictionary's words, handed to the analyzer as contextual strings.
///
/// Reads the store on the main actor because that's where it lives; the resulting array
/// of strings is plain value data and crosses back safely.
/// - Returns: nil when the dictionary is empty, so an empty context is never set for
/// nothing.
///
/// Hops to the main actor rather than asserting it. The store is main-actor isolated and
/// this runs on the engine's own executor — `MainActor.assumeIsolated` here doesn't check
/// that claim, it asserts it, and takes the whole process down when it's false.
private static func context() async -> AnalysisContext? {
let phrases = await MainActor.run { DictionaryStore.shared.biasPhrases }
guard !phrases.isEmpty else { return nil }
let context = AnalysisContext()
context.contextualStrings[.general] = phrases
Log.speech.info("biasing with \(phrases.count, privacy: .public) dictionary phrase(s)")
return context
}
private static func makeTranscriber(locale: Locale) -> SpeechTranscriber {
SpeechTranscriber(
locale: locale,
transcriptionOptions: [],
// `.volatileResults` is what makes live text appear while you're still talking.
reportingOptions: [.volatileResults],
attributeOptions: []
)
}
private static func ensureModelInstalled(for transcriber: SpeechTranscriber) async throws {
let installed = await SpeechTranscriber.installedLocales
let selected = transcriber.selectedLocales
let alreadyThere = selected.allSatisfy { locale in
installed.contains { $0.identifier(.bcp47) == locale.identifier(.bcp47) }
}
guard !alreadyThere else { return }
do {
if let request = try await AssetInventory.assetInstallationRequest(supporting: [transcriber]) {
Log.speech.info("downloading speech model…")
try await request.downloadAndInstall()
Log.speech.info("speech model installed")
}
} catch {
throw TranscriptionError.modelInstallFailed(error.localizedDescription)
}
}
}
@@ -0,0 +1,162 @@
import AVFoundation
import FluidAudio
import Foundation
/// NVIDIA Parakeet TDT 0.6B, compiled to CoreML and run on the Neural Engine via FluidAudio.
///
/// **Batch, not streaming.** Audio is accumulated while the key is held and transcribed in
/// one pass on release. That's a deliberate trade: at ~100× realtime a 30-second utterance
/// resolves in roughly a third of a second, which is imperceptible for push-to-talk — but
/// it means no live text in the HUD while you speak, unlike Apple's engine.
/// FluidAudio's `SlidingWindowAsrManager` would restore live partials at the cost of a
/// more complex integration; see the note in `docs`.
public actor ParakeetEngine: TranscriptionEngine {
private var samples: [Float] = []
private var continuation: AsyncThrowingStream<TranscriptionChunk, Error>.Continuation?
/// Defaults to 16 kHz mono float32 — exactly what Parakeet is trained on.
private let converter = AudioConverter()
public init() {}
public func preferredInputFormat() async -> AVAudioFormat? {
// Parakeet is trained on 16 kHz mono; AudioCapture converts to whatever we ask for.
AVAudioFormat(commonFormat: .pcmFormatFloat32, sampleRate: 16_000, channels: 1, interleaved: false)
}
public func start() async throws -> AsyncThrowingStream<TranscriptionChunk, Error> {
samples.removeAll(keepingCapacity: true)
let (stream, continuation) = AsyncThrowingStream<TranscriptionChunk, Error>.makeStream()
self.continuation = continuation
// Force the (possibly very slow) first load to happen here rather than on release,
// so the user waits before speaking instead of losing an utterance to a timeout.
_ = try await ParakeetModels.shared.manager()
return stream
}
public func feed(_ chunk: AudioChunk) async {
let buffer = chunk.buffer
guard buffer.frameLength > 0 else { return }
// Delegated to FluidAudio's own converter rather than hand-rolled, for one reason
// that matters more than tidiness: `AsrManager.transcribe(_ samples: [Float])`
// performs **no resampling and no rate validation**. Feed it the wrong sample rate
// and it doesn't throw — it silently transcribes garbage.
//
// That's a live risk here. In compare mode the capture format is dictated by
// Apple's analyzer, and `bestAvailableAudioFormat` may legitimately return 8 kHz
// as well as 16 kHz. `resampleBuffer` normalizes whatever arrives to the 16 kHz
// mono float32 the model expects, and its Int16→Float path is bit-identical to
// dividing by 32768, so nothing is lost versus doing it by hand.
do {
samples.append(contentsOf: try converter.resampleBuffer(buffer))
} catch {
Log.speech.error("Parakeet: audio conversion failed — \(error.localizedDescription)")
}
}
public func finish() async {
defer {
continuation?.finish()
continuation = nil
samples.removeAll(keepingCapacity: true)
}
// Parakeet's encoder needs a minimum window; a stray tap of the key isn't speech.
// Logged rather than silent — an unexpected drop to zero here is how the
// format bug above disguised itself as a fast, empty result.
guard samples.count >= 1_600 else {
Log.speech.info("Parakeet: skipped — only \(self.samples.count) samples captured")
return
}
do {
let manager = try await ParakeetModels.shared.manager()
var decoderState = try TdtDecoderState()
let started = Date()
let result = try await manager.transcribe(samples, decoderState: &decoderState)
let elapsed = Date().timeIntervalSince(started)
let audioSeconds = Double(samples.count) / 16_000
Log.speech.info("""
Parakeet: \(audioSeconds, format: .fixed(precision: 1))s audio in \
\(elapsed, format: .fixed(precision: 2))s (\(audioSeconds / max(elapsed, 0.0001), format: .fixed(precision: 0))× realtime)
""")
continuation?.yield(
TranscriptionChunk(
text: result.text.trimmingCharacters(in: .whitespacesAndNewlines),
isFinal: true
)
)
} catch {
Log.speech.error("Parakeet failed: \(error.localizedDescription)")
continuation?.finish(throwing: error)
continuation = nil
}
}
}
/// Process-wide model cache.
///
/// Loading is expensive — ~470 MB downloaded on first ever run, then a few seconds from
/// disk per process — and the models are immutable once loaded, so every dictation shares
/// one instance rather than paying that per utterance. Its own actor because `static var`
/// on `ParakeetEngine` would be unprotected global mutable state under Swift 6.
public actor ParakeetModels {
public static let shared = ParakeetModels()
/// Whether the models are already on disk, checked without loading them.
///
/// `nonisolated` and filesystem-based on purpose: the menu needs this synchronously
/// while drawing, and an in-memory "have I loaded yet" flag would wrongly report
/// "not downloaded" on every fresh launch.
nonisolated public static var isDownloaded: Bool {
let support = FileManager.default.urls(for: .applicationSupportDirectory, in: .userDomainMask)[0]
let encoder = support
.appendingPathComponent("FluidAudio/Models/parakeet-tdt-0.6b-v3/Encoder.mlmodelc")
return FileManager.default.fileExists(atPath: encoder.path)
}
private var loaded: AsrManager?
private var loadTask: Task<AsrManager, Error>?
var isLoaded: Bool { loaded != nil }
/// Loads once; concurrent callers await the same task rather than racing to download.
public func manager() async throws -> AsrManager {
if let loaded { return loaded }
if let loadTask { return try await loadTask.value }
let task = Task<AsrManager, Error> {
// Built as a value first: os.Logger requires a literal interpolation, so a
// ternary can't be passed directly as the argument.
let stage = Self.isDownloaded
? "loading models from disk"
: "downloading models (~470 MB, one time)"
Log.speech.info("Parakeet: \(stage, privacy: .public)")
let started = Date()
let models = try await AsrModels.downloadAndLoad(version: .v3, encoderPrecision: .int8)
let manager = AsrManager(config: .default)
try await manager.loadModels(models)
Log.speech.info("Parakeet: ready in \(Date().timeIntervalSince(started), format: .fixed(precision: 1))s")
return manager
}
loadTask = task
do {
let manager = try await task.value
loaded = manager
return manager
} catch {
// Don't cache a failed load — a transient download error shouldn't wedge the
// engine for the rest of the session.
loadTask = nil
throw error
}
}
}
@@ -0,0 +1,65 @@
import AVFoundation
import Foundation
/// One buffer of captured audio, in transit from the audio thread to the speech engine.
///
/// `AVAudioPCMBuffer` isn't `Sendable`, and `AVAudioEngine` recycles the buffer it hands
/// to a tap the moment the callback returns. The unchecked conformance is only sound
/// because `AudioCapture` allocates a **fresh** buffer for every chunk and never touches
/// it again after handing it over — don't construct one of these around a borrowed buffer.
public struct AudioChunk: @unchecked Sendable {
public let buffer: AVAudioPCMBuffer
public init(buffer: AVAudioPCMBuffer) {
self.buffer = buffer
}
}
/// A snapshot of the running transcript.
///
/// `text` is always the **full transcript so far**, not a delta — engines revise
/// earlier words as more audio arrives, so consumers should replace rather than append.
public struct TranscriptionChunk: Sendable {
public let text: String
/// `true` once the engine has committed everything it will emit for this session.
public let isFinal: Bool
}
/// The seam that keeps the app engine-agnostic.
///
/// Apple's `SpeechAnalyzer` ships with macOS 26 and needs no model download, so it is
/// the default. Parakeet (FluidAudio, CoreML/ANE) scores better on English and is the
/// intended upgrade — implementing this protocol is the whole cost of switching.
public protocol TranscriptionEngine: Actor {
/// Audio format the engine wants buffers delivered in. `AudioCapture` converts to it.
func preferredInputFormat() async -> AVAudioFormat?
/// Prepare models and open a session. Emits snapshots until `finish()` is called.
func start() async throws -> AsyncThrowingStream<TranscriptionChunk, Error>
/// Feed one buffer of captured microphone audio, already in `preferredInputFormat()`.
func feed(_ chunk: AudioChunk) async
/// Close the session and flush any pending final results.
func finish() async
}
public enum TranscriptionError: LocalizedError {
case localeUnsupported(Locale)
case modelInstallFailed(String)
case noAudioFormat
case notRunning
public var errorDescription: String? {
switch self {
case .localeUnsupported(let locale):
return "Dictation isn't available for \(locale.identifier) on this Mac."
case .modelInstallFailed(let detail):
return "Couldn't install the speech model: \(detail)"
case .noAudioFormat:
return "No compatible audio format available for the speech engine."
case .notRunning:
return "The transcription engine isn't running."
}
}
}