Initial commit of Edison Voice menu-bar dictation app
This commit is contained in:
@@ -0,0 +1,175 @@
|
||||
import AVFoundation
|
||||
import Foundation
|
||||
import Speech
|
||||
|
||||
/// Streaming on-device transcription via macOS 26's `SpeechAnalyzer` / `SpeechTranscriber`.
|
||||
///
|
||||
/// No model ships with the app — the OS downloads and manages the assets, so the first
|
||||
/// run for a given locale may block briefly while `AssetInstallationRequest` completes.
|
||||
public actor AppleSpeechEngine: TranscriptionEngine {
|
||||
private let locale: Locale
|
||||
|
||||
private var transcriber: SpeechTranscriber?
|
||||
private var analyzer: SpeechAnalyzer?
|
||||
private var inputContinuation: AsyncStream<AnalyzerInput>.Continuation?
|
||||
private var resultsTask: Task<Void, Never>?
|
||||
|
||||
/// Text the engine has committed. Volatile results are appended on top for display
|
||||
/// but discarded as soon as a final result covering the same range arrives.
|
||||
private var finalizedText = ""
|
||||
|
||||
public init(locale: Locale = Locale.current) {
|
||||
self.locale = locale
|
||||
}
|
||||
|
||||
public func preferredInputFormat() async -> AVAudioFormat? {
|
||||
let module = transcriber ?? Self.makeTranscriber(locale: locale)
|
||||
return await SpeechAnalyzer.bestAvailableAudioFormat(compatibleWith: [module])
|
||||
}
|
||||
|
||||
public func start() async throws -> AsyncThrowingStream<TranscriptionChunk, Error> {
|
||||
guard SpeechTranscriber.isAvailable else {
|
||||
throw TranscriptionError.localeUnsupported(locale)
|
||||
}
|
||||
|
||||
let resolvedLocale = await SpeechTranscriber.supportedLocale(equivalentTo: locale)
|
||||
?? Locale(identifier: "en-US")
|
||||
|
||||
let transcriber = Self.makeTranscriber(locale: resolvedLocale)
|
||||
self.transcriber = transcriber
|
||||
|
||||
try await Self.ensureModelInstalled(for: transcriber)
|
||||
|
||||
let (inputStream, inputContinuation) = AsyncStream<AnalyzerInput>.makeStream()
|
||||
self.inputContinuation = inputContinuation
|
||||
|
||||
// Bias the recognizer toward the dictionary's words before it hears anything. This
|
||||
// is a nudge, not a guarantee — `DictionaryCorrector` is the pass that actually
|
||||
// enforces spelling — but it's free and it catches things a post-hoc rewrite can't,
|
||||
// like a name the engine would otherwise split into two ordinary words.
|
||||
//
|
||||
// The list is capped at `DictionaryCorrector.biasLimit`. A long context list makes
|
||||
// these models drift: on quiet or ambiguous audio they start emitting the terms they
|
||||
// were primed with, which is a far worse failure than the misspelling it prevents.
|
||||
// Only the input-sequence initializers take a context up front, and this analyzer is
|
||||
// fed by `analyzer.start(inputSequence:)` later — so the context is applied here
|
||||
// instead. It must be set before any audio arrives to affect recognition.
|
||||
let analyzer = SpeechAnalyzer(modules: [transcriber])
|
||||
self.analyzer = analyzer
|
||||
if let context = await Self.context() {
|
||||
try? await analyzer.setContext(context)
|
||||
}
|
||||
|
||||
finalizedText = ""
|
||||
|
||||
let (chunks, chunkContinuation) = AsyncThrowingStream<TranscriptionChunk, Error>.makeStream()
|
||||
|
||||
// Drain the transcriber's results into our simpler chunk stream.
|
||||
resultsTask = Task { [weak self] in
|
||||
do {
|
||||
for try await result in transcriber.results {
|
||||
guard let self else { break }
|
||||
let snapshot = await self.absorb(result)
|
||||
chunkContinuation.yield(TranscriptionChunk(text: snapshot, isFinal: false))
|
||||
}
|
||||
let final = await self?.finalizedText ?? ""
|
||||
chunkContinuation.yield(TranscriptionChunk(text: final, isFinal: true))
|
||||
chunkContinuation.finish()
|
||||
} catch {
|
||||
Log.speech.error("results stream failed: \(error.localizedDescription)")
|
||||
chunkContinuation.finish(throwing: error)
|
||||
}
|
||||
}
|
||||
|
||||
try await analyzer.start(inputSequence: inputStream)
|
||||
Log.speech.info("SpeechAnalyzer started for \(resolvedLocale.identifier)")
|
||||
|
||||
return chunks
|
||||
}
|
||||
|
||||
public func feed(_ chunk: AudioChunk) async {
|
||||
inputContinuation?.yield(AnalyzerInput(buffer: chunk.buffer))
|
||||
}
|
||||
|
||||
public func finish() async {
|
||||
inputContinuation?.finish()
|
||||
inputContinuation = nil
|
||||
|
||||
do {
|
||||
try await analyzer?.finalizeAndFinishThroughEndOfInput()
|
||||
} catch {
|
||||
Log.speech.error("finalize failed: \(error.localizedDescription)")
|
||||
await analyzer?.cancelAndFinishNow()
|
||||
}
|
||||
|
||||
analyzer = nil
|
||||
transcriber = nil
|
||||
resultsTask = nil
|
||||
}
|
||||
|
||||
// MARK: - Result accumulation
|
||||
|
||||
/// Folds one result into the running transcript and returns the full text to display.
|
||||
///
|
||||
/// Final results are committed; a volatile result is shown appended to the committed
|
||||
/// text but never stored, so the next revision replaces it cleanly.
|
||||
private func absorb(_ result: SpeechTranscriber.Result) -> String {
|
||||
let text = String(result.text.characters)
|
||||
guard result.isFinal else {
|
||||
return (finalizedText + text).trimmingCharacters(in: .whitespaces)
|
||||
}
|
||||
finalizedText += text
|
||||
return finalizedText.trimmingCharacters(in: .whitespaces)
|
||||
}
|
||||
|
||||
// MARK: - Setup helpers
|
||||
|
||||
/// The dictionary's words, handed to the analyzer as contextual strings.
|
||||
///
|
||||
/// Reads the store on the main actor because that's where it lives; the resulting array
|
||||
/// of strings is plain value data and crosses back safely.
|
||||
/// - Returns: nil when the dictionary is empty, so an empty context is never set for
|
||||
/// nothing.
|
||||
///
|
||||
/// Hops to the main actor rather than asserting it. The store is main-actor isolated and
|
||||
/// this runs on the engine's own executor — `MainActor.assumeIsolated` here doesn't check
|
||||
/// that claim, it asserts it, and takes the whole process down when it's false.
|
||||
private static func context() async -> AnalysisContext? {
|
||||
let phrases = await MainActor.run { DictionaryStore.shared.biasPhrases }
|
||||
guard !phrases.isEmpty else { return nil }
|
||||
|
||||
let context = AnalysisContext()
|
||||
context.contextualStrings[.general] = phrases
|
||||
Log.speech.info("biasing with \(phrases.count, privacy: .public) dictionary phrase(s)")
|
||||
return context
|
||||
}
|
||||
|
||||
private static func makeTranscriber(locale: Locale) -> SpeechTranscriber {
|
||||
SpeechTranscriber(
|
||||
locale: locale,
|
||||
transcriptionOptions: [],
|
||||
// `.volatileResults` is what makes live text appear while you're still talking.
|
||||
reportingOptions: [.volatileResults],
|
||||
attributeOptions: []
|
||||
)
|
||||
}
|
||||
|
||||
private static func ensureModelInstalled(for transcriber: SpeechTranscriber) async throws {
|
||||
let installed = await SpeechTranscriber.installedLocales
|
||||
let selected = transcriber.selectedLocales
|
||||
let alreadyThere = selected.allSatisfy { locale in
|
||||
installed.contains { $0.identifier(.bcp47) == locale.identifier(.bcp47) }
|
||||
}
|
||||
guard !alreadyThere else { return }
|
||||
|
||||
do {
|
||||
if let request = try await AssetInventory.assetInstallationRequest(supporting: [transcriber]) {
|
||||
Log.speech.info("downloading speech model…")
|
||||
try await request.downloadAndInstall()
|
||||
Log.speech.info("speech model installed")
|
||||
}
|
||||
} catch {
|
||||
throw TranscriptionError.modelInstallFailed(error.localizedDescription)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
import AVFoundation
|
||||
import FluidAudio
|
||||
import Foundation
|
||||
|
||||
/// NVIDIA Parakeet TDT 0.6B, compiled to CoreML and run on the Neural Engine via FluidAudio.
|
||||
///
|
||||
/// **Batch, not streaming.** Audio is accumulated while the key is held and transcribed in
|
||||
/// one pass on release. That's a deliberate trade: at ~100× realtime a 30-second utterance
|
||||
/// resolves in roughly a third of a second, which is imperceptible for push-to-talk — but
|
||||
/// it means no live text in the HUD while you speak, unlike Apple's engine.
|
||||
/// FluidAudio's `SlidingWindowAsrManager` would restore live partials at the cost of a
|
||||
/// more complex integration; see the note in `docs`.
|
||||
public actor ParakeetEngine: TranscriptionEngine {
|
||||
private var samples: [Float] = []
|
||||
private var continuation: AsyncThrowingStream<TranscriptionChunk, Error>.Continuation?
|
||||
|
||||
/// Defaults to 16 kHz mono float32 — exactly what Parakeet is trained on.
|
||||
private let converter = AudioConverter()
|
||||
|
||||
public init() {}
|
||||
|
||||
public func preferredInputFormat() async -> AVAudioFormat? {
|
||||
// Parakeet is trained on 16 kHz mono; AudioCapture converts to whatever we ask for.
|
||||
AVAudioFormat(commonFormat: .pcmFormatFloat32, sampleRate: 16_000, channels: 1, interleaved: false)
|
||||
}
|
||||
|
||||
public func start() async throws -> AsyncThrowingStream<TranscriptionChunk, Error> {
|
||||
samples.removeAll(keepingCapacity: true)
|
||||
|
||||
let (stream, continuation) = AsyncThrowingStream<TranscriptionChunk, Error>.makeStream()
|
||||
self.continuation = continuation
|
||||
|
||||
// Force the (possibly very slow) first load to happen here rather than on release,
|
||||
// so the user waits before speaking instead of losing an utterance to a timeout.
|
||||
_ = try await ParakeetModels.shared.manager()
|
||||
|
||||
return stream
|
||||
}
|
||||
|
||||
public func feed(_ chunk: AudioChunk) async {
|
||||
let buffer = chunk.buffer
|
||||
guard buffer.frameLength > 0 else { return }
|
||||
|
||||
// Delegated to FluidAudio's own converter rather than hand-rolled, for one reason
|
||||
// that matters more than tidiness: `AsrManager.transcribe(_ samples: [Float])`
|
||||
// performs **no resampling and no rate validation**. Feed it the wrong sample rate
|
||||
// and it doesn't throw — it silently transcribes garbage.
|
||||
//
|
||||
// That's a live risk here. In compare mode the capture format is dictated by
|
||||
// Apple's analyzer, and `bestAvailableAudioFormat` may legitimately return 8 kHz
|
||||
// as well as 16 kHz. `resampleBuffer` normalizes whatever arrives to the 16 kHz
|
||||
// mono float32 the model expects, and its Int16→Float path is bit-identical to
|
||||
// dividing by 32768, so nothing is lost versus doing it by hand.
|
||||
do {
|
||||
samples.append(contentsOf: try converter.resampleBuffer(buffer))
|
||||
} catch {
|
||||
Log.speech.error("Parakeet: audio conversion failed — \(error.localizedDescription)")
|
||||
}
|
||||
}
|
||||
|
||||
public func finish() async {
|
||||
defer {
|
||||
continuation?.finish()
|
||||
continuation = nil
|
||||
samples.removeAll(keepingCapacity: true)
|
||||
}
|
||||
|
||||
// Parakeet's encoder needs a minimum window; a stray tap of the key isn't speech.
|
||||
// Logged rather than silent — an unexpected drop to zero here is how the
|
||||
// format bug above disguised itself as a fast, empty result.
|
||||
guard samples.count >= 1_600 else {
|
||||
Log.speech.info("Parakeet: skipped — only \(self.samples.count) samples captured")
|
||||
return
|
||||
}
|
||||
|
||||
do {
|
||||
let manager = try await ParakeetModels.shared.manager()
|
||||
var decoderState = try TdtDecoderState()
|
||||
let started = Date()
|
||||
let result = try await manager.transcribe(samples, decoderState: &decoderState)
|
||||
let elapsed = Date().timeIntervalSince(started)
|
||||
let audioSeconds = Double(samples.count) / 16_000
|
||||
|
||||
Log.speech.info("""
|
||||
Parakeet: \(audioSeconds, format: .fixed(precision: 1))s audio in \
|
||||
\(elapsed, format: .fixed(precision: 2))s (\(audioSeconds / max(elapsed, 0.0001), format: .fixed(precision: 0))× realtime)
|
||||
""")
|
||||
|
||||
continuation?.yield(
|
||||
TranscriptionChunk(
|
||||
text: result.text.trimmingCharacters(in: .whitespacesAndNewlines),
|
||||
isFinal: true
|
||||
)
|
||||
)
|
||||
} catch {
|
||||
Log.speech.error("Parakeet failed: \(error.localizedDescription)")
|
||||
continuation?.finish(throwing: error)
|
||||
continuation = nil
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
/// Process-wide model cache.
|
||||
///
|
||||
/// Loading is expensive — ~470 MB downloaded on first ever run, then a few seconds from
|
||||
/// disk per process — and the models are immutable once loaded, so every dictation shares
|
||||
/// one instance rather than paying that per utterance. Its own actor because `static var`
|
||||
/// on `ParakeetEngine` would be unprotected global mutable state under Swift 6.
|
||||
public actor ParakeetModels {
|
||||
public static let shared = ParakeetModels()
|
||||
|
||||
/// Whether the models are already on disk, checked without loading them.
|
||||
///
|
||||
/// `nonisolated` and filesystem-based on purpose: the menu needs this synchronously
|
||||
/// while drawing, and an in-memory "have I loaded yet" flag would wrongly report
|
||||
/// "not downloaded" on every fresh launch.
|
||||
nonisolated public static var isDownloaded: Bool {
|
||||
let support = FileManager.default.urls(for: .applicationSupportDirectory, in: .userDomainMask)[0]
|
||||
let encoder = support
|
||||
.appendingPathComponent("FluidAudio/Models/parakeet-tdt-0.6b-v3/Encoder.mlmodelc")
|
||||
return FileManager.default.fileExists(atPath: encoder.path)
|
||||
}
|
||||
|
||||
private var loaded: AsrManager?
|
||||
private var loadTask: Task<AsrManager, Error>?
|
||||
|
||||
var isLoaded: Bool { loaded != nil }
|
||||
|
||||
/// Loads once; concurrent callers await the same task rather than racing to download.
|
||||
public func manager() async throws -> AsrManager {
|
||||
if let loaded { return loaded }
|
||||
if let loadTask { return try await loadTask.value }
|
||||
|
||||
let task = Task<AsrManager, Error> {
|
||||
// Built as a value first: os.Logger requires a literal interpolation, so a
|
||||
// ternary can't be passed directly as the argument.
|
||||
let stage = Self.isDownloaded
|
||||
? "loading models from disk"
|
||||
: "downloading models (~470 MB, one time)"
|
||||
Log.speech.info("Parakeet: \(stage, privacy: .public)")
|
||||
let started = Date()
|
||||
let models = try await AsrModels.downloadAndLoad(version: .v3, encoderPrecision: .int8)
|
||||
let manager = AsrManager(config: .default)
|
||||
try await manager.loadModels(models)
|
||||
Log.speech.info("Parakeet: ready in \(Date().timeIntervalSince(started), format: .fixed(precision: 1))s")
|
||||
return manager
|
||||
}
|
||||
loadTask = task
|
||||
|
||||
do {
|
||||
let manager = try await task.value
|
||||
loaded = manager
|
||||
return manager
|
||||
} catch {
|
||||
// Don't cache a failed load — a transient download error shouldn't wedge the
|
||||
// engine for the rest of the session.
|
||||
loadTask = nil
|
||||
throw error
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
import AVFoundation
|
||||
import Foundation
|
||||
|
||||
/// One buffer of captured audio, in transit from the audio thread to the speech engine.
|
||||
///
|
||||
/// `AVAudioPCMBuffer` isn't `Sendable`, and `AVAudioEngine` recycles the buffer it hands
|
||||
/// to a tap the moment the callback returns. The unchecked conformance is only sound
|
||||
/// because `AudioCapture` allocates a **fresh** buffer for every chunk and never touches
|
||||
/// it again after handing it over — don't construct one of these around a borrowed buffer.
|
||||
public struct AudioChunk: @unchecked Sendable {
|
||||
public let buffer: AVAudioPCMBuffer
|
||||
|
||||
public init(buffer: AVAudioPCMBuffer) {
|
||||
self.buffer = buffer
|
||||
}
|
||||
}
|
||||
|
||||
/// A snapshot of the running transcript.
|
||||
///
|
||||
/// `text` is always the **full transcript so far**, not a delta — engines revise
|
||||
/// earlier words as more audio arrives, so consumers should replace rather than append.
|
||||
public struct TranscriptionChunk: Sendable {
|
||||
public let text: String
|
||||
/// `true` once the engine has committed everything it will emit for this session.
|
||||
public let isFinal: Bool
|
||||
}
|
||||
|
||||
/// The seam that keeps the app engine-agnostic.
|
||||
///
|
||||
/// Apple's `SpeechAnalyzer` ships with macOS 26 and needs no model download, so it is
|
||||
/// the default. Parakeet (FluidAudio, CoreML/ANE) scores better on English and is the
|
||||
/// intended upgrade — implementing this protocol is the whole cost of switching.
|
||||
public protocol TranscriptionEngine: Actor {
|
||||
/// Audio format the engine wants buffers delivered in. `AudioCapture` converts to it.
|
||||
func preferredInputFormat() async -> AVAudioFormat?
|
||||
|
||||
/// Prepare models and open a session. Emits snapshots until `finish()` is called.
|
||||
func start() async throws -> AsyncThrowingStream<TranscriptionChunk, Error>
|
||||
|
||||
/// Feed one buffer of captured microphone audio, already in `preferredInputFormat()`.
|
||||
func feed(_ chunk: AudioChunk) async
|
||||
|
||||
/// Close the session and flush any pending final results.
|
||||
func finish() async
|
||||
}
|
||||
|
||||
public enum TranscriptionError: LocalizedError {
|
||||
case localeUnsupported(Locale)
|
||||
case modelInstallFailed(String)
|
||||
case noAudioFormat
|
||||
case notRunning
|
||||
|
||||
public var errorDescription: String? {
|
||||
switch self {
|
||||
case .localeUnsupported(let locale):
|
||||
return "Dictation isn't available for \(locale.identifier) on this Mac."
|
||||
case .modelInstallFailed(let detail):
|
||||
return "Couldn't install the speech model: \(detail)"
|
||||
case .noAudioFormat:
|
||||
return "No compatible audio format available for the speech engine."
|
||||
case .notRunning:
|
||||
return "The transcription engine isn't running."
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user