Initial commit of Edison Voice menu-bar dictation app

This commit is contained in:
2026-09-09 23:49:51 +02:00
commit ad02cfb638
40 changed files with 3305 additions and 0 deletions
+107
View File
@@ -0,0 +1,107 @@
import Foundation
import Testing
@testable import EdisonCore
/// Runs the shared behavioural contract in `dictionary-test-vectors.json`.
///
/// The vectors are the specification for correction behaviour; they are copied verbatim
/// from the upstream Murmur YouTube project so the two stay honest against each other.
struct VectorTests {
struct Vectors: Decodable {
let version: Int
let cases: [Case]
}
struct Case: Decodable {
let name: String
let entries: [Entry]
let input: String
let expected: String
let expectedCorrections: [ExpectedCorrection]
}
struct Entry: Decodable {
let kind: String
var hear: String?
let write: String
var isEnabled: Bool?
var asEntry: DictionaryEntry {
DictionaryEntry(
kind: kind == "correction" ? .correction : .term,
write: write,
hear: hear ?? "",
isEnabled: isEnabled ?? true
)
}
}
struct ExpectedCorrection: Decodable {
let to: String
let count: Int
}
static func load() throws -> Vectors {
let url = try #require(
Bundle.module.url(forResource: "dictionary-test-vectors", withExtension: "json")
)
return try JSONDecoder().decode(Vectors.self, from: Data(contentsOf: url))
}
@Test("every shared vector produces the contracted output")
func vectors() throws {
let vectors = try Self.load()
#expect(vectors.cases.isEmpty == false)
for testCase in vectors.cases {
let corrector = DictionaryCorrector(entries: testCase.entries.map(\.asEntry))
let (text, applied) = corrector.apply(to: testCase.input)
#expect(text == testCase.expected, "\(testCase.name): text")
#expect(
applied.count == testCase.expectedCorrections.count,
"\(testCase.name): correction count — got \(applied.map(\.to))"
)
// Order-insensitive: which rule fires first is an implementation detail of the
// longest-first sort, but *what* fired and how often is contractual.
for expected in testCase.expectedCorrections {
let match = applied.first { $0.to == expected.to }
#expect(match != nil, "\(testCase.name): expected a correction to “\(expected.to)”")
#expect(match?.count == expected.count, "\(testCase.name): count for “\(expected.to)”")
}
}
}
@Test("bias list is capped and de-duplicated")
func biasList() {
let entries = (0..<100).map { DictionaryEntry.term("Word\($0)") }
+ [DictionaryEntry.term("Word0")]
let phrases = DictionaryCorrector.biasPhrases(from: entries)
#expect(phrases.count == DictionaryCorrector.biasLimit)
#expect(Set(phrases).count == phrases.count)
}
@Test("disabled entries are excluded from biasing")
func biasSkipsDisabled() {
let entries = [
DictionaryEntry(kind: .term, write: "Kept"),
DictionaryEntry(kind: .term, write: "Skipped", isEnabled: false),
]
#expect(DictionaryCorrector.biasPhrases(from: entries) == ["Kept"])
}
@Test("an ordinary word used as a trigger is flagged")
func warnsOnCommonWord() {
let entry = DictionaryEntry.correction(hear: "cloud", write: "Claude")
#expect(DictionaryWarning.check(entry).isEmpty == false)
}
@Test("a distinctive phrase is not flagged")
func doesNotWarnOnDistinctivePhrase() {
let entry = DictionaryEntry.correction(hear: "clawed code", write: "Claude Code")
#expect(DictionaryWarning.check(entry).isEmpty)
}
}
@@ -0,0 +1,353 @@
{
"$comment": [
"Shared behavioural contract for the dictionary correction pass.",
"",
"The macOS app (Swift) and the Windows app (C#) implement this logic independently.",
"Independent implementations drift. This file is the thing that stops them: both must",
"run every case below and produce exactly the stated output. A change to correction",
"semantics starts here, not in either implementation.",
"",
"Rules under test:",
" 1. Whole matches only - a rule must never bite into a longer word.",
" 2. Case-insensitive - the trigger matches in any case; output is verbatim.",
" 3. Longest match first - a longer trigger wins over a shorter overlapping one.",
" 4. Glued words match - parts may be separated by any number of spaces or hyphens,",
" including none at all.",
" 5. Disabled is inert - a disabled entry changes nothing."
],
"version": 1,
"cases": [
{
"name": "basic phrase replacement",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "I use cloud code every day.",
"expected": "I use Claude Code every day.",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
}
]
},
{
"name": "glued together with no separator",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "I use CloudCode every day.",
"expected": "I use Claude Code every day.",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
}
]
},
{
"name": "joined by a hyphen",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "I use Cloud-Code every day.",
"expected": "I use Claude Code every day.",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
}
]
},
{
"name": "multiple spaces between parts",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "I use cloud code every day.",
"expected": "I use Claude Code every day.",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
}
]
},
{
"name": "trigger matches regardless of case",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "CLOUD CODE is great.",
"expected": "Claude Code is great.",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
}
]
},
{
"name": "must not corrupt a longer word that starts the same way",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "Cloudflare fronts the cloudcodex service.",
"expected": "Cloudflare fronts the cloudcodex service.",
"expectedCorrections": []
},
{
"name": "must not touch the ordinary word alone",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "The cloud is fine and the code is fine.",
"expected": "The cloud is fine and the code is fine.",
"expectedCorrections": []
},
{
"name": "longest trigger wins over a shorter overlapping one",
"entries": [
{
"kind": "correction",
"hear": "cloud",
"write": "Claude"
},
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "cloud code",
"expected": "Claude Code",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
}
]
},
{
"name": "counts repeated hits",
"entries": [
{
"kind": "correction",
"hear": "clawed code",
"write": "Claude Code"
}
],
"input": "clawed code, then clawed code again, and clawed code.",
"expected": "Claude Code, then Claude Code again, and Claude Code.",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 3
}
]
},
{
"name": "punctuation is a boundary",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "(cloud code), \"cloud code\"; cloud code!",
"expected": "(Claude Code), \"Claude Code\"; Claude Code!",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 3
}
]
},
{
"name": "a possessive still matches - the apostrophe is not a letter",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "cloud code's output",
"expected": "Claude Code's output",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
}
]
},
{
"name": "a disabled entry changes nothing",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code",
"isEnabled": false
}
],
"input": "I use cloud code every day.",
"expected": "I use cloud code every day.",
"expectedCorrections": []
},
{
"name": "a term entry never rewrites text",
"entries": [
{
"kind": "term",
"write": "Anthropic"
}
],
"input": "anthropic makes models.",
"expected": "anthropic makes models.",
"expectedCorrections": []
},
{
"name": "single word trigger is honoured when asked for",
"entries": [
{
"kind": "correction",
"hear": "vercell",
"write": "Vercel"
}
],
"input": "Deployed on Vercell today.",
"expected": "Deployed on Vercel today.",
"expectedCorrections": [
{
"to": "Vercel",
"count": 1
}
]
},
{
"name": "several independent rules in one pass",
"entries": [
{
"kind": "correction",
"hear": "clawed code",
"write": "Claude Code"
},
{
"kind": "correction",
"hear": "whisper flow",
"write": "Wispr Flow"
}
],
"input": "clawed code beat whisper flow.",
"expected": "Claude Code beat Wispr Flow.",
"expectedCorrections": [
{
"to": "Claude Code",
"count": 1
},
{
"to": "Wispr Flow",
"count": 1
}
]
},
{
"name": "empty input is left alone",
"entries": [
{
"kind": "correction",
"hear": "cloud code",
"write": "Claude Code"
}
],
"input": "",
"expected": "",
"expectedCorrections": []
},
{
"name": "replacement containing regex metacharacters is literal",
"entries": [
{
"kind": "correction",
"hear": "c plus plus",
"write": "C++"
}
],
"input": "I write c plus plus.",
"expected": "I write C++.",
"expectedCorrections": [
{
"to": "C++",
"count": 1
}
]
},
{
"name": "trigger containing regex metacharacters is literal",
"entries": [
{
"kind": "correction",
"hear": "dot net",
"write": ".NET"
}
],
"input": "Built with dot net.",
"expected": "Built with .NET.",
"expectedCorrections": [
{
"to": ".NET",
"count": 1
}
]
},
{
"name": "accented trigger matches decomposed input (NFC normalization)",
"$comment": "macOS hands back NFD strings from several APIs. Without normalizing both sides, an accented trigger silently never fires. Input here is deliberately NFD.",
"entries": [
{
"kind": "correction",
"hear": "café racer",
"write": "Café Racer"
}
],
"input": "a café racer bike",
"expected": "a Café Racer bike",
"expectedCorrections": [
{
"to": "Café Racer",
"count": 1
}
]
}
]
}