Initial commit of Edison Voice menu-bar dictation app
This commit is contained in:
@@ -0,0 +1,107 @@
|
||||
import Foundation
|
||||
import Testing
|
||||
|
||||
@testable import EdisonCore
|
||||
|
||||
/// Runs the shared behavioural contract in `dictionary-test-vectors.json`.
|
||||
///
|
||||
/// The vectors are the specification for correction behaviour; they are copied verbatim
|
||||
/// from the upstream Murmur YouTube project so the two stay honest against each other.
|
||||
struct VectorTests {
|
||||
struct Vectors: Decodable {
|
||||
let version: Int
|
||||
let cases: [Case]
|
||||
}
|
||||
|
||||
struct Case: Decodable {
|
||||
let name: String
|
||||
let entries: [Entry]
|
||||
let input: String
|
||||
let expected: String
|
||||
let expectedCorrections: [ExpectedCorrection]
|
||||
}
|
||||
|
||||
struct Entry: Decodable {
|
||||
let kind: String
|
||||
var hear: String?
|
||||
let write: String
|
||||
var isEnabled: Bool?
|
||||
|
||||
var asEntry: DictionaryEntry {
|
||||
DictionaryEntry(
|
||||
kind: kind == "correction" ? .correction : .term,
|
||||
write: write,
|
||||
hear: hear ?? "",
|
||||
isEnabled: isEnabled ?? true
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
struct ExpectedCorrection: Decodable {
|
||||
let to: String
|
||||
let count: Int
|
||||
}
|
||||
|
||||
static func load() throws -> Vectors {
|
||||
let url = try #require(
|
||||
Bundle.module.url(forResource: "dictionary-test-vectors", withExtension: "json")
|
||||
)
|
||||
return try JSONDecoder().decode(Vectors.self, from: Data(contentsOf: url))
|
||||
}
|
||||
|
||||
@Test("every shared vector produces the contracted output")
|
||||
func vectors() throws {
|
||||
let vectors = try Self.load()
|
||||
#expect(vectors.cases.isEmpty == false)
|
||||
|
||||
for testCase in vectors.cases {
|
||||
let corrector = DictionaryCorrector(entries: testCase.entries.map(\.asEntry))
|
||||
let (text, applied) = corrector.apply(to: testCase.input)
|
||||
|
||||
#expect(text == testCase.expected, "\(testCase.name): text")
|
||||
#expect(
|
||||
applied.count == testCase.expectedCorrections.count,
|
||||
"\(testCase.name): correction count — got \(applied.map(\.to))"
|
||||
)
|
||||
|
||||
// Order-insensitive: which rule fires first is an implementation detail of the
|
||||
// longest-first sort, but *what* fired and how often is contractual.
|
||||
for expected in testCase.expectedCorrections {
|
||||
let match = applied.first { $0.to == expected.to }
|
||||
#expect(match != nil, "\(testCase.name): expected a correction to “\(expected.to)”")
|
||||
#expect(match?.count == expected.count, "\(testCase.name): count for “\(expected.to)”")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test("bias list is capped and de-duplicated")
|
||||
func biasList() {
|
||||
let entries = (0..<100).map { DictionaryEntry.term("Word\($0)") }
|
||||
+ [DictionaryEntry.term("Word0")]
|
||||
let phrases = DictionaryCorrector.biasPhrases(from: entries)
|
||||
|
||||
#expect(phrases.count == DictionaryCorrector.biasLimit)
|
||||
#expect(Set(phrases).count == phrases.count)
|
||||
}
|
||||
|
||||
@Test("disabled entries are excluded from biasing")
|
||||
func biasSkipsDisabled() {
|
||||
let entries = [
|
||||
DictionaryEntry(kind: .term, write: "Kept"),
|
||||
DictionaryEntry(kind: .term, write: "Skipped", isEnabled: false),
|
||||
]
|
||||
#expect(DictionaryCorrector.biasPhrases(from: entries) == ["Kept"])
|
||||
}
|
||||
|
||||
@Test("an ordinary word used as a trigger is flagged")
|
||||
func warnsOnCommonWord() {
|
||||
let entry = DictionaryEntry.correction(hear: "cloud", write: "Claude")
|
||||
#expect(DictionaryWarning.check(entry).isEmpty == false)
|
||||
}
|
||||
|
||||
@Test("a distinctive phrase is not flagged")
|
||||
func doesNotWarnOnDistinctivePhrase() {
|
||||
let entry = DictionaryEntry.correction(hear: "clawed code", write: "Claude Code")
|
||||
#expect(DictionaryWarning.check(entry).isEmpty)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,353 @@
|
||||
{
|
||||
"$comment": [
|
||||
"Shared behavioural contract for the dictionary correction pass.",
|
||||
"",
|
||||
"The macOS app (Swift) and the Windows app (C#) implement this logic independently.",
|
||||
"Independent implementations drift. This file is the thing that stops them: both must",
|
||||
"run every case below and produce exactly the stated output. A change to correction",
|
||||
"semantics starts here, not in either implementation.",
|
||||
"",
|
||||
"Rules under test:",
|
||||
" 1. Whole matches only - a rule must never bite into a longer word.",
|
||||
" 2. Case-insensitive - the trigger matches in any case; output is verbatim.",
|
||||
" 3. Longest match first - a longer trigger wins over a shorter overlapping one.",
|
||||
" 4. Glued words match - parts may be separated by any number of spaces or hyphens,",
|
||||
" including none at all.",
|
||||
" 5. Disabled is inert - a disabled entry changes nothing."
|
||||
],
|
||||
"version": 1,
|
||||
"cases": [
|
||||
{
|
||||
"name": "basic phrase replacement",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "I use cloud code every day.",
|
||||
"expected": "I use Claude Code every day.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "glued together with no separator",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "I use CloudCode every day.",
|
||||
"expected": "I use Claude Code every day.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "joined by a hyphen",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "I use Cloud-Code every day.",
|
||||
"expected": "I use Claude Code every day.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "multiple spaces between parts",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "I use cloud code every day.",
|
||||
"expected": "I use Claude Code every day.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "trigger matches regardless of case",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "CLOUD CODE is great.",
|
||||
"expected": "Claude Code is great.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "must not corrupt a longer word that starts the same way",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "Cloudflare fronts the cloudcodex service.",
|
||||
"expected": "Cloudflare fronts the cloudcodex service.",
|
||||
"expectedCorrections": []
|
||||
},
|
||||
{
|
||||
"name": "must not touch the ordinary word alone",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "The cloud is fine and the code is fine.",
|
||||
"expected": "The cloud is fine and the code is fine.",
|
||||
"expectedCorrections": []
|
||||
},
|
||||
{
|
||||
"name": "longest trigger wins over a shorter overlapping one",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud",
|
||||
"write": "Claude"
|
||||
},
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "cloud code",
|
||||
"expected": "Claude Code",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "counts repeated hits",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "clawed code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "clawed code, then clawed code again, and clawed code.",
|
||||
"expected": "Claude Code, then Claude Code again, and Claude Code.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 3
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "punctuation is a boundary",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "(cloud code), \"cloud code\"; cloud code!",
|
||||
"expected": "(Claude Code), \"Claude Code\"; Claude Code!",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 3
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "a possessive still matches - the apostrophe is not a letter",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "cloud code's output",
|
||||
"expected": "Claude Code's output",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "a disabled entry changes nothing",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code",
|
||||
"isEnabled": false
|
||||
}
|
||||
],
|
||||
"input": "I use cloud code every day.",
|
||||
"expected": "I use cloud code every day.",
|
||||
"expectedCorrections": []
|
||||
},
|
||||
{
|
||||
"name": "a term entry never rewrites text",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "term",
|
||||
"write": "Anthropic"
|
||||
}
|
||||
],
|
||||
"input": "anthropic makes models.",
|
||||
"expected": "anthropic makes models.",
|
||||
"expectedCorrections": []
|
||||
},
|
||||
{
|
||||
"name": "single word trigger is honoured when asked for",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "vercell",
|
||||
"write": "Vercel"
|
||||
}
|
||||
],
|
||||
"input": "Deployed on Vercell today.",
|
||||
"expected": "Deployed on Vercel today.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Vercel",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "several independent rules in one pass",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "clawed code",
|
||||
"write": "Claude Code"
|
||||
},
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "whisper flow",
|
||||
"write": "Wispr Flow"
|
||||
}
|
||||
],
|
||||
"input": "clawed code beat whisper flow.",
|
||||
"expected": "Claude Code beat Wispr Flow.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Claude Code",
|
||||
"count": 1
|
||||
},
|
||||
{
|
||||
"to": "Wispr Flow",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "empty input is left alone",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "cloud code",
|
||||
"write": "Claude Code"
|
||||
}
|
||||
],
|
||||
"input": "",
|
||||
"expected": "",
|
||||
"expectedCorrections": []
|
||||
},
|
||||
{
|
||||
"name": "replacement containing regex metacharacters is literal",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "c plus plus",
|
||||
"write": "C++"
|
||||
}
|
||||
],
|
||||
"input": "I write c plus plus.",
|
||||
"expected": "I write C++.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "C++",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "trigger containing regex metacharacters is literal",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "dot net",
|
||||
"write": ".NET"
|
||||
}
|
||||
],
|
||||
"input": "Built with dot net.",
|
||||
"expected": "Built with .NET.",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": ".NET",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "accented trigger matches decomposed input (NFC normalization)",
|
||||
"$comment": "macOS hands back NFD strings from several APIs. Without normalizing both sides, an accented trigger silently never fires. Input here is deliberately NFD.",
|
||||
"entries": [
|
||||
{
|
||||
"kind": "correction",
|
||||
"hear": "café racer",
|
||||
"write": "Café Racer"
|
||||
}
|
||||
],
|
||||
"input": "a café racer bike",
|
||||
"expected": "a Café Racer bike",
|
||||
"expectedCorrections": [
|
||||
{
|
||||
"to": "Café Racer",
|
||||
"count": 1
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
Reference in New Issue
Block a user