perf(asr): speed up local Flow dictation and land CLM/keyboard refactor

Reduce perceived latency from key release to final text:
- Adaptive chunking: 2.5s first chunk + 5s follow-ups so short
  utterances start on-device recognition while still recording.
- Session-level ASR warmup and audio-format cache reuse to remove
  per-utterance cold-start of SpeechAnalyzer.
- Mirror live pipelined partials to the keyboard transcript line via
  a new flow.transcriptionPartial App Group key + Darwin ping.

Also commits the accumulated custom language model, Flow session,
keyboard extension restructure, and Xiaomi MiMo provider work in
progress on this branch.
This commit is contained in:
Rocky
2026-07-06 00:00:19 +08:00
parent cfbfb542cc
commit 537a68552a
76 changed files with 3456 additions and 121086 deletions
@@ -10,12 +10,11 @@ public enum EngineServiceLabel {
engineMode: String,
providerId: String,
model: String,
localASRBackend: LocalASRBackend = .speechAnalyzer,
language: AppUILanguage? = nil
) -> String {
let lang = language ?? AppGroupStore().uiLanguage
if engineMode == "local" {
let asrName = asrDisplayName(for: localASRBackend, language: lang)
let asrName = SharedL10n.string("engine.asr.appleSpeech", language: lang)
return SharedL10n.format("engine.summary.local", language: lang, asrName)
}
let providerName = ProviderDisplayName.name(for: providerId, language: lang)
@@ -30,14 +29,4 @@ public enum EngineServiceLabel {
trimmedModel
)
}
private static func asrDisplayName(
for backend: LocalASRBackend,
language: AppUILanguage
) -> String {
// v0.2.0: only the iOS SpeechAnalyzer path remains. We keep the
// switch on `LocalASRBackend` so the next non-iOS backend can
// slot in without touching every call site.
return SharedL10n.string("engine.asr.appleSpeech", language: language)
}
}
}