537a68552a
Reduce perceived latency from key release to final text: - Adaptive chunking: 2.5s first chunk + 5s follow-ups so short utterances start on-device recognition while still recording. - Session-level ASR warmup and audio-format cache reuse to remove per-utterance cold-start of SpeechAnalyzer. - Mirror live pipelined partials to the keyboard transcript line via a new flow.transcriptionPartial App Group key + Darwin ping. Also commits the accumulated custom language model, Flow session, keyboard extension restructure, and Xiaomi MiMo provider work in progress on this branch.
67 lines
2.8 KiB
Swift
67 lines
2.8 KiB
Swift
// FlowSessionKeys.swift
|
|
// OSGKeyboard · Shared
|
|
//
|
|
// App Group keys for TypeWhisper-style Flow sessions between the
|
|
// keyboard extension and the host app (Session Owner).
|
|
|
|
import Foundation
|
|
|
|
public enum FlowSessionKeys {
|
|
public static let flowSessionActive = "flow.flowSessionActive"
|
|
public static let flowSessionExpires = "flow.flowSessionExpires"
|
|
public static let flowHeartbeat = "flow.flowHeartbeat"
|
|
public static let keyboardRecordingState = "flow.keyboardRecordingState"
|
|
public static let transcriptionLanguage = "flow.transcriptionLanguage"
|
|
public static let transcriptionResult = "flow.transcriptionResult"
|
|
/// Live pipelined ASR partial for the keyboard transcript line.
|
|
public static let transcriptionPartial = "flow.transcriptionPartial"
|
|
/// Soft warning when polish failed but raw transcript was delivered.
|
|
public static let transcriptionPolishWarning = "flow.transcriptionPolishWarning"
|
|
public static let transcriptionError = "flow.transcriptionError"
|
|
/// Structured kind paired with `transcriptionError` for keyboard UI.
|
|
public static let transcriptionErrorKind = "flow.transcriptionErrorKind"
|
|
public static let audioLevels = "flow.audioLevels"
|
|
|
|
/// Heartbeat older than this while the host is foreground → likely killed.
|
|
public static let heartbeatStaleInterval: TimeInterval = 3
|
|
|
|
/// Default Flow session length when started from the keyboard.
|
|
public static let defaultSessionDuration: TimeInterval = 480
|
|
|
|
/// Maximum duration for a single keyboard utterance (3.5 minutes).
|
|
public static let maxUtteranceDuration: TimeInterval = 210
|
|
|
|
/// Host polls for pipelined ASR drain after mic stop. Pipelining usually
|
|
/// finishes most chunks during recording; this is a soft deadline before
|
|
/// blocking on `asrTask.value` (which waits until the pipeline exits).
|
|
public static let localASRWaitTimeout: TimeInterval = 120
|
|
public static let cloudASRWaitTimeout: TimeInterval = 120
|
|
|
|
/// Keyboard watchdog after the user stops recording (not utterance max length).
|
|
/// Must cover worst-case post-stop backlog: remaining SpeechAnalyzer chunks
|
|
/// plus cloud LLM polish (see `PolishingService.effectiveTimeout` cap).
|
|
public static func keyboardResultTimeout(engineMode: String) -> TimeInterval {
|
|
if engineMode == "local" {
|
|
return 180
|
|
}
|
|
return 240
|
|
}
|
|
|
|
public enum RecordingState: String, Sendable, Equatable {
|
|
case idle
|
|
case recording
|
|
case stopped
|
|
case processing
|
|
case aborted
|
|
}
|
|
|
|
/// Structured host → keyboard transcription failure kind.
|
|
public enum TranscriptionErrorKind: String, Sendable, Equatable {
|
|
case noSpeech
|
|
case recognitionInterrupted
|
|
case audioUnavailable
|
|
case asrFailed
|
|
case generic
|
|
}
|
|
}
|