From c3bd4ab4cc655f08651a0f470765c33ac8eb9d4f Mon Sep 17 00:00:00 2001 From: lawrencecchen <54008264+lawrencecchen@users.noreply.github.com> Date: Tue, 16 Jun 2026 14:31:36 -0700 Subject: [PATCH 1/3] iOS dictation: validate audio format sample rate before installTap (mic crash) cmux Beta crashed when using the mic. beginRecognition() guarded only channelCount > 0 before installTap, but installTap raises an UNCATCHABLE Obj-C exception (IsFormatSampleRateAndChannelCountValid) on an invalid input format (zero sample rate), which do/catch can't trap. Also require sampleRate > 0 and failStart() gracefully. Addresses issue #6217; will confirm the exact frame against the device crash report and add an Obj-C exception shim if the crash is a format MISMATCH rather than an invalid format. (The conventions-lint fix this branch originally carried is superseded by main's ComposerDictationTextMerger refactor; reset to main and kept only the mic guard.) --- .../ComposerDictationController.swift | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift b/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift index b74e55eeea86..7fb39ef8d130 100644 --- a/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift +++ b/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift @@ -278,9 +278,14 @@ final class ComposerDictationController { let inputNode = audioEngine.inputNode let format = inputNode.outputFormat(forBus: 0) - // A zero-channel format means there is no usable input route; bail rather - // than crash installing a tap with an invalid format. - guard format.channelCount > 0 else { + // `installTap` raises an UNCATCHABLE Obj-C exception + // (`IsFormatSampleRateAndChannelCountValid`) when the input format is + // invalid, which a Swift `do/catch` cannot trap. The input node can hand + // back a format with a zero sample rate or zero channels when there is no + // usable input route yet (e.g. the session's input route is still + // settling, or the mic is claimed by another app), so validate BOTH + // dimensions and fail the start gracefully instead of crashing. + guard format.channelCount > 0, format.sampleRate > 0 else { failStart() return } From 5cfb50ea2d89267fae0a2509cc37987e27717e3f Mon Sep 17 00:00:00 2001 From: lawrencecchen <54008264+lawrencecchen@users.noreply.github.com> Date: Tue, 16 Jun 2026 14:45:57 -0700 Subject: [PATCH 2/3] Fix iOS mic-tap crash: nonisolated Swift 6 authorization callbacks Real root cause (device crash log, build 1.0.3 20260616105751): EXC_BREAKPOINT in swift_task_isCurrentExecutor -> dispatch_assert_queue_fail, inside the TCC permission callback. Tapping the mic requests speech+mic permission; SFSpeechRecognizer / AVAudioApplication invoke their completion handlers on their own (non-main) queues. The closures were inferred main-actor (written in this @MainActor controller), so under Swift 6 / iOS 26 the runtime asserts executor isolation when the system calls them off-main and traps. Fix (modern Swift 6 isolation): requestAuthorization + requestMicrophone- Permission are now nonisolated with Sendable completions, so no main-actor closure is invoked off-main; start() hops to the main actor exactly once via a Task @MainActor enqueue (not a synchronous executor assertion). Also keeps the installTap sample-rate guard (separate latent crash on an invalid input format). Compile verified on CI; runtime confirmed by iOS 26 device dogfood. Closes #6217. --- .../ComposerDictationController.swift | 74 +++++++++++-------- 1 file changed, 43 insertions(+), 31 deletions(-) diff --git a/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift b/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift index 7fb39ef8d130..f0fe024bf3df 100644 --- a/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift +++ b/Packages/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift @@ -138,21 +138,28 @@ final class ComposerDictationController { baseText = existingText self.onText = onText state = .requestingPermission + // The Speech/AVFoundation authorization callbacks fire on their own + // (non-main) queues, so `requestAuthorization` is nonisolated with a + // `@Sendable` completion. Hop to the main actor ONCE, here, via + // `Task { @MainActor in }` (an async enqueue, not a synchronous executor + // assertion) before touching any actor-isolated state. requestAuthorization { [weak self] granted in - guard let self else { return } - // A second tap may have cancelled (or otherwise moved on) while - // authorization resolved; if so this start is stale. Do nothing, so a - // cancel during the permission flow neither starts the engine nor - // overwrites the user's idle state with `unavailable`. - guard self.state == .requestingPermission else { return } - guard granted else { - // Denied or restricted: a terminal rest state that disables the - // mic. The captured callback is dropped. - self.onText = nil - self.state = .unavailable - return + Task { @MainActor in + guard let self else { return } + // A second tap may have cancelled (or otherwise moved on) while + // authorization resolved; if so this start is stale. Do nothing, so + // a cancel during the permission flow neither starts the engine nor + // overwrites the user's idle state with `unavailable`. + guard self.state == .requestingPermission else { return } + guard granted else { + // Denied or restricted: a terminal rest state that disables the + // mic. The captured callback is dropped. + self.onText = nil + self.state = .unavailable + return + } + self.beginRecognition() } - self.beginRecognition() } } @@ -210,34 +217,39 @@ final class ComposerDictationController { // MARK: - Authorization - /// Resolve both speech-recognition and microphone authorization, calling back - /// on the main actor with whether BOTH were granted. - private func requestAuthorization(_ completion: @escaping @MainActor (Bool) -> Void) { + /// Resolve speech-recognition then microphone authorization and report whether + /// BOTH were granted. + /// + /// `nonisolated` with a `@Sendable` completion ON PURPOSE: `SFSpeechRecognizer` + /// / `AVFoundation` invoke their completion handlers on their own (non-main) + /// queues. If those closures were main-actor-isolated (the default for a + /// closure written inside this `@MainActor` type), Swift 6 on iOS 26 asserts + /// executor isolation when the system calls them off-main and traps + /// (`EXC_BREAKPOINT` in `swift_task_isCurrentExecutor` -> + /// `dispatch_assert_queue_fail`), which is the mic-tap crash. Keeping the + /// whole authorization chain nonisolated means no `@MainActor` closure is ever + /// invoked off-main; the caller hops to the main actor once. + private nonisolated func requestAuthorization(_ completion: @escaping @Sendable (Bool) -> Void) { SFSpeechRecognizer.requestAuthorization { speechStatus in - // The Speech callback arrives off the main actor; hop back before - // touching any state or the microphone request. - Task { @MainActor in - guard speechStatus == .authorized else { - completion(false) - return - } - Self.requestMicrophonePermission { micGranted in - completion(micGranted) - } + guard speechStatus == .authorized else { + completion(false) + return } + Self.requestMicrophonePermission(completion) } } - /// Request microphone permission, bridging the iOS 17+ API to its - /// pre-17 fallback. Calls back on the main actor. - private static func requestMicrophonePermission(_ completion: @escaping @MainActor (Bool) -> Void) { + /// Request microphone permission, bridging the iOS 17+ API to its pre-17 + /// fallback. `nonisolated` + `@Sendable` for the same off-main-isolation + /// reason as ``requestAuthorization(_:)``; reports on the system's queue. + private nonisolated static func requestMicrophonePermission(_ completion: @escaping @Sendable (Bool) -> Void) { if #available(iOS 17.0, *) { AVAudioApplication.requestRecordPermission { granted in - Task { @MainActor in completion(granted) } + completion(granted) } } else { AVAudioSession.sharedInstance().requestRecordPermission { granted in - Task { @MainActor in completion(granted) } + completion(granted) } } } From 94bf8c6fd8d9d4cd41e89a2bd381b39ce066f208 Mon Sep 17 00:00:00 2001 From: lawrencecchen <54008264+lawrencecchen@users.noreply.github.com> Date: Tue, 16 Jun 2026 18:29:45 -0700 Subject: [PATCH 3/3] iOS dictation: fix mic-tap crash, text loss, and gate the permission prompt The mic-tap crash on iOS 26 was a Swift-concurrency executor trap: closures the compiler inferred as @MainActor (the permission completions, the AVAudioEngine tap block, and the SFSpeechRecognitionTask result handler) were invoked off the main thread by TCC / the realtime audio thread / the recognition queue, tripping swift_task_isCurrentExecutor -> dispatch_assert_queue_fail (EXC_BREAKPOINT). Fixes: - Build the tap block and the result handler in `nonisolated` factory methods so they are genuinely non-isolated and safe to invoke off-main. - Resolve speech+mic authorization synchronously when already determined (`resolvedAuthorization()`), so a repeat tap never re-invokes the crashing async TCC callback; only a never-determined permission uses the async request path. This also keeps the permission prompt gated to the first mic tap (status reads never prompt). - Make the authorization completions nonisolated/@Sendable. - Ignore an empty final transcript on stop so it cannot wipe the words the partial results already committed to the composer. Verified on-device (iPhone, iOS 26) through the full start -> listen -> stop cycle with no crash and text preserved. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../ComposerDictationController.swift | 125 ++++++++++++++---- 1 file changed, 100 insertions(+), 25 deletions(-) diff --git a/Packages/iOS/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift b/Packages/iOS/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift index f0fe024bf3df..44cf88156979 100644 --- a/Packages/iOS/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift +++ b/Packages/iOS/CmuxMobileShellUI/Sources/CmuxMobileShellUI/ComposerDictationController.swift @@ -137,6 +137,31 @@ final class ComposerDictationController { } baseText = existingText self.onText = onText + + // iOS 26 trap avoidance (the real mic-tap crash): when speech + mic + // authorization is ALREADY resolved (the common case after the first + // grant), decide synchronously and go straight to recognition. The async + // `SFSpeechRecognizer.requestAuthorization` / `requestRecordPermission` + // completions are dispatched by TCC on an XPC reply thread; a Swift + // closure the compiler treats as main-actor-isolated traps there in + // `swift_task_isCurrentExecutor` -> `dispatch_assert_queue_fail`. Reading + // the status synchronously never invokes that completion, so a repeat tap + // (Lawrence's repro) cannot hit the crashing callback. Only a genuinely + // undetermined permission falls through to the async request below. + switch Self.resolvedAuthorization() { + case .granted: + state = .requestingPermission + beginRecognition() + return + case .denied: + self.onText = nil + state = .unavailable + return + case .undetermined: + // First-ever request: fall through to the async prompt below. + break + } + state = .requestingPermission // The Speech/AVFoundation authorization callbacks fire on their own // (non-main) queues, so `requestAuthorization` is nonisolated with a @@ -217,6 +242,38 @@ final class ComposerDictationController { // MARK: - Authorization + /// Whether both authorizations are already resolved, and if so the verdict. + /// `undetermined` means at least one permission has never been requested, so a + /// first-time async prompt is still required. + private enum AuthResolution { case granted, denied, undetermined } + + /// Read the CURRENT speech + microphone authorization synchronously, without + /// invoking any async request completion. `nonisolated` and side-effect-free: + /// these status getters are plain synchronous reads, so they are safe to call + /// from the main actor and never touch the crashing TCC callback path. + private nonisolated static func resolvedAuthorization() -> AuthResolution { + let speech = SFSpeechRecognizer.authorizationStatus() + let micGranted: Bool + let micDetermined: Bool + if #available(iOS 17.0, *) { + switch AVAudioApplication.shared.recordPermission { + case .granted: micGranted = true; micDetermined = true + case .denied: micGranted = false; micDetermined = true + case .undetermined: micGranted = false; micDetermined = false + @unknown default: micGranted = false; micDetermined = false + } + } else { + switch AVAudioSession.sharedInstance().recordPermission { + case .granted: micGranted = true; micDetermined = true + case .denied: micGranted = false; micDetermined = true + case .undetermined: micGranted = false; micDetermined = false + @unknown default: micGranted = false; micDetermined = false + } + } + guard speech != .notDetermined, micDetermined else { return .undetermined } + return (speech == .authorized && micGranted) ? .granted : .denied + } + /// Resolve speech-recognition then microphone authorization and report whether /// BOTH were granted. /// @@ -277,10 +334,9 @@ final class ComposerDictationController { do { let session = AVAudioSession.sharedInstance() // Record-only category for speech-to-text. `.duckOthers` is NOT valid - // for `.record` (only Ambient/PlayAndRecord/Playback/MultiRoute), and - // `.notifyOthersOnDeactivation` is only valid on deactivation, so both - // are omitted here; passing them throws on OSes that enforce the - // documented restrictions and would permanently disable the mic. + // for `.record`, and `.notifyOthersOnDeactivation` is only valid on + // deactivation, so both are omitted here; passing them throws on OSes + // that enforce the documented restrictions. try session.setCategory(.record, mode: .measurement) try session.setActive(true) } catch { @@ -290,22 +346,13 @@ final class ComposerDictationController { let inputNode = audioEngine.inputNode let format = inputNode.outputFormat(forBus: 0) - // `installTap` raises an UNCATCHABLE Obj-C exception - // (`IsFormatSampleRateAndChannelCountValid`) when the input format is - // invalid, which a Swift `do/catch` cannot trap. The input node can hand - // back a format with a zero sample rate or zero channels when there is no - // usable input route yet (e.g. the session's input route is still - // settling, or the mic is claimed by another app), so validate BOTH - // dimensions and fail the start gracefully instead of crashing. + // Validate the input format; an invalid one makes `installTap` raise an + // uncatchable Obj-C exception. guard format.channelCount > 0, format.sampleRate > 0 else { failStart() return } - inputNode.installTap(onBus: 0, bufferSize: 1024, format: format) { [weak request] buffer, _ in - // The tap fires on a realtime audio thread. `append` is thread-safe on - // the request; do not touch main-actor state here. - request?.append(buffer) - } + inputNode.installTap(onBus: 0, bufferSize: 1024, format: format, block: Self.makeTapBlock(request: request)) audioEngine.prepare() do { @@ -315,18 +362,48 @@ final class ComposerDictationController { return } - task = recognizer.recognitionTask(with: request) { [weak self] result, error in - // The recognition callback arrives on an arbitrary queue with - // non-Sendable reference types (`SFSpeechRecognitionResult`, `Error`). - // Extract only Sendable value snapshots HERE, then hop to the main - // actor with those, so no non-Sendable reference crosses the actor - // boundary. `self` is weak so the task does not retain the controller. + task = recognizer.recognitionTask(with: request, resultHandler: makeRecognitionResultHandler()) + + state = .listening + } + + /// Build the audio-tap block. `nonisolated` so the returned closure is NOT + /// main-actor-isolated: `installTap` invokes it on the realtime audio render + /// thread, where a main-actor closure traps in `swift_task_isCurrentExecutor`. + /// `append` is thread-safe on the request; no main-actor state is touched. + private nonisolated static func makeTapBlock( + request: SFSpeechAudioBufferRecognitionRequest + ) -> (AVAudioPCMBuffer, AVAudioTime) -> Void { + return { [weak request] buffer, _ in + request?.append(buffer) + } + } + + /// Build the recognition result handler. `nonisolated` so the returned closure + /// is NOT main-actor-isolated: `SFSpeechRecognitionTask` delivers results on an + /// arbitrary queue. The closure extracts only `Sendable` value snapshots and + /// hops to the main actor via `Task { @MainActor in }` before touching any + /// actor-isolated state; `self` is weak so the task does not retain the + /// controller. + private nonisolated func makeRecognitionResultHandler() + -> @Sendable (SFSpeechRecognitionResult?, Error?) -> Void { + // `@Sendable` so the closure is its own isolation region (not main-actor): + // Speech invokes it off-main, where a main-actor closure traps. It captures + // only `[weak self]` (a Sendable, main-actor class) and reads Sendable + // snapshots from the non-Sendable result/error PARAMETERS, then hops to the + // main actor. This mirrors the authorization completion pattern above. + return { [weak self] result, error in let transcript = result?.bestTranscription.formattedString let isFinal = result?.isFinal ?? false let failed = error != nil Task { @MainActor in guard let self else { return } - if let transcript { + // Only apply a NON-EMPTY transcript. On stop, the recognizer can + // deliver a final result with an empty transcript; merging that + // (`merged(base, "")` -> `base`) would wipe the words the partials + // already committed. The latest non-empty partial is already in the + // field, so an empty final/partial must be ignored, not applied. + if let transcript, !transcript.isEmpty { self.onText?(self.textMerger.merged( base: self.baseText, transcript: transcript @@ -346,8 +423,6 @@ final class ComposerDictationController { } } } - - state = .listening } /// Tear down after a setup failure and disable the mic. Distinct from a clean