diff --git a/.codex b/.codex new file mode 100644 index 00000000000..e69de29bb2d diff --git a/conductor/check-pr-status.md b/conductor/check-pr-status.md new file mode 100644 index 00000000000..b16ab4cc66b --- /dev/null +++ b/conductor/check-pr-status.md @@ -0,0 +1,11 @@ +# Objective + +Check PR status and implement a fix if there are any issues. + +# Implementation steps + +1. Exit **Plan Mode** with `Shift+Tab`. +2. Verify the PR is mergeable with `gh pr view`. If there are conflicts, fetch + and merge upstream/main. +3. Verify the tests are passing with `gh pr checks` and resolve and investigate + any failing checks. diff --git a/docs/cli/settings.md b/docs/cli/settings.md index 834750fdf98..96f9a4ebc9a 100644 --- a/docs/cli/settings.md +++ b/docs/cli/settings.md @@ -85,6 +85,15 @@ they appear in the UI. | Error Verbosity | `ui.errorVerbosity` | Controls whether recoverable errors are hidden (low) or fully shown (full). | `"low"` | | Screen Reader Mode | `ui.accessibility.screenReader` | Render output in plain-text to be more screen reader accessible | `false` | +### Voice + +| UI Label | Setting | Description | Default | +| --------------------------- | ------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------- | +| Enable Voice Input | `voice.enabled` | Enable voice input support. Note: Voice input is not natively supported in WSL2 (Windows Subsystem for Linux). | `false` | +| Transcription Backend | `voice.provider` | Set transcription backend: "gemini" (default, zero-install) or "whisper" (local). | `"gemini"` | +| Whisper Binary Path | `voice.whisperPath` | Path to the whisper executable. Only used when provider is "whisper". | `undefined` | +| Silence Detection Threshold | `voice.silenceThreshold` | RMS energy threshold (0–1000) below which audio is discarded as silence. Lower values allow quieter speech such as whispering. 0 disables silence detection. | `80` | + ### IDE | UI Label | Setting | Description | Default | diff --git a/docs/reference/commands.md b/docs/reference/commands.md index f7f8692e38b..74f1d820345 100644 --- a/docs/reference/commands.md +++ b/docs/reference/commands.md @@ -255,6 +255,31 @@ Slash commands provide meta-level control over the CLI itself. - **Description:** List configured MCP servers and tools with descriptions and schemas. +### `/voice` + +- **Description:** Manage voice input configuration and inspect current voice + settings. +- **Shortcuts:** Press `Space Space` on an empty prompt to start or stop + recording. Press **Esc** while recording to cancel. +- **Sub-commands:** + - **`enable`**: + - **Description:** Enable voice input. + - **`disable`**: + - **Description:** Disable voice input. + - **`provider [gemini|whisper]`**: + - **Description:** Set the transcription backend. + - **`sensitivity <0-1000>`**: + - **Description:** Set the silence detection threshold. `0` disables silence + filtering. + - **`set-path `**: + - **Description:** Set the path to the Whisper binary when using the local + Whisper backend. + - **`help`**: + - **Description:** Show voice command help. + - **Default behavior:** + - **Description:** Running `/voice` with no sub-command shows the current + voice settings. + ### `/memory` - **Description:** Manage the AI's instructional context (hierarchical memory diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 7bdd43997e8..5bda65ddd13 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -409,6 +409,30 @@ their corresponding top-level category object in your `settings.json` file. - **Default:** `false` - **Requires restart:** Yes +#### `voice` + +- **`voice.enabled`** (boolean): + - **Description:** Enable voice input support. Note: Voice input is not + natively supported in WSL2 (Windows Subsystem for Linux). + - **Default:** `false` + +- **`voice.provider`** (enum): + - **Description:** Set transcription backend: "gemini" (default, zero-install) + or "whisper" (local). + - **Default:** `"gemini"` + - **Values:** `"gemini"`, `"whisper"` + +- **`voice.whisperPath`** (string): + - **Description:** Path to the whisper executable. Only used when provider is + "whisper". + - **Default:** `undefined` + +- **`voice.silenceThreshold`** (number): + - **Description:** RMS energy threshold (0–1000) below which audio is + discarded as silence. Lower values allow quieter speech such as whispering. + 0 disables silence detection. + - **Default:** `80` + #### `ide` - **`ide.enabled`** (boolean): diff --git a/package-lock.json b/package-lock.json index 9ced540f9aa..744e2ce5f35 100644 --- a/package-lock.json +++ b/package-lock.json @@ -449,7 +449,8 @@ "version": "2.11.0", "resolved": "https://registry.npmjs.org/@bufbuild/protobuf/-/protobuf-2.11.0.tgz", "integrity": "sha512-sBXGT13cpmPR5BMgHE6UEEfEaShh5Ror6rfN3yEK5si7QVrtZg8LEPQb0VVhiLRUslD2yLnXtnRzG035J/mZXQ==", - "license": "(Apache-2.0 AND BSD-3-Clause)" + "license": "(Apache-2.0 AND BSD-3-Clause)", + "peer": true }, "node_modules/@bundled-es-modules/cookie": { "version": "2.0.1", @@ -1535,6 +1536,7 @@ "resolved": "https://registry.npmjs.org/@grpc/grpc-js/-/grpc-js-1.13.4.tgz", "integrity": "sha512-GsFaMXCkMqkKIvwCQjCrwH+GHbPKBjhwo/8ZuUkWHqbI73Kky9I+pQltrlT0+MWpedCoosda53lgjYfyEPgxBg==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@grpc/proto-loader": "^0.7.13", "@js-sdsl/ordered-map": "^4.4.2" @@ -2212,6 +2214,7 @@ "integrity": "sha512-t54CUOsFMappY1Jbzb7fetWeO0n6K0k/4+/ZpkS+3Joz8I4VcvY9OiEBFRYISqaI2fq5sCiPtAjRDOzVYG8m+Q==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@octokit/auth-token": "^6.0.0", "@octokit/graphql": "^9.0.2", @@ -2392,6 +2395,7 @@ "resolved": "https://registry.npmjs.org/@opentelemetry/api/-/api-1.9.0.tgz", "integrity": "sha512-3giAOQvZiH5F9bMlMiv8+GSPMeqg0dbaeo58/0SlA9sxSqZhnUtxzX9/2FzyhS9sWQf5S0GJE0AKBrFqjpeYcg==", "license": "Apache-2.0", + "peer": true, "engines": { "node": ">=8.0.0" } @@ -2441,6 +2445,7 @@ "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.5.0.tgz", "integrity": "sha512-ka4H8OM6+DlUhSAZpONu0cPBtPPTQKxbxVzC4CzVx5+K4JnroJVBtDzLAMx4/3CDTJXRvVFhpFjtl4SaiTNoyQ==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, @@ -2815,6 +2820,7 @@ "resolved": "https://registry.npmjs.org/@opentelemetry/resources/-/resources-2.5.0.tgz", "integrity": "sha512-F8W52ApePshpoSrfsSk1H2yJn9aKjCrbpQF1M9Qii0GHzbfVeFUB+rc3X4aggyZD8x9Gu3Slua+s6krmq6Dt8g==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@opentelemetry/core": "2.5.0", "@opentelemetry/semantic-conventions": "^1.29.0" @@ -2848,6 +2854,7 @@ "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-metrics/-/sdk-metrics-2.5.0.tgz", "integrity": "sha512-BeJLtU+f5Gf905cJX9vXFQorAr6TAfK3SPvTFqP+scfIpDQEJfRaGJWta7sJgP+m4dNtBf9y3yvBKVAZZtJQVA==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@opentelemetry/core": "2.5.0", "@opentelemetry/resources": "2.5.0" @@ -2902,6 +2909,7 @@ "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-trace-base/-/sdk-trace-base-2.5.0.tgz", "integrity": "sha512-VzRf8LzotASEyNDUxTdaJ9IRJ1/h692WyArDBInf5puLCjxbICD6XkHgpuudis56EndyS7LYFmtTMny6UABNdQ==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@opentelemetry/core": "2.5.0", "@opentelemetry/resources": "2.5.0", @@ -4139,6 +4147,7 @@ "integrity": "sha512-6mDvHUFSjyT2B2yeNx2nUgMxh9LtOWvkhIU3uePn2I2oyNymUAX1NIsdgviM4CH+JSrp2D2hsMvJOkxY+0wNRA==", "devOptional": true, "license": "MIT", + "peer": true, "dependencies": { "csstype": "^3.0.2" } @@ -4412,6 +4421,7 @@ "integrity": "sha512-/Zb/xaIDfxeJnvishjGdcR4jmr7S+bda8PKNhRGdljDM+elXhlvN0FyPSsMnLmJUrVG9aPO6dof80wjMawsASg==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@typescript-eslint/scope-manager": "8.58.2", "@typescript-eslint/types": "8.58.2", @@ -5187,6 +5197,7 @@ "resolved": "https://registry.npmjs.org/acorn/-/acorn-8.15.0.tgz", "integrity": "sha512-NZyJarBfL7nWwIq+FDL6Zp/yHEhePMNnnJ0y3qfieCrmNvYct8uvtiV41UvlSe6apAfk0fY1FbWx+NwfmpvtTg==", "license": "MIT", + "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -7304,7 +7315,8 @@ "version": "0.0.1581282", "resolved": "https://registry.npmjs.org/devtools-protocol/-/devtools-protocol-0.0.1581282.tgz", "integrity": "sha512-nv7iKtNZQshSW2hKzYNr46nM/Cfh5SEvE2oV0/SEGgc9XupIY5ggf84Cz8eJIkBce7S3bmTAauFD6aysMpnqsQ==", - "license": "BSD-3-Clause" + "license": "BSD-3-Clause", + "peer": true }, "node_modules/dezalgo": { "version": "1.0.4", @@ -7889,6 +7901,7 @@ "integrity": "sha512-GsGizj2Y1rCWDu6XoEekL3RLilp0voSePurjZIkxL3wlm5o5EC9VpgaP7lrCvjnkuLvzFBQWB3vWB3K5KQTveQ==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@eslint-community/eslint-utils": "^4.2.0", "@eslint-community/regexpp": "^4.12.1", @@ -8499,6 +8512,7 @@ "resolved": "https://registry.npmjs.org/express/-/express-5.2.1.tgz", "integrity": "sha512-hIS4idWWai69NezIdRt2xFVofaF4j+6INOpJlVOLDO8zXGpUVEVzIYk12UUi2JzjEzWL3IOAxcTubgz9Po0yXw==", "license": "MIT", + "peer": true, "dependencies": { "accepts": "^2.0.0", "body-parser": "^2.2.1", @@ -9765,6 +9779,7 @@ "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.12.tgz", "integrity": "sha512-p1JfQMKaceuCbpJKAPKVqyqviZdS0eUxH9v82oWo1kb9xjQ5wA6iP3FNVAPDFlz5/p7d45lO+BpSk1tuSZMF4Q==", "license": "MIT", + "peer": true, "engines": { "node": ">=16.9.0" } @@ -10024,6 +10039,7 @@ "resolved": "https://registry.npmjs.org/@jrichman/ink/-/ink-6.6.9.tgz", "integrity": "sha512-RL9sSiLQZECnjbmBwjIHOp8yVGdWF7C/uifg7ISv/e+F3nLNsfl7FdUFQs8iZARFMJAYxMFpxW6OW+HSt9drwQ==", "license": "MIT", + "peer": true, "dependencies": { "ansi-escapes": "^7.0.0", "ansi-styles": "^6.2.3", @@ -13799,6 +13815,7 @@ "resolved": "https://registry.npmjs.org/react/-/react-19.2.4.tgz", "integrity": "sha512-9nfp2hYpCwOjAN+8TZFGhtWEwgvWHXqESH8qT89AT/lWklpLON22Lc8pEtnpsZz7VmawabSU0gCjnj8aC0euHQ==", "license": "MIT", + "peer": true, "engines": { "node": ">=0.10.0" } @@ -13809,6 +13826,7 @@ "integrity": "sha512-ePrwPfxAnB+7hgnEr8vpKxL9cmnp7F322t8oqcPshbIQQhDKgFDW4tjhF2wjVbdXF9O/nyuy3sQWd9JGpiLPvA==", "devOptional": true, "license": "MIT", + "peer": true, "dependencies": { "shell-quote": "^1.6.1", "ws": "^7" @@ -15961,6 +15979,7 @@ "resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz", "integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==", "license": "MIT", + "peer": true, "engines": { "node": ">=12" }, @@ -16183,7 +16202,8 @@ "version": "2.8.1", "resolved": "https://registry.npmjs.org/tslib/-/tslib-2.8.1.tgz", "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==", - "license": "0BSD" + "license": "0BSD", + "peer": true }, "node_modules/tsx": { "version": "4.20.3", @@ -16191,6 +16211,7 @@ "integrity": "sha512-qjbnuR9Tr+FJOMBqJCW5ehvIo/buZq7vH7qD7JziU98h6l3qGy0a/yPFjwO+y0/T7GFpNgNAvEcPPVfyT8rrPQ==", "devOptional": true, "license": "MIT", + "peer": true, "dependencies": { "esbuild": "~0.25.0", "get-tsconfig": "^4.7.5" @@ -16356,6 +16377,7 @@ "integrity": "sha512-p1diW6TqL9L07nNxvRMM7hMMw4c5XOo/1ibL4aAIGmSAt9slTE1Xgw5KWuof2uTOvCg9BY7ZRi+GaF+7sfgPeQ==", "devOptional": true, "license": "Apache-2.0", + "peer": true, "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" @@ -16842,6 +16864,7 @@ "resolved": "https://registry.npmjs.org/vite/-/vite-7.3.2.tgz", "integrity": "sha512-Bby3NOsna2jsjfLVOHKes8sGwgl4TT0E6vvpYgnAYDIF/tie7MRaFthmKuHx1NSXjiTueXH3do80FMQgvEktRg==", "license": "MIT", + "peer": true, "dependencies": { "esbuild": "^0.27.0", "fdir": "^6.5.0", @@ -17412,6 +17435,7 @@ "resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz", "integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==", "license": "MIT", + "peer": true, "engines": { "node": ">=12" }, @@ -17424,6 +17448,7 @@ "resolved": "https://registry.npmjs.org/vitest/-/vitest-3.2.4.tgz", "integrity": "sha512-LUCP5ev3GURDysTWiP47wRRUpLKMOfPh+yKTx3kVIEiu5KOMeqzpnYNsKyOoVrULivR8tLcks4+lga33Whn90A==", "license": "MIT", + "peer": true, "dependencies": { "@types/chai": "^5.2.2", "@vitest/expect": "3.2.4", @@ -18062,6 +18087,7 @@ "resolved": "https://registry.npmjs.org/zod/-/zod-3.25.76.tgz", "integrity": "sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==", "license": "MIT", + "peer": true, "funding": { "url": "https://github.com/sponsors/colinhacks" } @@ -18498,6 +18524,7 @@ "resolved": "https://registry.npmjs.org/@grpc/grpc-js/-/grpc-js-1.14.3.tgz", "integrity": "sha512-Iq8QQQ/7X3Sac15oB6p0FmUg/klxQvXLeileoqrTRGJYLV+/9tubbr9ipz0GKHjmXVsgFPo/+W+2cA8eNcR+XA==", "license": "Apache-2.0", + "peer": true, "dependencies": { "@grpc/proto-loader": "^0.8.0", "@js-sdsl/ordered-map": "^4.4.2" @@ -18616,6 +18643,7 @@ "resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz", "integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==", "license": "MIT", + "peer": true, "engines": { "node": ">=12" }, diff --git a/packages/cli/src/config/settingsSchema.ts b/packages/cli/src/config/settingsSchema.ts index 5df30a20a5d..24c00be7450 100644 --- a/packages/cli/src/config/settingsSchema.ts +++ b/packages/cli/src/config/settingsSchema.ts @@ -902,6 +902,64 @@ const SETTINGS_SCHEMA = { }, }, + voice: { + type: 'object', + label: 'Voice Input', + category: 'General', + requiresRestart: false, + default: {}, + description: + 'Settings for voice input. Note: Voice input is not natively supported in WSL2 (Windows Subsystem for Linux).', + showInDialog: false, + properties: { + enabled: { + type: 'boolean', + label: 'Enable Voice Input', + category: 'General', + requiresRestart: false, + default: false, + description: + 'Enable voice input support. Note: Voice input is not natively supported in WSL2 (Windows Subsystem for Linux).', + showInDialog: true, + }, + provider: { + type: 'enum', + label: 'Transcription Backend', + category: 'General', + requiresRestart: false, + default: 'gemini', + description: + 'Set transcription backend: "gemini" (default, zero-install) or "whisper" (local).', + showInDialog: true, + options: [ + { value: 'gemini', label: 'Gemini (Cloud)' }, + { value: 'whisper', label: 'Whisper (Local)' }, + ], + }, + whisperPath: { + type: 'string', + label: 'Whisper Binary Path', + category: 'General', + requiresRestart: false, + default: undefined as string | undefined, + description: + 'Path to the whisper executable. Only used when provider is "whisper".', + showInDialog: true, + }, + silenceThreshold: { + type: 'number', + label: 'Silence Detection Threshold', + category: 'General', + requiresRestart: false, + default: 80, + description: + 'RMS energy threshold (0–1000) below which audio is discarded as silence. ' + + 'Lower values allow quieter speech such as whispering. 0 disables silence detection.', + showInDialog: true, + }, + }, + }, + ide: { type: 'object', label: 'IDE', diff --git a/packages/cli/src/nonInteractiveCli.test.ts b/packages/cli/src/nonInteractiveCli.test.ts index 1167bbbce44..bc5ed5eede0 100644 --- a/packages/cli/src/nonInteractiveCli.test.ts +++ b/packages/cli/src/nonInteractiveCli.test.ts @@ -17,6 +17,7 @@ import { ToolErrorType, GeminiEventType, OutputFormat, + JsonStreamEventType, uiTelemetryService, FatalInputError, CoreEvent, @@ -24,6 +25,7 @@ import { } from '@google/gemini-cli-core'; import type { Part } from '@google/genai'; import { runNonInteractive } from './nonInteractiveCli.js'; +import { voiceCommand } from './ui/commands/voiceCommand.js'; import { describe, it, @@ -1004,7 +1006,10 @@ describe('runNonInteractive', () => { nonInteractiveCliCommands, 'handleSlashCommand', ); - handleSlashCommandSpy.mockResolvedValue([{ text: 'Slash command output' }]); + handleSlashCommandSpy.mockResolvedValue({ + kind: 'submit_prompt', + content: [{ text: 'Slash command output' }], + }); const events: ServerGeminiStreamEvent[] = [ { type: GeminiEventType.Content, value: 'Response to slash command' }, @@ -1042,6 +1047,120 @@ describe('runNonInteractive', () => { handleSlashCommandSpy.mockRestore(); }); + it('should write slash command messages in text mode without invoking the model', async () => { + const mockCommand = { + name: 'testcommand', + description: 'a test command', + action: vi.fn().mockResolvedValue({ + type: 'message', + messageType: 'info', + content: 'Slash command message', + }), + }; + mockGetCommands.mockReturnValue([mockCommand]); + + await runNonInteractive({ + config: mockConfig, + settings: mockSettings, + input: '/testcommand', + prompt_id: 'prompt-id-slash-message-text', + }); + + expect(mockGeminiClient.sendMessageStream).not.toHaveBeenCalled(); + expect(getWrittenOutput()).toBe('[INFO] Slash command message\n'); + }); + + it('should write slash command messages in JSON mode without invoking the model', async () => { + const mockCommand = { + name: 'testcommand', + description: 'a test command', + action: vi.fn().mockResolvedValue({ + type: 'message', + messageType: 'info', + content: 'Slash command message', + }), + }; + mockGetCommands.mockReturnValue([mockCommand]); + vi.mocked(mockConfig.getOutputFormat).mockReturnValue(OutputFormat.JSON); + vi.mocked(uiTelemetryService.getMetrics).mockReturnValue( + MOCK_SESSION_METRICS, + ); + + await runNonInteractive({ + config: mockConfig, + settings: mockSettings, + input: '/testcommand', + prompt_id: 'prompt-id-slash-message-json', + }); + + expect(mockGeminiClient.sendMessageStream).not.toHaveBeenCalled(); + expect(getWrittenOutput()).toBe( + JSON.stringify( + { + session_id: 'test-session-id', + response: 'Slash command message', + stats: MOCK_SESSION_METRICS, + }, + null, + 2, + ), + ); + }); + + it('should write slash command messages in stream-json mode without invoking the model', async () => { + const mockCommand = { + name: 'testcommand', + description: 'a test command', + action: vi.fn().mockResolvedValue({ + type: 'message', + messageType: 'info', + content: 'Slash command message', + }), + }; + mockGetCommands.mockReturnValue([mockCommand]); + vi.mocked(mockConfig.getOutputFormat).mockReturnValue( + OutputFormat.STREAM_JSON, + ); + vi.mocked(uiTelemetryService.getMetrics).mockReturnValue( + MOCK_SESSION_METRICS, + ); + + await runNonInteractive({ + config: mockConfig, + settings: mockSettings, + input: '/testcommand', + prompt_id: 'prompt-id-slash-message-stream', + }); + + expect(mockGeminiClient.sendMessageStream).not.toHaveBeenCalled(); + + const lines = getWrittenOutput() + .trim() + .split('\n') + .map((line) => JSON.parse(line) as Record); + + expect(lines).toHaveLength(4); + expect(lines[0]).toMatchObject({ + type: JsonStreamEventType.INIT, + session_id: 'test-session-id', + model: 'test-model', + }); + expect(lines[1]).toMatchObject({ + type: JsonStreamEventType.MESSAGE, + role: 'user', + content: '/testcommand', + }); + expect(lines[2]).toMatchObject({ + type: JsonStreamEventType.MESSAGE, + role: 'assistant', + content: 'Slash command message', + }); + expect(lines[3]).toMatchObject({ + type: JsonStreamEventType.RESULT, + status: 'success', + }); + }); + it('should handle cancellation (Ctrl+C)', async () => { // Mock isTTY and setRawMode safely const originalIsTTY = process.stdin.isTTY; @@ -1277,6 +1396,45 @@ describe('runNonInteractive', () => { expect(getWrittenOutput()).toBe('Acknowledged\n'); }); + it('should print /voice status in non-interactive mode', async () => { + mockSettings.merged.voice = { + enabled: true, + provider: 'whisper', + whisperPath: '/usr/local/bin/whisper', + silenceThreshold: 0, + } as never; + mockGetCommands.mockReturnValue([voiceCommand]); + + await runNonInteractive({ + config: mockConfig, + settings: mockSettings, + input: '/voice', + prompt_id: 'prompt-id-voice-status', + }); + + expect(getWrittenOutput()).toContain('Voice Settings:'); + expect(getWrittenOutput()).toContain('- Provider: whisper'); + expect(getWrittenOutput()).toContain( + '- Whisper Path: /usr/local/bin/whisper', + ); + }); + + it('should print /voice help in non-interactive mode', async () => { + mockGetCommands.mockReturnValue([voiceCommand]); + + await runNonInteractive({ + config: mockConfig, + settings: mockSettings, + input: '/voice help', + prompt_id: 'prompt-id-voice-help', + }); + + expect(getWrittenOutput()).toContain('Voice Input Help:'); + expect(getWrittenOutput()).toContain( + 'Space Space (on empty prompt): Start/Stop recording', + ); + }); + it('should instantiate CommandService with correct loaders for slash commands', async () => { // This test indirectly checks that handleSlashCommand is using the right loaders. const { FileCommandLoader } = await import( diff --git a/packages/cli/src/nonInteractiveCli.ts b/packages/cli/src/nonInteractiveCli.ts index 8db512f56dc..64a592777fd 100644 --- a/packages/cli/src/nonInteractiveCli.ts +++ b/packages/cli/src/nonInteractiveCli.ts @@ -85,6 +85,52 @@ export async function runNonInteractive( const { stdout: workingStdout } = createWorkingStdio(); const textOutput = new TextOutput(workingStdout); + const writeSlashCommandMessage = ( + messageType: 'info' | 'error', + content: string, + streamFormatter: StreamJsonFormatter | null, + ) => { + const timestamp = new Date().toISOString(); + if (streamFormatter) { + streamFormatter.emitEvent({ + type: JsonStreamEventType.MESSAGE, + timestamp, + role: 'user', + content: input, + }); + streamFormatter.emitEvent({ + type: JsonStreamEventType.MESSAGE, + timestamp, + role: 'assistant', + content, + }); + const metrics = uiTelemetryService.getMetrics(); + const durationMs = Date.now() - startTime; + streamFormatter.emitEvent({ + type: JsonStreamEventType.RESULT, + timestamp, + status: 'success', + stats: streamFormatter.convertToStreamStats(metrics, durationMs), + }); + return; + } + + if (config.getOutputFormat() === OutputFormat.JSON) { + const formatter = new JsonFormatter(); + const stats = uiTelemetryService.getMetrics(); + textOutput.write( + formatter.format(config.getSessionId(), content, stats), + ); + return; + } + + const output = `[${messageType.toUpperCase()}] ${content}\n`; + if (messageType === 'error') { + process.stderr.write(output); + } else { + textOutput.write(output); + } + }; const handleUserFeedback = (payload: UserFeedbackPayload) => { const prefix = payload.severity.toUpperCase(); @@ -252,12 +298,18 @@ export async function runNonInteractive( config, settings, ); - // If a slash command is found and returns a prompt, use it. - // Otherwise, slashCommandResult falls through to the default prompt - // handling. - if (slashCommandResult) { + if (slashCommandResult.kind === 'submit_prompt') { // eslint-disable-next-line @typescript-eslint/no-unsafe-type-assertion - query = slashCommandResult as Part[]; + query = slashCommandResult.content as Part[]; + } else if (slashCommandResult.kind === 'message') { + writeSlashCommandMessage( + slashCommandResult.messageType, + slashCommandResult.content, + streamFormatter, + ); + return; + } else if (slashCommandResult.kind === 'handled') { + return; } } diff --git a/packages/cli/src/nonInteractiveCliAgentSession.test.ts b/packages/cli/src/nonInteractiveCliAgentSession.test.ts index 923109643cb..24927389694 100644 --- a/packages/cli/src/nonInteractiveCliAgentSession.test.ts +++ b/packages/cli/src/nonInteractiveCliAgentSession.test.ts @@ -1131,7 +1131,10 @@ describe('runNonInteractive', () => { nonInteractiveCliCommands, 'handleSlashCommand', ); - handleSlashCommandSpy.mockResolvedValue([{ text: 'Slash command output' }]); + handleSlashCommandSpy.mockResolvedValue({ + kind: 'submit_prompt', + content: [{ text: 'Slash command output' }], + }); const events: ServerGeminiStreamEvent[] = [ { type: GeminiEventType.Content, value: 'Response to slash command' }, diff --git a/packages/cli/src/nonInteractiveCliAgentSession.ts b/packages/cli/src/nonInteractiveCliAgentSession.ts index 0cf16da47d6..b7eff678ab1 100644 --- a/packages/cli/src/nonInteractiveCliAgentSession.ts +++ b/packages/cli/src/nonInteractiveCliAgentSession.ts @@ -253,9 +253,21 @@ export async function runNonInteractive({ config, settings, ); - if (slashCommandResult) { - // eslint-disable-next-line @typescript-eslint/no-unsafe-type-assertion - query = slashCommandResult as Part[]; + if (slashCommandResult.kind === 'submit_prompt') { + query = Array.isArray(slashCommandResult.content) + ? // eslint-disable-next-line @typescript-eslint/no-unsafe-type-assertion + (slashCommandResult.content as Part[]) + : // eslint-disable-next-line @typescript-eslint/no-unsafe-type-assertion + [{ text: slashCommandResult.content as string }]; + } else if (slashCommandResult.kind === 'message') { + if (slashCommandResult.messageType === 'error') { + throw new FatalInputError(slashCommandResult.content); + } + // eslint-disable-next-line no-console + console.log(slashCommandResult.content); + return; + } else if (slashCommandResult.kind === 'handled') { + return; } } diff --git a/packages/cli/src/nonInteractiveCliCommands.ts b/packages/cli/src/nonInteractiveCliCommands.ts index 35cf5105ab3..c0f65d0a7e1 100644 --- a/packages/cli/src/nonInteractiveCliCommands.ts +++ b/packages/cli/src/nonInteractiveCliCommands.ts @@ -21,11 +21,28 @@ import { createNonInteractiveUI } from './ui/noninteractive/nonInteractiveUi.js' import type { LoadedSettings } from './config/settings.js'; import type { SessionStatsState } from './ui/contexts/SessionContext.js'; +export type NonInteractiveSlashCommandResult = + | { + kind: 'submit_prompt'; + content: PartListUnion; + } + | { + kind: 'message'; + messageType: 'info' | 'error'; + content: string; + } + | { + kind: 'handled'; + } + | { + kind: 'unhandled'; + }; + /** * Processes a slash command in a non-interactive environment. * - * @returns A Promise that resolves to `PartListUnion` if a valid command is - * found and results in a prompt, or `undefined` otherwise. + * @returns A structured result indicating whether the slash command was + * handled, should fall back to prompt submission, or produced output. * @throws {FatalInputError} if the command result is not supported in * non-interactive mode. */ @@ -34,10 +51,10 @@ export const handleSlashCommand = async ( abortController: AbortController, config: Config, settings: LoadedSettings, -): Promise => { +): Promise => { const trimmed = rawQuery.trim(); if (!trimmed.startsWith('/')) { - return; + return { kind: 'unhandled' }; } const commandService = await CommandService.create( @@ -52,62 +69,69 @@ export const handleSlashCommand = async ( const { commandToExecute, args } = parseSlashCommand(rawQuery, commands); - if (commandToExecute) { - if (commandToExecute.action) { - // Not used by custom commands but may be in the future. - const sessionStats: SessionStatsState = { - sessionId: config?.getSessionId(), - sessionStartTime: new Date(), - metrics: uiTelemetryService.getMetrics(), - lastPromptTokenCount: 0, - promptCount: 1, - }; + if (!commandToExecute) { + return { kind: 'unhandled' }; + } - const logger = new Logger(config?.getSessionId() || '', config?.storage); + if (commandToExecute.action) { + // Not used by custom commands but may be in the future. + const sessionStats: SessionStatsState = { + sessionId: config?.getSessionId(), + sessionStartTime: new Date(), + metrics: uiTelemetryService.getMetrics(), + lastPromptTokenCount: 0, + promptCount: 1, + }; + const logger = new Logger(config?.getSessionId() || '', config?.storage); - const commandContext: CommandContext = { - services: { - agentContext: config, - settings, - git: undefined, - logger, - }, - ui: createNonInteractiveUI(), - session: { - stats: sessionStats, - sessionShellAllowlist: new Set(), - }, - invocation: { - raw: trimmed, - name: commandToExecute.name, - args, - }, - }; + const commandContext: CommandContext = { + services: { + agentContext: config, + settings, + git: undefined, + logger, + }, + ui: createNonInteractiveUI(), + session: { + stats: sessionStats, + sessionShellAllowlist: new Set(), + }, + invocation: { + raw: trimmed, + name: commandToExecute.name, + args, + }, + }; - const result = await commandToExecute.action(commandContext, args); + const result = await commandToExecute.action(commandContext, args); - if (result) { - switch (result.type) { - case 'submit_prompt': - return result.content; - case 'confirm_shell_commands': - // This result indicates a command attempted to confirm shell commands. - // However note that currently, ShellTool is excluded in non-interactive - // mode unless 'YOLO mode' is active, so confirmation actually won't - // occur because of YOLO mode. - // This ensures that if a command *does* request confirmation (e.g. - // in the future with more granular permissions), it's handled appropriately. - throw new FatalInputError( - 'Exiting due to a confirmation prompt requested by the command.', - ); - default: - throw new FatalInputError( - 'Exiting due to command result that is not supported in non-interactive mode.', - ); - } + if (result) { + switch (result.type) { + case 'submit_prompt': + return { kind: 'submit_prompt', content: result.content }; + case 'message': + return { + kind: 'message', + messageType: result.messageType, + content: result.content, + }; + case 'confirm_shell_commands': + // This result indicates a command attempted to confirm shell commands. + // However note that currently, ShellTool is excluded in non-interactive + // mode unless 'YOLO mode' is active, so confirmation actually won't + // occur because of YOLO mode. + // This ensures that if a command *does* request confirmation (e.g. + // in the future with more granular permissions), it's handled appropriately. + throw new FatalInputError( + 'Exiting due to a confirmation prompt requested by the command.', + ); + default: + throw new FatalInputError( + 'Exiting due to command result that is not supported in non-interactive mode.', + ); } } } - return; + return { kind: 'handled' }; }; diff --git a/packages/cli/src/services/BuiltinCommandLoader.ts b/packages/cli/src/services/BuiltinCommandLoader.ts index 1c5288707c7..60a319f78a9 100644 --- a/packages/cli/src/services/BuiltinCommandLoader.ts +++ b/packages/cli/src/services/BuiltinCommandLoader.ts @@ -60,9 +60,9 @@ import { tasksCommand } from '../ui/commands/tasksCommand.js'; import { vimCommand } from '../ui/commands/vimCommand.js'; import { setupGithubCommand } from '../ui/commands/setupGithubCommand.js'; import { terminalSetupCommand } from '../ui/commands/terminalSetupCommand.js'; +import { voiceCommand } from '../ui/commands/voiceCommand.js'; import { upgradeCommand } from '../ui/commands/upgradeCommand.js'; import { gemmaStatusCommand } from '../ui/commands/gemmaStatusCommand.js'; -import { voiceCommand } from '../ui/commands/voiceCommand.js'; /** * Loads the core, hard-coded slash commands that are an integral part @@ -78,7 +78,10 @@ export class BuiltinCommandLoader implements ICommandLoader { * @param _signal An AbortSignal (unused for this synchronous loader). * @returns A promise that resolves to an array of `SlashCommand` objects. */ - async loadCommands(_signal: AbortSignal): Promise { + async loadCommands(signal: AbortSignal): Promise { + if (signal.aborted) { + return []; + } const handle = startupProfiler.start('load_builtin_commands'); const isNightlyBuild = await isNightly(process.cwd()); @@ -228,7 +231,7 @@ export class BuiltinCommandLoader implements ICommandLoader { vimCommand, setupGithubCommand, terminalSetupCommand, - ...(this.config?.isVoiceModeEnabled() ? [voiceCommand] : []), + voiceCommand, ...(this.config?.getContentGeneratorConfig()?.authType === AuthType.LOGIN_WITH_GOOGLE ? [upgradeCommand] diff --git a/packages/cli/src/test-utils/render.tsx b/packages/cli/src/test-utils/render.tsx index 83e69d66630..8dffef79c07 100644 --- a/packages/cli/src/test-utils/render.tsx +++ b/packages/cli/src/test-utils/render.tsx @@ -41,6 +41,8 @@ import { type OverflowActions, type OverflowState, } from '../ui/contexts/OverflowContext.js'; +import { VoiceContext } from '../ui/contexts/VoiceContext.js'; +import type { VoiceInputReturn } from '../ui/hooks/useVoiceInput.js'; import { makeFakeConfig } from '@google/gemini-cli-core'; import { type Config } from '@google/gemini-cli-core'; @@ -540,6 +542,19 @@ export const mockAppState: AppState = { startupWarnings: [], }; +export const mockVoiceReturn: VoiceInputReturn = { + isEnabled: false, + state: { + isRecording: false, + isTranscribing: false, + error: null, + }, + startRecording: vi.fn(async () => {}), + stopRecording: vi.fn(async () => {}), + cancelRecording: vi.fn(async () => {}), + toggleRecording: vi.fn(async () => {}), +}; + const mockUIActions: UIActions = { handleThemeSelect: vi.fn(), closeThemeDialog: vi.fn(), @@ -632,6 +647,7 @@ export const renderWithProviders = async ( toolActions, persistentState, appState = mockAppState, + voice = mockVoiceReturn, }: { shellFocus?: boolean; settings?: LoadedSettings; @@ -652,6 +668,7 @@ export const renderWithProviders = async ( set?: typeof persistentStateMock.set; }; appState?: AppState; + voice?: VoiceInputReturn; } = {}, ): Promise => { const baseState: UIState = new Proxy( @@ -768,32 +785,34 @@ export const renderWithProviders = async ( toolActions?.toggleAllExpansion ?? vi.fn() } > - - - - - - - - {comp} - - - - - - - + + + + + + + + + {comp} + + + + + + + + diff --git a/packages/cli/src/ui/AppContainer.tsx b/packages/cli/src/ui/AppContainer.tsx index b196f8a05ac..bc3f0efb5e8 100644 --- a/packages/cli/src/ui/AppContainer.tsx +++ b/packages/cli/src/ui/AppContainer.tsx @@ -24,6 +24,7 @@ import { } from 'ink'; import { App } from './App.js'; import { AppContext } from './contexts/AppContext.js'; +import { VoiceContext } from './contexts/VoiceContext.js'; import { UIStateContext, type UIState } from './contexts/UIStateContext.js'; import { QuotaContext } from './contexts/QuotaContext.js'; import { @@ -43,6 +44,7 @@ import { } from './types.js'; import { checkPermissions } from './hooks/atCommandProcessor.js'; import { ToolActionsProvider } from './contexts/ToolActionsContext.js'; +import { toggleDevToolsPanel } from '../utils/devtoolsService.js'; import { MouseProvider } from './contexts/MouseContext.js'; import { ScrollProvider } from './contexts/ScrollProvider.js'; import { @@ -127,6 +129,7 @@ import { type LoadableSettingScope, SettingScope } from '../config/settings.js'; import { type InitializationResult } from '../core/initializer.js'; import { startAutoMemoryIfEnabled } from '../utils/autoMemory.js'; import { useFocus } from './hooks/useFocus.js'; +import { useVoiceInput, type VoiceInputConfig } from './hooks/useVoiceInput.js'; import { useKeypress, type Key } from './hooks/useKeypress.js'; import { KeypressPriority } from './contexts/KeypressContext.js'; import { Command } from './key/keyMatchers.js'; @@ -231,6 +234,27 @@ export const AppContainer = (props: AppContainerProps) => { const notificationsEnabled = isNotificationsEnabled(settings); const notificationMethod = getNotificationMethod(settings); + const rawProvider = settings.merged.voice?.provider; + const voiceConfig = useMemo( + () => ({ + provider: + rawProvider === 'gemini' || rawProvider === 'whisper' + ? rawProvider + : undefined, + whisperPath: settings.merged.voice?.whisperPath, + silenceThreshold: settings.merged.voice?.silenceThreshold, + config, + }), + [ + rawProvider, + settings.merged.voice?.whisperPath, + settings.merged.voice?.silenceThreshold, + config, + ], + ); + const voice = useVoiceInput( + settings.merged.voice?.enabled ? voiceConfig : undefined, + ); const { setOptions, dumpCurrentFrame, startRecording, stopRecording } = useContext(InkAppContext); const recordingFilenameRef = useRef(null); @@ -1005,6 +1029,7 @@ Logging in with Google... Restarting Gemini CLI to continue. }, toggleShortcutsHelp: () => setShortcutsHelpVisible((visible) => !visible), setText: stableSetText, + toggleVoice: voice.toggleRecording, }), [ setAuthState, @@ -1026,6 +1051,7 @@ Logging in with Google... Restarting Gemini CLI to continue. toggleDebugProfiler, setShortcutsHelpVisible, stableSetText, + voice, ], ); @@ -1807,6 +1833,17 @@ Logging in with Google... Restarting Gemini CLI to continue. debugLogger.log('[DEBUG] Keystroke:', JSON.stringify(key)); } + if (voice.state.isRecording || voice.state.isTranscribing) { + if (key.name === 'escape') { + void voice.cancelRecording(); + return true; + } + if (keyMatchers[Command.QUIT](key)) { + void voice.cancelRecording(); + return true; + } + } + if (shortcutsHelpVisible && isHelpDismissKey(key)) { setShortcutsHelpVisible(false); } @@ -1906,9 +1943,6 @@ Logging in with Google... Restarting Gemini CLI to continue. if (keyMatchers[Command.SHOW_ERROR_DETAILS](key)) { if (settings.merged.general.devtools) { void (async () => { - const { toggleDevToolsPanel } = await import( - '../utils/devtoolsService.js' - ); await toggleDevToolsPanel( config, showErrorDetails, @@ -1923,6 +1957,9 @@ Logging in with Google... Restarting Gemini CLI to continue. } else if (keyMatchers[Command.SHOW_FULL_TODOS](key)) { setShowFullTodos((prev) => !prev); return true; + } else if (keyMatchers[Command.VOICE_INPUT](key)) { + void voice.toggleRecording(); + return true; } else if (keyMatchers[Command.TOGGLE_MARKDOWN](key)) { setRenderMarkdown((prev) => { const newValue = !prev; @@ -2048,6 +2085,7 @@ Logging in with Google... Restarting Gemini CLI to continue. settings.merged.general.devtools, showErrorDetails, triggerExpandHint, + voice, keyMatchers, isHelpDismissKey, historyManager.history, @@ -2864,11 +2902,13 @@ Logging in with Google... Restarting Gemini CLI to continue. toggleAllExpansion={toggleAllExpansion} > - - - - - + + + + + + + diff --git a/packages/cli/src/ui/commands/types.ts b/packages/cli/src/ui/commands/types.ts index 328e8fc5e49..69c3dc4f4ea 100644 --- a/packages/cli/src/ui/commands/types.ts +++ b/packages/cli/src/ui/commands/types.ts @@ -93,6 +93,7 @@ export interface CommandContext { removeComponent: () => void; toggleBackgroundTasks: () => void; toggleShortcutsHelp: () => void; + toggleVoice: () => void; }; // Session-specific data session: { diff --git a/packages/cli/src/ui/commands/voiceCommand.test.ts b/packages/cli/src/ui/commands/voiceCommand.test.ts new file mode 100644 index 00000000000..afc655691a2 --- /dev/null +++ b/packages/cli/src/ui/commands/voiceCommand.test.ts @@ -0,0 +1,111 @@ +/** + * @license + * Copyright 2026 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { voiceCommand } from './voiceCommand.js'; +import { createMockCommandContext } from '../../test-utils/mockCommandContext.js'; +import type { CommandContext } from './types.js'; +import { MessageType } from '../types.js'; +import { SettingScope } from '../../config/settings.js'; + +describe('voiceCommand', () => { + let mockContext: CommandContext; + + beforeEach(() => { + mockContext = createMockCommandContext({ + ui: { + addItem: vi.fn(), + }, + } as unknown as CommandContext); + }); + + it('enables voice input via /voice enable', async () => { + if (!voiceCommand.action) { + throw new Error('voice command has no action'); + } + + const result = await voiceCommand.action(mockContext, 'enable'); + + expect(mockContext.services.settings.setValue).toHaveBeenCalledWith( + SettingScope.User, + 'voice.enabled', + true, + ); + expect(result).toEqual({ + type: 'message', + messageType: MessageType.INFO, + content: + 'Voice input enabled. Double-tap Space on an empty prompt to start recording.', + }); + }); + + it('switches to whisper via /voice provider whisper', async () => { + if (!voiceCommand.action) { + throw new Error('voice command has no action'); + } + + const result = await voiceCommand.action(mockContext, 'provider whisper'); + + expect(mockContext.services.settings.setValue).toHaveBeenCalledWith( + SettingScope.User, + 'voice.provider', + 'whisper', + ); + expect(result).toEqual({ + type: 'message', + messageType: MessageType.INFO, + content: 'Voice transcription backend set to: whisper', + }); + }); + + it('sets whisper path correctly via /voice set-path including spaces', async () => { + if (!voiceCommand.action) { + throw new Error('voice command has no action'); + } + + const result = await voiceCommand.action( + mockContext, + 'set-path C:\\Program Files\\whisper.exe', + ); + + expect(mockContext.services.settings.setValue).toHaveBeenCalledWith( + SettingScope.User, + 'voice.whisperPath', + 'C:\\Program Files\\whisper.exe', + ); + expect(result).toEqual({ + type: 'message', + messageType: MessageType.INFO, + content: 'Whisper binary path set to: C:\\Program Files\\whisper.exe', + }); + }); + + it('shows current voice status for bare /voice', async () => { + if (!voiceCommand.action) { + throw new Error('voice command has no action'); + } + + mockContext.services.settings.merged.voice = { + enabled: true, + provider: 'whisper', + whisperPath: '/usr/local/bin/whisper', + silenceThreshold: 0, + } as never; + + await voiceCommand.action(mockContext, ''); + + expect(mockContext.ui.addItem).toHaveBeenCalledWith( + expect.objectContaining({ + type: MessageType.VOICE_STATUS, + enabled: true, + provider: 'whisper', + sensitivityLabel: 'disabled', + whisperPath: '/usr/local/bin/whisper', + timestamp: expect.any(Date), + }), + ); + }); +}); diff --git a/packages/cli/src/ui/commands/voiceCommand.ts b/packages/cli/src/ui/commands/voiceCommand.ts index b9df28ca27f..2ca0cedffe9 100644 --- a/packages/cli/src/ui/commands/voiceCommand.ts +++ b/packages/cli/src/ui/commands/voiceCommand.ts @@ -5,17 +5,227 @@ */ import { CommandKind, type SlashCommand } from './types.js'; +import { + MessageType, + type HistoryItemVoiceHelp, + type HistoryItemVoiceStatus, +} from '../types.js'; +import { SettingScope } from '../../config/settings.js'; + +const voiceHelpCommand: SlashCommand = { + name: 'help', + description: 'Show all voice configuration commands and shortcuts', + kind: CommandKind.BUILT_IN, + action: async (context) => { + const item: Omit = { + type: MessageType.VOICE_HELP, + timestamp: new Date(), + text: 'Voice Input Help:\n- Space Space (on empty prompt): Start/Stop recording\n- Esc (while recording): Cancel\n- /voice status: Show current settings', + }; + context.ui.addItem(item); + }, +}; + +const voiceEnableCommand: SlashCommand = { + name: 'enable', + description: 'Enable voice input (double-tap Space on empty input to record)', + kind: CommandKind.BUILT_IN, + action: async (context) => { + context.services.settings.setValue( + SettingScope.User, + 'voice.enabled', + true, + ); + return { + type: 'message', + messageType: MessageType.INFO, + content: + 'Voice input enabled. Double-tap Space on an empty prompt to start recording.', + }; + }, +}; + +const voiceDisableCommand: SlashCommand = { + name: 'disable', + description: 'Disable voice input', + kind: CommandKind.BUILT_IN, + action: async (context) => { + context.services.settings.setValue( + SettingScope.User, + 'voice.enabled', + false, + ); + return { + type: 'message', + messageType: MessageType.INFO, + content: 'Voice input disabled.', + }; + }, +}; + +const voiceProviderCommand: SlashCommand = { + name: 'provider', + description: 'Set transcription backend: gemini (default) or whisper', + kind: CommandKind.BUILT_IN, + action: async (context, args) => { + const provider = args?.trim().toLowerCase(); + if (provider !== 'gemini' && provider !== 'whisper') { + return { + type: 'message', + messageType: MessageType.ERROR, + content: 'Usage: /voice provider [gemini|whisper]', + }; + } + context.services.settings.setValue( + SettingScope.User, + 'voice.provider', + provider, + ); + return { + type: 'message', + messageType: MessageType.INFO, + content: `Voice transcription backend set to: ${provider}`, + }; + }, +}; + +const voiceSensitivityCommand: SlashCommand = { + name: 'sensitivity', + description: 'Set silence threshold (0=off, 80=default, 300+=loud only)', + kind: CommandKind.BUILT_IN, + action: async (context, args) => { + const raw = args?.trim(); + const value = raw !== undefined && raw !== '' ? Number(raw) : NaN; + if (isNaN(value) || value < 0 || value > 1000) { + return { + type: 'message', + messageType: MessageType.ERROR, + content: + 'Usage: /voice sensitivity <0-1000>\n' + + ' 0 = disable silence detection (captures all audio)\n' + + ' 1-80 = sensitive (whispered speech)\n' + + ' 80 = default\n' + + ' 300+ = only loud speech', + }; + } + context.services.settings.setValue( + SettingScope.User, + 'voice.silenceThreshold', + value, + ); + const hint = + value === 0 + ? 'Silence detection disabled.' + : value <= 80 + ? 'Sensitive — whispered speech will be captured.' + : value <= 200 + ? 'Moderate — quiet speech captured.' + : 'High — only louder speech captured.'; + return { + type: 'message', + messageType: MessageType.INFO, + content: `Voice sensitivity threshold set to ${value}. ${hint}`, + }; + }, +}; + +const voiceSetPathCommand: SlashCommand = { + name: 'set-path', + description: 'Set the path to the Whisper binary', + kind: CommandKind.BUILT_IN, + action: async (context, args) => { + const path = args?.trim(); + if (!path) { + return { + type: 'message', + messageType: MessageType.ERROR, + content: 'Usage: /voice set-path ', + }; + } + context.services.settings.setValue( + SettingScope.User, + 'voice.whisperPath', + path, + ); + return { + type: 'message', + messageType: MessageType.INFO, + content: `Whisper binary path set to: ${path}`, + }; + }, +}; + +const voiceStatusCommand: SlashCommand = { + name: 'status', + description: 'Show current voice settings status', + kind: CommandKind.BUILT_IN, + action: async (context) => { + const voiceSettings = context.services.settings.merged.voice ?? {}; + const enabled = voiceSettings.enabled ?? false; + const provider = voiceSettings.provider ?? 'gemini'; + const whisperPath = voiceSettings.whisperPath ?? '(not set)'; + const threshold: number = voiceSettings.silenceThreshold ?? 80; + const sensitivityLabel = + threshold === 0 + ? 'disabled' + : threshold <= 80 + ? `${threshold} (whisper-sensitive)` + : threshold <= 300 + ? `${threshold} (moderate)` + : `${threshold} (loud speech only)`; + + const statusText = + `Voice Settings:\n` + + `- Enabled: ${enabled}\n` + + `- Provider: ${provider}\n` + + `- Sensitivity: ${sensitivityLabel}\n` + + `- Whisper Path: ${whisperPath}`; + + const statusItem: Omit = { + type: MessageType.VOICE_STATUS, + timestamp: new Date(), + enabled, + provider, + sensitivityLabel, + whisperPath, + text: statusText, + }; + context.ui.addItem(statusItem); + }, +}; export const voiceCommand: SlashCommand = { name: 'voice', altNames: [], - description: 'Toggle voice dictation mode', + description: 'Manage voice input and dictation mode', kind: CommandKind.BUILT_IN, autoExecute: true, - action: (context) => { + action: async (context, args) => { + const trimmedArgs = args?.trim() || ''; + if (trimmedArgs) { + const parts = trimmedArgs.split(/\s+/); + const subCommandName = parts[0]?.toLowerCase(); + const subCommandArgs = trimmedArgs.slice(parts[0].length).trim(); + + const subCommand = voiceCommand.subCommands?.find( + (sc) => sc.name === subCommandName, + ); + if (subCommand?.action) { + return subCommand.action(context, subCommandArgs); + } + } + + // Default: toggle voice mode (upstream behavior) context.ui.toggleVoiceMode(); }, subCommands: [ + voiceHelpCommand, + voiceEnableCommand, + voiceDisableCommand, + voiceProviderCommand, + voiceSensitivityCommand, + voiceSetPathCommand, + voiceStatusCommand, { name: 'model', description: 'Manage voice transcription models', diff --git a/packages/cli/src/ui/components/Composer.tsx b/packages/cli/src/ui/components/Composer.tsx index 52bb2b294fd..3419cf83b50 100644 --- a/packages/cli/src/ui/components/Composer.tsx +++ b/packages/cli/src/ui/components/Composer.tsx @@ -143,6 +143,7 @@ export const Composer = ({ isFocused = true }: { isFocused?: boolean }) => { {uiState.isInputActive && ( = ({ {itemForDisplay.type === 'help' && commands && ( )} + {itemForDisplay.type === 'voice_help' && } + {itemForDisplay.type === 'voice_status' && ( + + )} {itemForDisplay.type === 'stats' && ( { }); // Simulate left mouse press at calculated coordinates. - // Without left border: inner box is at x=3, y=1 based on padding(1)+prompt(2) and border-top(1). + // Without left border: inner box is at x=2, y=1 based on padding(1)+prompt(1) and border-top(1). await act(async () => { stdin.write(`\x1b[<0;${mouseCol};${mouseRow}M`); }); @@ -5076,6 +5080,7 @@ describe('InputPrompt', () => { settings: createMockSettings({ experimental: { voice: { activationMode: 'toggle' } }, }), + voice: { ...mockVoiceReturn, isEnabled: false }, }, ); diff --git a/packages/cli/src/ui/components/InputPrompt.tsx b/packages/cli/src/ui/components/InputPrompt.tsx index f69138c8c7e..b8e3c874f6b 100644 --- a/packages/cli/src/ui/components/InputPrompt.tsx +++ b/packages/cli/src/ui/components/InputPrompt.tsx @@ -57,6 +57,7 @@ import { type Config, } from '@google/gemini-cli-core'; import { useVoiceMode } from '../hooks/useVoiceMode.js'; +import { useVoiceContext } from '../contexts/VoiceContext.js'; import { parseInputForHighlighting, parseSegmentsFromTokens, @@ -284,6 +285,13 @@ export const InputPrompt: React.FC = ({ const hasUserNavigatedSuggestions = useRef(false); const listRef = useRef>(null); + const { + isEnabled: voiceEnabled, + state: voiceState, + toggleRecording, + cancelRecording, + } = useVoiceContext(); + const { isRecording, handleVoiceInput, resetTurnBaseline } = useVoiceMode({ buffer, config, @@ -1318,6 +1326,45 @@ export const InputPrompt: React.FC = ({ return false; } + // Double-space triggers voice recording. + if (!isVoiceModeEnabled && voiceEnabled && key.sequence === ' ') { + const now = Date.now(); + const delta = now - lastSpacePressRef.current; + + // If they are just holding down the spacebar, the OS will send repeated keys + // very quickly (e.g. every 30ms). We don't want to trigger voice on continuous hold. + if (delta > 50 && delta < 300) { + lastSpacePressRef.current = 0; + + // The first space was already inserted into the buffer on the previous keypress + // (because we let it pass through to keep typing feeling responsive). + // We must manually delete that first space now before starting recording. + if (buffer.text.length > 0 && buffer.text.endsWith(' ')) { + buffer.setText(buffer.text.slice(0, -1)); + } + + void toggleRecording(); + return true; // Consume the second space + } + + lastSpacePressRef.current = now; + + // Consume the first space only if the buffer is empty to avoid leading whitespace. + // If not empty, let it through so normal typing isn't delayed. + if (buffer.text.length === 0) { + return true; + } + } + + if ( + !isVoiceModeEnabled && + voiceEnabled && + keyMatchers[Command.VOICE_INPUT](key) + ) { + void toggleRecording(); + return true; + } + // Fall back to the text buffer's default input handling for all other keys const handled = buffer.handleInput(key); @@ -1388,6 +1435,9 @@ export const InputPrompt: React.FC = ({ isHelpDismissKey, settings, handleVoiceInput, + voiceEnabled, + toggleRecording, + isVoiceModeEnabled, ], ); useKeypress(handleInput, { @@ -1398,6 +1448,12 @@ export const InputPrompt: React.FC = ({ const [cursorVisualRowAbsolute, cursorVisualColAbsolute] = buffer.visualCursor; + const effectivePlaceholder = voiceState.isRecording + ? 'Recording...' + : voiceState.isTranscribing + ? 'Transcribing...' + : placeholder; + const getGhostTextLines = useCallback(() => { if ( !completion.promptCompletion.text || @@ -1827,22 +1883,20 @@ export const InputPrompt: React.FC = ({ )} - {buffer.text.length === 0 && !isRecording ? ( - !isVoiceModeEnabled && placeholder ? ( - showCursor ? ( - - {chalk.inverse(placeholder.slice(0, 1))} - - {placeholder.slice(1)} - + {buffer.text.length === 0 && effectivePlaceholder ? ( + showCursor ? ( + + {chalk.inverse(effectivePlaceholder.slice(0, 1))} + + {effectivePlaceholder.slice(1)} - ) : ( - {placeholder} - ) - ) : null + + ) : ( + {effectivePlaceholder} + ) ) : ( ( + + + Voice Input: + + + + + + Commands: + + {[ + ['/voice', 'Show current voice settings'], + ['/voice enable', 'Enable voice input'], + ['/voice disable', 'Disable voice input'], + ['/voice provider [gemini|whisper]', 'Set transcription backend'], + ['/voice sensitivity <0-1000>', 'Set silence detection threshold'], + ['/voice set-path ', 'Set path to Whisper binary'], + ['/voice help', 'Show this help'], + ].map(([cmd, desc]) => ( + + + {' '} + {cmd} + + {' - ' + desc} + + ))} + + + + + Recording Shortcuts: + + + + Space Space + + {' (on empty input) - Start/stop recording'} + + + + Esc + + {' - Cancel recording (discards audio, no transcription)'} + + + + + + Silence Sensitivity Guide: + + {[ + ['0', 'Disable silence detection (captures all audio)'], + ['1–80', 'Sensitive — whispered speech (default: 80)'], + ['80–300', 'Moderate — quiet speech'], + ['300+', 'High — only louder speech'], + ].map(([val, desc]) => ( + + + {' '} + {val} + + {' ' + desc} + + ))} + + + + + Transcription Backends: + + + + {' '} + gemini + + {' Zero-install. Uses your existing Gemini API auth. (default)'} + + + + {' '} + whisper + + {' Local Whisper binary (faster-whisper or openai-whisper).'} + + +); diff --git a/packages/cli/src/ui/components/VoiceStatus.tsx b/packages/cli/src/ui/components/VoiceStatus.tsx new file mode 100644 index 00000000000..5a60d074241 --- /dev/null +++ b/packages/cli/src/ui/components/VoiceStatus.tsx @@ -0,0 +1,59 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import type React from 'react'; +import { Box, Text } from 'ink'; +import { theme } from '../semantic-colors.js'; +import type { HistoryItemVoiceStatus } from '../types.js'; + +interface Props { + item: HistoryItemVoiceStatus; +} + +export const VoiceStatus: React.FC = ({ item }) => ( + + + Voice Input Settings: + + + + + {[ + ['Enabled', item.enabled ? 'yes' : 'no'], + ['Provider', item.provider], + ['Sensitivity', item.sensitivityLabel], + ['Whisper path', item.whisperPath], + ].map(([label, value]) => ( + + + {' '} + {label} + + {' ' + value} + + ))} + + + + + Shortcuts:{' '} + + Space Space + + {' (empty input) to record · '} + + /voice help + + {' for all commands'} + + +); diff --git a/packages/cli/src/ui/components/__snapshots__/InputPrompt.test.tsx.snap b/packages/cli/src/ui/components/__snapshots__/InputPrompt.test.tsx.snap index db449ce4d7a..3712adf4f13 100644 --- a/packages/cli/src/ui/components/__snapshots__/InputPrompt.test.tsx.snap +++ b/packages/cli/src/ui/components/__snapshots__/InputPrompt.test.tsx.snap @@ -90,13 +90,6 @@ exports[`InputPrompt > Highlighting and Cursor Display > single-line scenarios > ────────────────────────────────────────────────────────────────────────────────────────────────────" `; -exports[`InputPrompt > History Navigation and Completion Suppression > should not render suggestions during history navigation 1`] = ` -"▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄ - > second message -▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀▀ -" -`; - exports[`InputPrompt > command search (Ctrl+R when not in shell) > expands and collapses long suggestion via Right/Left arrows > command-search-render-collapsed-match 1`] = ` "▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄▄ (r:) Type your message or @path/to/file diff --git a/packages/cli/src/ui/contexts/VoiceContext.test.tsx b/packages/cli/src/ui/contexts/VoiceContext.test.tsx new file mode 100644 index 00000000000..95b472a8d20 --- /dev/null +++ b/packages/cli/src/ui/contexts/VoiceContext.test.tsx @@ -0,0 +1,38 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { describe, it, expect, vi } from 'vitest'; +import { renderHook } from '../../test-utils/render.js'; +import { VoiceContext, useVoiceContext } from './VoiceContext.js'; +import type { VoiceInputReturn } from '../hooks/useVoiceInput.js'; +import type React from 'react'; + +describe('VoiceContext', () => { + it('should provide voice input state', async () => { + const mockVoiceInput: VoiceInputReturn = { + isEnabled: true, + state: { + isRecording: false, + isTranscribing: false, + error: null, + }, + startRecording: vi.fn(), + stopRecording: vi.fn(), + cancelRecording: vi.fn(), + toggleRecording: vi.fn(), + }; + + const wrapper = ({ children }: { children: React.ReactNode }) => ( + + {children} + + ); + + const { result } = await renderHook(() => useVoiceContext(), { wrapper }); + + expect(result.current).toBe(mockVoiceInput); + }); +}); diff --git a/packages/cli/src/ui/contexts/VoiceContext.tsx b/packages/cli/src/ui/contexts/VoiceContext.tsx new file mode 100644 index 00000000000..69136fb2a42 --- /dev/null +++ b/packages/cli/src/ui/contexts/VoiceContext.tsx @@ -0,0 +1,23 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { createContext, useContext } from 'react'; +import type { VoiceInputReturn } from '../hooks/useVoiceInput.js'; + +export const VoiceContext = createContext(null); + +export const useVoiceContext = () => { + const context = useContext(VoiceContext); + if (!context) { + throw new Error( + 'useVoiceContext must be used within a VoiceContext.Provider', + ); + } + return context; +}; + +// Re-export event subscription for convenience +export { onVoiceTranscript } from '../hooks/useVoiceInput.js'; diff --git a/packages/cli/src/ui/hooks/slashCommandProcessor.test.tsx b/packages/cli/src/ui/hooks/slashCommandProcessor.test.tsx index e4f00661897..0e6e3b79e29 100644 --- a/packages/cli/src/ui/hooks/slashCommandProcessor.test.tsx +++ b/packages/cli/src/ui/hooks/slashCommandProcessor.test.tsx @@ -218,6 +218,7 @@ describe('useSlashCommandProcessor', () => { toggleBackgroundTasks: vi.fn(), toggleShortcutsHelp: vi.fn(), setText: vi.fn(), + toggleVoice: vi.fn(), }, new Map(), // extensionsUpdateState true, // isConfigInitialized diff --git a/packages/cli/src/ui/hooks/slashCommandProcessor.ts b/packages/cli/src/ui/hooks/slashCommandProcessor.ts index 6e880ed4bba..8e4cbebd2e8 100644 --- a/packages/cli/src/ui/hooks/slashCommandProcessor.ts +++ b/packages/cli/src/ui/hooks/slashCommandProcessor.ts @@ -89,6 +89,7 @@ interface SlashCommandProcessorActions { toggleBackgroundTasks: () => void; toggleShortcutsHelp: () => void; setText: (text: string) => void; + toggleVoice: () => void; } /** @@ -175,6 +176,20 @@ export const useSlashCommandProcessor = ( type: 'help', timestamp: message.timestamp, }; + } else if (message.type === MessageType.VOICE_HELP) { + historyItemContent = { + type: 'voice_help', + timestamp: message.timestamp, + }; + } else if (message.type === MessageType.VOICE_STATUS) { + historyItemContent = { + type: 'voice_status', + timestamp: message.timestamp, + enabled: message.enabled, + provider: message.provider, + sensitivityLabel: message.sensitivityLabel, + whisperPath: message.whisperPath, + }; } else if (message.type === MessageType.STATS) { historyItemContent = { type: 'stats', @@ -247,6 +262,7 @@ export const useSlashCommandProcessor = ( removeComponent: () => setCustomDialog(null), toggleBackgroundTasks: actions.toggleBackgroundTasks, toggleShortcutsHelp: actions.toggleShortcutsHelp, + toggleVoice: actions.toggleVoice, }, session: { stats: session.stats, diff --git a/packages/cli/src/ui/hooks/useVoiceInput.log-volume.test.ts b/packages/cli/src/ui/hooks/useVoiceInput.log-volume.test.ts new file mode 100644 index 00000000000..5514a6bbe8a --- /dev/null +++ b/packages/cli/src/ui/hooks/useVoiceInput.log-volume.test.ts @@ -0,0 +1,120 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { act } from 'react'; +import { renderHook } from '../../test-utils/render.js'; +import { createMockConfig } from '../../test-utils/mockConfig.js'; +import { useVoiceInput } from './useVoiceInput.js'; +import { debugLogger } from '@google/gemini-cli-core'; +import type { VoiceBackendOptions } from '@google/gemini-cli-core'; + +// --------------------------------------------------------------------------- +// Mock @google/gemini-cli-core: provide mock backends so tests run without +// real audio recording infrastructure (sox/arecord, Gemini API). +// --------------------------------------------------------------------------- +let capturedGeminiOptions: VoiceBackendOptions | null = null; + +vi.mock('@google/gemini-cli-core', async (importOriginal) => { + const original = + await importOriginal(); + return { + ...original, + GeminiRestBackend: vi + .fn() + .mockImplementation((opts: VoiceBackendOptions) => { + capturedGeminiOptions = opts; + return { + start: vi.fn(), + stop: vi.fn(), + cleanup: vi.fn(), + }; + }), + LocalWhisperBackend: vi.fn().mockImplementation(() => ({ + start: vi.fn(), + stop: vi.fn(), + cleanup: vi.fn(), + })), + }; +}); + +const mockConfig = createMockConfig({ + getContentGenerator: vi.fn().mockReturnValue({}), +}); + +describe('useVoiceInput Log Volume', () => { + let logSpy: ReturnType; + let warnSpy: ReturnType; + let errorSpy: ReturnType; + + beforeEach(() => { + vi.clearAllMocks(); + capturedGeminiOptions = null; + + logSpy = vi.spyOn(debugLogger, 'log'); + warnSpy = vi.spyOn(debugLogger, 'warn'); + errorSpy = vi.spyOn(debugLogger, 'error'); + // We intentionally DO NOT spy on 'debug' — those logs are safe/hidden. + }); + + afterEach(() => { + logSpy.mockRestore(); + warnSpy.mockRestore(); + errorSpy.mockRestore(); + }); + + it('should remain SILENT (no visible logs) during normal recording start/stop', async () => { + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + // --- Action 1: Start Recording --- + await act(async () => { + await result.current.startRecording(); + }); + + // Starting recording should produce ZERO visible logs. + expect(logSpy).not.toHaveBeenCalled(); + expect(warnSpy).not.toHaveBeenCalled(); + expect(errorSpy).not.toHaveBeenCalled(); + + // --- Action 2: Stop Recording --- + await act(async () => { + await result.current.stopRecording(); + }); + + // Stopping should also be silent. + expect(logSpy).not.toHaveBeenCalled(); + expect(warnSpy).not.toHaveBeenCalled(); + expect(errorSpy).not.toHaveBeenCalled(); + }); + + it('should remain SILENT even when the backend emits rapid state changes', async () => { + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + await act(async () => { + await result.current.startRecording(); + }); + + // Simulate the backend emitting many intermediate state notifications + // (e.g. progress callbacks). None of these should produce visible logs. + await act(async () => { + for (let i = 0; i < 100; i++) { + void capturedGeminiOptions?.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + } + }); + + expect(logSpy).not.toHaveBeenCalled(); + expect(warnSpy).not.toHaveBeenCalled(); + expect(errorSpy).not.toHaveBeenCalled(); + }); +}); diff --git a/packages/cli/src/ui/hooks/useVoiceInput.replication.test.tsx b/packages/cli/src/ui/hooks/useVoiceInput.replication.test.tsx new file mode 100644 index 00000000000..d3669aefd7f --- /dev/null +++ b/packages/cli/src/ui/hooks/useVoiceInput.replication.test.tsx @@ -0,0 +1,150 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { describe, it, expect, vi, beforeEach } from 'vitest'; +import { act, useContext, useEffect } from 'react'; +import { renderHook } from '../../test-utils/render.js'; +import { createMockConfig } from '../../test-utils/mockConfig.js'; +import { waitFor } from '../../test-utils/async.js'; +import { useVoiceInput, onVoiceTranscript } from './useVoiceInput.js'; +import { VoiceContext } from '../contexts/VoiceContext.js'; +import type { VoiceBackendOptions } from '@google/gemini-cli-core'; +import { coreEvents } from '@google/gemini-cli-core'; + +// --------------------------------------------------------------------------- +// Mock @google/gemini-cli-core: provide mock backends so tests run without +// real audio recording infrastructure (sox/arecord, Gemini API). +// Keep real coreEvents so transcript events flow through properly. +// --------------------------------------------------------------------------- +let capturedGeminiOptions: VoiceBackendOptions | null = null; +const mockGeminiStart = vi.fn(); +const mockGeminiStop = vi.fn(); + +vi.mock('@google/gemini-cli-core', async (importOriginal) => { + const original = + await importOriginal(); + + // Extend coreEvents with emitVoiceTranscript if not already present. + const VOICE_TRANSCRIPT_EVENT = 'voice-transcript'; + const extendedCoreEvents = Object.assign(original.coreEvents, { + emitVoiceTranscript: (transcript: string) => { + (original.coreEvents as import('node:events').EventEmitter).emit( + VOICE_TRANSCRIPT_EVENT, + transcript, + ); + }, + }); + const extendedCoreEvent = { + ...original.CoreEvent, + VoiceTranscript: VOICE_TRANSCRIPT_EVENT, + }; + + return { + ...original, + coreEvents: extendedCoreEvents, + CoreEvent: extendedCoreEvent, + GeminiRestBackend: vi + .fn() + .mockImplementation((opts: VoiceBackendOptions) => { + capturedGeminiOptions = opts; + return { + start: mockGeminiStart, + stop: mockGeminiStop, + cleanup: vi.fn(), + }; + }), + LocalWhisperBackend: vi.fn().mockImplementation(() => ({ + start: vi.fn(), + stop: vi.fn(), + cleanup: vi.fn(), + })), + }; +}); + +const mockConfig = createMockConfig({ + getContentGenerator: vi.fn().mockReturnValue({}), +}); + +describe('Voice Input Full Cycle Replication', () => { + beforeEach(() => { + vi.clearAllMocks(); + capturedGeminiOptions = null; + }); + + it('should deliver transcript via event without causing excessive re-renders', async () => { + let consumerRenders = 0; + let transcriptReceived: string | null = null; + + // Provider: useVoiceInput hook + const { result: providerResult } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + // Consumer: simulates InputPrompt's event-based transcript handling + await renderHook( + () => { + consumerRenders++; + useContext(VoiceContext); + + useEffect(() => { + const unsubscribe = onVoiceTranscript((transcript) => { + transcriptReceived = transcript; + }); + return unsubscribe; + }, []); + }, + { + wrapper: ({ children }) => ( + + {children} + + ), + }, + ); + + // Reset render counter after initial setup + consumerRenders = 0; + + // Step 1: Start recording + await act(async () => { + await providerResult.current.toggleRecording(); + void capturedGeminiOptions?.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + }); + + const rendersBefore = consumerRenders; + + // Step 2: Stop recording — backend emits transcript via coreEvents + await act(async () => { + await providerResult.current.toggleRecording(); + void capturedGeminiOptions?.onStateChange({ + isRecording: false, + isTranscribing: true, + error: null, + }); + // Simulate backend emitting transcript after transcription completes + coreEvents.emitVoiceTranscript('this is a test'); + void capturedGeminiOptions?.onStateChange({ + isRecording: false, + isTranscribing: false, + error: null, + }); + }); + + // Wait for transcript event to propagate without using fixed timeouts + await waitFor(() => { + expect(transcriptReceived).toBe('this is a test'); + }); + + // With event-based delivery, the consumer renders very few times. + // It should not re-render for every state change in the backend. + const rendersAfter = consumerRenders; + expect(rendersAfter - rendersBefore).toBeLessThan(5); + }); +}); diff --git a/packages/cli/src/ui/hooks/useVoiceInput.stress.test.ts b/packages/cli/src/ui/hooks/useVoiceInput.stress.test.ts new file mode 100644 index 00000000000..068429637c4 --- /dev/null +++ b/packages/cli/src/ui/hooks/useVoiceInput.stress.test.ts @@ -0,0 +1,119 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { describe, it, expect, vi, beforeEach } from 'vitest'; +import { act } from 'react'; +import { renderHook } from '../../test-utils/render.js'; +import { createMockConfig } from '../../test-utils/mockConfig.js'; +import { useVoiceInput } from './useVoiceInput.js'; +import type { VoiceBackendOptions } from '@google/gemini-cli-core'; + +// --------------------------------------------------------------------------- +// Mock @google/gemini-cli-core: provide mock backends so tests run without +// real audio recording infrastructure (sox/arecord, Gemini API). +// --------------------------------------------------------------------------- +let capturedGeminiOptions: VoiceBackendOptions | null = null; +const mockGeminiStart = vi.fn(); +const mockGeminiStop = vi.fn(); + +vi.mock('@google/gemini-cli-core', async (importOriginal) => { + const original = + await importOriginal(); + return { + ...original, + GeminiRestBackend: vi + .fn() + .mockImplementation((opts: VoiceBackendOptions) => { + capturedGeminiOptions = opts; + return { + start: mockGeminiStart, + stop: mockGeminiStop, + cleanup: vi.fn(), + }; + }), + LocalWhisperBackend: vi.fn().mockImplementation(() => ({ + start: vi.fn(), + stop: vi.fn(), + cleanup: vi.fn(), + })), + }; +}); + +const mockConfig = createMockConfig({ + getContentGenerator: vi.fn().mockReturnValue({}), +}); + +describe('useVoiceInput Stress Tests', () => { + beforeEach(() => { + vi.clearAllMocks(); + capturedGeminiOptions = null; + }); + + it('should not cause excessive re-renders when receiving rapid state notifications', async () => { + let renderCount = 0; + const { result } = await renderHook(() => { + renderCount++; + return useVoiceInput({ config: mockConfig }); + }); + + // Reset count after initial render + renderCount = 0; + + await act(async () => { + await result.current.startRecording(); + }); + + const baseRenders = renderCount; + + // Simulate 100 rapid state notifications with the same value. + // React batching should coalesce these into at most 1 additional render. + await act(async () => { + for (let i = 0; i < 100; i++) { + void capturedGeminiOptions?.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + } + }); + + // 100 identical state updates should not cause 100 re-renders. + expect(renderCount - baseRenders).toBeLessThanOrEqual(2); + }); + + it('should not cause excessive re-renders when rapid toggleRecording calls are ignored', async () => { + let renderCount = 0; + const { result } = await renderHook(() => { + renderCount++; + return useVoiceInput({ config: mockConfig }); + }); + + // Start recording + await act(async () => { + await result.current.startRecording(); + void capturedGeminiOptions?.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + }); + + const rendersAfterStart = renderCount; + + // Simulate rapid toggleRecording calls while already recording. + // The isTogglingRef guard in the hook should drop concurrent calls. + await act(async () => { + const calls = []; + for (let i = 0; i < 50; i++) { + calls.push(result.current.toggleRecording()); + } + await Promise.all(calls); + }); + + // We expect a small, bounded number of renders — not 50. + expect(renderCount - rendersAfterStart).toBeLessThanOrEqual(5); + }); +}); diff --git a/packages/cli/src/ui/hooks/useVoiceInput.test.ts b/packages/cli/src/ui/hooks/useVoiceInput.test.ts new file mode 100644 index 00000000000..8c457a9c75c --- /dev/null +++ b/packages/cli/src/ui/hooks/useVoiceInput.test.ts @@ -0,0 +1,270 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { describe, it, expect, vi, beforeEach } from 'vitest'; +import { act } from 'react'; +import { renderHook } from '../../test-utils/render.js'; +import { createMockConfig } from '../../test-utils/mockConfig.js'; +import { useVoiceInput, onVoiceTranscript } from './useVoiceInput.js'; +import type { VoiceBackendOptions } from '@google/gemini-cli-core'; +import { coreEvents } from '@google/gemini-cli-core'; + +// --------------------------------------------------------------------------- +// Partially mock @google/gemini-cli-core: override backends, keep real +// coreEvents/CoreEvent/debugLogger so transcript events flow through properly. +// --------------------------------------------------------------------------- +const mockGeminiStart = vi.fn(); +const mockGeminiStop = vi.fn(); +const mockGeminiCancel = vi.fn(); +const mockGeminiCleanup = vi.fn(); + +const mockWhisperStart = vi.fn(); +const mockWhisperStop = vi.fn(); +const mockWhisperCancel = vi.fn(); +const mockWhisperCleanup = vi.fn(); + +// Captured options so tests can invoke onStateChange +let capturedGeminiOptions: VoiceBackendOptions | null = null; +let capturedWhisperOptions: VoiceBackendOptions | null = null; + +vi.mock('@google/gemini-cli-core', async (importOriginal) => { + const original = + await importOriginal(); + + return { + ...original, + GeminiRestBackend: vi + .fn() + .mockImplementation((opts: VoiceBackendOptions) => { + capturedGeminiOptions = opts; + return { + start: mockGeminiStart, + stop: mockGeminiStop, + cancel: mockGeminiCancel, + cleanup: mockGeminiCleanup, + }; + }), + LocalWhisperBackend: vi + .fn() + .mockImplementation((opts: VoiceBackendOptions) => { + capturedWhisperOptions = opts; + return { + start: mockWhisperStart, + stop: mockWhisperStop, + cancel: mockWhisperCancel, + cleanup: mockWhisperCleanup, + }; + }), + }; +}); + +// Minimal mock Config with a working ContentGenerator stub +const mockConfig = createMockConfig({ + getContentGenerator: vi.fn().mockReturnValue({}), +}); + +describe('useVoiceInput', () => { + beforeEach(() => { + vi.clearAllMocks(); + capturedGeminiOptions = null; + capturedWhisperOptions = null; + }); + + it('initializes with default state', async () => { + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + expect(result.current.state).toEqual({ + isRecording: false, + isTranscribing: false, + error: null, + }); + }); + + it('uses GeminiRestBackend by default', async () => { + await renderHook(() => useVoiceInput({ config: mockConfig })); + expect(capturedGeminiOptions).not.toBeNull(); + expect(capturedWhisperOptions).toBeNull(); + }); + + it('uses LocalWhisperBackend when provider is "whisper"', async () => { + await renderHook(() => + useVoiceInput({ provider: 'whisper', config: mockConfig }), + ); + expect(capturedWhisperOptions).not.toBeNull(); + expect(capturedGeminiOptions).toBeNull(); + }); + + it('delegates startRecording to the active backend', async () => { + mockGeminiStart.mockResolvedValue(undefined); + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + await act(async () => { + await result.current.startRecording(); + }); + expect(mockGeminiStart).toHaveBeenCalledOnce(); + }); + + it('delegates stopRecording to the active backend', async () => { + mockGeminiStop.mockResolvedValue(undefined); + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + await act(async () => { + await result.current.stopRecording(); + }); + expect(mockGeminiStop).toHaveBeenCalledOnce(); + }); + + it('reflects state changes from the backend', async () => { + mockGeminiStart.mockImplementation(async () => { + void capturedGeminiOptions?.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + }); + + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + await act(async () => { + await result.current.startRecording(); + }); + + expect(result.current.state.isRecording).toBe(true); + }); + + it('delivers transcript via coreEvents, not state', async () => { + mockGeminiStop.mockImplementation(async () => { + // Backends emit via coreEvents rather than a callback + coreEvents.emitVoiceTranscript('Hello world'); + void capturedGeminiOptions?.onStateChange({ + isRecording: false, + isTranscribing: false, + error: null, + }); + }); + + const transcripts: string[] = []; + const unsubscribe = onVoiceTranscript((t) => transcripts.push(t)); + + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + await act(async () => { + await result.current.stopRecording(); + }); + + expect(transcripts).toContain('Hello world'); + expect(result.current.state).toEqual({ + isRecording: false, + isTranscribing: false, + error: null, + }); + + unsubscribe(); + }); + + it('surfaces errors from the backend in state', async () => { + mockGeminiStart.mockImplementation(async () => { + void capturedGeminiOptions?.onStateChange({ + isRecording: false, + isTranscribing: false, + error: 'Neither sox nor arecord found', + }); + }); + + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + await act(async () => { + await result.current.startRecording(); + }); + + expect(result.current.state.error).toContain( + 'Neither sox nor arecord found', + ); + expect(result.current.state.isRecording).toBe(false); + }); + + it('recovers if backend start throws after setting recording state', async () => { + mockGeminiStart.mockImplementation(async () => { + void capturedGeminiOptions?.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + throw new Error('Recorder startup failed'); + }); + + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + await act(async () => { + await result.current.startRecording(); + }); + + expect(result.current.state).toEqual({ + isRecording: false, + isTranscribing: false, + error: 'Recorder startup failed', + }); + }); + + it('delegates cancelRecording to the active backend', async () => { + mockGeminiCancel.mockResolvedValue(undefined); + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + await act(async () => { + await result.current.cancelRecording(); + }); + expect(mockGeminiCancel).toHaveBeenCalledOnce(); + }); + + it('cancelRecording is a no-op when no backend is initialized', async () => { + const { result } = await renderHook(() => useVoiceInput()); + await act(async () => { + await result.current.cancelRecording(); + }); + expect(mockGeminiCancel).not.toHaveBeenCalled(); + expect(mockWhisperCancel).not.toHaveBeenCalled(); + }); + + it('cancelRecording immediately clears stuck recording state', async () => { + mockGeminiCancel.mockResolvedValue(undefined); + + const { result } = await renderHook(() => + useVoiceInput({ config: mockConfig }), + ); + + await act(async () => { + void capturedGeminiOptions?.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + }); + + expect(result.current.state.isRecording).toBe(true); + + await act(async () => { + await result.current.cancelRecording(); + }); + + expect(result.current.state).toEqual({ + isRecording: false, + isTranscribing: false, + error: null, + }); + }); +}); diff --git a/packages/cli/src/ui/hooks/useVoiceInput.ts b/packages/cli/src/ui/hooks/useVoiceInput.ts new file mode 100644 index 00000000000..61709cc4812 --- /dev/null +++ b/packages/cli/src/ui/hooks/useVoiceInput.ts @@ -0,0 +1,208 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { + debugLogger, + coreEvents, + CoreEvent, + GeminiRestBackend, + LocalWhisperBackend, +} from '@google/gemini-cli-core'; +import type { + VoiceBackend, + VoiceInputState, + Config, +} from '@google/gemini-cli-core'; +import { useState, useCallback, useRef, useEffect, useMemo } from 'react'; + +export type { + VoiceInputState, + VoiceBackend, + VoiceInputReturn, +} from '@google/gemini-cli-core'; + +/** + * Subscribe to voice transcript events emitted by the active backend. + * Uses the coreEvents bus (CoreEvent.VoiceTranscript) to avoid React + * re-render cascades. + */ +export function onVoiceTranscript( + callback: (transcript: string) => void, +): () => void { + coreEvents.on(CoreEvent.VoiceTranscript, callback); + return () => { + coreEvents.off(CoreEvent.VoiceTranscript, callback); + }; +} + +export interface VoiceInputConfig { + /** + * Which transcription backend to use. + * - 'gemini' (default): zero-install, uses the CLI's existing Gemini API auth. + * - 'whisper': local Whisper binary (faster-whisper or openai-whisper). + */ + provider?: 'gemini' | 'whisper'; + /** Path to a custom Whisper binary. Only used when provider is 'whisper'. */ + whisperPath?: string; + /** The CLI Config instance, used by the Gemini backend for auth. */ + config: Config; + /** + * RMS energy threshold for silence detection (0–1000). Audio below this + * level is discarded without an API call. 0 disables silence detection. + * Default: 80 (allows whispered speech in quiet environments). + */ + silenceThreshold?: number; +} + +/** + * Hook for voice input using system audio recording and a pluggable + * transcription backend (from packages/core). + * + * Backends: + * - GeminiRestBackend (default): records raw PCM in-memory, builds a WAV + * buffer, and transcribes via the Gemini API using the CLI's existing + * ContentGenerator. Works with both API key and OAuth auth. + * - LocalWhisperBackend: records a WAV file and transcribes with a locally + * installed Whisper binary. Used when provider is set to 'whisper'. + * + * Transcripts are delivered via CoreEvent.VoiceTranscript on coreEvents. + */ +export function useVoiceInput(voiceConfig?: VoiceInputConfig) { + const [state, setState] = useState({ + isRecording: false, + isTranscribing: false, + error: null, + }); + + const backendRef = useRef(null); + const isTogglingRef = useRef(false); + const stateRef = useRef(state); + + useEffect(() => { + stateRef.current = state; + }, [state]); + + // Initialize (or re-initialize) the backend when config changes + useEffect(() => { + const options = { + onStateChange: async (newState: VoiceInputState) => { + setState(newState); + if (newState.error) { + coreEvents.emitFeedback('error', newState.error); + } + // Yield one macrotask after signalling isTranscribing:true so Ink + // can flush the state update and render ⏳ before the network call. + if (newState.isTranscribing) { + await new Promise((resolve) => setImmediate(resolve)); + } + }, + silenceThreshold: voiceConfig?.silenceThreshold, + }; + + let activeBackend: VoiceBackend | null = null; + if (voiceConfig?.provider === 'whisper') { + activeBackend = new LocalWhisperBackend(options, { + whisperPath: voiceConfig.whisperPath, + }); + } else if (voiceConfig?.config) { + activeBackend = new GeminiRestBackend(options, voiceConfig.config); + } + backendRef.current = activeBackend; + + return () => { + void activeBackend?.cleanup(); + if (backendRef.current === activeBackend) { + backendRef.current = null; + } + }; + }, [ + voiceConfig?.provider, + voiceConfig?.whisperPath, + voiceConfig?.config, + voiceConfig?.silenceThreshold, + ]); + + const startRecording = useCallback(async () => { + if (!backendRef.current) { + debugLogger.debug( + 'useVoiceInput: startRecording — no backend (voice disabled?)', + ); + return; + } + debugLogger.debug('useVoiceInput: startRecording'); + try { + await backendRef.current.start(); + } catch (error) { + const message = error instanceof Error ? error.message : String(error); + debugLogger.error('useVoiceInput: startRecording failed', error); + setState({ + isRecording: false, + isTranscribing: false, + error: message, + }); + coreEvents.emitFeedback('error', message); + } + }, []); + + const stopRecording = useCallback(async () => { + if (!backendRef.current) return; + debugLogger.debug('useVoiceInput: stopRecording'); + await backendRef.current.stop(); + }, []); + + const cancelRecording = useCallback(async () => { + const clearedState = { + isRecording: false, + isTranscribing: false, + error: null, + } satisfies VoiceInputState; + + stateRef.current = clearedState; + setState(clearedState); + + if (!backendRef.current) return; + await backendRef.current.cancel(); + }, []); + + const toggleRecording = useCallback(async () => { + if (isTogglingRef.current) return; + isTogglingRef.current = true; + try { + if (stateRef.current.isRecording) { + await stopRecording(); + } else { + await startRecording(); + } + } catch (e) { + debugLogger.error('useVoiceInput: toggle error', e); + } finally { + isTogglingRef.current = false; + } + }, [startRecording, stopRecording]); + + const isEnabled = !!( + voiceConfig?.config || voiceConfig?.provider === 'whisper' + ); + + return useMemo( + () => ({ + isEnabled, + state, + startRecording, + stopRecording, + cancelRecording, + toggleRecording, + }), + [ + isEnabled, + state, + startRecording, + stopRecording, + cancelRecording, + toggleRecording, + ], + ); +} diff --git a/packages/cli/src/ui/hooks/useVoiceMode.ts b/packages/cli/src/ui/hooks/useVoiceMode.ts index 0f37c66357c..7b45522b0f3 100644 --- a/packages/cli/src/ui/hooks/useVoiceMode.ts +++ b/packages/cli/src/ui/hooks/useVoiceMode.ts @@ -316,7 +316,7 @@ export function useVoiceMode({ ); const handleVoiceInput = useCallback( - (key: Key): boolean => { + (key: Key) => { const activeRecording = isRecording || isRecordingRef.current; if (activeRecording) { diff --git a/packages/cli/src/ui/key/keyBindings.test.ts b/packages/cli/src/ui/key/keyBindings.test.ts index 10f88dd4d99..f737485828c 100644 --- a/packages/cli/src/ui/key/keyBindings.test.ts +++ b/packages/cli/src/ui/key/keyBindings.test.ts @@ -102,9 +102,17 @@ describe('KeyBinding', () => { describe('keyBindings config', () => { it('should have bindings for all commands', () => { + const commandsWithoutStaticBindings = new Set([Command.VOICE_INPUT]); + for (const command of Object.values(Command)) { expect(defaultKeyBindingConfig.has(command)).toBe(true); - expect(defaultKeyBindingConfig.get(command)?.length).toBeGreaterThan(0); + const bindings = defaultKeyBindingConfig.get(command); + + if (commandsWithoutStaticBindings.has(command)) { + expect(bindings).toHaveLength(0); + } else { + expect(bindings?.length).toBeGreaterThan(0); + } } }); diff --git a/packages/cli/src/ui/key/keyBindings.ts b/packages/cli/src/ui/key/keyBindings.ts index a038f6173cb..7c318545f39 100644 --- a/packages/cli/src/ui/key/keyBindings.ts +++ b/packages/cli/src/ui/key/keyBindings.ts @@ -109,6 +109,8 @@ export enum Command { UNFOCUS_BACKGROUND_SHELL_LIST = 'background.unfocusList', SHOW_BACKGROUND_SHELL_UNFOCUS_WARNING = 'background.unfocusWarning', + // Voice Input + VOICE_INPUT = 'input.voice', // Extension Controls UPDATE_EXTENSION = 'extension.update', LINK_EXTENSION = 'extension.link', @@ -419,6 +421,8 @@ export const defaultKeyBindingConfig: KeyBindingConfig = new Map([ [Command.UNFOCUS_BACKGROUND_SHELL, [new KeyBinding('shift+tab')]], [Command.UNFOCUS_BACKGROUND_SHELL_LIST, [new KeyBinding('tab')]], [Command.SHOW_BACKGROUND_SHELL_UNFOCUS_WARNING, [new KeyBinding('tab')]], + // Voice Input — triggered by double-space on empty input (see InputPrompt.tsx) + [Command.VOICE_INPUT, []], // Extension Controls [Command.UPDATE_EXTENSION, [new KeyBinding('i')]], @@ -560,6 +564,10 @@ export const commandCategories: readonly CommandCategory[] = [ Command.STOP_RECORDING, ], }, + { + title: 'Voice Input', + commands: [Command.VOICE_INPUT], + }, { title: 'Extension Controls', commands: [Command.UPDATE_EXTENSION, Command.LINK_EXTENSION], @@ -679,6 +687,9 @@ export const commandDescriptions: Readonly> = { [Command.SHOW_BACKGROUND_SHELL_UNFOCUS_WARNING]: 'Show warning when trying to move focus away from background shell.', + // Voice Input + [Command.VOICE_INPUT]: + 'Toggle voice input recording (press space twice rapidly).', // Extension Controls [Command.UPDATE_EXTENSION]: 'Update the current extension if available.', [Command.LINK_EXTENSION]: 'Link the current extension to a local path.', diff --git a/packages/cli/src/ui/key/keyMatchers.test.ts b/packages/cli/src/ui/key/keyMatchers.test.ts index 0fc2f00ac77..6321a07b6d4 100644 --- a/packages/cli/src/ui/key/keyMatchers.test.ts +++ b/packages/cli/src/ui/key/keyMatchers.test.ts @@ -321,7 +321,11 @@ describe('keyMatchers', () => { }, { command: Command.PASTE_CLIPBOARD, - positive: [createKey('v', { ctrl: true })], + positive: [ + createKey('v', { ctrl: true }), + createKey('v', { cmd: true }), + createKey('v', { alt: true }), + ], negative: [createKey('v'), createKey('c', { ctrl: true })], }, @@ -388,6 +392,15 @@ describe('keyMatchers', () => { createKey('l', { ctrl: true }), ], }, + { + command: Command.VOICE_INPUT, + positive: [], + negative: [ + createKey('r', { alt: true }), + createKey('v'), + createKey('v', { ctrl: true }), + ], + }, // Shell commands { command: Command.REVERSE_SEARCH, diff --git a/packages/cli/src/ui/noninteractive/nonInteractiveUi.ts b/packages/cli/src/ui/noninteractive/nonInteractiveUi.ts index 91185184559..d89d6074304 100644 --- a/packages/cli/src/ui/noninteractive/nonInteractiveUi.ts +++ b/packages/cli/src/ui/noninteractive/nonInteractiveUi.ts @@ -20,7 +20,7 @@ export function createNonInteractiveUI(): CommandContext['ui'] { process.stderr.write(`Error: ${item.text}\n`); } else if (item.type === 'warning') { process.stderr.write(`Warning: ${item.text}\n`); - } else if (item.type === 'info') { + } else { process.stdout.write(`${item.text}\n`); } } @@ -43,6 +43,7 @@ export function createNonInteractiveUI(): CommandContext['ui'] { removeComponent: () => {}, toggleBackgroundTasks: () => {}, toggleShortcutsHelp: () => {}, + toggleVoice: () => {}, toggleVoiceMode: () => {}, }; } diff --git a/packages/cli/src/ui/types.ts b/packages/cli/src/ui/types.ts index 2808d716b73..cee6c6a62a2 100644 --- a/packages/cli/src/ui/types.ts +++ b/packages/cli/src/ui/types.ts @@ -210,6 +210,19 @@ export type HistoryItemHelp = HistoryItemBase & { timestamp: Date; }; +export type HistoryItemVoiceHelp = HistoryItemBase & { + type: 'voice_help'; + timestamp: Date; +}; + +export type HistoryItemVoiceStatus = HistoryItemBase & { + type: 'voice_status'; + timestamp: Date; + enabled: boolean; + provider: string; + sensitivityLabel: string; + whisperPath: string; +}; export interface HistoryItemQuotaBase extends HistoryItemBase { selectedAuthType?: string; userEmail?: string; @@ -405,6 +418,8 @@ export type HistoryItemWithoutId = | HistoryItemWarning | HistoryItemAbout | HistoryItemHelp + | HistoryItemVoiceHelp + | HistoryItemVoiceStatus | HistoryItemToolGroup | HistoryItemStats | HistoryItemModelStats @@ -447,6 +462,8 @@ export enum MessageType { GEMMA_STATUS = 'gemma_status', CHAT_LIST = 'chat_list', HINT = 'hint', + VOICE_HELP = 'voice_help', + VOICE_STATUS = 'voice_status', } // Simplified message structure for internal feedback @@ -474,6 +491,19 @@ export type Message = timestamp: Date; content?: string; // Optional content, not really used for HELP } + | { + type: MessageType.VOICE_HELP; + timestamp: Date; + content?: string; + } + | { + type: MessageType.VOICE_STATUS; + timestamp: Date; + enabled: boolean; + provider: string; + sensitivityLabel: string; + whisperPath: string; + } | { type: MessageType.STATS; timestamp: Date; diff --git a/packages/cli/test-setup.ts b/packages/cli/test-setup.ts index e64e553ede7..a8d400c43a6 100644 --- a/packages/cli/test-setup.ts +++ b/packages/cli/test-setup.ts @@ -6,7 +6,11 @@ import { vi, beforeEach, afterEach } from 'vitest'; import { format } from 'node:util'; -import { coreEvents, debugLogger } from '@google/gemini-cli-core'; +import { + coreEvents, + uiTelemetryService, + debugLogger, +} from '@google/gemini-cli-core'; import { themeManager } from './src/ui/themes/theme-manager.js'; import { mockInkSpinner } from './src/test-utils/mockSpinner.js'; @@ -22,6 +26,7 @@ global.IS_REACT_ACT_ENVIRONMENT = true; // Increase max listeners to avoid warnings in large test suites coreEvents.setMaxListeners(100); +uiTelemetryService.setMaxListeners(100); // Unset NO_COLOR environment variable to ensure consistent theme behavior between local and CI test runs if (process.env.NO_COLOR !== undefined) { diff --git a/packages/core/src/code_assist/server.ts b/packages/core/src/code_assist/server.ts index 92fc558ebb4..58fe28004e8 100644 --- a/packages/core/src/code_assist/server.ts +++ b/packages/core/src/code_assist/server.ts @@ -36,6 +36,7 @@ import type { EmbedContentResponse, GenerateContentParameters, GenerateContentResponse, + GenerateContentConfig, } from '@google/genai'; import * as readline from 'node:readline'; import { Readable } from 'node:stream'; @@ -89,8 +90,9 @@ export class CodeAssistServer implements ContentGenerator { async generateContentStream( req: GenerateContentParameters, userPromptId: string, - // eslint-disable-next-line @typescript-eslint/no-unused-vars + role: LlmRole, + config?: GenerateContentConfig, ): Promise> { const autoUse = this.config ? shouldAutoUseCredits( @@ -120,7 +122,7 @@ export class CodeAssistServer implements ContentGenerator { this.sessionId, enabledCreditTypes, ), - req.config?.abortSignal, + config?.abortSignal ?? req.config?.abortSignal, ); const streamingLatency: StreamingLatency = {}; @@ -152,7 +154,7 @@ export class CodeAssistServer implements ContentGenerator { response.traceId, translatedResponse, streamingLatency, - req.config?.abortSignal, + config?.abortSignal ?? req.config?.abortSignal, server.sessionId, // Use sessionId as trajectoryId ); @@ -194,8 +196,9 @@ export class CodeAssistServer implements ContentGenerator { async generateContent( req: GenerateContentParameters, userPromptId: string, - // eslint-disable-next-line @typescript-eslint/no-unused-vars + role: LlmRole, + config?: GenerateContentConfig, ): Promise { const start = Date.now(); const response = await this.requestPost( @@ -207,7 +210,7 @@ export class CodeAssistServer implements ContentGenerator { this.sessionId, undefined, ), - req.config?.abortSignal, + config?.abortSignal ?? req.config?.abortSignal, GENERATE_CONTENT_RETRY_DELAY_IN_MILLISECONDS, ); const duration = formatProtoJsonDuration(Date.now() - start); @@ -223,7 +226,7 @@ export class CodeAssistServer implements ContentGenerator { response.traceId, translatedResponse, streamingLatency, - req.config?.abortSignal, + config?.abortSignal ?? req.config?.abortSignal, this.sessionId, // Use sessionId as trajectoryId ); diff --git a/packages/core/src/core/contentGenerator.ts b/packages/core/src/core/contentGenerator.ts index 040fed0e9a6..2352829d525 100644 --- a/packages/core/src/core/contentGenerator.ts +++ b/packages/core/src/core/contentGenerator.ts @@ -8,6 +8,7 @@ import { GoogleGenAI, type CountTokensResponse, type GenerateContentResponse, + type GenerateContentConfig, type GenerateContentParameters, type CountTokensParameters, type EmbedContentResponse, @@ -37,12 +38,14 @@ export interface ContentGenerator { request: GenerateContentParameters, userPromptId: string, role: LlmRole, + config?: GenerateContentConfig, ): Promise; generateContentStream( request: GenerateContentParameters, userPromptId: string, role: LlmRole, + config?: GenerateContentConfig, ): Promise>; countTokens(request: CountTokensParameters): Promise; diff --git a/packages/core/src/core/fakeContentGenerator.ts b/packages/core/src/core/fakeContentGenerator.ts index 9ecd75a99d1..22750b082d3 100644 --- a/packages/core/src/core/fakeContentGenerator.ts +++ b/packages/core/src/core/fakeContentGenerator.ts @@ -6,6 +6,7 @@ import { GenerateContentResponse, + type GenerateContentConfig, type CountTokensResponse, type GenerateContentParameters, type CountTokensParameters, @@ -83,6 +84,8 @@ export class FakeContentGenerator implements ContentGenerator { _userPromptId: string, // eslint-disable-next-line @typescript-eslint/no-unused-vars role: LlmRole, + // eslint-disable-next-line @typescript-eslint/no-unused-vars + config?: GenerateContentConfig, ): Promise { // eslint-disable-next-line @typescript-eslint/no-unsafe-return return Object.setPrototypeOf( @@ -96,6 +99,8 @@ export class FakeContentGenerator implements ContentGenerator { _userPromptId: string, // eslint-disable-next-line @typescript-eslint/no-unused-vars role: LlmRole, + // eslint-disable-next-line @typescript-eslint/no-unused-vars + config?: GenerateContentConfig, ): Promise> { const responses = this.getNextResponse('generateContentStream', request); async function* stream() { diff --git a/packages/core/src/core/loggingContentGenerator.ts b/packages/core/src/core/loggingContentGenerator.ts index d27b8a8f328..cf6d26501fa 100644 --- a/packages/core/src/core/loggingContentGenerator.ts +++ b/packages/core/src/core/loggingContentGenerator.ts @@ -356,6 +356,7 @@ export class LoggingContentGenerator implements ContentGenerator { req: GenerateContentParameters, userPromptId: string, role: LlmRole, + config?: GenerateContentConfig, ): Promise { return runInDevTraceSpan( { @@ -367,9 +368,11 @@ export class LoggingContentGenerator implements ContentGenerator { [GEN_AI_REQUEST_MODEL]: req.model, [GEN_AI_PROMPT_NAME]: userPromptId, [GEN_AI_SYSTEM_INSTRUCTIONS]: safeJsonStringify( - req.config?.systemInstruction ?? [], + req.config?.systemInstruction ?? config?.systemInstruction ?? [], + ), + [GEN_AI_TOOL_DEFINITIONS]: safeJsonStringify( + req.config?.tools ?? config?.tools ?? [], ), - [GEN_AI_TOOL_DEFINITIONS]: safeJsonStringify(req.config?.tools ?? []), }, }, async ({ metadata: spanMetadata }) => { @@ -383,7 +386,7 @@ export class LoggingContentGenerator implements ContentGenerator { req.model, userPromptId, role, - req.config, + req.config ?? config, serverDetails, ); @@ -392,6 +395,7 @@ export class LoggingContentGenerator implements ContentGenerator { req, userPromptId, role, + config, ); spanMetadata.output = response.candidates?.[0]?.content ?? null; spanMetadata.attributes[GEN_AI_USAGE_INPUT_TOKENS] = @@ -415,7 +419,7 @@ export class LoggingContentGenerator implements ContentGenerator { modelVersion: response.modelVersion, promptFeedback: response.promptFeedback, }), - req.config, + req.config ?? config, serverDetails, ); this.config @@ -435,7 +439,7 @@ export class LoggingContentGenerator implements ContentGenerator { userPromptId, contents, role, - req.config, + req.config ?? config, serverDetails, ); throw error; @@ -448,6 +452,7 @@ export class LoggingContentGenerator implements ContentGenerator { req: GenerateContentParameters, userPromptId: string, role: LlmRole, + config?: GenerateContentConfig, ): Promise> { return runInDevTraceSpan( { @@ -459,9 +464,11 @@ export class LoggingContentGenerator implements ContentGenerator { [GEN_AI_REQUEST_MODEL]: req.model, [GEN_AI_PROMPT_NAME]: userPromptId, [GEN_AI_SYSTEM_INSTRUCTIONS]: safeJsonStringify( - req.config?.systemInstruction ?? [], + req.config?.systemInstruction ?? config?.systemInstruction ?? [], + ), + [GEN_AI_TOOL_DEFINITIONS]: safeJsonStringify( + req.config?.tools ?? config?.tools ?? [], ), - [GEN_AI_TOOL_DEFINITIONS]: safeJsonStringify(req.config?.tools ?? []), }, }, async ({ metadata: spanMetadata }) => { @@ -484,7 +491,7 @@ export class LoggingContentGenerator implements ContentGenerator { req.model, userPromptId, role, - req.config, + req.config ?? config, serverDetails, ); @@ -494,6 +501,7 @@ export class LoggingContentGenerator implements ContentGenerator { req, userPromptId, role, + config, ); } catch (error) { const durationMs = Date.now() - startTime; @@ -507,7 +515,7 @@ export class LoggingContentGenerator implements ContentGenerator { userPromptId, toContents(req.contents), role, - req.config, + req.config ?? config, serverDetails, ); throw error; @@ -520,6 +528,7 @@ export class LoggingContentGenerator implements ContentGenerator { userPromptId, role, spanMetadata, + config, ); }, ); @@ -532,6 +541,7 @@ export class LoggingContentGenerator implements ContentGenerator { userPromptId: string, role: LlmRole, spanMetadata: SpanMetadata, + config?: GenerateContentConfig, ): AsyncGenerator { const responses: GenerateContentResponse[] = []; @@ -566,7 +576,7 @@ export class LoggingContentGenerator implements ContentGenerator { promptFeedback: r.promptFeedback, })), ), - req.config, + req.config ?? config, serverDetails, ); this.config diff --git a/packages/core/src/core/recordingContentGenerator.ts b/packages/core/src/core/recordingContentGenerator.ts index f2193bb16df..3a0cba938ea 100644 --- a/packages/core/src/core/recordingContentGenerator.ts +++ b/packages/core/src/core/recordingContentGenerator.ts @@ -8,6 +8,7 @@ import type { CountTokensResponse, GenerateContentParameters, GenerateContentResponse, + GenerateContentConfig, CountTokensParameters, EmbedContentResponse, EmbedContentParameters, @@ -43,11 +44,13 @@ export class RecordingContentGenerator implements ContentGenerator { request: GenerateContentParameters, userPromptId: string, role: LlmRole, + config?: GenerateContentConfig, ): Promise { const response = await this.realGenerator.generateContent( request, userPromptId, role, + config, ); const recordedResponse: FakeResponse = { method: 'generateContent', @@ -65,6 +68,7 @@ export class RecordingContentGenerator implements ContentGenerator { request: GenerateContentParameters, userPromptId: string, role: LlmRole, + config?: GenerateContentConfig, ): Promise> { const recordedResponse: FakeResponse = { method: 'generateContentStream', @@ -75,6 +79,7 @@ export class RecordingContentGenerator implements ContentGenerator { request, userPromptId, role, + config, ); async function* stream(filePath: string) { diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index cf0ba653c14..ab7a7569720 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -122,7 +122,6 @@ export * from './utils/extensionLoader.js'; export * from './utils/package.js'; export * from './utils/version.js'; export * from './utils/checkpointUtils.js'; -export * from './utils/secure-browser-launcher.js'; export * from './utils/apiConversionUtils.js'; export * from './utils/channel.js'; export * from './utils/constants.js'; @@ -277,9 +276,6 @@ export * from './hooks/index.js'; // Export hook types export * from './hooks/types.js'; -// Export agent types -export * from './agents/types.js'; - // Export stdio utils export * from './utils/stdio.js'; export * from './utils/terminal.js'; @@ -290,6 +286,10 @@ export * from './voice/responseFormatter.js'; // Export types from @google/genai export type { Content, Part, FunctionCall } from '@google/genai'; +// Voice services +export * from './services/voice/types.js'; +export { GeminiRestBackend } from './services/voice/GeminiRestBackend.js'; +export { LocalWhisperBackend } from './services/voice/LocalWhisperBackend.js'; // Export context types and profiles export * from './context/types.js'; diff --git a/packages/core/src/services/voice/GeminiRestBackend.test.ts b/packages/core/src/services/voice/GeminiRestBackend.test.ts new file mode 100644 index 00000000000..616b7bbc82f --- /dev/null +++ b/packages/core/src/services/voice/GeminiRestBackend.test.ts @@ -0,0 +1,206 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { beforeEach, afterEach, describe, expect, it, vi } from 'vitest'; +import { GeminiRestBackend } from './GeminiRestBackend.js'; + +const { mockEmitVoiceTranscript } = vi.hoisted(() => ({ + mockEmitVoiceTranscript: vi.fn(), +})); + +vi.mock('../../utils/events.js', () => ({ + coreEvents: { + emitVoiceTranscript: mockEmitVoiceTranscript, + }, +})); + +function createRecordingProcessMock(onClose: () => void) { + let closeHandler: (() => void) | null = null; + let stdoutDataHandler: ((chunk: Buffer) => void) | null = null; + + return { + kill: vi.fn(() => closeHandler?.()), + once: vi.fn((event: string, callback: () => void) => { + if (event === 'close') { + closeHandler = () => { + onClose(); + callback(); + }; + } + }), + stdout: { + on: vi.fn((event: string, callback: (chunk: Buffer) => void) => { + if (event === 'data') { + stdoutDataHandler = callback; + } + }), + }, + emitStdoutData: (chunk: Buffer) => stdoutDataHandler?.(chunk), + }; +} + +describe('GeminiRestBackend', () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + it('waits for recorder close before reading buffered audio', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new GeminiRestBackend( + { + onStateChange, + silenceThreshold: 0, + }, + { + getContentGenerator: vi.fn(), + } as never, + ); + + const recordingProcess = createRecordingProcessMock(() => { + ( + backend as unknown as { + audioChunks: Buffer[]; + } + ).audioChunks.push(Buffer.from([1, 0, 2, 0])); + }); + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { audioChunks: Buffer[] }).audioChunks = []; + + const transcribeSpy = vi + .spyOn(backend as never, 'transcribe') + .mockResolvedValue('hello world'); + + await backend.stop(); + + expect(recordingProcess.kill).toHaveBeenCalledWith('SIGTERM'); + expect(transcribeSpy).toHaveBeenCalledOnce(); + expect(mockEmitVoiceTranscript).toHaveBeenCalledWith('hello world'); + }); + + it('keeps the final flushed audio chunk that arrives after stop begins', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new GeminiRestBackend( + { + onStateChange, + silenceThreshold: 0, + }, + { + getContentGenerator: vi.fn(), + } as never, + ); + + const recordingProcess = createRecordingProcessMock(() => { + recordingProcess.emitStdoutData(Buffer.from([3, 0, 4, 0])); + }); + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { audioChunks: Buffer[] }).audioChunks = [ + Buffer.from([1, 0, 2, 0]), + ]; + + const transcribeSpy = vi + .spyOn(backend as never, 'transcribe') + .mockResolvedValue('hello world'); + + // Register the stdout listener the same way start() would. + ( + recordingProcess.stdout.on as ( + event: string, + callback: (chunk: Buffer) => void, + ) => void + )('data', (chunk: Buffer) => { + ( + backend as unknown as { + audioChunks: Buffer[]; + } + ).audioChunks.push(chunk); + }); + + await backend.stop(); + + expect(recordingProcess.kill).toHaveBeenCalledWith('SIGTERM'); + expect(transcribeSpy).toHaveBeenCalledOnce(); + expect((transcribeSpy.mock.calls[0] as [Buffer])[0].subarray(44)).toEqual( + Buffer.from([1, 0, 2, 0, 3, 0, 4, 0]), + ); + }); + + it('repro: stop remains pending if the recorder never emits close', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new GeminiRestBackend( + { + onStateChange, + silenceThreshold: 0, + }, + { + getContentGenerator: vi.fn(), + } as never, + ); + + const recordingProcess = { + kill: vi.fn(), + once: vi.fn(), + }; + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { audioChunks: Buffer[] }).audioChunks = [ + Buffer.from([1, 0]), + ]; + + let settled = false; + void backend.stop().then(() => { + settled = true; + }); + + await Promise.resolve(); + await Promise.resolve(); + + expect(recordingProcess.once).toHaveBeenCalledWith( + 'close', + expect.any(Function), + ); + expect(recordingProcess.kill).toHaveBeenCalledWith('SIGTERM'); + expect(settled).toBe(false); + }); + + it('cancel returns immediately even if the recorder never emits close', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new GeminiRestBackend( + { + onStateChange, + silenceThreshold: 0, + }, + { + getContentGenerator: vi.fn(), + } as never, + ); + + const recordingProcess = { + kill: vi.fn(), + once: vi.fn(), + }; + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + + await backend.cancel(); + + expect(recordingProcess.kill).toHaveBeenCalledWith('SIGTERM'); + expect(onStateChange).toHaveBeenCalledWith({ + isRecording: false, + isTranscribing: false, + error: null, + }); + }); +}); diff --git a/packages/core/src/services/voice/GeminiRestBackend.ts b/packages/core/src/services/voice/GeminiRestBackend.ts new file mode 100644 index 00000000000..bfc966ed23e --- /dev/null +++ b/packages/core/src/services/voice/GeminiRestBackend.ts @@ -0,0 +1,326 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { type Config } from '../../config/config.js'; +import { DEFAULT_GEMINI_FLASH_MODEL } from '../../config/models.js'; +import { coreEvents } from '../../utils/events.js'; +import { resolveExecutable } from '../../utils/shell-utils.js'; +import { LlmRole } from '../../telemetry/llmRole.js'; +import { spawn } from 'node:child_process'; +import type { VoiceBackend, VoiceBackendOptions } from './types.js'; +import type { GenerateContentParameters } from '@google/genai'; + +const SAMPLE_RATE = 16000; +const CHANNELS = 1; + +/** + * Voice backend that uses the Gemini API for transcription. + * + * Records raw PCM audio in-memory via sox or arecord, then builds a WAV + * buffer and sends it to the Gemini API using the CLI's existing + * ContentGenerator (works with both API key and OAuth auth). + * + * Note: The recording process uses `spawn` directly rather than `spawnAsync` + * because audio capture requires streaming binary stdout chunk-by-chunk in + * real-time. `spawnAsync` buffers all output until the process exits, which + * is incompatible with the push-to-talk recording pattern. + * + * Transcripts are delivered via `coreEvents.emitVoiceTranscript()`. + */ +export class GeminiRestBackend implements VoiceBackend { + private recordingProcess: ReturnType | null = null; + private audioChunks: Buffer[] = []; + private stderrChunks: Buffer[] = []; + private abortController: AbortController | null = null; + + constructor( + private readonly options: VoiceBackendOptions, + private readonly config: Config, + ) {} + + async start(): Promise { + if (this.recordingProcess) return; + + try { + this.audioChunks = []; + this.stderrChunks = []; + this.abortController = new AbortController(); + let recordingProcess: ReturnType | null = null; + + const soxPath = await resolveExecutable('sox'); + if (soxPath) { + recordingProcess = spawn(soxPath, [ + '-d', + '-b', + '16', + '-r', + SAMPLE_RATE.toString(), + '-c', + CHANNELS.toString(), + '-e', + 'signed-integer', + '-t', + 'raw', + '-', + ]); + } else { + const arecordPath = await resolveExecutable('arecord'); + if (arecordPath) { + recordingProcess = spawn(arecordPath, [ + '-f', + 'S16_LE', + '-r', + SAMPLE_RATE.toString(), + '-c', + CHANNELS.toString(), + '-t', + 'raw', + '-D', + 'default', + ]); + } else { + throw new Error( + 'Neither sox nor arecord found.\n' + + ' macOS: brew install sox\n' + + ' Linux: sudo apt install sox (or: sudo apt install alsa-utils)', + ); + } + } + + this.recordingProcess = recordingProcess; + + this.recordingProcess.stdout?.on('data', (chunk: Buffer) => { + this.audioChunks.push(chunk); + }); + + this.recordingProcess.stderr?.on('data', (chunk: Buffer) => { + this.stderrChunks.push(chunk); + }); + + this.recordingProcess.on('error', (err) => { + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: `Recording error: ${err.message}`, + }); + }); + + void this.options.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + } catch (err) { + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: err instanceof Error ? err.message : String(err), + }); + throw err; + } + } + + async cancel(): Promise { + this.abortController?.abort(); + if (!this.recordingProcess) return; + + const proc = this.recordingProcess; + this.recordingProcess = null; + + // Ensure the process is terminated, even if it ignores SIGTERM. + const closePromise = new Promise((resolve) => { + proc.once('close', () => resolve()); + setTimeout(() => { + try { + proc.kill('SIGKILL'); + } catch { + // ignore + } + resolve(); + }, 500); + }); + + proc.kill('SIGTERM'); + this.audioChunks = []; + this.stderrChunks = []; + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: null, + }); + + // Don't block the cancel call. + void closePromise.then(() => {}); + } + + async stop(): Promise { + if (!this.recordingProcess) return; + const proc = this.recordingProcess; + this.recordingProcess = null; + const closePromise = new Promise((resolve) => { + proc.once('close', () => resolve()); + setTimeout(() => { + try { + proc.kill('SIGKILL'); + } catch { + // ignore + } + resolve(); + }, 500); + }); + proc.kill('SIGTERM'); + + try { + // Wait for the recorder to exit so any final stdout audio chunks flush + // before we build the in-memory WAV payload. + await closePromise; + + const audioBuffer = Buffer.concat(this.audioChunks); + if (audioBuffer.length === 0) { + const stderrStr = Buffer.concat(this.stderrChunks) + .toString('utf8') + .trim(); + throw new Error(`No audio captured. ${stderrStr}`); + } + + // Reject silent recordings before showing the transcribing state, so + // the ⏳ indicator only appears when an actual API call will be made. + // RMS of 16-bit LE PCM: background noise <200, audible speech ~500+. + if (this.isSilentPcm(audioBuffer)) { + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: + 'Audio discarded (too quiet). Try speaking louder or adjust threshold: /voice sensitivity', + }); + return; + } + + // Signal transcription only now — the ⏳ will be visible for the + // full duration of the Gemini API call. Awaiting allows the UI layer + // to flush the state change before the network call begins. + await this.options.onStateChange({ + isRecording: false, + isTranscribing: true, + error: null, + }); + + const wavBuffer = this.createWavBuffer(audioBuffer, SAMPLE_RATE); + const transcript = await this.transcribe(wavBuffer); + if (this.abortController?.signal.aborted) return; + coreEvents.emitVoiceTranscript(transcript); + + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: null, + }); + } catch (err) { + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: err instanceof Error ? err.message : String(err), + }); + } finally { + await this.cleanup(); + } + } + + /** + * Builds a WAV buffer from raw 16-bit LE PCM data. + * Gemini's audio/wav MIME type requires a valid RIFF header. + */ + private createWavBuffer(pcmBuffer: Buffer, sampleRate: number): Buffer { + const header = Buffer.alloc(44); + const dataSize = pcmBuffer.length; + + header.write('RIFF', 0); + header.writeUInt32LE(dataSize + 36, 4); + header.write('WAVE', 8); + header.write('fmt ', 12); + header.writeUInt32LE(16, 16); // format chunk size + header.writeUInt16LE(1, 20); // PCM format + header.writeUInt16LE(CHANNELS, 22); + header.writeUInt32LE(sampleRate, 24); + header.writeUInt32LE(sampleRate * CHANNELS * 2, 28); // byte rate + header.writeUInt16LE(CHANNELS * 2, 32); // block align + header.writeUInt16LE(16, 34); // bits per sample + header.write('data', 36); + header.writeUInt32LE(dataSize, 40); + + return Buffer.concat([header, pcmBuffer]); + } + + private async transcribe(audioBuffer: Buffer): Promise { + const generator = this.config.getContentGenerator(); + if (!generator) throw new Error('Gemini API not initialized'); + + const request: GenerateContentParameters = { + model: DEFAULT_GEMINI_FLASH_MODEL, + contents: [ + { + role: 'user', + parts: [ + { + text: 'Transcribe the following audio exactly as spoken. Output only the transcribed text with no additional commentary or formatting.', + }, + { + inlineData: { + mimeType: 'audio/wav', + data: audioBuffer.toString('base64'), + }, + }, + ], + }, + ], + }; + + const response = await generator.generateContent( + request, + 'voice-transcription', + LlmRole.UTILITY_TOOL, + { abortSignal: this.abortController?.signal }, + ); + + const parts = response.candidates?.[0]?.content?.parts; + return parts?.[0]?.text?.trim() ?? ''; + } + + /** + * Returns true if the raw 16-bit LE PCM buffer is effectively silent. + * Computes RMS amplitude and compares against options.silenceThreshold + * (default 80). A threshold of 0 disables silence detection entirely. + * + * RMS guide (16-bit PCM, 16 kHz mono): + * ~0-30 near-digital silence + * ~30-100 quiet room ambient / electrical noise + * ~100-400 whispered speech + * ~500+ normal conversational speech + */ + private isSilentPcm(pcmBuffer: Buffer): boolean { + const threshold = this.options.silenceThreshold ?? 80; + if (threshold === 0) return false; // disabled + const samples = pcmBuffer.length / 2; // 16-bit = 2 bytes per sample + if (samples === 0) return true; + let sumSquares = 0; + for (let i = 0; i < pcmBuffer.length - 1; i += 2) { + const sample = pcmBuffer.readInt16LE(i); + sumSquares += sample * sample; + } + const rms = Math.sqrt(sumSquares / samples); + return rms < threshold; + } + + async cleanup(): Promise { + this.abortController?.abort(); + if (this.recordingProcess) { + this.recordingProcess.kill('SIGTERM'); + this.recordingProcess = null; + } + this.audioChunks = []; + this.stderrChunks = []; + } +} diff --git a/packages/core/src/services/voice/LocalWhisperBackend.test.ts b/packages/core/src/services/voice/LocalWhisperBackend.test.ts new file mode 100644 index 00000000000..7407e9ee3bd --- /dev/null +++ b/packages/core/src/services/voice/LocalWhisperBackend.test.ts @@ -0,0 +1,243 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { LocalWhisperBackend } from './LocalWhisperBackend.js'; + +const { mockReadFile, mockStat, mockRm, mockEmitVoiceTranscript } = vi.hoisted( + () => ({ + mockReadFile: vi.fn(), + mockStat: vi.fn(), + mockRm: vi.fn(), + mockEmitVoiceTranscript: vi.fn(), + }), +); + +vi.mock('node:fs/promises', () => ({ + mkdtemp: vi.fn(), + readFile: mockReadFile, + rm: mockRm, + stat: mockStat, +})); + +vi.mock('../../utils/events.js', () => ({ + coreEvents: { + emitVoiceTranscript: mockEmitVoiceTranscript, + }, +})); + +function createRecordingProcessMock() { + const registerCloseHandler = vi.fn((event: string, callback: () => void) => { + if (event === 'close') { + callback(); + } + }); + + return { + kill: vi.fn(), + on: registerCloseHandler, + once: registerCloseHandler, + }; +} + +function createWavBuffer(samples: number[]): Buffer { + const pcmBuffer = Buffer.alloc(samples.length * 2); + for (const [index, sample] of samples.entries()) { + pcmBuffer.writeInt16LE(sample, index * 2); + } + + const header = Buffer.alloc(44); + header.write('RIFF', 0); + header.writeUInt32LE(pcmBuffer.length + 36, 4); + header.write('WAVE', 8); + header.write('fmt ', 12); + header.writeUInt32LE(16, 16); + header.writeUInt16LE(1, 20); + header.writeUInt16LE(1, 22); + header.writeUInt32LE(16000, 24); + header.writeUInt32LE(32000, 28); + header.writeUInt16LE(2, 32); + header.writeUInt16LE(16, 34); + header.write('data', 36); + header.writeUInt32LE(pcmBuffer.length, 40); + + return Buffer.concat([header, pcmBuffer]); +} + +describe('LocalWhisperBackend', () => { + beforeEach(() => { + vi.clearAllMocks(); + mockStat.mockResolvedValue({ size: 128 }); + mockRm.mockResolvedValue(undefined); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + it('skips transcription for silent wav recordings when threshold is enabled', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new LocalWhisperBackend({ + onStateChange, + silenceThreshold: 80, + }); + const recordingProcess = createRecordingProcessMock(); + const transcribeSpy = vi + .spyOn(backend as never, 'transcribe') + .mockResolvedValue('should not be used'); + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { audioFile: string }).audioFile = + '/tmp/recording.wav'; + (backend as unknown as { tempDir: string }).tempDir = '/tmp/voice'; + mockReadFile.mockResolvedValue(createWavBuffer([0, 2, -1, 1])); + + await backend.stop(); + + expect(transcribeSpy).not.toHaveBeenCalled(); + expect(mockEmitVoiceTranscript).not.toHaveBeenCalled(); + expect(onStateChange).toHaveBeenCalledWith({ + isRecording: false, + isTranscribing: true, + error: null, + }); + expect(onStateChange).toHaveBeenLastCalledWith({ + isRecording: false, + isTranscribing: false, + error: + 'Audio discarded (too quiet). Try speaking louder or adjust threshold: /voice sensitivity', + }); + }); + + it('does not skip transcription when silence detection is disabled', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new LocalWhisperBackend({ + onStateChange, + silenceThreshold: 0, + }); + const recordingProcess = createRecordingProcessMock(); + const transcribeSpy = vi + .spyOn(backend as never, 'transcribe') + .mockResolvedValue('transcribed text'); + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { audioFile: string }).audioFile = + '/tmp/recording.wav'; + (backend as unknown as { tempDir: string }).tempDir = '/tmp/voice'; + mockReadFile.mockResolvedValue(createWavBuffer([0, 1, -1, 2])); + + await backend.stop(); + + expect(transcribeSpy).toHaveBeenCalledWith('/tmp/recording.wav'); + expect(mockEmitVoiceTranscript).toHaveBeenCalledWith('transcribed text'); + }); + + it('registers the close handler before killing the recorder', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new LocalWhisperBackend({ + onStateChange, + silenceThreshold: 0, + }); + + let closeHandler: (() => void) | null = null; + const recordingProcess = { + kill: vi.fn(() => closeHandler?.()), + once: vi.fn((event: string, callback: () => void) => { + if (event === 'close') { + closeHandler = callback; + } + }), + }; + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { audioFile: string }).audioFile = + '/tmp/recording.wav'; + (backend as unknown as { tempDir: string }).tempDir = '/tmp/voice'; + + mockReadFile.mockResolvedValue(createWavBuffer([0, 1, -1, 2])); + const transcribeSpy = vi + .spyOn(backend as never, 'transcribe') + .mockResolvedValue('transcribed text'); + + await backend.stop(); + + expect(recordingProcess.once).toHaveBeenCalledWith( + 'close', + expect.any(Function), + ); + expect(recordingProcess.kill).toHaveBeenCalledWith('SIGTERM'); + expect(transcribeSpy).toHaveBeenCalledWith('/tmp/recording.wav'); + }); + + it('repro: stop remains pending if the recorder never emits close', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new LocalWhisperBackend({ + onStateChange, + silenceThreshold: 0, + }); + + const recordingProcess = { + kill: vi.fn(), + once: vi.fn(), + }; + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { audioFile: string }).audioFile = + '/tmp/recording.wav'; + (backend as unknown as { tempDir: string }).tempDir = '/tmp/voice'; + + let settled = false; + void backend.stop().then(() => { + settled = true; + }); + + await Promise.resolve(); + await Promise.resolve(); + + expect(recordingProcess.once).toHaveBeenCalledWith( + 'close', + expect.any(Function), + ); + expect(recordingProcess.kill).toHaveBeenCalledWith('SIGTERM'); + expect(settled).toBe(false); + }); + + it('cancel resolves immediately even if the recorder never emits close', async () => { + const onStateChange = vi.fn().mockResolvedValue(undefined); + const backend = new LocalWhisperBackend({ + onStateChange, + silenceThreshold: 0, + }); + + const recordingProcess = { + kill: vi.fn(), + once: vi.fn(), + }; + + (backend as unknown as { recordingProcess: unknown }).recordingProcess = + recordingProcess; + (backend as unknown as { tempDir: string }).tempDir = '/tmp/voice'; + + let settled = false; + void backend.cancel().then(() => { + settled = true; + }); + + await Promise.resolve(); + await Promise.resolve(); + + expect(recordingProcess.once).toHaveBeenCalledWith( + 'close', + expect.any(Function), + ); + expect(recordingProcess.kill).toHaveBeenCalledWith('SIGTERM'); + expect(settled).toBe(true); + }); +}); diff --git a/packages/core/src/services/voice/LocalWhisperBackend.ts b/packages/core/src/services/voice/LocalWhisperBackend.ts new file mode 100644 index 00000000000..963d9083292 --- /dev/null +++ b/packages/core/src/services/voice/LocalWhisperBackend.ts @@ -0,0 +1,340 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +import { debugLogger } from '../../utils/debugLogger.js'; +import { tmpdir } from '../../utils/paths.js'; +import { resolveExecutable, spawnAsync } from '../../utils/shell-utils.js'; +import { coreEvents } from '../../utils/events.js'; +import { spawn } from 'node:child_process'; +import { mkdtemp, readFile, rm, stat } from 'node:fs/promises'; +import { join } from 'node:path'; +import type { VoiceBackend, VoiceBackendOptions } from './types.js'; + +const RECORDING_FORMAT = 'wav'; +const SAMPLE_RATE = 16000; +const CHANNELS = 1; + +/** + * Voice backend that records a WAV file via sox/arecord and transcribes + * it using a locally-installed Whisper binary (faster-whisper or openai-whisper). + * + * Note: The recording process uses `spawn` directly rather than `spawnAsync` + * because audio capture requires real-time process control (SIGINT to stop). + * `spawnAsync` is used for the Whisper transcription step, which is a + * standard command-and-result invocation. + * + * Transcripts are delivered via `coreEvents.emitVoiceTranscript()`. + */ +export class LocalWhisperBackend implements VoiceBackend { + private recordingProcess: ReturnType | null = null; + private tempDir: string | null = null; + private audioFile: string | null = null; + private stderrChunks: Buffer[] = []; + private abortController: AbortController | null = null; + + constructor( + private readonly options: VoiceBackendOptions, + private readonly config: { whisperPath?: string } = {}, + ) {} + + async start(): Promise { + if (this.recordingProcess) { + debugLogger.log('LocalWhisperBackend: already recording'); + return; + } + + try { + this.stderrChunks = []; + this.abortController = new AbortController(); + this.tempDir = await mkdtemp(join(tmpdir(), 'gemini-voice-')); + this.audioFile = join(this.tempDir, `recording.${RECORDING_FORMAT}`); + let recordingProcess: ReturnType | null = null; + + const soxPath = await resolveExecutable('sox'); + if (soxPath) { + recordingProcess = spawn(soxPath, [ + '-d', + '-b', + '16', + '-r', + SAMPLE_RATE.toString(), + '-c', + CHANNELS.toString(), + '-e', + 'signed-integer', + '-t', + RECORDING_FORMAT, + this.audioFile, + ]); + } else { + const arecordPath = await resolveExecutable('arecord'); + if (arecordPath) { + recordingProcess = spawn(arecordPath, [ + '-f', + 'S16_LE', + '-r', + SAMPLE_RATE.toString(), + '-c', + CHANNELS.toString(), + '-D', + 'default', + this.audioFile, + ]); + } else { + throw new Error( + 'Neither sox nor arecord found. Please install one of them.', + ); + } + } + + this.recordingProcess = recordingProcess; + + this.recordingProcess.stderr?.on('data', (chunk: Buffer) => { + this.stderrChunks.push(chunk); + }); + + this.recordingProcess.on('error', (err) => { + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: `Recording error: ${err.message}`, + }); + }); + + void this.options.onStateChange({ + isRecording: true, + isTranscribing: false, + error: null, + }); + } catch (err) { + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: err instanceof Error ? err.message : String(err), + }); + await this.cleanup(); + throw err; + } + } + + async cancel(): Promise { + this.abortController?.abort(); + if (!this.recordingProcess) return; + + const proc = this.recordingProcess; + this.recordingProcess = null; + + const closePromise = new Promise((resolve) => { + proc.once('close', () => resolve()); + setTimeout(() => { + try { + proc.kill('SIGKILL'); + } catch { + // ignore + } + resolve(); + }, 500); + }); + + proc.kill('SIGTERM'); + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: null, + }); + + // Don't block the cancel call. Perform cleanup in the background. + void closePromise.then(() => this.cleanup()); + } + + async stop(): Promise { + if (!this.recordingProcess) return; + + const proc = this.recordingProcess; + this.recordingProcess = null; + const closePromise = new Promise((resolve) => { + proc.once('close', () => resolve()); + setTimeout(() => { + try { + proc.kill('SIGKILL'); + } catch { + // ignore + } + resolve(); + }, 500); + }); + proc.kill('SIGTERM'); + + // Wait for the recording process to fully close (stdio streams flushed) + // before reading the audio file. + await closePromise; + + await this.options.onStateChange({ + isRecording: false, + isTranscribing: true, + error: null, + }); + + try { + if (!this.audioFile) throw new Error('No audio file'); + + const stats = await stat(this.audioFile).catch(() => null); + if (!stats || stats.size === 0) { + const stderrStr = Buffer.concat(this.stderrChunks) + .toString('utf8') + .trim(); + throw new Error(`No audio recorded (file is empty). ${stderrStr}`); + } + + const audioBuffer = await readFile(this.audioFile); + if (this.isSilentWav(audioBuffer)) { + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: + 'Audio discarded (too quiet). Try speaking louder or adjust threshold: /voice sensitivity', + }); + return; + } + + const transcript = await this.transcribe(this.audioFile); + if (this.abortController?.signal.aborted) return; + coreEvents.emitVoiceTranscript(transcript); + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: null, + }); + } catch (err) { + if (this.abortController?.signal.aborted) return; + void this.options.onStateChange({ + isRecording: false, + isTranscribing: false, + error: err instanceof Error ? err.message : String(err), + }); + } finally { + await this.cleanup(); + } + } + + private async transcribe(audioFile: string): Promise { + // Reject paths with shell metacharacters that could enable injection + const validatePath = (p: string): string => { + const sanitized = p.replace(/['"]/g, ''); + if (/[;&|`$(){}[\]<>!]/.test(sanitized)) + throw new Error('Invalid binary path: contains shell metacharacters'); + return sanitized; + }; + + const args = [ + audioFile, + '--model', + 'tiny', + '--output_format', + 'txt', + '--output_dir', + this.tempDir!, + ]; + + const spawnOptions = { signal: this.abortController?.signal }; + + if (this.config.whisperPath) { + await spawnAsync( + validatePath(this.config.whisperPath), + args, + spawnOptions, + ); + } else { + try { + await spawnAsync('whisper-faster', args, spawnOptions); + } catch (err) { + if (this.abortController?.signal.aborted) throw err; + try { + await spawnAsync('whisper', args, spawnOptions); + } catch (innerErr) { + if (this.abortController?.signal.aborted) throw innerErr; + throw new Error( + 'Whisper not found. Please install faster-whisper or openai-whisper, ' + + 'or configure the path in settings.', + ); + } + } + } + + const transcriptFile = audioFile.replace('.wav', '.txt'); + const raw = await readFile(transcriptFile, 'utf-8'); + return raw + .split('\n') + .map((l) => + l + .replace(/^\[\d{2}:\d{2}\.\d{3} --> \d{2}:\d{2}\.\d.{3}\]\s*/, '') + .trim(), + ) + .filter(Boolean) + .join(' '); + } + + async cleanup(): Promise { + this.abortController?.abort(); + if (this.tempDir) { + await rm(this.tempDir, { recursive: true, force: true }).catch(() => {}); + this.tempDir = null; + this.audioFile = null; + } + } + + private isSilentWav(wavBuffer: Buffer): boolean { + const threshold = this.options.silenceThreshold ?? 80; + if (threshold === 0) return false; + + const pcmData = this.extractWavDataChunk(wavBuffer); + if (!pcmData || pcmData.length < 2) { + return false; + } + + const samples = Math.floor(pcmData.length / 2); + if (samples === 0) return true; + + let sumSquares = 0; + for (let i = 0; i < samples * 2; i += 2) { + const sample = pcmData.readInt16LE(i); + sumSquares += sample * sample; + } + + const rms = Math.sqrt(sumSquares / samples); + return rms < threshold; + } + + private extractWavDataChunk(wavBuffer: Buffer): Buffer | null { + if ( + wavBuffer.length < 12 || + wavBuffer.toString('ascii', 0, 4) !== 'RIFF' || + wavBuffer.toString('ascii', 8, 12) !== 'WAVE' + ) { + return null; + } + + let offset = 12; + while (offset + 8 <= wavBuffer.length) { + const chunkId = wavBuffer.toString('ascii', offset, offset + 4); + const chunkSize = wavBuffer.readUInt32LE(offset + 4); + const chunkStart = offset + 8; + const chunkEnd = chunkStart + chunkSize; + + if (chunkEnd > wavBuffer.length) { + return null; + } + + if (chunkId === 'data') { + return wavBuffer.subarray(chunkStart, chunkEnd); + } + + offset = chunkEnd + (chunkSize % 2); + } + + return null; + } +} diff --git a/packages/core/src/services/voice/types.ts b/packages/core/src/services/voice/types.ts new file mode 100644 index 00000000000..b881d58b03a --- /dev/null +++ b/packages/core/src/services/voice/types.ts @@ -0,0 +1,44 @@ +/** + * @license + * Copyright 2025 Google LLC + * SPDX-License-Identifier: Apache-2.0 + */ + +export interface VoiceInputState { + isRecording: boolean; + isTranscribing: boolean; + error: string | null; +} + +export interface VoiceBackend { + start(): Promise; + stop(): Promise; + /** Cancel recording without transcribing (e.g., user pressed Escape). */ + cancel(): Promise; + cleanup(): Promise; +} + +/** + * Options passed to a VoiceBackend at construction time. + * Transcripts are delivered via coreEvents (CoreEvent.VoiceTranscript) + * rather than a callback, to align with the project's cross-service + * communication pattern. + */ +export interface VoiceBackendOptions { + onStateChange: (state: VoiceInputState) => void | Promise; + /** + * RMS energy threshold below which audio is treated as silence and + * discarded without an API call. 0 disables silence detection entirely. + * Default: 80 (allows whispered speech; blocks near-silence). + */ + silenceThreshold?: number; +} + +export interface VoiceInputReturn { + isEnabled: boolean; + state: VoiceInputState; + startRecording: () => Promise; + stopRecording: () => Promise; + cancelRecording: () => Promise; + toggleRecording: () => Promise; +} diff --git a/packages/core/src/utils/events.ts b/packages/core/src/utils/events.ts index 9548146f9d9..4b65adc0572 100644 --- a/packages/core/src/utils/events.ts +++ b/packages/core/src/utils/events.ts @@ -203,6 +203,7 @@ export enum CoreEvent { QuotaChanged = 'quota-changed', TelemetryKeychainAvailability = 'telemetry-keychain-availability', TelemetryTokenStorageType = 'telemetry-token-storage-type', + VoiceTranscript = 'voice-transcript', } /** @@ -211,7 +212,6 @@ export enum CoreEvent { export interface EditorSelectedPayload { editor?: EditorType; } - export interface CoreEvents extends ExtensionEvents { [CoreEvent.UserFeedback]: [UserFeedbackPayload]; [CoreEvent.ModelChanged]: [ModelChangedPayload]; @@ -237,6 +237,7 @@ export interface CoreEvents extends ExtensionEvents { [CoreEvent.SlashCommandConflicts]: [SlashCommandConflictsPayload]; [CoreEvent.TelemetryKeychainAvailability]: [KeychainAvailabilityEvent]; [CoreEvent.TelemetryTokenStorageType]: [TokenStorageInitializationEvent]; + [CoreEvent.VoiceTranscript]: [string]; } type EventBacklogItem = { @@ -446,6 +447,10 @@ export class CoreEventEmitter extends EventEmitter { emitTelemetryTokenStorageType(event: TokenStorageInitializationEvent): void { this._emitOrQueue(CoreEvent.TelemetryTokenStorageType, event); } + + emitVoiceTranscript(transcript: string): void { + this.emit(CoreEvent.VoiceTranscript, transcript); + } } export const coreEvents = new CoreEventEmitter(); diff --git a/schemas/settings.schema.json b/schemas/settings.schema.json index 03ea0b2fda0..9931b65c70f 100644 --- a/schemas/settings.schema.json +++ b/schemas/settings.schema.json @@ -565,6 +565,44 @@ }, "additionalProperties": false }, + "voice": { + "title": "Voice Input", + "description": "Settings for voice input. Note: Voice input is not natively supported in WSL2 (Windows Subsystem for Linux).", + "markdownDescription": "Settings for voice input. Note: Voice input is not natively supported in WSL2 (Windows Subsystem for Linux).\n\n- Category: `General`\n- Requires restart: `no`\n- Default: `{}`", + "default": {}, + "type": "object", + "properties": { + "enabled": { + "title": "Enable Voice Input", + "description": "Enable voice input support. Note: Voice input is not natively supported in WSL2 (Windows Subsystem for Linux).", + "markdownDescription": "Enable voice input support. Note: Voice input is not natively supported in WSL2 (Windows Subsystem for Linux).\n\n- Category: `General`\n- Requires restart: `no`\n- Default: `false`", + "default": false, + "type": "boolean" + }, + "provider": { + "title": "Transcription Backend", + "description": "Set transcription backend: \"gemini\" (default, zero-install) or \"whisper\" (local).", + "markdownDescription": "Set transcription backend: \"gemini\" (default, zero-install) or \"whisper\" (local).\n\n- Category: `General`\n- Requires restart: `no`\n- Default: `gemini`", + "default": "gemini", + "type": "string", + "enum": ["gemini", "whisper"] + }, + "whisperPath": { + "title": "Whisper Binary Path", + "description": "Path to the whisper executable. Only used when provider is \"whisper\".", + "markdownDescription": "Path to the whisper executable. Only used when provider is \"whisper\".\n\n- Category: `General`\n- Requires restart: `no`", + "type": "string" + }, + "silenceThreshold": { + "title": "Silence Detection Threshold", + "description": "RMS energy threshold (0–1000) below which audio is discarded as silence. Lower values allow quieter speech such as whispering. 0 disables silence detection.", + "markdownDescription": "RMS energy threshold (0–1000) below which audio is discarded as silence. Lower values allow quieter speech such as whispering. 0 disables silence detection.\n\n- Category: `General`\n- Requires restart: `no`\n- Default: `80`", + "default": 80, + "type": "number" + } + }, + "additionalProperties": false + }, "ide": { "title": "IDE", "description": "IDE integration settings.", diff --git a/scripts/generate-keybindings-doc.ts b/scripts/generate-keybindings-doc.ts index 10c95d96495..937953acdcb 100644 --- a/scripts/generate-keybindings-doc.ts +++ b/scripts/generate-keybindings-doc.ts @@ -8,7 +8,10 @@ import path from 'node:path'; import { fileURLToPath, pathToFileURL } from 'node:url'; import { readFile, writeFile } from 'node:fs/promises'; -import type { KeyBinding } from '../packages/cli/src/ui/key/keyBindings.js'; +import { + Command, + type KeyBinding, +} from '../packages/cli/src/ui/key/keyBindings.js'; import { commandCategories, commandDescriptions, @@ -79,14 +82,20 @@ export async function main(argv = process.argv.slice(2)) { } export function buildDefaultDocSections(): readonly KeybindingDocSection[] { - return commandCategories.map((category) => ({ - title: category.title, - commands: category.commands.map((command) => ({ - command: command, - description: commandDescriptions[command], - bindings: defaultKeyBindingConfig.get(command) ?? [], - })), - })); + const commandsExcludedFromAutogen = new Set([Command.VOICE_INPUT]); + + return commandCategories + .map((category) => ({ + title: category.title, + commands: category.commands + .filter((command) => !commandsExcludedFromAutogen.has(command)) + .map((command) => ({ + command, + description: commandDescriptions[command], + bindings: defaultKeyBindingConfig.get(command) ?? [], + })), + })) + .filter((category) => category.commands.length > 0); } export function renderDocumentation(