From 1e4be5f9c9c273499e3487bbdb62d361ea590841 Mon Sep 17 00:00:00 2001 From: Sachin Sharma Date: Fri, 1 May 2026 15:45:31 +0530 Subject: [PATCH] feat(voice): add multi-provider TTS, STT, and realtime voice integration MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add Speech-to-Text (STT) as a new capability alongside existing TTS, with multi-provider support for both. Everything flows through the existing generate() and stream() JSON config pattern. New TTS providers (via generate({ tts: { provider: "..." } })): - openai-tts: OpenAI TTS API (tts-1, tts-1-hd), 6 voices - elevenlabs: ElevenLabs (eleven_multilingual_v2) - azure-tts: Azure Cognitive Services Speech New STT providers (via generate({ stt: { enabled: true, audio, provider } })): - whisper/openai-stt: OpenAI Whisper API - google-stt: Google Cloud Speech-to-Text - deepgram: Deepgram Nova-2/Nova-3 - azure-stt: Azure Cognitive Services Speech Realtime providers (registered for future SDK use): - openai-realtime: OpenAI Realtime API (WebSocket) - gemini-live: Google Gemini Live (WebSocket) Infrastructure: - STTProcessor (mirrors TTSProcessor) with SpanType.STT observability - Audio utilities: format detection, WAV creation, PCM resampling - ChunkedAudioStream with backpressure and validation - 30-second fetch timeout on all voice provider API calls - CLI flags: --stt, --stt-provider, --input-audio, --stt-language, --tts-provider STT pipeline in generate(): - When stt.audio provided without text: transcription becomes the prompt - When stt.audio provided with text: transcription prepended as context - result.transcription contains STTResult with text + confidence Tested end-to-end with real API calls: - Google TTS, OpenAI TTS, ElevenLabs, Azure TTS (valid MP3 output) - Whisper STT (0.95), Deepgram (1.0), Google STT (0.98), Azure STT (0.9) - Full round-trip: Whisper→Vertex LLM→ElevenLabs (audio in → audio out) --- .env.example | 90 +- CHANGELOG.md | 112 - README.md | 40 +- docs-site/scripts/sync-docs.ts | 4 + docs-site/sidebars.ts | 11 + docs/features/audio-input.md | 68 +- docs/features/tts.md | 68 +- docs/getting-started/environment-variables.md | 110 + docs/getting-started/provider-setup.md | 193 ++ docs/getting-started/providers/deepgram.md | 495 +++++ docs/getting-started/providers/elevenlabs.md | 431 ++++ docs/getting-started/providers/index.md | 140 ++ docs/getting-started/providers/openai-tts.md | 366 ++++ .../14-voice-speech-integration.md | 452 ++++ docs/provider-integration/README.md | 18 +- docs/reference/provider-comparison.md | 152 ++ docs/reference/provider-selection.md | 161 +- examples/cli-examples.sh | 46 + .../voice-bridge-implementation-plan.md | 453 ++++ memory-bank/voice-cleanup-plan.md | 251 +++ package.json | 34 +- pnpm-lock.yaml | 68 +- src/cli/factories/commandFactory.ts | 83 +- src/cli/factories/sagemakerCommandFactory.ts | 33 +- src/cli/loop/optionsSchema.ts | 1 + src/lib/core/baseProvider.ts | 19 +- src/lib/factories/providerRegistry.ts | 115 ++ src/lib/neurolink.ts | 72 +- .../exporters/laminarExporter.ts | 1 + .../exporters/posthogExporter.ts | 1 + src/lib/observability/utils/spanSerializer.ts | 1 + src/lib/server/voice/voiceWebSocketHandler.ts | 882 ++++---- src/lib/types/generate.ts | 44 + src/lib/types/index.ts | 2 +- src/lib/types/realtime.ts | 322 +++ src/lib/types/server.ts | 11 - src/lib/types/span.ts | 2 + src/lib/types/stream.ts | 8 + src/lib/types/stt.ts | 772 +++++++ src/lib/types/tts.ts | 22 +- src/lib/types/voice.ts | 484 +++++ src/lib/utils/sttProcessor.ts | 319 +++ src/lib/voice/RealtimeVoiceAPI.ts | 516 +++++ src/lib/voice/audio-utils.ts | 552 +++++ src/lib/voice/errors.ts | 464 +++++ src/lib/voice/index.ts | 125 ++ src/lib/voice/providers/AzureSTT.ts | 374 ++++ src/lib/voice/providers/AzureTTS.ts | 357 ++++ src/lib/voice/providers/DeepgramSTT.ts | 564 +++++ src/lib/voice/providers/ElevenLabsTTS.ts | 326 +++ src/lib/voice/providers/GeminiLive.ts | 418 ++++ src/lib/voice/providers/GoogleSTT.ts | 482 +++++ src/lib/voice/providers/OpenAIRealtime.ts | 471 +++++ src/lib/voice/providers/OpenAISTT.ts | 317 +++ src/lib/voice/providers/OpenAITTS.ts | 262 +++ src/lib/voice/stream-handler.ts | 546 +++++ test/continuous-test-suite-voice.ts | 1822 +++++++++++++++++ 57 files changed, 13708 insertions(+), 845 deletions(-) create mode 100644 docs/getting-started/providers/deepgram.md create mode 100644 docs/getting-started/providers/elevenlabs.md create mode 100644 docs/getting-started/providers/openai-tts.md create mode 100644 docs/provider-integration/14-voice-speech-integration.md create mode 100644 memory-bank/voice-bridge-implementation-plan.md create mode 100644 memory-bank/voice-cleanup-plan.md create mode 100644 src/lib/types/realtime.ts create mode 100644 src/lib/types/stt.ts create mode 100644 src/lib/types/voice.ts create mode 100644 src/lib/utils/sttProcessor.ts create mode 100644 src/lib/voice/RealtimeVoiceAPI.ts create mode 100644 src/lib/voice/audio-utils.ts create mode 100644 src/lib/voice/errors.ts create mode 100644 src/lib/voice/index.ts create mode 100644 src/lib/voice/providers/AzureSTT.ts create mode 100644 src/lib/voice/providers/AzureTTS.ts create mode 100644 src/lib/voice/providers/DeepgramSTT.ts create mode 100644 src/lib/voice/providers/ElevenLabsTTS.ts create mode 100644 src/lib/voice/providers/GeminiLive.ts create mode 100644 src/lib/voice/providers/GoogleSTT.ts create mode 100644 src/lib/voice/providers/OpenAIRealtime.ts create mode 100644 src/lib/voice/providers/OpenAISTT.ts create mode 100644 src/lib/voice/providers/OpenAITTS.ts create mode 100644 src/lib/voice/stream-handler.ts create mode 100644 test/continuous-test-suite-voice.ts diff --git a/.env.example b/.env.example index 3a95abd77..bb569a878 100644 --- a/.env.example +++ b/.env.example @@ -405,16 +405,6 @@ NEUROLINK_IMAGE_MAX_SIZE=10485760 # NEUROLINK_IMAGE_CACHE_TTL_MS=3600000 # 1 hour TTL (longer retention) # NEUROLINK_IMAGE_MAX_SIZE=5242880 # 5MB max per image (smaller limit) -# ============================================================================= -# CONVERSATION TITLE GENERATION (Optional) -# ============================================================================= -# Customize the prompt used to generate conversation titles from user messages. -# Use ${userMessage} placeholder to include the user's message in the prompt. -# If not set, uses default prompt optimized for 20-character titles. -# -NEUROLINK_TITLE_PROMPT='Create a short 3-word title for: ${userMessage}' -# - # ============================================================================= # CONVERSATION MEMORY CONFIGURATION (Optional) # ============================================================================= @@ -806,61 +796,33 @@ PICOVOICE_ACCESS_KEY= VOICE_LLM_MODEL=gpt-4o-automatic VOICE_LLM_PROVIDER=azure -# ============================================================================= -# DEEPSEEK CONFIGURATION -# ============================================================================= -# Get an API key at https://platform.deepseek.com/api_keys -DEEPSEEK_API_KEY= -# Optional: override default model (deepseek-chat | deepseek-reasoner) -DEEPSEEK_MODEL=deepseek-chat -# Optional: override default base URL -# DEEPSEEK_BASE_URL=https://api.deepseek.com +# === Voice Provider Credentials (TTS / STT) === +# ElevenLabs TTS (https://elevenlabs.io) +ELEVENLABS_API_KEY= -# ============================================================================= -# NVIDIA NIM CONFIGURATION -# ============================================================================= -# Get an API key at https://build.nvidia.com/settings/api-keys -NVIDIA_NIM_API_KEY= -# Optional: override default model (browse https://build.nvidia.com/models) -NVIDIA_NIM_MODEL=meta/llama-3.3-70b-instruct -# Optional: override default base URL (use for self-hosted NIM) -# NVIDIA_NIM_BASE_URL=https://integrate.api.nvidia.com/v1 -# Optional NIM extras (rarely needed, leave commented) -# NVIDIA_NIM_TOP_K= -# NVIDIA_NIM_MIN_P= -# NVIDIA_NIM_REPETITION_PENALTY= -# NVIDIA_NIM_MIN_TOKENS= -# NVIDIA_NIM_CHAT_TEMPLATE= +# Deepgram STT (https://deepgram.com) +DEEPGRAM_API_KEY= -# ============================================================================= -# LM STUDIO CONFIGURATION (local provider) -# ============================================================================= -# Install LM Studio: https://lmstudio.ai/ -# Load a model in the app and click "Start Server" -LM_STUDIO_BASE_URL=http://localhost:1234/v1 -# Optional: explicit model id (blank = auto-discover from /v1/models) -LM_STUDIO_MODEL= -# Optional: API key (not required for stock LM Studio; use only when running -# behind an auth-proxying reverse-proxy) -# LM_STUDIO_API_KEY= +# Azure Cognitive Services Speech (TTS + STT) +# Get from: Azure Portal → Cognitive Services → Speech → Keys and Endpoint +AZURE_SPEECH_KEY= +AZURE_SPEECH_REGION=eastus -# ============================================================================= -# LLAMA.CPP CONFIGURATION (local provider) -# ============================================================================= -# Run: ./llama-server -m model.gguf --port 8080 (add --jinja for tool support) -LLAMACPP_BASE_URL=http://localhost:8080/v1 -# Optional: explicit model id (blank = use whatever model llama-server has loaded) -LLAMACPP_MODEL= -# Optional: API key (not required for stock llama-server; use only when running -# behind an auth-proxying reverse-proxy) -# LLAMACPP_API_KEY= +# Google API Key (for Gemini Live realtime voice) +GOOGLE_API_KEY= + +# === Voice Provider Credentials (TTS / STT) === +# ElevenLabs TTS (https://elevenlabs.io) +ELEVENLABS_API_KEY= + +# Deepgram STT (https://deepgram.com) +DEEPGRAM_API_KEY= + +# Azure Cognitive Services Speech (TTS + STT) +# Get from: Azure Portal → Cognitive Services → Speech → Keys and Endpoint +AZURE_SPEECH_KEY= +AZURE_SPEECH_REGION=eastus + +# Google API Key (for Gemini Live realtime voice) +GOOGLE_API_KEY= -# ============================================================================= -# TEST-ONLY CREDENTIALS (used by test/continuous-test-suite-credentials.ts and -# test/continuous-test-suite-new-providers.ts to verify per-call overrides -# without depending on the runtime env vars above) -# ============================================================================= -# TEST_DEEPSEEK_API_KEY= -# TEST_NVIDIA_NIM_API_KEY= -# TEST_LM_STUDIO_BASE_URL= -# TEST_LLAMACPP_BASE_URL= diff --git a/CHANGELOG.md b/CHANGELOG.md index a0ae96b6c..7b9f2b4d1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,115 +1,3 @@ -## [9.60.1](https://github.com/juspay/neurolink/compare/v9.60.0...v9.60.1) (2026-04-30) - -### Bug Fixes - -- **(proxy):** validate pnpm global store compatibility before auto-update install ([ac573ad](https://github.com/juspay/neurolink/commit/ac573adc688aa28376d4b39ffa8e6bb7539cb40e)) - -## [9.60.0](https://github.com/juspay/neurolink/compare/v9.59.6...v9.60.0) (2026-04-30) - -### Features - -- **(providers):** integrate DeepSeek, NVIDIA NIM, LM Studio, llama.cpp ([c829f4d](https://github.com/juspay/neurolink/commit/c829f4dea09bf3a6eae08c4902f9293bfb6c05f6)) - -## [9.59.6](https://github.com/juspay/neurolink/compare/v9.59.5...v9.59.6) (2026-04-30) - -### Bug Fixes - -- **(tools):** start execution timeout after HITL approval ([1e6d3e0](https://github.com/juspay/neurolink/commit/1e6d3e044a216b67674116830a3917cf1df7171a)) - -## [9.59.5](https://github.com/juspay/neurolink/compare/v9.59.4...v9.59.5) (2026-04-29) - -### Bug Fixes - -- **(routing):** dual-mode image text fallback + skip video-frame hijack on structured output ([97b2373](https://github.com/juspay/neurolink/commit/97b2373a793e56414a2ca41009efc72d9a574999)) - -## [9.59.4](https://github.com/juspay/neurolink/compare/v9.59.3...v9.59.4) (2026-04-27) - -### Bug Fixes - -- **(proxy):** replace blocking quiet-gate with best-effort wait for auto-updates ([defd6e0](https://github.com/juspay/neurolink/commit/defd6e0f177abea19e490ebe8bd8ea492f6bebca)) - -## [9.59.3](https://github.com/juspay/neurolink/compare/v9.59.2...v9.59.3) (2026-04-27) - -### Bug Fixes - -- **(observability):** enrich NoOutputGeneratedError sentinel chunk metadata + actually trigger the catch path the production bug needs ([6854af1](https://github.com/juspay/neurolink/commit/6854af103688dd20093bbfef15e2252b4b502b51)) - -## [9.59.2](https://github.com/juspay/neurolink/compare/v9.59.1...v9.59.2) (2026-04-26) - -### Bug Fixes - -- **(context):** pre-dispatch compaction + hard cap for inline conversationMessages on both generate and stream paths + compaction.insufficient event ([d39739f](https://github.com/juspay/neurolink/commit/d39739fc6ac01aa64e309481e0a8fe53525e9c7f)) - -## [9.59.1](https://github.com/juspay/neurolink/compare/v9.59.0...v9.59.1) (2026-04-26) - -### Bug Fixes - -- **(observability):** emit generation:end exactly once on stream finalize ([9bd2cd0](https://github.com/juspay/neurolink/commit/9bd2cd0a16484fa93ef8c5aaa018b4323632f094)) - -## [9.59.0](https://github.com/juspay/neurolink/compare/v9.58.0...v9.59.0) (2026-04-26) - -### Features - -- **(errors):** typed ModelAccessDeniedError + sdk.checkCredentials() API ([1ffc5bc](https://github.com/juspay/neurolink/commit/1ffc5bc44ce411086f130cc7ee33cb094290b108)) - -## [9.58.0](https://github.com/juspay/neurolink/compare/v9.57.1...v9.58.0) (2026-04-26) - -### Features - -- **(fallback):** providerFallback callback + modelChain config for centralized policy ([92e5026](https://github.com/juspay/neurolink/commit/92e5026ac48ca98c640cd1793b4c194c8b84a128)) - -## [9.57.1](https://github.com/juspay/neurolink/compare/v9.57.0...v9.57.1) (2026-04-25) - -### Bug Fixes - -- **(conversation-memory):** stop persisting abort sentinel; add typed AbortError + read-time filter (SI-069/SI-071) ([595b355](https://github.com/juspay/neurolink/commit/595b3558d9bc4eeb4121f0504b9f077ef5f01729)) - -## [9.57.0](https://github.com/juspay/neurolink/compare/v9.56.2...v9.57.0) (2026-04-25) - -### Features - -- **(dynamic-args):** add dynamic argument resolution with context-aware utilities ([673b2a2](https://github.com/juspay/neurolink/commit/673b2a213f6ac095645c670280ae4a2bb22946b5)) - -## [9.56.2](https://github.com/juspay/neurolink/compare/v9.56.1...v9.56.2) (2026-04-24) - -### Bug Fixes - -- **(files):** honor caller-provided mimetype hint for extension-less buffers ([40276cc](https://github.com/juspay/neurolink/commit/40276cc9abad565089b8161a1e7a9c2eb533df1f)) - -## [9.56.1](https://github.com/juspay/neurolink/compare/v9.56.0...v9.56.1) (2026-04-21) - -### Bug Fixes - -- **(context):** Add support to filter out empty content chunks ([5f13d91](https://github.com/juspay/neurolink/commit/5f13d919cb5342dce3c2796fa22436ad6aceb318)) - -## [9.56.0](https://github.com/juspay/neurolink/compare/v9.55.11...v9.56.0) (2026-04-20) - -### Features - -- **(logs):** add logs in stream function flow ([730efdc](https://github.com/juspay/neurolink/commit/730efdcca0a509480d0e41c2ee1d0ee25f6b9931)) - -## [9.55.11](https://github.com/juspay/neurolink/compare/v9.55.10...v9.55.11) (2026-04-20) - -### Bug Fixes - -- **(observability):** close Curator-reported Langfuse telemetry gaps ([42ed72a](https://github.com/juspay/neurolink/commit/42ed72acf59cca32138b4441c1331f4ed7497454)) - -## [9.55.10](https://github.com/juspay/neurolink/compare/v9.55.9...v9.55.10) (2026-04-19) - -## [9.55.9](https://github.com/juspay/neurolink/compare/v9.55.8...v9.55.9) (2026-04-19) - -## [9.55.8](https://github.com/juspay/neurolink/compare/v9.55.7...v9.55.8) (2026-04-19) - -## [9.55.7](https://github.com/juspay/neurolink/compare/v9.55.6...v9.55.7) (2026-04-19) - -## [9.55.6](https://github.com/juspay/neurolink/compare/v9.55.5...v9.55.6) (2026-04-18) - -## [9.55.5](https://github.com/juspay/neurolink/compare/v9.55.4...v9.55.5) (2026-04-18) - -## [9.55.4](https://github.com/juspay/neurolink/compare/v9.55.3...v9.55.4) (2026-04-18) - -## [9.55.3](https://github.com/juspay/neurolink/compare/v9.55.2...v9.55.3) (2026-04-18) - ## [9.55.2](https://github.com/juspay/neurolink/compare/v9.55.1...v9.55.2) (2026-04-18) ## [9.55.1](https://github.com/juspay/neurolink/compare/v9.55.0...v9.55.1) (2026-04-18) diff --git a/README.md b/README.md index 1419a17a5..47db6ebc8 100644 --- a/README.md +++ b/README.md @@ -40,23 +40,25 @@ Extracted from production systems at Juspay and battle-tested at enterprise scal ## What's New (Q1 2026) -| Feature | Version | Description | Guide | -| ----------------------------------- | ------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------- | -| **Gemini 3 Multi-turn Tool Fix** | v9.49.0 | Fixed multi-step agentic tool calling on Vertex AI Gemini 3 models. Correct `thoughtSignature` replay, `stepIndex` parallel-call grouping, `executionId` session isolation, 5-min timeout, silent-timeout surfacing. | [Vertex AI Guide](docs/getting-started/providers/google-vertex.md) | -| **AutoResearch** | v9.17.0 | Autonomous AI experiment engine: proposes code changes, runs experiments, evaluates metrics, keeps improvements — unattended for hours. | [AutoResearch Guide](docs/features/autoresearch.md) | -| **MCP Enhancements** | v9.16.0 | Advanced MCP features: tool routing, result caching, request batching, annotations, elicitation, custom server base, multi-server management | [MCP Enhancements Guide](docs/features/mcp-enhancements.md) | -| **Memory** | v9.12.0 | Per-user condensed memory that persists across conversations. LLM-powered condensation with S3, Redis, or SQLite backends. | [Memory Guide](docs/features/memory.md) | -| **Context Window Management** | v9.2.0 | 4-stage compaction pipeline with auto-detection, budget gate at 80% usage, per-provider token estimation | [Context Compaction Guide](docs/features/context-compaction.md) | -| **Tool Execution Control** | v9.3.0 | `prepareStep` and `toolChoice` support for per-step tool enforcement in multi-step agentic loops. API-level control over tool calls. | [API Reference](docs/api/type-aliases/GenerateOptions.md#preparestep) | -| **File Processor System** | v9.1.0 | 17+ file type processors with ProcessorRegistry, security sanitization, SVG text injection | [File Processors Guide](docs/features/file-processors.md) | -| **RAG with generate()/stream()** | v9.2.0 | Pass `rag: { files }` to generate/stream for automatic document chunking, embedding, and AI-powered search. 10 chunking strategies, hybrid search, reranking. | [RAG Guide](docs/features/rag.md) | -| **External TracerProvider Support** | v8.43.0 | Integrate NeuroLink with existing OpenTelemetry instrumentation. Prevents duplicate registration conflicts. | [Observability Guide](docs/features/observability.md) | -| **Server Adapters** | v8.43.0 | Multi-framework HTTP server with Hono, Express, Fastify, Koa support. Full CLI for server management with foreground/background modes. | [Server Adapters Guide](docs/guides/server-adapters/index.md) | -| **Title Generation Events** | v8.38.0 | Emit `conversation:titleGenerated` event when conversation title is generated. Supports custom title prompts via `NEUROLINK_TITLE_PROMPT`. | [Conversation Memory Guide](docs/conversation-memory.md) | -| **Video Generation with Veo** | v8.32.0 | Video generation using Veo 3.1 (`veo-3.1`). Realistic video generation with many parameter options | [Video Generation Guide](docs/features/video-generation.md) | -| **Image Generation with Gemini** | v8.31.0 | Native image generation using Gemini 2.0 Flash Experimental (`imagen-3.0-generate-002`). High-quality image synthesis directly from Google AI. | [Image Generation Guide](docs/image-generation-streaming.md) | -| **HTTP/Streamable HTTP Transport** | v8.29.0 | Connect to remote MCP servers via HTTP with authentication headers, automatic retry with exponential backoff, and configurable rate limiting. | [HTTP Transport Guide](docs/mcp-http-transport.md) | - +| Feature | Version | Description | Guide | +| ----------------------------------- | ------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------ | +| **Multi-Provider Voice (TTS/STT)** | v8.42.0 | 6 TTS providers (OpenAI TTS, ElevenLabs, Google TTS, Azure Speech, Gemini Live) + 6 STT providers (Whisper, Deepgram, AssemblyAI, Azure, Google STT, Gladia). Realtime voice APIs. CLI `--tts` / `--stt` flags. | [TTS Guide](docs/features/tts.md) \| [STT Guide](docs/features/audio-input.md) | +| **Gemini 3 Multi-turn Tool Fix** | v9.49.0 | Fixed multi-step agentic tool calling on Vertex AI Gemini 3 models. Correct `thoughtSignature` replay, `stepIndex` parallel-call grouping, `executionId` session isolation, 5-min timeout, silent-timeout surfacing. | [Vertex AI Guide](docs/getting-started/providers/google-vertex.md) | +| **AutoResearch** | v9.17.0 | Autonomous AI experiment engine: proposes code changes, runs experiments, evaluates metrics, keeps improvements — unattended for hours. | [AutoResearch Guide](docs/features/autoresearch.md) | +| **MCP Enhancements** | v9.16.0 | Advanced MCP features: tool routing, result caching, request batching, annotations, elicitation, custom server base, multi-server management | [MCP Enhancements Guide](docs/features/mcp-enhancements.md) | +| **Memory** | v9.12.0 | Per-user condensed memory that persists across conversations. LLM-powered condensation with S3, Redis, or SQLite backends. | [Memory Guide](docs/features/memory.md) | +| **Context Window Management** | v9.2.0 | 4-stage compaction pipeline with auto-detection, budget gate at 80% usage, per-provider token estimation | [Context Compaction Guide](docs/features/context-compaction.md) | +| **Tool Execution Control** | v9.3.0 | `prepareStep` and `toolChoice` support for per-step tool enforcement in multi-step agentic loops. API-level control over tool calls. | [API Reference](docs/api/type-aliases/GenerateOptions.md#preparestep) | +| **File Processor System** | v9.1.0 | 17+ file type processors with ProcessorRegistry, security sanitization, SVG text injection | [File Processors Guide](docs/features/file-processors.md) | +| **RAG with generate()/stream()** | v9.2.0 | Pass `rag: { files }` to generate/stream for automatic document chunking, embedding, and AI-powered search. 10 chunking strategies, hybrid search, reranking. | [RAG Guide](docs/features/rag.md) | +| **External TracerProvider Support** | v8.43.0 | Integrate NeuroLink with existing OpenTelemetry instrumentation. Prevents duplicate registration conflicts. | [Observability Guide](docs/features/observability.md) | +| **Server Adapters** | v8.43.0 | Multi-framework HTTP server with Hono, Express, Fastify, Koa support. Full CLI for server management with foreground/background modes. | [Server Adapters Guide](docs/guides/server-adapters/index.md) | +| **Title Generation Events** | v8.38.0 | Emit `conversation:titleGenerated` event when conversation title is generated. Supports custom title prompts via `NEUROLINK_TITLE_PROMPT`. | [Conversation Memory Guide](docs/conversation-memory.md) | +| **Video Generation with Veo** | v8.32.0 | Video generation using Veo 3.1 (`veo-3.1`). Realistic video generation with many parameter options | [Video Generation Guide](docs/features/video-generation.md) | +| **Image Generation with Gemini** | v8.31.0 | Native image generation using Gemini 2.0 Flash Experimental (`imagen-3.0-generate-002`). High-quality image synthesis directly from Google AI. | [Image Generation Guide](docs/image-generation-streaming.md) | +| **HTTP/Streamable HTTP Transport** | v8.29.0 | Connect to remote MCP servers via HTTP with authentication headers, automatic retry with exponential backoff, and configurable rate limiting. | [HTTP Transport Guide](docs/mcp-http-transport.md) | + +- **Multi-Provider Voice (TTS/STT)** – Full voice pipeline with 6 TTS providers (OpenAI TTS, ElevenLabs, Google TTS, Azure Speech, Gemini Live) and 6 STT providers (Whisper, Deepgram, AssemblyAI, Azure Speech, Google STT, Gladia). Realtime voice APIs for OpenAI and Gemini Live. Pass `--tts` / `--stt` flags in the CLI or configure via `voice` in the SDK. → [TTS Guide](docs/features/tts.md) | [STT Guide](docs/features/audio-input.md) | [Realtime Guide](docs/features/real-time-services.md) - **AutoResearch** – Autonomous AI experiment engine inspired by Karpathy's autoresearch. Phase-gated tool access, git-backed safety, deterministic metric evaluation, and TaskManager integration for continuous unattended research. 12 research tools, 10 typed events, 9 CLI subcommands. → [AutoResearch Guide](docs/features/autoresearch.md) - **Memory** – Per-user condensed memory that persists across all conversations. Automatically retrieves and stores memory on each `generate()`/`stream()` call. Supports S3, Redis, and SQLite storage with LLM-powered condensation. → [Memory Guide](docs/features/memory.md) - **External TracerProvider Support** – Integrate NeuroLink with applications that already have OpenTelemetry instrumentation. Supports auto-detection and manual configuration. → [Observability Guide](docs/features/observability.md) @@ -435,6 +437,10 @@ NeuroLink is a comprehensive AI development platform. Every feature below is pro | **NVIDIA NIM** | Llama 3.3 70B, 400+ catalog models | ❌ | ✅ Full | ✅ Production | [Setup Guide](docs/getting-started/provider-setup.md#nvidia-nim) | | **LM Studio** | Any model loaded in LM Studio (local) | ✅ Free (Local) | ✅ Full | ✅ Production | [Setup Guide](docs/getting-started/provider-setup.md#lm-studio) | | **llama.cpp** | Any GGUF model served by llama-server (local) | ✅ Free (Local) | ✅ Full | ✅ Production | [Setup Guide](docs/getting-started/provider-setup.md#llamacpp) | +| **OpenAI TTS** | TTS-1, TTS-1-HD, GPT-4o Audio | ❌ | N/A | ✅ Production | [Setup Guide](docs/getting-started/provider-setup.md#openai-tts) | +| **ElevenLabs** | Multilingual v2, Turbo v2.5, Flash v2.5 | ✅ Free Tier | N/A | ✅ Production | [Setup Guide](docs/getting-started/provider-setup.md#elevenlabs) | +| **Deepgram** | Nova-3, Nova-2, Enhanced, Base (STT) | ✅ Free Tier | N/A | ✅ Production | [Setup Guide](docs/getting-started/provider-setup.md#deepgram) | +| **Azure Speech** | Azure Cognitive Services TTS + STT | ❌ | N/A | ✅ Production | [Setup Guide](docs/getting-started/provider-setup.md#azure-speech) | **[📖 Provider Comparison Guide](docs/reference/provider-comparison.md)** - Detailed feature matrix and selection criteria **[🔬 Provider Feature Compatibility](docs/reference/provider-feature-compatibility.md)** - Test-based compatibility reference for all 19 features across 13 providers diff --git a/docs-site/scripts/sync-docs.ts b/docs-site/scripts/sync-docs.ts index 9f6e14e35..0862bfbd7 100644 --- a/docs-site/scripts/sync-docs.ts +++ b/docs-site/scripts/sync-docs.ts @@ -1096,6 +1096,10 @@ const LINK_MAPPINGS: Record = { "builtin-middleware": "/advanced/builtin-middleware", "google-vertex": "/getting-started/providers/google-vertex", huggingface: "/getting-started/providers/huggingface", + "openai-tts": "/getting-started/providers/openai-tts", + elevenlabs: "/getting-started/providers/elevenlabs", + deepgram: "/getting-started/providers/deepgram", + "azure-speech": "/getting-started/providers/azure-speech", // API documentation paths - these need /api/ prefix "enumerations/AIProviderName": "/api/enumerations/AIProviderName", diff --git a/docs-site/sidebars.ts b/docs-site/sidebars.ts index a79159e04..0c4490c49 100644 --- a/docs-site/sidebars.ts +++ b/docs-site/sidebars.ts @@ -33,6 +33,17 @@ const sidebars: SidebarsConfig = { "getting-started/providers/sagemaker", "getting-started/providers/openrouter", "getting-started/providers/openai-compatible", + { + type: "category", + label: "Voice Providers", + collapsed: true, + items: [ + "getting-started/providers/openai-tts", + "getting-started/providers/elevenlabs", + "getting-started/providers/deepgram", + "getting-started/providers/azure-speech", + ], + }, ], }, ], diff --git a/docs/features/audio-input.md b/docs/features/audio-input.md index 731898c93..115809ca4 100644 --- a/docs/features/audio-input.md +++ b/docs/features/audio-input.md @@ -15,7 +15,8 @@ NeuroLink provides comprehensive audio input capabilities, enabling real-time vo NeuroLink supports the following audio capabilities today: - **Real-time voice conversations** via Gemini Live (Google AI Studio) -- **Text-to-Speech (TTS) output** via Google Cloud TTS integration +- **Text-to-Speech (TTS) output** via Google Cloud TTS, OpenAI TTS, ElevenLabs, and Azure TTS +- **Speech-to-Text (STT)** via `generate()` and `stream()` options (Whisper/OpenAI STT, Google STT, Deepgram, Azure STT) - **WebSocket-based voice streaming** for web applications - **Bidirectional audio** - speak and hear AI responses in real-time @@ -25,22 +26,22 @@ The following features are planned for future releases: - CLI commands: `neurolink audio transcribe`, `neurolink audio analyze`, `neurolink audio summarize` - CLI commands: `neurolink voice chat`, `neurolink voice demo` -- OpenAI Whisper transcription integration -- Cross-provider audio support (Anthropic, Azure, AWS) +- Cross-provider audio support (Anthropic, AWS Transcribe still planned) - File-based audio input processing --- ## Provider Support Matrix -| Provider | Real-time Voice | TTS Output | Audio Transcription | Status | -| -------------------- | --------------- | ---------- | ------------------- | ---------------- | -| **Google AI Studio** | Yes | Yes | Planned | Production Ready | -| **Google Vertex AI** | Planned | Yes | Planned | TTS Available | -| **OpenAI** | Planned | Planned | Planned | Planned | -| **Anthropic** | Planned | Planned | Planned | Planned | -| **Azure OpenAI** | Planned | Planned | Planned | Planned | -| **AWS Bedrock** | Planned | Planned | Planned | Planned | +| Provider | Real-time Voice | TTS Output | Audio Transcription | Status | +| -------------------- | --------------- | ---------- | ---------------------------- | ---------------- | +| **Google AI Studio** | Yes | Yes | Yes (via Google STT) | Production Ready | +| **Google Vertex AI** | Planned | Yes | Yes (via Google STT) | Available | +| **OpenAI** | Planned | Yes | Yes (via Whisper/OpenAI STT) | Available | +| **Deepgram** | Planned | No | Yes | Available | +| **Azure** | Planned | Yes | Yes (via Azure STT) | Available | +| **Anthropic** | Planned | Planned | Planned | Planned | +| **AWS Bedrock** | Planned | Planned | Planned | Planned | **Supported Model for Real-time Voice:** @@ -585,23 +586,42 @@ type AudioContent = { neurolink audio analyze podcast.mp3 --prompt "Summarize key points" ``` -### Phase 3 (Planned) +### Phase 3 (Partially Available) -- **OpenAI Whisper Integration** +- **Speech-to-Text via `generate()` / `stream()`** — Available now via the `stt` option. There is no standalone `neurolink.transcribe()` method; STT is integrated directly into `generate()` and `stream()`: ```typescript - const transcription = await neurolink.transcribe({ - audioFile: "./recording.mp3", - provider: "openai", - model: "whisper-1", - language: "en", + const result = await neurolink.generate({ + input: { text: "Respond to the audio" }, + provider: "vertex", + stt: { + enabled: true, + provider: "google-stt", + audio: audioBuffer, + language: "en-US", + }, }); + console.log("Transcription:", result.transcription?.text); + console.log("AI Response:", result.content); ``` + **CLI equivalent:** + + ```bash + neurolink generate "Respond to the audio" \ + --provider vertex \ + --stt \ + --stt-provider google-stt \ + --input-audio recording.wav + ``` + + **Available STT providers:** `whisper` / `openai-stt`, `google-stt`, `deepgram`, `azure-stt` + + **CLI STT flags:** `--stt`, `--stt-provider `, `--input-audio `, `--stt-language ` + - **Cross-provider Audio Support** - - Anthropic voice capabilities - - Azure Speech Services - - AWS Transcribe + - Anthropic voice capabilities — Planned + - AWS Transcribe — Planned - **File-based Audio Input** ```typescript @@ -724,15 +744,15 @@ NeuroLink's audio input capabilities provide: - Real-time voice conversations via Gemini Live - Bidirectional audio streaming (speak and hear) -- TTS output via Google Cloud +- TTS output via Google Cloud, OpenAI TTS, ElevenLabs, and Azure TTS +- STT via `generate({ stt: { ... } })` — Whisper/OpenAI STT, Google STT, Deepgram, Azure STT - Voice demo example application - PCM16LE audio format support **Planned:** - CLI voice commands (`voice chat`, `audio transcribe`) -- OpenAI Whisper transcription -- Cross-provider audio support +- Anthropic and AWS Transcribe audio support - File-based audio processing **Next Steps:** diff --git a/docs/features/tts.md b/docs/features/tts.md index 7102e0638..a249f06d8 100644 --- a/docs/features/tts.md +++ b/docs/features/tts.md @@ -12,12 +12,13 @@ NeuroLink provides integrated Text-to-Speech (TTS) capabilities, allowing you to **Key Features:** -- **High-quality voices** - Neural, Wavenet, and Standard voice types +- **Multiple providers** - Google Cloud TTS, OpenAI TTS, ElevenLabs, and Azure TTS +- **High-quality voices** - Neural, Wavenet, Standard, and multilingual voice types - **Multiple languages** - 50+ voices across 10+ languages - **Flexible audio formats** - MP3, WAV, OGG/Opus - **Voice customization** - Adjust speed, pitch, and volume - **Two synthesis modes** - Direct text-to-speech OR AI response synthesis -- **Production-ready** - Google Cloud TTS integration +- **Production-ready** - Works with Google Cloud, OpenAI, ElevenLabs, and Azure --- @@ -29,19 +30,29 @@ TTS support is built into NeuroLink. No additional installation required. ### Environment Setup -TTS requires Google Cloud credentials: +Set the appropriate environment variables for your chosen TTS provider: ```bash -# Option 1: Service account (recommended for production) +# Google AI Studio (google-ai) +export GOOGLE_AI_API_KEY="your-api-key" + +# Google Vertex AI (vertex) — service account recommended for production export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service-account.json" -# Option 2: API key (simpler for development) -export GOOGLE_AI_API_KEY="your-api-key" +# OpenAI TTS (openai-tts) +export OPENAI_API_KEY="your-openai-api-key" + +# ElevenLabs (elevenlabs) +export ELEVENLABS_API_KEY="your-elevenlabs-api-key" + +# Azure TTS (azure-tts) +export AZURE_SPEECH_KEY="your-azure-speech-key" +export AZURE_SPEECH_REGION="eastus" # or your Azure region ``` -**API Key Configuration:** +**Google API Key Configuration:** -If using API key authentication, enable both APIs in Google Cloud Console: +If using API key authentication for Google, enable both APIs in Google Cloud Console: 1. Navigate to "APIs & Services" > "Credentials" 2. Create or select your API key @@ -93,17 +104,18 @@ console.log("Audio format:", result.tts?.format); ## Supported Providers -TTS is currently available through Google Cloud Text-to-Speech API: +TTS is available through the following providers: -| Provider | Authentication | Voices | Notes | -| ------------- | -------------------------------------------------- | ---------- | ------------------------------------ | -| **google-ai** | API Key (`GOOGLE_AI_API_KEY`) | 50+ voices | Simplest setup, good for development | -| **vertex** | Service Account (`GOOGLE_APPLICATION_CREDENTIALS`) | 50+ voices | Recommended for production | +| Provider | Authentication | Voices / Models | Notes | +| -------------- | ----------------------------------------------------------- | -------------------------------------------------------------------------- | ------------------------------------ | +| **google-ai** | API Key (`GOOGLE_AI_API_KEY`) | 50+ voices (Neural2, Wavenet, Standard) | Simplest setup, good for development | +| **vertex** | Service Account (`GOOGLE_APPLICATION_CREDENTIALS`) | 50+ voices (Neural2, Wavenet, Standard) | Recommended for production | +| **openai-tts** | API Key (`OPENAI_API_KEY`) | 6 voices: alloy, echo, fable, onyx, nova, shimmer; models: tts-1, tts-1-hd | Good default quality | +| **elevenlabs** | API Key (`ELEVENLABS_API_KEY`) | Multilingual voices; model: eleven_multilingual_v2 | High-quality multilingual synthesis | +| **azure-tts** | API Key (`AZURE_SPEECH_KEY` + region `AZURE_SPEECH_REGION`) | Neural voices with SSML support | Enterprise-grade Azure Speech | **Planned for future releases:** -- OpenAI TTS (GPT-4 voices: alloy, echo, fable, onyx, nova, shimmer) -- Azure Speech Services - AWS Polly --- @@ -413,6 +425,7 @@ if (result.tts?.buffer) { neurolink generate "Your text" \ --provider google-ai \ --tts-voice \ # Required to enable TTS + --tts-provider \ # TTS provider: google-ai|vertex|openai-tts|elevenlabs|azure-tts --tts-format \ # mp3|wav|ogg (default: mp3) --tts-speed \ # 0.25-4.0 (default: 1.0) --tts-pitch \ # -20.0 to 20.0 (default: 0.0) @@ -420,6 +433,19 @@ neurolink generate "Your text" \ --tts-use-ai-response # Synthesize AI response instead of input ``` +**Selecting a specific TTS provider:** + +```bash +# Use OpenAI TTS +neurolink generate "Hello" --tts --tts-provider openai-tts + +# Use ElevenLabs +neurolink generate "Hello" --tts --tts-provider elevenlabs + +# Use Azure TTS +neurolink generate "Hello" --tts --tts-provider azure-tts +``` + --- ## Use Cases & Examples @@ -830,12 +856,12 @@ For detailed pricing, see [Google Cloud TTS Pricing](https://cloud.google.com/te NeuroLink's TTS integration provides: -✅ **High-quality voices** - Neural2, Wavenet, and Standard options -✅ **Multiple languages** - 50+ voices across 10+ languages -✅ **Flexible synthesis modes** - Direct text or AI response -✅ **Voice customization** - Speed, pitch, volume control -✅ **Production-ready** - Google Cloud TTS integration -✅ **Easy integration** - Works seamlessly with CLI and SDK +- **Multiple TTS providers** - Google Cloud TTS, OpenAI TTS, ElevenLabs, Azure TTS +- **High-quality voices** - Neural2, Wavenet, Standard, and multilingual options +- **Multiple languages** - 50+ voices across 10+ languages +- **Flexible synthesis modes** - Direct text or AI response +- **Voice customization** - Speed, pitch, volume control +- **Easy integration** - Works seamlessly with CLI and SDK via `--tts-provider` flag **Next Steps:** diff --git a/docs/getting-started/environment-variables.md b/docs/getting-started/environment-variables.md index 591a671a3..5fd3c8349 100644 --- a/docs/getting-started/environment-variables.md +++ b/docs/getting-started/environment-variables.md @@ -987,6 +987,116 @@ LLAMACPP_MODEL="" # Blank = use whatever model ll --- +### 16. OpenAI TTS + +OpenAI TTS uses the same `OPENAI_API_KEY` as the OpenAI LLM provider. No additional credentials are required. + +#### Required Variables + +```bash +OPENAI_API_KEY="sk-proj-your-openai-api-key" # Shared with the OpenAI LLM provider +``` + +#### How to Get the API Key + +See [### 1. OpenAI](#1-openai) above — the same key is used for both LLM and TTS. + +#### Supported Models + +- `tts-1` (default) - Optimized for speed +- `tts-1-hd` - Optimized for audio quality + +--- + +### 17. ElevenLabs TTS + +#### Required Variables + +```bash +ELEVENLABS_API_KEY="your-elevenlabs-api-key" +``` + +#### How to Get ElevenLabs API Key + +1. Visit [ElevenLabs](https://elevenlabs.io) +2. Sign up or log in to your account +3. Navigate to **Profile → API Key** +4. Copy the key + +#### Supported Models + +- `eleven_multilingual_v2` (default) - Best quality, 29 languages +- `eleven_turbo_v2_5` - Low-latency streaming, 32 languages +- `eleven_flash_v2_5` - Fastest, suitable for real-time use + +--- + +### 18. Deepgram STT + +#### Required Variables + +```bash +DEEPGRAM_API_KEY="your-deepgram-api-key" +``` + +#### How to Get Deepgram API Key + +1. Visit [Deepgram Console](https://console.deepgram.com) +2. Sign up or log in to your account +3. Navigate to **API Keys** +4. Click **Create a New API Key** +5. Copy the key + +#### Supported Models + +- `nova-3` (default) - Latest, highest accuracy +- `nova-2` - High accuracy, broad language support +- `base` - Balanced accuracy and speed + +--- + +### 19. Azure Speech Services (TTS + STT) + +Azure Speech Services provides both text-to-speech and speech-to-text through Microsoft Azure Cognitive Services. + +#### Required Variables + +```bash +AZURE_SPEECH_KEY="your-azure-speech-key" +AZURE_SPEECH_REGION="eastus" # Azure region where your Speech resource is deployed +``` + +#### Optional Variables + +```bash +GOOGLE_API_KEY="AIza-your-google-api-key" # Required if also using Google STT alongside Azure +``` + +#### How to Set Up Azure Speech Services + +1. Sign in to [Azure Portal](https://portal.azure.com) +2. Create a **Speech** resource under **Azure AI services** +3. Go to **Keys and Endpoint** in your Speech resource +4. Copy **Key 1** and note the **Location/Region** +5. Set `AZURE_SPEECH_KEY` and `AZURE_SPEECH_REGION` + +#### Supported Capabilities + +- **TTS**: Azure Neural TTS with 400+ voices across 140+ languages +- **STT**: Azure Speech-to-Text with real-time and batch transcription + +#### Environment Variables Reference + +| Variable | Required | Default | Description | +| --------------------- | -------- | ------- | --------------------------------------------- | +| `AZURE_SPEECH_KEY` | ✅ | - | Azure Speech Services API key | +| `AZURE_SPEECH_REGION` | ✅ | - | Azure region (e.g., `eastus`, `westeurope`) | +| `GOOGLE_API_KEY` | ❌ | - | Google API key, if using Google STT alongside | +| `ELEVENLABS_API_KEY` | ❌ | - | ElevenLabs key, if using ElevenLabs alongside | +| `DEEPGRAM_API_KEY` | ❌ | - | Deepgram key, if using Deepgram alongside | + +--- + ## 🔧 Configuration Examples ### Complete .env File Example diff --git a/docs/getting-started/provider-setup.md b/docs/getting-started/provider-setup.md index 8b59ebd5c..3b66dd0bc 100644 --- a/docs/getting-started/provider-setup.md +++ b/docs/getting-started/provider-setup.md @@ -2020,4 +2020,197 @@ Authentication failed --- +## OpenAI TTS Configuration {#openai-tts} + +OpenAI TTS provides text-to-speech synthesis using the same API key as the OpenAI LLM provider. No additional credentials are required. + +### Basic Setup + +```bash +export OPENAI_API_KEY="sk-your-openai-api-key" +``` + +**Note:** `OPENAI_API_KEY` is shared with the OpenAI LLM provider. No separate key is needed. + +### Supported Models + +- `tts-1` (default) - Optimized for speed, lower latency +- `tts-1-hd` - Optimized for quality, higher fidelity audio + +### Supported Voices + +`alloy`, `echo`, `fable`, `onyx`, `nova`, `shimmer` + +### Supported Output Formats + +`mp3` (default), `opus`, `aac`, `flac`, `wav`, `pcm` + +### Usage Example + +```typescript +import { NeuroLink } from "@juspay/neurolink"; + +const neurolink = new NeuroLink(); + +const result = await neurolink.generate({ + input: { text: "Hello, world!" }, + tts: { + enabled: true, + provider: "openai-tts", + voice: "alloy", + format: "mp3", + }, +}); +``` + +### CLI Usage + +```bash +npx @juspay/neurolink generate "Hello, world!" --tts --tts-provider openai-tts +``` + +### Environment Variables Reference + +| Variable | Required | Default | Description | +| ---------------- | -------- | ------- | ----------------------------------- | +| `OPENAI_API_KEY` | ✅ | - | Shared with the OpenAI LLM provider | + +### Provider ID and Aliases + +- **Provider ID**: `openai-tts` + +--- + +## ElevenLabs Configuration {#elevenlabs} + +ElevenLabs provides high-quality, multilingual text-to-speech synthesis with a wide selection of voices and voice cloning support. + +### Basic Setup + +```bash +export ELEVENLABS_API_KEY="your-elevenlabs-api-key" +``` + +### How to Get ElevenLabs API Key + +1. Visit [ElevenLabs](https://elevenlabs.io) +2. Sign up or log in to your account +3. Navigate to **Profile → API Key** +4. Copy the key + +### Supported Models + +- `eleven_multilingual_v2` (default) - Best quality, 29 languages +- `eleven_turbo_v2_5` - Low-latency streaming, 32 languages +- `eleven_flash_v2_5` - Fastest, suitable for real-time applications + +### Usage Example + +```typescript +import { NeuroLink } from "@juspay/neurolink"; + +const neurolink = new NeuroLink(); + +const result = await neurolink.generate({ + input: { text: "Bonjour le monde!" }, + tts: { + enabled: true, + provider: "elevenlabs", + voice: "Rachel", + model: "eleven_multilingual_v2", + }, +}); +``` + +### CLI Usage + +```bash +npx @juspay/neurolink generate "Hello, world!" --tts --tts-provider elevenlabs +``` + +### Notes + +- **Multilingual support**: ElevenLabs models support up to 32 languages with natural prosody +- **Voice cloning**: ElevenLabs supports custom voice IDs from your ElevenLabs account + +### Environment Variables Reference + +| Variable | Required | Default | Description | +| -------------------- | -------- | ------- | ------------------ | +| `ELEVENLABS_API_KEY` | ✅ | - | ElevenLabs API key | + +### Provider ID and Aliases + +- **Provider ID**: `elevenlabs` + +--- + +## Deepgram STT Configuration {#deepgram} + +Deepgram provides fast, accurate speech-to-text transcription with support for real-time streaming and pre-recorded audio. + +### Basic Setup + +```bash +export DEEPGRAM_API_KEY="your-deepgram-api-key" +``` + +### How to Get Deepgram API Key + +1. Visit [Deepgram Console](https://console.deepgram.com) +2. Sign up or log in to your account +3. Navigate to **API Keys** +4. Click **Create a New API Key** +5. Copy the key + +### Supported Models + +- `nova-3` (default) - Latest, highest accuracy +- `nova-2` - High accuracy, broad language support +- `base` - Balanced accuracy and speed + +### Usage Example + +```typescript +import { NeuroLink } from "@juspay/neurolink"; +import { readFileSync } from "fs"; + +const neurolink = new NeuroLink(); +const audioBuffer = readFileSync("audio.wav"); + +const result = await neurolink.generate({ + input: { text: "Respond to what was said" }, + stt: { + enabled: true, + provider: "deepgram", + audio: audioBuffer, + model: "nova-3", + language: "en", + }, +}); +``` + +### CLI Usage + +```bash +npx @juspay/neurolink generate "Respond to this" --stt --stt-provider deepgram --input-audio file.wav +``` + +### Notes + +- **Streaming transcription**: Deepgram supports real-time audio streaming for live transcription +- **Language support**: Deepgram nova models support 30+ languages + +### Environment Variables Reference + +| Variable | Required | Default | Description | +| ------------------ | -------- | ------- | ---------------- | +| `DEEPGRAM_API_KEY` | ✅ | - | Deepgram API key | + +### Provider ID and Aliases + +- **Provider ID**: `deepgram` + +--- + [← Back to Main README](../index.md) | [Next: API Reference →](./api-reference.md) diff --git a/docs/getting-started/providers/deepgram.md b/docs/getting-started/providers/deepgram.md new file mode 100644 index 000000000..bdf427d48 --- /dev/null +++ b/docs/getting-started/providers/deepgram.md @@ -0,0 +1,495 @@ +--- +title: Deepgram Provider Guide +description: Transcribe audio to text using Deepgram's Nova speech recognition models through NeuroLink, with streaming, speaker diarization, and smart formatting +keywords: deepgram, stt, speech-to-text, transcription, nova-2, nova-3, diarization, smart formatting, streaming, audio +--- + +# Deepgram Provider Guide + +**Fast, accurate speech-to-text with streaming, speaker diarization, and smart formatting** + +--- + +## Overview + +Deepgram is a speech recognition provider optimised for speed and accuracy in production environments. NeuroLink wraps Deepgram's Listen API, giving you access to the Nova-2 and Nova-3 model families through the standard `generate()` call. Deepgram's strengths include real-time streaming transcription over WebSocket, speaker diarization for multi-speaker audio, and smart formatting that cleans up dates, currency, and numbers automatically. + +### Key Facts + +| Property | Value | +| ---------------------- | --------------------------------------------- | +| **Provider ID** | `deepgram` | +| **API endpoint** | `https://api.deepgram.com/v1/listen` | +| **Streaming endpoint** | `wss://api.deepgram.com/v1/listen` | +| **Default model** | `nova-2` | +| **Formats** | mp3, wav, ogg, opus | +| **Max audio** | 2 hours (7,200 seconds) per request | +| **Languages** | 40+ languages and dialects | +| **Streaming** | Yes (WebSocket-based real-time transcription) | + +--- + +## Quick Start + +### 1. Get an API Key + +Sign up at [https://console.deepgram.com](https://console.deepgram.com) and create an API key under **Settings → API Keys**. + +### 2. Configure Environment + +Add to your `.env` file: + +```bash +# Required +DEEPGRAM_API_KEY=your-deepgram-api-key + +# Optional: default model (default: nova-2) +DEEPGRAM_MODEL=nova-2 + +# Optional: default language (default: en-US) +DEEPGRAM_LANGUAGE=en-US +``` + +### 3. Install NeuroLink + +```bash +npm install @juspay/neurolink +# or +pnpm add @juspay/neurolink +``` + +### 4. Transcribe Your First Audio File + +```typescript +import { NeuroLink } from "@juspay/neurolink"; +import { readFileSync } from "fs"; + +const ai = new NeuroLink(); +const audioBuffer = readFileSync("./recording.wav"); + +const result = await ai.generate({ + input: { text: "Transcribe the following audio." }, + stt: { + enabled: true, + provider: "deepgram", + audio: audioBuffer, + format: "wav", + }, +}); + +if (result.stt) { + console.log("Transcript:", result.stt.text); + console.log("Confidence:", result.stt.confidence); + console.log("Duration:", result.stt.duration, "seconds"); +} +``` + +--- + +## Supported Models + +| Model ID | Description | Best For | +| ------------------ | -------------------------------------------------- | --------------------------------------- | +| `nova-2` (default) | Fastest, lowest Word Error Rate in the Nova family | General transcription, production use | +| `nova-2-general` | General-purpose variant, same as `nova-2` | Broad use cases | +| `nova-2-meeting` | Optimised for multi-speaker meeting audio | Video conferences, recordings | +| `nova-2-phonecall` | Tuned for telephone audio quality | Call centre, PSTN audio | +| `nova-2-voicemail` | Handles background noise and compressed audio | Voicemail transcription | +| `nova-2-finance` | Finance-domain vocabulary boost | Earnings calls, financial content | +| `nova-2-medical` | Medical terminology | Clinical notes, consultations | +| `nova-3` | Next-generation model with improved accuracy | Demanding accuracy requirements | +| `nova` | Previous generation Nova | Legacy compatibility | +| `enhanced` | High accuracy, slower processing | Archival, quality-critical paths | +| `base` | Fastest, lower accuracy | Draft transcriptions, cost optimisation | + +--- + +## SDK Usage + +### Basic Transcription + +```typescript +import { NeuroLink } from "@juspay/neurolink"; +import { readFileSync } from "fs"; + +const ai = new NeuroLink(); +const audio = readFileSync("./meeting.wav"); + +const result = await ai.generate({ + input: { text: "Transcribe this audio." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + language: "en-US", + }, +}); + +if (result.stt) { + console.log(result.stt.text); +} +``` + +### Choosing a Model + +```typescript +import type { DeepgramSTTOptions } from "@juspay/neurolink"; + +const result = await ai.generate({ + input: { text: "Transcribe this meeting recording." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + model: "nova-2-meeting", + } as DeepgramSTTOptions, +}); +``` + +### Smart Formatting + +Smart formatting cleans up numbers, currency, dates, and other structured data automatically: + +```typescript +import type { DeepgramSTTOptions } from "@juspay/neurolink"; + +const result = await ai.generate({ + input: { text: "Transcribe with formatting." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + smartFormat: true, // Formats "twenty five dollars" → "$25" + } as DeepgramSTTOptions, +}); +``` + +### Speaker Diarization + +Identify who spoke when in multi-speaker audio: + +```typescript +const result = await ai.generate({ + input: { text: "Transcribe and identify speakers." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + speakerDiarization: true, + }, +}); + +if (result.stt) { + console.log("Transcript:", result.stt.text); + console.log("Speakers found:", result.stt.speakers); + + // Word-level speaker attribution + for (const word of result.stt.words ?? []) { + console.log( + `${word.speaker ?? "?"}: "${word.word}" [${word.startTime}s–${word.endTime}s]`, + ); + } +} +``` + +### Utterance Segmentation + +Split audio into utterance-level segments with speaker and timing information: + +```typescript +import type { DeepgramSTTOptions } from "@juspay/neurolink"; + +const result = await ai.generate({ + input: { text: "Segment into utterances." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + utterances: true, + speakerDiarization: true, + } as DeepgramSTTOptions, +}); + +if (result.stt?.segments) { + for (const seg of result.stt.segments) { + console.log(`[${seg.startTime}s] ${seg.speaker ?? "Speaker"}: ${seg.text}`); + } +} +``` + +### Word-Level Timestamps + +```typescript +const result = await ai.generate({ + input: { text: "Transcribe with word timings." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + wordTimestamps: true, + }, +}); + +if (result.stt?.words) { + for (const word of result.stt.words) { + console.log( + `"${word.word}" at ${word.startTime}s (confidence: ${word.confidence?.toFixed(2)})`, + ); + } +} +``` + +### Custom Vocabulary / Keyword Boosting + +Improve recognition of domain-specific terms: + +```typescript +import type { DeepgramSTTOptions } from "@juspay/neurolink"; + +const result = await ai.generate({ + input: { text: "Transcribe technical content." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + keywords: ["NeuroLink", "EulerHS", "Juspay", "HyperSDK"], + keywordBoost: "high", + } as DeepgramSTTOptions, +}); +``` + +### Content Redaction + +Automatically redact sensitive data from transcripts: + +```typescript +import type { DeepgramSTTOptions } from "@juspay/neurolink"; + +const result = await ai.generate({ + input: { text: "Transcribe and redact PII." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + redact: ["pci", "ssn"], // Redact credit card and SSN numbers + } as DeepgramSTTOptions, +}); +``` + +### Real-Time Streaming Transcription + +Use the `DeepgramSTT` handler directly for WebSocket-based streaming: + +```typescript +import { DeepgramSTT } from "@juspay/neurolink/voice"; +import { createReadStream } from "fs"; + +const handler = new DeepgramSTT(process.env.DEEPGRAM_API_KEY); + +async function* readAudioStream(filePath: string): AsyncIterable { + const stream = createReadStream(filePath, { highWaterMark: 4096 }); + for await (const chunk of stream) { + yield chunk as Buffer; + } +} + +const audioStream = readAudioStream("./live-audio.wav"); + +for await (const segment of handler.transcribeStream(audioStream, { + language: "en-US", + smartFormat: true, + speakerDiarization: true, +})) { + const status = segment.isFinal ? "[FINAL]" : "[partial]"; + console.log(`${status} ${segment.text}`); +} +``` + +### Per-Call Credential Override + +```typescript +const result = await ai.generate({ + input: { text: "Transcribe with a per-request key." }, + stt: { + enabled: true, + provider: "deepgram", + audio, + format: "wav", + }, + credentials: { + deepgram: { + apiKey: "user-specific-deepgram-key", + }, + }, +}); +``` + +--- + +## CLI Usage + +### Basic Transcription + +```bash +# Transcribe an audio file +neurolink generate "Respond to audio" \ + --stt --stt-provider deepgram \ + --input-audio recording.wav + +# Specify model +neurolink generate "Transcribe this meeting" \ + --stt --stt-provider deepgram \ + --stt-model nova-2-meeting \ + --input-audio meeting.mp3 +``` + +### Language Selection + +```bash +neurolink generate "Transcribe Spanish audio" \ + --stt --stt-provider deepgram \ + --stt-language es \ + --input-audio audio-es.wav +``` + +### Smart Formatting + +```bash +neurolink generate "Transcribe with smart formatting" \ + --stt --stt-provider deepgram \ + --stt-smart-format \ + --input-audio recording.wav +``` + +### Speaker Diarization + +```bash +neurolink generate "Identify speakers" \ + --stt --stt-provider deepgram \ + --stt-diarize \ + --input-audio meeting.wav +``` + +--- + +## Supported Languages + +Deepgram supports 40+ languages and regional dialects. Key languages available with diarization and punctuation: + +| Code | Language | +| ------- | ------------ | +| `en` | English | +| `en-US` | English (US) | +| `en-GB` | English (UK) | +| `es` | Spanish | +| `fr` | French | +| `de` | German | +| `it` | Italian | +| `pt` | Portuguese | +| `nl` | Dutch | +| `ja` | Japanese | +| `ko` | Korean | +| `zh` | Chinese | +| `hi` | Hindi | +| `ru` | Russian | + +For the full language list, see the [Deepgram language support docs](https://developers.deepgram.com/docs/models-languages-overview). + +--- + +## Configuration Reference + +| Environment Variable | Required | Default | Description | +| -------------------- | -------- | -------- | ------------------------------ | +| `DEEPGRAM_API_KEY` | Yes | — | Deepgram API key | +| `DEEPGRAM_MODEL` | No | `nova-2` | Default transcription model | +| `DEEPGRAM_LANGUAGE` | No | `en-US` | Default transcription language | + +--- + +## Feature Support Matrix + +| Feature | Supported | Notes | +| ---------------------- | --------- | ---------------------------------------------- | +| Batch transcription | Yes | Up to 2 hours per request | +| Real-time streaming | Yes | WebSocket via `transcribeStream()` | +| Speaker diarization | Yes | `speakerDiarization: true` | +| Word-level timestamps | Yes | Included by default when words are returned | +| Smart formatting | Yes | `smartFormat: true` — numbers, dates, currency | +| Utterance segmentation | Yes | `utterances: true` | +| Keyword boosting | Yes | `keywords` + `keywordBoost` | +| Content redaction | Yes | PCI, SSN number redaction | +| Profanity filter | Yes | `profanityFilter: true` | +| Custom vocabulary | Yes | `keywords` array | +| Multi-format input | Yes | mp3, wav, ogg, opus | +| Confidence scores | Yes | Per-transcript and per-word | +| 40+ languages | Yes | `language` option | + +--- + +## Troubleshooting + +### "deepgram provider not configured" + +The `DEEPGRAM_API_KEY` environment variable is missing or not loaded. + +```bash +echo $DEEPGRAM_API_KEY + +export DEEPGRAM_API_KEY=your-key-here +``` + +Create or rotate keys at [https://console.deepgram.com](https://console.deepgram.com). + +### "HTTP 401" — Invalid API key + +Your key is invalid or has been revoked. Generate a new one from the Deepgram console. + +### "HTTP 402" — Insufficient credits + +Your account balance is exhausted. Top up at [https://console.deepgram.com/billing](https://console.deepgram.com/billing). + +### "HTTP 429" — Rate limit exceeded + +Too many concurrent requests. Implement exponential backoff or reduce concurrency. Rate limits are documented in the [Deepgram API docs](https://developers.deepgram.com/docs/rate-limits). + +### Empty transcript returned + +Audio may be silent, below detection threshold, or in the wrong language. Verify: + +1. The audio buffer is not empty (`audioBuffer.length > 0`). +2. The `format` matches the actual audio encoding. +3. The `language` matches the audio's spoken language. + +### "Deepgram STT request timed out after 30 seconds" + +The request took longer than 30 seconds — typically due to very long audio or network issues. For audio over 30 minutes, consider splitting into chunks. + +### Streaming WebSocket disconnects + +Check that `DEEPGRAM_API_KEY` is valid and that your network allows outbound WebSocket connections to `wss://api.deepgram.com`. Firewall or proxy configurations may block WebSocket upgrades. + +### Diarization not appearing in results + +Diarization requires multi-speaker audio with clearly separated voices. Single-speaker audio will return no speaker labels. Also confirm `speakerDiarization: true` is set, and that you are using a model that supports it (Nova-2 and above). + +--- + +## See Also + +- [Audio Input (STT) Guide](/docs/features/audio-input) — complete multi-provider STT reference +- [Voice Agent Guide](/docs/features/voice-agent) — building full voice assistants +- [OpenAI TTS Provider Guide](/docs/getting-started/providers/openai-tts) — text-to-speech counterpart +- [ElevenLabs Provider Guide](/docs/getting-started/providers/elevenlabs) — alternative TTS with voice cloning + +--- + +**Need Help?** Join the [GitHub Discussions](https://github.com/juspay/neurolink/discussions) or open an [issue](https://github.com/juspay/neurolink/issues). diff --git a/docs/getting-started/providers/elevenlabs.md b/docs/getting-started/providers/elevenlabs.md new file mode 100644 index 000000000..565ccd0f4 --- /dev/null +++ b/docs/getting-started/providers/elevenlabs.md @@ -0,0 +1,431 @@ +--- +title: ElevenLabs Provider Guide +description: Generate studio-quality multilingual speech using ElevenLabs' neural TTS through NeuroLink, with dynamic voice discovery and voice cloning support +keywords: elevenlabs, tts, text-to-speech, audio, multilingual, voice cloning, eleven_multilingual_v2, neural speech +--- + +# ElevenLabs Provider Guide + +**Studio-quality, multilingual text-to-speech with dynamic voice discovery and voice cloning** + +--- + +## Overview + +ElevenLabs is a specialist voice AI provider known for exceptionally natural-sounding speech synthesis and extensive multilingual support. NeuroLink integrates their TTS API, giving you access to their full voice library — including custom and cloned voices — through the same `generate()` call used for all other TTS providers. + +The default model, `eleven_multilingual_v2`, produces high-fidelity audio across 29 languages with a single voice. ElevenLabs voices are dynamically fetched from the API and cached for five minutes, so newly added or cloned voices are always available without restarting your application. + +### Key Facts + +| Property | Value | +| ----------------- | --------------------------------------------------- | +| **Provider ID** | `elevenlabs` | +| **API endpoint** | `https://api.elevenlabs.io/v1` | +| **Default model** | `eleven_multilingual_v2` | +| **Default voice** | Rachel (`21m00Tcm4TlvDq8ikWAM`) | +| **Formats** | mp3 (44.1 kHz), wav (PCM 44.1 kHz), ogg (22 kHz) | +| **Max input** | 5,000 characters per request | +| **Languages** | 29+ languages per voice (auto-detected from input) | +| **Streaming** | Not supported in NeuroLink integration (batch only) | + +--- + +## Quick Start + +### 1. Get an API Key + +Sign up at [https://elevenlabs.io](https://elevenlabs.io) and copy your API key from **Profile → API Key**. + +### 2. Configure Environment + +Add to your `.env` file: + +```bash +# Required +ELEVENLABS_API_KEY=your-api-key-here + +# Optional: default voice ID (default: Rachel — 21m00Tcm4TlvDq8ikWAM) +ELEVENLABS_VOICE_ID=21m00Tcm4TlvDq8ikWAM + +# Optional: default model (default: eleven_multilingual_v2) +ELEVENLABS_MODEL=eleven_multilingual_v2 +``` + +### 3. Install NeuroLink + +```bash +npm install @juspay/neurolink +# or +pnpm add @juspay/neurolink +``` + +### 4. Synthesise Your First Audio + +```typescript +import { NeuroLink } from "@juspay/neurolink"; +import { writeFileSync } from "fs"; + +const ai = new NeuroLink(); + +const result = await ai.generate({ + input: { text: "Hello! This is ElevenLabs speaking through NeuroLink." }, + tts: { + enabled: true, + provider: "elevenlabs", + format: "mp3", + }, +}); + +if (result.tts) { + writeFileSync("output.mp3", result.tts.buffer); + console.log(`Saved ${result.tts.size} bytes to output.mp3`); +} +``` + +--- + +## Supported Models + +| Model ID | Description | Use Case | +| ------------------------ | ------------------------------------------------ | --------------------------------- | +| `eleven_multilingual_v2` | Default; 29 languages, highest quality | General use, multilingual content | +| `eleven_monolingual_v1` | English-only, optimised for English naturalness | English-only apps | +| `eleven_multilingual_v1` | First-generation multilingual (superseded by v2) | Legacy compatibility | +| `eleven_turbo_v2` | Fast, lower latency variant | Real-time applications | + +Pass the model ID explicitly via the `ElevenLabsTTSOptions.model` field or let the integration default to `eleven_multilingual_v2`. + +--- + +## SDK Usage + +### Direct Text Synthesis + +Synthesise the input text without calling an AI model: + +```typescript +import { NeuroLink } from "@juspay/neurolink"; + +const ai = new NeuroLink(); + +const result = await ai.generate({ + input: { + text: "ElevenLabs produces natural-sounding speech in 29 languages.", + }, + tts: { + enabled: true, + provider: "elevenlabs", + format: "mp3", + }, +}); + +if (result.tts) { + console.log("Format:", result.tts.format); + console.log("Size:", result.tts.size, "bytes"); + console.log("Provider:", result.tts.metadata?.provider); +} +``` + +### Specifying a Voice + +Voices are identified by their `voice_id` string. Use a known ID directly, or list available voices programmatically (see [Voice Discovery](#voice-discovery)): + +```typescript +const result = await ai.generate({ + input: { text: "A specific voice selected by ID." }, + tts: { + enabled: true, + provider: "elevenlabs", + voice: "21m00Tcm4TlvDq8ikWAM", // Rachel — the default + format: "mp3", + }, +}); +``` + +### AI Response Synthesis + +Generate a response with an AI model and then synthesise it: + +```typescript +const result = await ai.generate({ + provider: "openai", + input: { text: "Explain quantum computing in two sentences." }, + tts: { + enabled: true, + provider: "elevenlabs", + useAiResponse: true, // Synthesise the AI-generated text, not the prompt + voice: "21m00Tcm4TlvDq8ikWAM", + format: "mp3", + }, +}); +``` + +### Multilingual Synthesis + +ElevenLabs detects the language of your input automatically. No extra configuration is needed: + +```typescript +// Spanish +await ai.generate({ + input: { text: "Buenos días. ¿Cómo puedo ayudarte hoy?" }, + tts: { enabled: true, provider: "elevenlabs", format: "mp3" }, +}); + +// French +await ai.generate({ + input: { text: "Bonjour! Comment puis-je vous aider?" }, + tts: { enabled: true, provider: "elevenlabs", format: "mp3" }, +}); + +// Hindi +await ai.generate({ + input: { text: "नमस्ते! मैं आपकी कैसे सहायता कर सकता हूँ?" }, + tts: { enabled: true, provider: "elevenlabs", format: "mp3" }, +}); +``` + +### Voice Settings Tuning + +Fine-tune the voice character using ElevenLabs-specific options: + +```typescript +import type { ElevenLabsTTSOptions } from "@juspay/neurolink"; + +const result = await ai.generate({ + input: { text: "Fine-tuned voice output." }, + tts: { + enabled: true, + provider: "elevenlabs", + voice: "21m00Tcm4TlvDq8ikWAM", + format: "mp3", + // ElevenLabs-specific settings (cast required for typed access) + stability: 0.6, // 0–1: higher = more consistent, lower = more expressive + similarityBoost: 0.8, // 0–1: how closely to match the original voice + style: 0.2, // 0–1: style exaggeration (v2 models only) + useSpeakerBoost: true, // Boost speaker clarity + } as ElevenLabsTTSOptions, +}); +``` + +### Save to File + +```typescript +const result = await ai.generate({ + input: { text: "Saving ElevenLabs audio to disk." }, + tts: { + enabled: true, + provider: "elevenlabs", + voice: "21m00Tcm4TlvDq8ikWAM", + format: "mp3", + output: "./audio/output.mp3", // NeuroLink saves automatically if set + }, +}); +``` + +### Per-Call Credential Override + +```typescript +const result = await ai.generate({ + input: { text: "Using a per-request API key." }, + tts: { + enabled: true, + provider: "elevenlabs", + }, + credentials: { + elevenlabs: { + apiKey: "user-specific-elevenlabs-key", + }, + }, +}); +``` + +--- + +## CLI Usage + +### Basic TTS + +```bash +# Synthesise text using ElevenLabs +neurolink generate "Hello from ElevenLabs!" --tts --tts-provider elevenlabs + +# Save to file +neurolink generate "Saving to disk." \ + --tts --tts-provider elevenlabs \ + --tts-output output.mp3 +``` + +### Choose a Voice + +```bash +neurolink generate "Custom voice ID." \ + --tts --tts-provider elevenlabs \ + --tts-voice 21m00Tcm4TlvDq8ikWAM +``` + +### Synthesise AI Response + +```bash +neurolink generate "Write a product tagline for a fintech app." \ + --provider openai \ + --tts --tts-provider elevenlabs \ + --tts-use-ai-response \ + --tts-output tagline.mp3 +``` + +### Multilingual + +```bash +neurolink generate "Bonjour! Comment puis-je vous aider?" \ + --tts --tts-provider elevenlabs \ + --tts-output french.mp3 +``` + +--- + +## Voice Discovery + +ElevenLabs voices are fetched dynamically from your account. The result includes both the ElevenLabs library voices and any custom or cloned voices in your account. + +```typescript +import { ElevenLabsTTS } from "@juspay/neurolink/voice"; + +const handler = new ElevenLabsTTS(process.env.ELEVENLABS_API_KEY); +const voices = await handler.getVoices(); + +for (const voice of voices) { + console.log(`${voice.id} — ${voice.name} (${voice.gender})`); +} +``` + +Voices are cached for **5 minutes** per handler instance to avoid redundant API calls. + +--- + +## Supported Languages + +`eleven_multilingual_v2` supports 29 languages. The following are recognised by the NeuroLink voice metadata: + +| Code | Language | +| ---- | ---------- | +| `en` | English | +| `es` | Spanish | +| `fr` | French | +| `de` | German | +| `it` | Italian | +| `pt` | Portuguese | +| `pl` | Polish | +| `hi` | Hindi | +| `ar` | Arabic | +| `zh` | Chinese | +| `ja` | Japanese | +| `ko` | Korean | + +For the full language list, refer to the [ElevenLabs documentation](https://elevenlabs.io/docs/api-reference/how-to-use-tts-with-streaming). + +--- + +## Audio Formats + +| Format | Extension | ElevenLabs internal format | Sample Rate | +| ------ | --------- | -------------------------- | ----------- | +| `mp3` | `.mp3` | `mp3_44100_128` | 44,100 Hz | +| `wav` | `.wav` | `pcm_44100` | 44,100 Hz | +| `ogg` | `.ogg` | `ogg_22050` | 22,050 Hz | +| `opus` | `.opus` | `ogg_22050` | 22,050 Hz | + +--- + +## Configuration Reference + +| Environment Variable | Required | Default | Description | +| --------------------- | -------- | ------------------------ | ---------------------- | +| `ELEVENLABS_API_KEY` | Yes | — | ElevenLabs API key | +| `ELEVENLABS_VOICE_ID` | No | `21m00Tcm4TlvDq8ikWAM` | Default voice (Rachel) | +| `ELEVENLABS_MODEL` | No | `eleven_multilingual_v2` | Default TTS model | + +--- + +## Feature Support Matrix + +| Feature | Supported | Notes | +| ---------------------- | --------- | ------------------------------------------ | +| Text synthesis | Yes | | +| AI response synthesis | Yes | Set `useAiResponse: true` | +| Multilingual support | Yes | 29 languages, auto-detected | +| Voice discovery | Yes | Dynamic API fetch, 5-minute cache | +| Custom / cloned voices | Yes | Pass voice ID from your ElevenLabs account | +| Voice stability tuning | Yes | `stability`, `similarityBoost`, `style` | +| Multiple formats | Yes | mp3, wav, ogg, opus | +| Streaming TTS | No | Batch synthesis only in NeuroLink | +| Speed control | No | Not supported by this integration | + +--- + +## Troubleshooting + +### "ElevenLabs API key not configured" + +The `ELEVENLABS_API_KEY` environment variable is missing or was not loaded. + +```bash +echo $ELEVENLABS_API_KEY + +export ELEVENLABS_API_KEY=your-key-here +``` + +Retrieve your key from [https://elevenlabs.io/app/settings/api-keys](https://elevenlabs.io/app/settings/api-keys). + +### "HTTP 401" — Unauthorised + +Your API key is invalid or has been revoked. Generate a new key from the ElevenLabs dashboard. + +### "HTTP 429" — Rate limit or quota exceeded + +You have reached your character quota for the billing period, or exceeded the per-minute request rate. Check your usage at [https://elevenlabs.io/app/subscription](https://elevenlabs.io/app/subscription). + +### "HTTP 400" — Request too long + +The input text exceeds 5,000 characters. Split the content into chunks: + +```typescript +function chunkText(text: string, maxLen = 4500): string[] { + const chunks: string[] = []; + for (let i = 0; i < text.length; i += maxLen) { + chunks.push(text.slice(i, i + maxLen)); + } + return chunks; +} +``` + +### "ElevenLabs TTS request timed out after 30 seconds" + +A slow network or high server load caused the request to time out. This error is marked retriable — retry with backoff. + +### Voice not found + +You passed a `voice` ID that does not exist in your account. List available voices to confirm: + +```typescript +const handler = new ElevenLabsTTS(); +const voices = await handler.getVoices(); +console.log(voices.map((v) => `${v.id}: ${v.name}`).join("\n")); +``` + +### "Failed to get voices" + +Voice discovery failed (network error or invalid key). The 5-minute cache shields against transient failures, but a hard failure at startup will propagate. Ensure `ELEVENLABS_API_KEY` is valid and the ElevenLabs API is reachable. + +--- + +## See Also + +- [TTS Integration Guide](/docs/features/tts) — complete multi-provider TTS reference +- [OpenAI TTS Provider Guide](/docs/getting-started/providers/openai-tts) — alternative TTS provider +- [Audio Input (STT)](/docs/features/audio-input) — speech-to-text counterpart +- [Voice Agent Guide](/docs/features/voice-agent) — building full voice assistants + +--- + +**Need Help?** Join the [GitHub Discussions](https://github.com/juspay/neurolink/discussions) or open an [issue](https://github.com/juspay/neurolink/issues). diff --git a/docs/getting-started/providers/index.md b/docs/getting-started/providers/index.md index 251238731..078a300b2 100644 --- a/docs/getting-started/providers/index.md +++ b/docs/getting-started/providers/index.md @@ -211,6 +211,134 @@ Access multiple providers through unified interfaces: --- +## 🎙️ Voice Providers + +Synthesize speech, transcribe audio, or run live voice sessions. Voice providers are separate from LLM providers — they handle audio I/O rather than text generation. + +### Text-to-Speech (TTS) + +#### [OpenAI TTS](../../guides/voice/openai-tts.md) + +**Highest-quality text-to-speech** + +- 🎙️ Voices: alloy, echo, fable, onyx, nova, shimmer +- 🎵 Models: tts-1 (fast) and tts-1-hd (high quality) +- 🎼 Formats: MP3, WAV, OGG, Opus +- 🔑 Auth: API Key (`OPENAI_API_KEY`) + +[Setup Guide →](../../guides/voice/openai-tts.md) + +#### [ElevenLabs](../../guides/voice/elevenlabs.md) + +**Best multilingual and voice-cloning TTS** + +- 🌍 Supports 30+ languages with natural prosody +- 🎭 Custom voice cloning from short audio samples +- 🎼 Formats: MP3 +- 🔑 Auth: API Key (`ELEVENLABS_API_KEY`) + +[Setup Guide →](../../guides/voice/elevenlabs.md) + +#### [Google TTS](../../guides/voice/google-tts.md) + +**1M characters/month free tier** + +- 💰 Generous free tier for standard voices +- 🌍 380+ voices across 50+ languages +- 🎼 Formats: MP3, WAV, OGG +- 🔑 Auth: Service Account + +[Setup Guide →](../../guides/voice/google-tts.md) + +#### [Azure TTS](../../guides/voice/azure-tts.md) + +**Enterprise TTS with full SSML support** + +- 🏢 Fine-grained prosody control via SSML +- 🌍 400+ neural voices, 140+ languages +- 🎼 Formats: MP3 +- 🔑 Auth: API Key + Region + +[Setup Guide →](../../guides/voice/azure-tts.md) + +--- + +### Speech-to-Text (STT) + +#### [Whisper (OpenAI)](../../guides/voice/whisper.md) + +**Highest transcription accuracy** + +- 🎯 Best-in-class accuracy on diverse audio +- 🌍 Multilingual with automatic language detection +- 🎼 Formats: WAV, MP3, M4A, FLAC +- 🔑 Auth: API Key (`OPENAI_API_KEY`) + +[Setup Guide →](../../guides/voice/whisper.md) + +#### [Deepgram](../../guides/voice/deepgram.md) + +**Real-time streaming transcription via WebSocket** + +- ⚡ Sub-300 ms word-level results over WebSocket +- 🌊 REST batch and WebSocket streaming modes +- 🎼 Formats: WAV, MP3, OGG, FLAC +- 🔑 Auth: API Key (`DEEPGRAM_API_KEY`) + +[Setup Guide →](../../guides/voice/deepgram.md) + +#### [Google STT](../../guides/voice/google-stt.md) + +**125+ languages with speaker diarization** + +- 🌍 Best fit for existing Google Cloud users +- 👥 Speaker diarization and multi-channel audio +- 🎼 Formats: WAV, FLAC, MP3, OGG +- 🔑 Auth: Service Account + +[Setup Guide →](../../guides/voice/google-stt.md) + +#### [Azure STT](../../guides/voice/azure-stt.md) + +**Enterprise STT with custom model training** + +- 🏢 Batch transcription and custom model support +- 🔒 Compliance controls for regulated industries +- 🎼 Formats: WAV, MP3 +- 🔑 Auth: API Key + Region + +[Setup Guide →](../../guides/voice/azure-stt.md) + +--- + +### Realtime Voice + +Realtime providers maintain a persistent bidirectional WebSocket connection, enabling low-latency spoken conversation with the AI model. + +#### [OpenAI Realtime](../../guides/voice/openai-realtime.md) + +**Low-latency bidirectional voice over WebSocket** + +- ⚡ Full-duplex audio stream with GPT-4o +- 🎵 Voice activity detection (VAD) built-in +- 🎼 Formats: WAV, Opus +- 🔑 Auth: API Key (`OPENAI_API_KEY`) + +[Setup Guide →](../../guides/voice/openai-realtime.md) + +#### [Gemini Live](../../guides/voice/gemini-live.md) + +**Google's native realtime voice API** + +- ⚡ Native multimodal realtime session with Gemini +- 🎵 Supports audio + video input simultaneously +- 🎼 Formats: WAV +- 🔑 Auth: API Key (`GOOGLE_AI_KEY`) + +[Setup Guide →](../../guides/voice/gemini-live.md) + +--- + ## Quick Comparison | Provider | Free Tier | Enterprise | GDPR | Latency | Best For | @@ -229,6 +357,16 @@ Access multiple providers through unified interfaces: | [NVIDIA NIM](../../getting-started/provider-setup.md#nvidia-nim) | ❌ | ✅ | Varies | Low | NVIDIA-hosted or self-hosted LLMs | | [LM Studio](../../getting-started/provider-setup.md#lm-studio) | ✅ (Local) | ❌ | ✅ | Varies | Local GUI model management | | [llama.cpp](../../getting-started/provider-setup.md#llamacpp) | ✅ (Local) | ❌ | ✅ | Varies | High-performance local GGUF inference | +| [OpenAI TTS](../../guides/voice/openai-tts.md) | ❌ | ✅ | ✅ | Low | High-quality TTS (tts-1-hd) | +| [ElevenLabs](../../guides/voice/elevenlabs.md) | ❌ | ✅ | Varies | Low | Multilingual TTS, voice cloning | +| [Google TTS](../../guides/voice/google-tts.md) | ✅ | ✅ | ✅ | Low | Cost-effective TTS, 1M chars free | +| [Azure TTS](../../guides/voice/azure-tts.md) | ❌ | ✅ | ✅ | Low | Enterprise TTS, SSML support | +| [Whisper](../../guides/voice/whisper.md) | ❌ | ✅ | ✅ | Low | Best STT accuracy | +| [Deepgram](../../guides/voice/deepgram.md) | ❌ | ✅ | Varies | Low | Real-time STT streaming (WebSocket) | +| [Google STT](../../guides/voice/google-stt.md) | ❌ | ✅ | ✅ | Low | STT for GCP users, 125+ languages | +| [Azure STT](../../guides/voice/azure-stt.md) | ❌ | ✅ | ✅ | Low | Enterprise STT, custom models | +| [OpenAI Realtime](../../guides/voice/openai-realtime.md) | ❌ | ✅ | ✅ | Low | Realtime bidirectional voice | +| [Gemini Live](../../guides/voice/gemini-live.md) | ❌ | ✅ | ✅ | Low | Realtime voice + video (Gemini) | --- @@ -347,3 +485,5 @@ const ai = new NeuroLink({ - **[Cost Optimization](../../guides/enterprise/cost-optimization.md)** - Reduce costs by 80-95% - **[Compliance & Security](../../guides/enterprise/compliance.md)** - GDPR, SOC2, HIPAA - **[Load Balancing](../../guides/enterprise/load-balancing.md)** - Distribution strategies +- **[Voice Providers Comparison](../../reference/provider-comparison.md#voice-providers)** - TTS, STT, and Realtime capability matrix +- **[Voice Provider Selection](../../reference/provider-selection.md#text-to-speech-tts)** - Choosing the right voice provider diff --git a/docs/getting-started/providers/openai-tts.md b/docs/getting-started/providers/openai-tts.md new file mode 100644 index 000000000..706da0eeb --- /dev/null +++ b/docs/getting-started/providers/openai-tts.md @@ -0,0 +1,366 @@ +--- +title: OpenAI TTS Provider Guide +description: Generate high-quality speech audio using OpenAI's Text-to-Speech API through NeuroLink, with six neural voices and multiple audio formats +keywords: openai, tts, text-to-speech, audio, speech synthesis, neural voice, tts-1, tts-1-hd +--- + +# OpenAI TTS Provider Guide + +**High-quality neural text-to-speech with six distinct voices and HD quality option** + +--- + +## Overview + +NeuroLink integrates OpenAI's Text-to-Speech API, giving you access to six expressive neural voices across two model tiers. The standard model (`tts-1`) optimises for low latency, while the HD model (`tts-1-hd`) delivers higher audio fidelity for production use cases such as podcasts, voice assistants, and narration. + +OpenAI TTS works with any NeuroLink text generation call — you can synthesise the raw prompt directly or synthesise the AI-generated response, controlled by the `useAiResponse` flag. + +### Key Facts + +| Property | Value | +| ---------------- | --------------------------------------------- | +| **Provider ID** | `openai-tts` | +| **API endpoint** | `https://api.openai.com/v1/audio/speech` | +| **Models** | `tts-1` (standard), `tts-1-hd` (high quality) | +| **Voices** | alloy, echo, fable, onyx, nova, shimmer | +| **Formats** | mp3, wav, ogg (opus), opus | +| **Max input** | 4,096 characters per request | +| **Languages** | Follows input text language automatically | +| **Streaming** | Not supported (batch synthesis only) | + +--- + +## Quick Start + +### 1. Get an API Key + +Sign up or log in at [https://platform.openai.com](https://platform.openai.com) and create a new secret key under **API keys**. + +### 2. Configure Environment + +Add to your `.env` file: + +```bash +# Required +OPENAI_API_KEY=sk-... + +# Optional: default model (default: tts-1) +OPENAI_TTS_MODEL=tts-1 + +# Optional: default voice (default: alloy) +OPENAI_TTS_VOICE=alloy +``` + +### 3. Install NeuroLink + +```bash +npm install @juspay/neurolink +# or +pnpm add @juspay/neurolink +``` + +### 4. Synthesise Your First Audio + +```typescript +import { NeuroLink } from "@juspay/neurolink"; +import { writeFileSync } from "fs"; + +const ai = new NeuroLink(); + +const result = await ai.generate({ + provider: "openai", + input: { text: "Hello! Welcome to NeuroLink." }, + tts: { + enabled: true, + provider: "openai-tts", + format: "mp3", + }, +}); + +if (result.tts) { + writeFileSync("output.mp3", result.tts.buffer); + console.log(`Saved ${result.tts.size} bytes to output.mp3`); +} +``` + +--- + +## Supported Models + +| Model ID | Quality | Latency | Use Case | +| ---------- | -------- | ------- | ---------------------------------------------- | +| `tts-1` | Standard | Lower | Default; real-time apps, interactive voice UIs | +| `tts-1-hd` | HD | Higher | Podcasts, narration, production audio assets | + +Select the HD model by passing `quality: "hd"` in TTS options — NeuroLink maps this automatically to `tts-1-hd`. + +--- + +## SDK Usage + +### Direct Text Synthesis + +Synthesise the input text directly without calling an AI model: + +```typescript +import { NeuroLink } from "@juspay/neurolink"; + +const ai = new NeuroLink(); + +const result = await ai.generate({ + input: { text: "NeuroLink makes AI development simple." }, + tts: { + enabled: true, + provider: "openai-tts", + voice: "nova", + format: "mp3", + }, +}); + +if (result.tts) { + console.log("Format:", result.tts.format); + console.log("Size:", result.tts.size, "bytes"); + console.log("Latency:", result.tts.metadata?.latency, "ms"); +} +``` + +### AI Response Synthesis + +Generate a response with an AI model and then synthesise it: + +```typescript +const result = await ai.generate({ + provider: "openai", + input: { text: "Greet the user in a warm and friendly tone." }, + tts: { + enabled: true, + provider: "openai-tts", + useAiResponse: true, // Synthesise the AI-generated text, not the prompt + voice: "shimmer", + format: "mp3", + }, +}); +``` + +### HD Quality Audio + +```typescript +const result = await ai.generate({ + input: { text: "This is high-definition audio narration." }, + tts: { + enabled: true, + provider: "openai-tts", + quality: "hd", // Maps to tts-1-hd + voice: "onyx", + format: "wav", + }, +}); +``` + +### Adjusting Playback Speed + +```typescript +const result = await ai.generate({ + input: { text: "Speaking at 80% normal speed." }, + tts: { + enabled: true, + provider: "openai-tts", + voice: "alloy", + speed: 0.8, // Range: 0.25 to 4.0 (default: 1.0) + format: "mp3", + }, +}); +``` + +### Save to File + +```typescript +import { writeFileSync } from "fs"; + +const result = await ai.generate({ + input: { text: "Saving audio to disk." }, + tts: { + enabled: true, + provider: "openai-tts", + voice: "echo", + format: "mp3", + output: "./audio/output.mp3", // NeuroLink saves automatically if set + }, +}); +``` + +### Per-Call Credential Override + +```typescript +const result = await ai.generate({ + input: { text: "Hello!" }, + tts: { + enabled: true, + provider: "openai-tts", + }, + credentials: { + openai: { + apiKey: "sk-user-specific-key", + }, + }, +}); +``` + +--- + +## CLI Usage + +### Basic TTS + +```bash +# Synthesise text directly +neurolink generate "Hello, world!" --tts --tts-provider openai-tts + +# Choose a voice +neurolink generate "Good morning!" --tts --tts-provider openai-tts --tts-voice nova + +# Save to file +neurolink generate "Save this audio." \ + --tts --tts-provider openai-tts \ + --tts-voice shimmer \ + --tts-output greeting.mp3 +``` + +### HD Quality + +```bash +neurolink generate "Professional narration." \ + --tts --tts-provider openai-tts \ + --tts-quality hd \ + --tts-voice onyx \ + --tts-output narration.mp3 +``` + +### Synthesise AI Response + +```bash +neurolink generate "Tell me a short story." \ + --provider openai \ + --tts --tts-provider openai-tts \ + --tts-voice fable \ + --tts-use-ai-response +``` + +### Speed Adjustment + +```bash +neurolink generate "Slow and clear narration." \ + --tts --tts-provider openai-tts \ + --tts-voice alloy \ + --tts-speed 0.75 +``` + +--- + +## Available Voices + +| Voice ID | Gender | Character | Best For | +| --------- | ------- | -------------------------------- | --------------------------------- | +| `alloy` | Neutral | Balanced, clear, versatile | General purpose, default | +| `echo` | Male | Crisp, authoritative | Announcements, business content | +| `fable` | Neutral | Warm, expressive, storytelling | Narration, audiobooks | +| `onyx` | Male | Deep, confident, professional | Voiceovers, documentary | +| `nova` | Female | Bright, friendly, conversational | Voice assistants, customer-facing | +| `shimmer` | Female | Soft, gentle, calm | Wellness apps, guided meditation | + +OpenAI voices are language-agnostic — they follow the language of the input text automatically, supporting English, Spanish, French, German, Japanese, and many more. + +--- + +## Audio Formats + +| Format | Extension | Use Case | Notes | +| ------ | --------- | --------------------------------------- | -------------------- | +| `mp3` | `.mp3` | Default; web, mobile, general storage | 24 kHz sample rate | +| `wav` | `.wav` | Uncompressed; audio editors, processing | 24 kHz sample rate | +| `ogg` | `.ogg` | Browser streaming, web apps | Opus codec at 48 kHz | +| `opus` | `.opus` | Low-bandwidth streaming | Opus codec at 48 kHz | + +--- + +## Configuration Reference + +| Environment Variable | Required | Default | Description | +| -------------------- | -------- | ------- | ------------------------------------- | +| `OPENAI_API_KEY` | Yes | — | OpenAI API key (starts with `sk-`) | +| `OPENAI_TTS_MODEL` | No | `tts-1` | Default model (`tts-1` or `tts-1-hd`) | +| `OPENAI_TTS_VOICE` | No | `alloy` | Default voice ID | + +--- + +## Feature Support Matrix + +| Feature | Supported | Notes | +| ---------------------- | --------- | ---------------------------------- | +| Text synthesis | Yes | | +| AI response synthesis | Yes | Set `useAiResponse: true` | +| HD quality | Yes | `quality: "hd"` maps to `tts-1-hd` | +| Speed control | Yes | 0.25 – 4.0 | +| Voice selection | Yes | 6 neural voices | +| Multiple formats | Yes | mp3, wav, ogg, opus | +| Streaming TTS | No | Batch synthesis only | +| Pitch / volume control | No | Not supported by OpenAI TTS API | +| Custom voices | No | Only built-in voices supported | + +--- + +## Troubleshooting + +### "OpenAI TTS API key not configured" + +The `OPENAI_API_KEY` environment variable is missing or was not loaded. + +```bash +# Check the variable is set +echo $OPENAI_API_KEY + +# Set it for the current session +export OPENAI_API_KEY=sk-... +``` + +Create or rotate keys at [https://platform.openai.com/api-keys](https://platform.openai.com/api-keys). + +### "HTTP 429" — Rate limit exceeded + +You have hit OpenAI's TTS rate limits. Implement exponential backoff or reduce request concurrency. Rate limits are per-key and depend on your usage tier. + +### "HTTP 400" — Request too long + +The input text exceeds 4,096 characters. Split long content into smaller chunks and synthesise each separately. + +```typescript +function chunkText(text: string, maxLen = 4000): string[] { + const chunks: string[] = []; + for (let i = 0; i < text.length; i += maxLen) { + chunks.push(text.slice(i, i + maxLen)); + } + return chunks; +} +``` + +### "OpenAI TTS request timed out after 30 seconds" + +A network issue or overloaded API caused the request to time out. Retry the request — the error is marked retriable by NeuroLink's error system. + +### Audio sounds distorted at high speed + +Speeds above 2.0 can introduce artifacts. Use `speed: 1.0` – `1.5` for natural-sounding output. + +--- + +## See Also + +- [TTS Integration Guide](/docs/features/tts) — complete multi-provider TTS reference +- [Audio Input (STT)](/docs/features/audio-input) — speech-to-text counterpart +- [OpenAI Provider Guide](/docs/getting-started/providers/openai) — full OpenAI text generation provider +- [ElevenLabs Provider Guide](/docs/getting-started/providers/elevenlabs) — alternative TTS provider with voice cloning + +--- + +**Need Help?** Join the [GitHub Discussions](https://github.com/juspay/neurolink/discussions) or open an [issue](https://github.com/juspay/neurolink/issues). diff --git a/docs/provider-integration/14-voice-speech-integration.md b/docs/provider-integration/14-voice-speech-integration.md new file mode 100644 index 000000000..6c88bc15a --- /dev/null +++ b/docs/provider-integration/14-voice-speech-integration.md @@ -0,0 +1,452 @@ +# 14 · Voice / Speech Integration — Implementation Journal + +Commit: `27a31c32` — `feat(voice): add multi-provider TTS, STT, and realtime voice integration` + +--- + +## Architecture + +### How voice plugs into Factory + Registry + +The voice integration does not add AI providers (it adds no entries to `AIProviderName`). Instead it introduces **three parallel static registries** that mirror the `ProviderFactory` / `ProviderRegistry` pattern for non-LLM capabilities: + +``` +ProviderFactory → creates LLM provider instances +ProviderRegistry → holds LLM factory functions (dynamic imports) + +TTSProcessor → static Map (text-to-speech) +STTProcessor → static Map (speech-to-text) +RealtimeProcessor → static Map (bidirectional voice) +``` + +Each processor exposes `registerHandler(name, handler)` and the appropriate operation (`synthesize`, `transcribe`, `connect`). The same `O(1)` Map lookup and lazy-instantiation pattern used by `ProviderRegistry` applies here. + +### Registration location + +All handler registration happens at the **bottom** of `ProviderRegistry.registerAllProviders()` in `src/lib/factories/providerRegistry.ts`, after all LLM providers are registered. The order is: + +1. LLM providers (existing) +2. TTS handler registration block +3. STT handler registration block +4. Realtime handler registration block + +Each block uses a separate `try/catch` so a missing API key or a broken import cannot prevent the LLM providers from registering. Registration is fire-and-forget: failures log a `warn` and continue. + +All imports inside the registration blocks are **dynamic** (`await import(...)`), matching CLAUDE.md rule #1 and preventing circular dependencies. + +```ts +// Pattern used for every voice handler: +try { + const { TTSProcessor } = await import("../utils/ttsProcessor.js"); + const { OpenAITTS } = await import("../voice/providers/OpenAITTS.js"); + TTSProcessor.registerHandler("openai-tts", new OpenAITTS()); +} catch { + /* Optional provider — skip if unavailable */ +} +``` + +### STT preprocessing in `neurolink.ts runStandardGenerateRequest()` + +When a caller passes `{ stt: { enabled: true, audio: buffer } }` to `generate()`, the following happens inside `runStandardGenerateRequest()` before the LLM call: + +1. `ProviderRegistry.isRegistered()` is checked; if false, `registerAllProviders()` is awaited. +2. `STTProcessor` is dynamically imported and `transcribe(audio, providerName, sttOptions)` is called. +3. The transcription text is injected into the LLM prompt: + - If no user text exists, the transcription becomes the prompt directly. + - If user text exists, the transcription is prepended as `[Transcribed audio]: \n\n`. +4. `generateResult.transcription` is set to the `STTResult` object (available to callers). +5. Transcription failures log an error but **do not block** generation — STT is optional. + +### Type organisation + +Three new canonical type files added to `src/lib/types/` (CLAUDE.md rule #8 compliant — no "Types" suffix): + +| File | Contents | +| --------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `src/lib/types/tts.ts` | Extended `AudioFormat` (added `m4a`, `flac`, `webm`, `mp4`, `mpeg`, `mpga`); added `TTSOptions.provider` field | +| `src/lib/types/stt.ts` | `STTHandler`, `STTOptions`, `STTResult`, `STTLanguage`, `WordTiming`, `TranscriptionSegment`, `STT_ERROR_CODES`, `DEFAULT_STT_OPTIONS`, guards | +| `src/lib/types/realtime.ts` | `RealtimeHandler`, `RealtimeConfig`, `RealtimeSession`, `RealtimeAudioChunk`, `RealtimeSessionState`, `REALTIME_ERROR_CODES`, `DEFAULT_REALTIME_CONFIG` | +| `src/lib/types/voice.ts` | Aggregator: re-exports all of `tts.ts`, `stt.ts`, `realtime.ts`; adds `VoiceCapability`, `VoiceProviderName`, `VoiceProviderConfig`, `VoiceErrorOptions` | + +`src/lib/types/index.ts` gets two new `export *` lines (for `stt.ts` and `realtime.ts`; `voice.ts` is already present). All rules 9 and 10 apply: type names are globally unique, barrel uses `export *` only. + +--- + +## TTS Providers Added + +### `openai-tts` + +- **File:** `src/lib/voice/providers/OpenAITTS.ts` (253 lines, NEW) +- **Class:** `OpenAITTS implements TTSHandler` +- **API:** `POST https://api.openai.com/v1/audio/speech` +- **Auth:** `Authorization: Bearer $OPENAI_API_KEY` +- **Models:** `tts-1` (standard, default) and `tts-1-hd` (high quality; selected when `options.quality === "hd"`) +- **Voices (6):** `alloy`, `echo`, `fable`, `onyx`, `nova`, `shimmer` +- **Output formats:** `mp3` (default), `wav`, `opus`/`ogg` (mapped to OpenAI's `opus`) +- **Max text:** 4 096 characters +- **Registered as:** `"openai-tts"` in `TTSProcessor` +- **Timeout:** 30-second `AbortController` on every `fetch` call; throws `TTSError` with `TTS_ERROR_CODES.SYNTHESIS_FAILED` on abort + +### `elevenlabs` + +- **File:** `src/lib/voice/providers/ElevenLabsTTS.ts` (326 lines, NEW) +- **Class:** `ElevenLabsTTS implements TTSHandler` +- **API:** `POST https://api.elevenlabs.io/v1/text-to-speech/{voice_id}?output_format=...` +- **Auth:** `xi-api-key: $ELEVENLABS_API_KEY` +- **Model:** `eleven_multilingual_v2` (default) +- **Voices:** Dynamic — fetched from `/v1/voices` and cached for 5 minutes. Default voice: `21m00Tcm4TlvDq8ikWAM` (Rachel). +- **Output formats:** `mp3_44100_128` (mp3), `pcm_44100` (wav), `ogg_22050` (ogg/opus) +- **Voice settings:** `stability` (default 0.5), `similarity_boost` (0.75), `style` (0.0), `use_speaker_boost` (true) +- **Max text:** 5 000 characters +- **Registered as:** `"elevenlabs"` and `"elevenlabs-tts"` in `TTSProcessor` +- **Timeout:** 30-second `AbortController` on `synthesize` and `getVoices` calls + +### `azure-tts` + +- **File:** `src/lib/voice/providers/AzureTTS.ts` (357 lines, NEW) +- **Class:** `AzureTTS implements TTSHandler` +- **API:** `POST https://{region}.tts.speech.microsoft.com/cognitiveservices/v1` +- **Auth:** `Ocp-Apim-Subscription-Key: $AZURE_SPEECH_KEY` +- **Region:** `$AZURE_SPEECH_REGION` (default `"eastus"`) +- **Default voice:** `en-US-JennyNeural` +- **Output format (default):** `audio-24khz-96kbitrate-mono-mp3` +- **SSML:** The handler builds SSML automatically from `text`, `voice`, `speed`, and `pitch` options. Callers can pass raw SSML by setting `text` to a string starting with ` controller.abort(), 30000); +try { + response = await fetch(url, { ..., signal: controller.signal }); +} finally { + clearTimeout(timeoutId); +} +``` + +`AbortError` is caught and re-thrown as a typed `TTSError` / `STTError` with a human-readable message. This pattern is consistent across all 7 new providers. + +### Audio utilities (`src/lib/voice/audio-utils.ts`) + +552-line utility module with no external dependencies beyond Node.js built-ins: + +| Export | Purpose | +| ---------------------------------------------------------------- | ------------------------------------------------------- | +| `detectAudioFormat(buffer)` | Identifies `wav`, `mp3`, `ogg`, `opus` from magic bytes | +| `createWavHeader(samples, sampleRate, channels, bitsPerSample)` | Builds a 44-byte RIFF/WAV header | +| `createWavFile(pcmData, sampleRate, channels, bitsPerSample)` | Header + PCM data | +| `createPcmBuffer(durationMs, sampleRate, frequency)` | Generates a sine-wave PCM buffer (for testing) | +| `extractPcmSamples(buffer)` | Reads 16-bit LE PCM samples from a WAV | +| `normalizeAudio(samples)` | Scales to peak 0.9 | +| `resamplePcm(samples, fromRate, toRate)` | Linear interpolation resampling | +| `calculateDuration(buffer, sampleRate, channels, bitsPerSample)` | Duration in seconds | +| `splitIntoChunks(buffer, chunkSize)` | Splits a Buffer into equal-size chunks | +| `convertAudioFormat(buffer, from, to)` | Best-effort format conversion (wav↔pcm only for now) | +| `getMimeType(format)` / `getFileExtension(format)` | Format → MIME / extension | +| `AUDIO_SIGNATURES` | Magic-byte constants per format | +| `MIME_TYPES` | Format → MIME map constant | + +### Stream infrastructure (`src/lib/voice/stream-handler.ts`) + +546-line module providing: + +| Export | Purpose | +| ----------------------------------------- | ---------------------------------------------------------------------------------------------- | +| `ChunkedAudioStream extends EventEmitter` | Slices incoming audio into fixed-duration chunks (default 100 ms) with backpressure management | +| `StreamHandler extends EventEmitter` | Generic event-driven handler with start/stop and error propagation | +| `StreamSplitter` | Fan-out: one input → multiple output streams | +| `StreamMerger` | Fan-in: multiple input streams → one output | +| `asyncIterableToStream(iterable)` | Converts `AsyncIterable` → Node `Readable` | +| `streamToAsyncIterable(stream)` | Converts Node `Readable` → `AsyncIterable` | + +`ChunkedAudioStream` defaults: `chunkDurationMs=100`, `sampleRate=16000`, `bytesPerSample=2`, `highWaterMark=64KB`. + +--- + +## Error Handling + +Three new error classes in `src/lib/voice/errors.ts` (all extend `NeuroLinkError`): + +| Class | Default category | Default severity | +| --------------- | ---------------- | ---------------- | +| `VoiceError` | `EXECUTION` | `MEDIUM` | +| `STTError` | `VALIDATION` | `MEDIUM` | +| `RealtimeError` | `EXECUTION` | `HIGH` | + +`TTSError` lives in `src/lib/utils/ttsProcessor.ts` (pre-existing; not in `errors.ts`). + +`STTError` includes static factory methods: `audioEmpty`, `audioTooLong`, `invalidFormat`, `languageNotSupported`, `transcriptionFailed`, `providerNotConfigured`, `providerNotSupported`, `streamError`. + +`RealtimeError` includes: `connectionFailed`, `sessionTimeout`, `protocolError`, `audioStreamError`, `providerNotConfigured`, `sessionAlreadyActive`, `sessionNotActive`, `invalidMessage`. + +--- + +## CLI Changes + +New flags added to `src/cli/commands/voice.ts` and propagated via `src/cli/factories/commandFactory.ts`: + +| Flag | Purpose | +| ---------------- | --------------------------------------------------------------------- | +| `--stt` | Enable STT preprocessing | +| `--stt-provider` | Which STT provider to use (default: `whisper`) | +| `--input-audio` | Path to audio file for STT | +| `--stt-language` | BCP-47 language code for transcription | +| `--tts-provider` | Override TTS provider (e.g., `openai-tts`, `elevenlabs`, `azure-tts`) | + +The `--tts` and `--tts-voice` flags are pre-existing. + +--- + +## Testing + +Test suite: `test/continuous-test-suite-voice.ts` (1 822 lines, NEW) + +The suite is invoked as: + +```bash +npx tsx test/continuous-test-suite-voice.ts --provider=vertex +``` + +It covers 15 test items via the consumer API only — no direct provider class calls: + +| # | Test | Notes | +| ---- | --------------------------- | ---------------------------------------------------------------------------------------- | +| 1 | `generate()` + TTS MP3 | Validates MP3 magic bytes (`0xFF 0xFB` or `0x49 0x44 0x33`) | +| 2 | `generate()` + TTS WAV | Validates RIFF header (`0x52 0x49 0x46 0x46`) | +| 3 | Unconfigured TTS provider | Verifies `azure-tts` without keys errors gracefully | +| 4 | `generate()` + STT | Validates `result.transcription.confidence` is numeric | +| 5 | STT + TTS round-trip | Audio in → LLM → audio out; validates both transcription and MP3 output | +| 6–8 | `stream()` + TTS | Validates `StreamResult` with audio chunks | +| 9–10 | CLI `--tts` / `--stt` flags | Spawns CLI subprocess, validates exit code and JSON output | +| 11 | Handler registration check | Verifies `TTSProcessor`, `STTProcessor`, `RealtimeProcessor` have expected provider keys | +| 12 | Audio utility validation | `detectAudioFormat`, `createWavHeader`, `splitIntoChunks`, `resamplePcm` | +| 13 | `ChunkedAudioStream` | Validates chunking and event emission | +| 14 | Barrel exports | `VOICE_ERROR_CODES`, `STT_ERROR_CODES`, `REALTIME_ERROR_CODES`, `DEFAULT_STT_OPTIONS` | +| 15 | Removed method guard | Asserts `synthesize`, `transcribe`, `startRealtimeVoice` do NOT exist on `NeuroLink` | + +**Real API results logged in commit message:** + +| Provider | Phrase | Confidence | +| -------------------- | --------------------------------- | ----------------- | +| Whisper (openai-stt) | "The quick brown fox..." | 0.95 | +| Deepgram | same | 1.0 | +| Google STT | same | 0.98 | +| Azure STT | same | 0.9 | +| Full round-trip | Whisper → Vertex LLM → ElevenLabs | 126 KB MP3 output | + +--- + +## Files Changed + +### New files (11) + +| File | Lines | Purpose | +| ------------------------------------------- | ----- | ----------------------------------------- | +| `src/lib/voice/providers/OpenAITTS.ts` | 253 | OpenAI TTS handler | +| `src/lib/voice/providers/ElevenLabsTTS.ts` | 326 | ElevenLabs TTS handler | +| `src/lib/voice/providers/AzureTTS.ts` | 357 | Azure Cognitive Services TTS handler | +| `src/lib/voice/providers/OpenAISTT.ts` | 317 | Whisper / OpenAI STT handler | +| `src/lib/voice/providers/DeepgramSTT.ts` | 547 | Deepgram STT handler | +| `src/lib/voice/providers/GoogleSTT.ts` | 481 | Google Cloud STT handler | +| `src/lib/voice/providers/AzureSTT.ts` | 374 | Azure Cognitive Services STT handler | +| `src/lib/voice/providers/OpenAIRealtime.ts` | 475 | OpenAI Realtime (WebSocket) handler | +| `src/lib/voice/providers/GeminiLive.ts` | 413 | Gemini Live (WebSocket) handler | +| `src/lib/voice/audio-utils.ts` | 552 | Audio format detection, WAV/PCM utilities | +| `src/lib/voice/stream-handler.ts` | 546 | Chunked streaming, fan-out/fan-in | + +### Substantially extended files (4) + +| File | Change | +| ----------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `src/lib/voice/RealtimeVoiceAPI.ts` | 516 lines added — `BaseRealtimeHandler` (abstract) and `RealtimeProcessor` (static handler registry with connect/send/disconnect) | +| `src/lib/voice/errors.ts` | 464 lines added — `VoiceError`, `STTError`, `RealtimeError` with full static factory methods | +| `src/lib/voice/index.ts` | 125 lines added — barrel for all voice exports | +| `src/lib/utils/sttProcessor.ts` | 319 lines added — `STTProcessor` static registry with `transcribe`, `getHandler`, `supports`, `registerHandler`, span instrumentation matching `TTSProcessor` | + +### New type files (2) + +| File | Lines | Purpose | +| --------------------------- | ----- | -------------------------------------------------- | +| `src/lib/types/stt.ts` | 772 | All STT types, error codes, constants, type guards | +| `src/lib/types/realtime.ts` | 322 | All Realtime types, error codes, constants, guards | + +### Modified files + +| File | Change | +| ----------------------------------------------- | ---------------------------------------------------------------------------------------------------------- | +| `src/lib/types/tts.ts` | Extended `AudioFormat` union with 6 additional formats; added `TTSOptions.provider` | +| `src/lib/types/voice.ts` | Now re-exports `stt.ts` and `realtime.ts`; adds voice-level union types | +| `src/lib/types/index.ts` | New `export *` for `stt.ts` and `realtime.ts` | +| `src/lib/types/generate.ts` | Added `stt` option block to `GenerateOptions`; added `transcription: STTResult` to `GenerateResult` | +| `src/lib/types/stream.ts` | Minor additions for audio stream result types | +| `src/lib/types/span.ts` | Added `SpanType.STT` enum value | +| `src/lib/factories/providerRegistry.ts` | TTS, STT, and Realtime handler registration blocks at end of `registerAllProviders()` | +| `src/lib/neurolink.ts` | STT preprocessing in `runStandardGenerateRequest()`; TTS option threading to stream/generate | +| `src/cli/commands/voice.ts` | New `--stt`, `--stt-provider`, `--input-audio`, `--stt-language`, `--tts-provider` flags | +| `src/lib/server/voice/voiceWebSocketHandler.ts` | Refactored to use `STTProcessor` / `TTSProcessor` / `RealtimeProcessor` instead of direct provider classes | +| `.env.example` | Added `DEEPGRAM_API_KEY`, `ELEVENLABS_API_KEY`, `AZURE_SPEECH_KEY`, `AZURE_SPEECH_REGION` | +| `test/continuous-test-suite-voice.ts` | 1 822-line new test suite | + +--- + +## Smoke Tests + +```bash +# Build first +pnpm run build:cli + +# TTS: OpenAI +export OPENAI_API_KEY="sk-..." +pnpm run cli generate "Hello world" --tts --tts-provider openai-tts --tts-voice nova + +# TTS: ElevenLabs +export ELEVENLABS_API_KEY="..." +pnpm run cli generate "Hello world" --tts --tts-provider elevenlabs + +# STT: Whisper +export OPENAI_API_KEY="sk-..." +pnpm run cli generate --stt --stt-provider whisper --input-audio recording.wav + +# STT + TTS round-trip +pnpm run cli generate --stt --stt-provider whisper --input-audio recording.wav \ + --tts --tts-provider openai-tts --provider openai + +# Full test suite (requires Vertex credentials) +npx tsx test/continuous-test-suite-voice.ts --provider=vertex +``` + +--- + +## Backward Compatibility + +- No changes to `AIProviderName` enum — existing provider callers unaffected. +- No new public `NeuroLink` methods — interface extends only through option fields. +- `AudioFormat` type extended additively — existing `"mp3" | "wav" | "ogg" | "opus"` values unchanged. +- `GenerateOptions.stt` and `GenerateResult.transcription` are optional — callers not passing `stt` see no change in behaviour. +- `TTSProcessor` pre-existing registration for `google-ai` and `vertex` (via `GoogleTTSHandler`) is unmodified. diff --git a/docs/provider-integration/README.md b/docs/provider-integration/README.md index e17d028b6..2ca3d2483 100644 --- a/docs/provider-integration/README.md +++ b/docs/provider-integration/README.md @@ -23,6 +23,7 @@ into Neurolink: 6. **`03-nvidia-nim.md`** — most complex (extra body params, retry-on-400). Tackle last. 7. **`06-testing.md`** — test additions and validation strategy. 8. **`07-implementation-order.md`** — ordered task list, milestone gates, risk mitigations. +9. **`14-voice-speech-integration.md`** — voice/speech integration journal: TTS (OpenAI, ElevenLabs, Azure), STT (Whisper, Deepgram, Google, Azure), Realtime (OpenAI Realtime, Gemini Live). Feature integration, not a new LLM provider. ## Source-of-truth references @@ -36,14 +37,15 @@ Recent commits used as templates: ## Status -| Task | Status | -| ------------------------------ | ------------------------------------------------------------------------------------------- | -| DeepSeek implementation | ✅ Implemented · live in `src/lib/providers/deepseek.ts` | -| NVIDIA NIM implementation | ✅ Implemented · live in `src/lib/providers/nvidiaNim.ts` | -| LM Studio implementation | ✅ Implemented · live in `src/lib/providers/lmStudio.ts` | -| llama.cpp implementation | ✅ Implemented · live in `src/lib/providers/llamaCpp.ts` | -| Shared changes (types/CLI/etc) | ✅ Implemented · see [`01-shared-changes.md`](/docs/provider-integration/01-shared-changes) | -| Tests | ✅ Implemented · `test/continuous-test-suite-new-providers.ts` | +| Task | Status | +| ------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------- | +| DeepSeek implementation | ✅ Implemented · live in `src/lib/providers/deepseek.ts` | +| NVIDIA NIM implementation | ✅ Implemented · live in `src/lib/providers/nvidiaNim.ts` | +| LM Studio implementation | ✅ Implemented · live in `src/lib/providers/lmStudio.ts` | +| llama.cpp implementation | ✅ Implemented · live in `src/lib/providers/llamaCpp.ts` | +| Shared changes (types/CLI/etc) | ✅ Implemented · see [`01-shared-changes.md`](/docs/provider-integration/01-shared-changes) | +| Tests | ✅ Implemented · `test/continuous-test-suite-new-providers.ts` | +| Voice/Speech integration | ✅ Implemented · see [`14-voice-speech-integration.md`](/docs/provider-integration/14-voice-speech-integration) · commit `27a31c32` | Each per-provider doc lists: diff --git a/docs/reference/provider-comparison.md b/docs/reference/provider-comparison.md index 752cb9b81..7fc812546 100644 --- a/docs/reference/provider-comparison.md +++ b/docs/reference/provider-comparison.md @@ -972,6 +972,35 @@ _Subscription Pricing (via OAuth):_ --- +--- + +## Voice Providers + +Voice providers handle audio I/O and are distinct from LLM text-generation providers. They are categorised by function: Text-to-Speech (TTS), Speech-to-Text (STT), and Realtime (bidirectional audio over WebSocket). + +| Provider | Type | Protocol | Streaming | Formats | Auth | +| ------------------- | -------- | ---------------- | --------- | ------------------- | ---------------- | +| **google-ai** (TTS) | TTS | REST (gRPC SDK) | No | MP3, WAV, OGG | Service Account | +| **openai-tts** | TTS | REST | No | MP3, WAV, OGG, Opus | API Key | +| **elevenlabs** | TTS | REST | No | MP3 | API Key | +| **azure-tts** | TTS | REST | No | MP3 | API Key + Region | +| **whisper** | STT | REST | No | WAV, MP3, M4A, FLAC | API Key | +| **google-stt** | STT | REST | No | WAV, FLAC, MP3, OGG | Service Account | +| **deepgram** | STT | REST + WebSocket | Yes | WAV, MP3, OGG, FLAC | API Key | +| **azure-stt** | STT | REST | No | WAV, MP3 | API Key + Region | +| **openai-realtime** | Realtime | WebSocket | Yes | WAV, Opus | API Key | +| **gemini-live** | Realtime | WebSocket | Yes | WAV | API Key | + +**Legend:** + +- **TTS** — Text-to-Speech: converts text to audio +- **STT** — Speech-to-Text: transcribes audio to text +- **Realtime** — Bidirectional voice session with the model over a persistent WebSocket + +See also: [Voice Provider Selection](#voice-provider-selection) | [Voice Providers Index](../getting-started/providers/index.md#voice-providers) + +--- + ## Use Case Recommendations ### For Startups (Limited Budget) @@ -1304,6 +1333,122 @@ const result = await neurolink.generate({ --- +## Voice Provider Selection + +### Text-to-Speech (TTS) + +**Best quality: `openai-tts` with model tts-1-hd** + +```typescript +import { NeuroLink } from "@juspay/neurolink"; + +const neurolink = new NeuroLink(); + +const audio = await neurolink.tts({ + input: "Hello, world!", + voice: { provider: "openai-tts", model: "tts-1-hd", voiceId: "nova" }, +}); +``` + +**Best multilingual: `elevenlabs`** + +ElevenLabs supports the widest range of languages and voice cloning, making it the default choice for multilingual or branded voice experiences. + +```typescript +const audio = await neurolink.tts({ + input: "Hola, ¿cómo estás?", + voice: { provider: "elevenlabs", voiceId: "your-voice-id" }, +}); +``` + +**Most cost-effective: `google-ai` (1M chars free tier)** + +Google Cloud Text-to-Speech provides a generous free tier (1M characters/month for standard voices) and is ideal for high-volume applications on GCP. + +```typescript +const audio = await neurolink.tts({ + input: "Cost-effective synthesis at scale.", + voice: { provider: "google-ai", voiceId: "en-US-Standard-A" }, +}); +``` + +**Enterprise: `azure-tts` (SSML support)** + +Azure Cognitive Services TTS has the most comprehensive SSML support, including fine-grained prosody control, making it the standard choice for enterprise IVR and accessibility pipelines. + +```typescript +const audio = await neurolink.tts({ + input: "Welcome to Neurolink.", + voice: { + provider: "azure-tts", + voiceId: "en-US-AriaNeural", + credentials: { + azureApiKey: process.env.AZURE_SPEECH_KEY, + azureRegion: "eastus", + }, + }, +}); +``` + +--- + +### Speech-to-Text (STT) + +**Best accuracy: `whisper` (OpenAI)** + +OpenAI Whisper consistently ranks highest on transcription benchmarks across languages and noisy environments. + +```typescript +const transcript = await neurolink.stt({ + audio: { file: "./recording.mp3" }, + speech: { provider: "whisper", model: "whisper-1" }, +}); +``` + +**Best streaming: `deepgram` (WebSocket real-time)** + +Deepgram is the only STT provider with native WebSocket streaming support, enabling sub-300 ms word-level transcription for live audio. + +```typescript +const transcript = await neurolink.stt({ + audio: { stream: microphoneStream }, + speech: { provider: "deepgram", model: "nova-3", streaming: true }, +}); +``` + +**Best for Google Cloud users: `google-stt`** + +Tight integration with GCP infrastructure, support for 125+ languages, and speaker diarization make `google-stt` the natural choice when already on Google Cloud. + +```typescript +const transcript = await neurolink.stt({ + audio: { file: "./meeting.flac" }, + speech: { + provider: "google-stt", + credentials: { googleServiceAccountKey: process.env.GOOGLE_SA_KEY }, + }, +}); +``` + +**Enterprise: `azure-stt`** + +Azure Cognitive Services STT offers custom model training, batch transcription, and fine-grained compliance controls for regulated industries. + +```typescript +const transcript = await neurolink.stt({ + audio: { file: "./call-recording.wav" }, + speech: { + provider: "azure-stt", + credentials: { + azureApiKey: process.env.AZURE_SPEECH_KEY, + azureRegion: "eastus", + }, + }, +}); +``` + +--- + ## Conclusion **Choose based on priorities:** @@ -1317,6 +1462,11 @@ const result = await neurolink.generate({ 7. **Flexibility Priority** → OpenRouter (300+ models) or NVIDIA NIM (curated NVIDIA-hosted models) 8. **Flat-Rate Pricing** → Anthropic subscription (Pro $20/mo, Max $100+/mo) 9. **Zero Cloud Cost** → LM Studio or llama.cpp (local execution) +10. **TTS Quality** → `openai-tts` (tts-1-hd) or `elevenlabs` (multilingual) +11. **TTS Cost** → `google-ai` TTS (1M chars/month free tier) +12. **STT Accuracy** → `whisper` (OpenAI) +13. **STT Streaming** → `deepgram` (WebSocket, sub-300 ms) +14. **Realtime Voice** → `openai-realtime` or `gemini-live` **NeuroLink Advantage:** @@ -1330,3 +1480,5 @@ See also: - [Provider Capabilities Audit](./provider-capabilities-audit.md) - Detailed technical capabilities - [Provider Selection Wizard](../guides/provider-selection.md) - Interactive decision guide - [Claude Subscription Support](../features/claude-subscription.md) - OAuth authentication and subscription tiers for Anthropic +- [Voice Provider Selection](./provider-comparison.md#voice-provider-selection) - TTS, STT, and Realtime provider recommendations +- [Voice Providers Index](../getting-started/providers/index.md#voice-providers) - Voice provider setup cards diff --git a/docs/reference/provider-selection.md b/docs/reference/provider-selection.md index ebd5866fb..2c126911c 100644 --- a/docs/reference/provider-selection.md +++ b/docs/reference/provider-selection.md @@ -16,26 +16,31 @@ This guide helps you choose the optimal AI provider for your specific use case, Use this matrix to quickly identify the best provider for your primary requirement: -| Primary Need | Best Choice | Alternative | Budget Option | -| ------------------------- | -------------------- | -------------------- | ----------------------- | -| **Highest Quality** | OpenAI GPT-4o/GPT-5 | Anthropic Claude 4.5 | Google Gemini 2.5 Pro | -| **Extended Thinking** | Anthropic Claude 4.5 | Google Gemini 2.5+ | Google AI Studio (Free) | -| **PDF Processing** | Anthropic | Google AI Studio | Google Vertex | -| **Complete Privacy** | Ollama (Local) | Self-hosted LiteLLM | - | -| **Enterprise Security** | Azure OpenAI | Amazon Bedrock | Google Vertex | -| **GDPR Compliance** | Mistral | Ollama (Local) | - | -| **Free Tier** | Google AI Studio | OpenRouter | HuggingFace | -| **Multi-Provider Access** | OpenRouter | LiteLLM | - | -| **AWS Integration** | Amazon Bedrock | Amazon SageMaker | - | -| **Azure Integration** | Azure OpenAI | - | - | -| **GCP Integration** | Google Vertex | Google AI Studio | - | -| **Vision/Multimodal** | OpenAI GPT-4o | Anthropic Claude 4.5 | Google Gemini | -| **Tool Calling** | OpenAI | Anthropic | Google AI Studio | -| **Custom Models** | Amazon SageMaker | OpenAI Compatible | Ollama | -| **Budget Reasoning** | DeepSeek (R1) | NVIDIA NIM | llama.cpp (local) | -| **Local GUI Inference** | LM Studio | Ollama | llama.cpp | -| **Local CLI Inference** | llama.cpp | Ollama | LM Studio | -| **NVIDIA GPU Cloud** | NVIDIA NIM | - | - | +| Primary Need | Best Choice | Alternative | Budget Option | +| ------------------------- | --------------------- | -------------------- | ----------------------- | +| **Highest Quality** | OpenAI GPT-4o/GPT-5 | Anthropic Claude 4.5 | Google Gemini 2.5 Pro | +| **Extended Thinking** | Anthropic Claude 4.5 | Google Gemini 2.5+ | Google AI Studio (Free) | +| **PDF Processing** | Anthropic | Google AI Studio | Google Vertex | +| **Complete Privacy** | Ollama (Local) | Self-hosted LiteLLM | - | +| **Enterprise Security** | Azure OpenAI | Amazon Bedrock | Google Vertex | +| **GDPR Compliance** | Mistral | Ollama (Local) | - | +| **Free Tier** | Google AI Studio | OpenRouter | HuggingFace | +| **Multi-Provider Access** | OpenRouter | LiteLLM | - | +| **AWS Integration** | Amazon Bedrock | Amazon SageMaker | - | +| **Azure Integration** | Azure OpenAI | - | - | +| **GCP Integration** | Google Vertex | Google AI Studio | - | +| **Vision/Multimodal** | OpenAI GPT-4o | Anthropic Claude 4.5 | Google Gemini | +| **Tool Calling** | OpenAI | Anthropic | Google AI Studio | +| **Custom Models** | Amazon SageMaker | OpenAI Compatible | Ollama | +| **Budget Reasoning** | DeepSeek (R1) | NVIDIA NIM | llama.cpp (local) | +| **Local GUI Inference** | LM Studio | Ollama | llama.cpp | +| **Local CLI Inference** | llama.cpp | Ollama | LM Studio | +| **NVIDIA GPU Cloud** | NVIDIA NIM | - | - | +| **TTS Quality** | openai-tts (tts-1-hd) | elevenlabs | google-ai (free tier) | +| **TTS Multilingual** | elevenlabs | openai-tts | azure-tts | +| **STT Accuracy** | whisper | deepgram | google-stt | +| **STT Streaming** | deepgram | - | - | +| **Realtime Voice** | openai-realtime | gemini-live | - | --- @@ -846,10 +851,20 @@ START: What's your primary constraint? │ ├─ GCP → Google Vertex AI │ └─ Multi-cloud → LiteLLM or OpenRouter │ -└─ PERFORMANCE → What matters most? - ├─ Latency → Ollama (local) or Google AI Studio - ├─ Throughput → OpenAI or Google - └─ Quality → OpenAI GPT-4o or Anthropic Claude +├─ PERFORMANCE → What matters most? +│ ├─ Latency → Ollama (local) or Google AI Studio +│ ├─ Throughput → OpenAI or Google +│ └─ Quality → OpenAI GPT-4o or Anthropic Claude +│ +└─ VOICE → What kind of audio I/O do you need? + ├─ Text-to-Speech (quality) → openai-tts (tts-1-hd) or elevenlabs (multilingual) + ├─ Text-to-Speech (cost) → google-ai (1M chars/month free) + ├─ Text-to-Speech (enterprise) → azure-tts (full SSML) + ├─ Speech-to-Text (accuracy) → whisper + ├─ Speech-to-Text (real-time streaming) → deepgram (WebSocket) + ├─ Speech-to-Text (GCP users) → google-stt + ├─ Speech-to-Text (enterprise) → azure-stt + └─ Realtime bidirectional voice → openai-realtime or gemini-live ``` --- @@ -880,6 +895,102 @@ START: What's your primary constraint? **Use NVIDIA NIM** - Curated Llama, Nemotron, and DeepSeek-R1 models served at scale via NVIDIA's cloud. +### Text-to-Speech (TTS) + +**Best quality: `openai-tts` with model tts-1-hd** + +```typescript +import { NeuroLink } from "@juspay/neurolink"; + +const neurolink = new NeuroLink(); + +const audio = await neurolink.tts({ + input: "Hello, world!", + voice: { provider: "openai-tts", model: "tts-1-hd", voiceId: "nova" }, +}); +``` + +**Best multilingual: `elevenlabs`** + +```typescript +const audio = await neurolink.tts({ + input: "Hola, ¿cómo estás?", + voice: { provider: "elevenlabs", voiceId: "your-voice-id" }, +}); +``` + +**Most cost-effective: `google-ai` (1M chars free tier)** + +```typescript +const audio = await neurolink.tts({ + input: "Cost-effective synthesis at scale.", + voice: { provider: "google-ai", voiceId: "en-US-Standard-A" }, +}); +``` + +**Enterprise: `azure-tts` (SSML support)** + +```typescript +const audio = await neurolink.tts({ + input: "Welcome to Neurolink.", + voice: { + provider: "azure-tts", + voiceId: "en-US-AriaNeural", + credentials: { + azureApiKey: process.env.AZURE_SPEECH_KEY, + azureRegion: "eastus", + }, + }, +}); +``` + +### Speech-to-Text (STT) + +**Best accuracy: `whisper` (OpenAI)** + +```typescript +const transcript = await neurolink.stt({ + audio: { file: "./recording.mp3" }, + speech: { provider: "whisper", model: "whisper-1" }, +}); +``` + +**Best streaming: `deepgram` (WebSocket real-time)** + +```typescript +const transcript = await neurolink.stt({ + audio: { stream: microphoneStream }, + speech: { provider: "deepgram", model: "nova-3", streaming: true }, +}); +``` + +**Best for Google Cloud users: `google-stt`** + +```typescript +const transcript = await neurolink.stt({ + audio: { file: "./meeting.flac" }, + speech: { + provider: "google-stt", + credentials: { googleServiceAccountKey: process.env.GOOGLE_SA_KEY }, + }, +}); +``` + +**Enterprise: `azure-stt`** + +```typescript +const transcript = await neurolink.stt({ + audio: { file: "./call-recording.wav" }, + speech: { + provider: "azure-stt", + credentials: { + azureApiKey: process.env.AZURE_SPEECH_KEY, + azureRegion: "eastus", + }, + }, +}); +``` + ### For Cost Optimization **Implement multi-provider routing** - Use free/cheap providers for simple tasks, premium for complex ones. @@ -894,3 +1005,5 @@ START: What's your primary constraint? - **[Troubleshooting](troubleshooting.md)** - Common issues and solutions - **[Multi-Provider Fallback Cookbook](../cookbook/multi-provider-fallback.md)** - Implementation patterns - **[Cost Optimization Cookbook](../cookbook/cost-optimization.md)** - Strategies to reduce costs +- **[Voice Providers Comparison](provider-comparison.md#voice-providers)** - TTS, STT, and Realtime provider matrix +- **[Voice Providers Index](../getting-started/providers/index.md#voice-providers)** - Voice provider setup cards diff --git a/examples/cli-examples.sh b/examples/cli-examples.sh index b9f2f03ab..b529a52fd 100644 --- a/examples/cli-examples.sh +++ b/examples/cli-examples.sh @@ -154,6 +154,48 @@ echo "Custom base URL for llama-server (if running on a different port)..." echo " Set LLAMACPP_BASE_URL=http://localhost:8080/v1 and run:" echo " npx @juspay/neurolink generate \"Hello\" --provider llamacpp" +echo "" +echo "==========================================" +echo "18. 🔊 TTS (Text-to-Speech) Examples" +echo "==========================================" +echo "Note: TTS flags convert the AI's text response to an audio file." +echo "" +echo "Google TTS (default, requires GOOGLE_AI_API_KEY or Vertex credentials)..." +npx @juspay/neurolink generate "Hello world" --provider vertex --tts --tts-voice en-US-Neural2-C --tts-output hello.mp3 + +echo "" +echo "OpenAI TTS (requires OPENAI_API_KEY)..." +npx @juspay/neurolink generate "Hello world" --provider vertex --tts --tts-provider openai-tts --tts-output hello.mp3 + +echo "" +echo "ElevenLabs TTS (requires ELEVENLABS_API_KEY)..." +npx @juspay/neurolink generate "Hello world" --provider vertex --tts --tts-provider elevenlabs --tts-output hello.mp3 + +echo "" +echo "Azure Speech TTS (requires AZURE_SPEECH_KEY and AZURE_SPEECH_REGION)..." +npx @juspay/neurolink generate "Hello world" --provider vertex --tts --tts-provider azure-speech --tts-output hello.mp3 + +echo "" +echo "==========================================" +echo "19. 🎙️ STT (Speech-to-Text) Examples" +echo "==========================================" +echo "Note: STT flags transcribe an audio file and feed the text to the AI." +echo "" +echo "Whisper STT (requires OPENAI_API_KEY)..." +npx @juspay/neurolink generate "Respond to audio" --provider vertex --stt --stt-provider whisper --input-audio recording.wav + +echo "" +echo "Deepgram STT (requires DEEPGRAM_API_KEY)..." +npx @juspay/neurolink generate "Respond to audio" --provider vertex --stt --stt-provider deepgram --input-audio recording.wav + +echo "" +echo "Google STT (requires GOOGLE_AI_API_KEY or Vertex credentials)..." +npx @juspay/neurolink generate "Respond to audio" --provider vertex --stt --stt-provider google-stt --input-audio recording.wav + +echo "" +echo "AssemblyAI STT (requires ASSEMBLYAI_API_KEY)..." +npx @juspay/neurolink generate "Respond to audio" --provider vertex --stt --stt-provider assemblyai --input-audio recording.wav + echo "" echo "✅ CLI Examples Complete!" echo "" @@ -169,5 +211,9 @@ echo "- Use --provider deepseek with DEEPSEEK_API_KEY for cost-efficient reasoni echo "- Use --provider nvidia-nim with NVIDIA_NIM_API_KEY for NVIDIA-hosted models" echo "- Use --provider lm-studio for local LM Studio inference (no API key needed)" echo "- Use --provider llamacpp for local llama-server inference (no API key needed)" +echo "- Use --tts / --stt flags for voice pipeline (TTS/STT providers)" +echo "- Set ELEVENLABS_API_KEY for ElevenLabs TTS" +echo "- Set DEEPGRAM_API_KEY for Deepgram STT" +echo "- Set AZURE_SPEECH_KEY + AZURE_SPEECH_REGION for Azure Speech TTS/STT" echo "- Built-in tools work in v1.7.1!" echo "- External server discovery working!" diff --git a/memory-bank/voice-bridge-implementation-plan.md b/memory-bank/voice-bridge-implementation-plan.md new file mode 100644 index 000000000..fc3594c0e --- /dev/null +++ b/memory-bank/voice-bridge-implementation-plan.md @@ -0,0 +1,453 @@ +# Voice Bridge Implementation Plan + +## Status: In Progress +## Created: 2026-04-16 +## Author: Sachin Sharma / Claude Code + +--- + +## 1. Goals & Non-Goals + +**Goals:** +- New TTS providers (OpenAI, ElevenLabs, Azure) work with the existing `generate({ tts: ... })` API +- New STT capability is added with the same shape as TTS (`generate({ stt: ... })`) +- Realtime voice (OpenAI Realtime, Gemini Live) works through a clean SDK API +- The voice server (`neurolink voice-server`) becomes provider-pluggable +- One registry per concern (TTS, STT, Realtime) — no parallel registries +- Full observability coverage matching `TTSProcessor` (spans, metrics, error categorization) +- Test coverage in `continuous-test-suite-*.ts` format + +**Non-Goals:** +- No new top-level SDK methods unless absolutely needed (consumers use `generate()`/`stream()`) +- No breaking changes to existing `TTSOptions`/`TTSResult`/voice-server APIs +- No browser-specific changes (we're SDK + Node CLI only) +- Don't expose `VoiceFactory`, `STTProcessor`, etc. as public API — internal only + +--- + +## 2. Project Conventions Checklist + +Every change must follow these (from CLAUDE.md): + +| Rule | Application | +|------|-------------| +| Dynamic imports in registry | All new provider registrations use `await import(...)` | +| Types in `src/lib/types/` | All new types go in `voice.ts`, `stt.ts`, or `realtime.ts` (under `types/`) | +| `type` not `interface` | New STT/Realtime contracts are `type` aliases | +| Barrel imports for internal types | All `import type {...} from "../types/index.js"` | +| `formatProviderError` returns | Error formatters return, never throw | +| `logger.shouldLog("debug")` guards | Wrap expensive serialization | +| Factory + Registry pattern | `STTProcessor`, `RealtimeProcessor` mirror `TTSProcessor` exactly | +| Observability spans | Every external call wrapped in a span with `SpanType.*` | +| Tests in `continuous-test-suite-*.ts` | tsx-based, no vitest | + +--- + +## 3. Architecture: The Bridge + +``` +┌─────────────────────────────────────────────────────────────────┐ +│ Consumer Surface (unchanged for TTS, additive for STT/Realtime) │ +├─────────────────────────────────────────────────────────────────┤ +│ neurolink.generate({ tts: {...}, stt: {...} }) │ +│ neurolink.stream({ tts: {...}, stt: {...} }) │ +│ neurolink.transcribe(audio, { provider, ... }) ← thin wrapper│ +│ neurolink.startRealtimeSession({ provider, ... })← thin wrapper│ +└──────────────────────────┬──────────────────────────────────────┘ + │ +┌──────────────────────────▼──────────────────────────────────────┐ +│ BaseProvider.generate() / .stream() │ +│ ├─ if (options.stt?.enabled) → STTProcessor.transcribe() │ +│ ├─ executeStandardGenerateFlow() (LLM call) │ +│ └─ if (options.tts?.enabled) → TTSProcessor.synthesize() │ +└──────────────────────────┬──────────────────────────────────────┘ + │ +┌─────────────┬────────────▼────────────┬──────────────────────┐ +│ TTSProcessor│ STTProcessor (NEW) │ RealtimeProcessor │ +│ (existing) │ mirror of TTSProcessor│ (rewired) │ +├─────────────┼─────────────────────────┼──────────────────────┤ +│ TTSHandler │ STTHandler │ RealtimeHandler │ +└──────┬──────┴────────────┬────────────┴──────────┬───────────┘ + │ │ │ +┌──────▼──────────┬────────▼────────┬──────────────▼───────────┐ +│ src/lib/ │ src/lib/ │ src/lib/ │ +│ adapters/tts/ │ adapters/stt/ │ adapters/realtime/ (NEW) │ +│ ───────────────│ ────────────── │ ─────────────────────────│ +│ google.ts │ google.ts │ openai.ts │ +│ openai.ts (NEW)│ whisper.ts │ gemini.ts │ +│ elevenlabs(NEW)│ deepgram.ts │ │ +│ azure.ts (NEW) │ azure.ts │ │ +│ cartesia.ts* │ assemblyai.ts │ │ +│ (* not handler;│ gladia.ts │ │ +│ streaming WS) │ │ │ +└────────────────┴─────────────────┴──────────────────────────┘ + ▲ ▲ ▲ + │ │ │ +┌──────┴──────────────────┴──────────────────────┴──────────────┐ +│ Voice Server (now pluggable) │ +│ STTProcessor.transcribe() TTSProcessor.synthesize() │ +│ Configurable via VOICE_STT_PROVIDER / VOICE_TTS_PROVIDER │ +└────────────────────────────────────────────────────────────────┘ +``` + +**Three things go away:** +- `VoiceFactory` and `VoiceRegistry` (parallel registry — replaced by Processor pattern) +- `CompositeVoice` and `VoiceAgent` (orchestrators — replaced by `generate({ stt, tts })`) +- All duplicated provider classes in `src/lib/voice/providers/` (canonical homes are `src/lib/adapters/{tts,stt,realtime}/`) + +--- + +## 4. Implementation Phases + +### Phase 0 — Type Consolidation (foundation) + +**Goal:** Single source of truth for voice types, eliminate duplicates. + +**Files:** + +| Action | File | Purpose | +|--------|------|---------| +| Keep | `src/lib/types/tts.ts` | TTS types (already canonical) | +| Move | `TTSHandler` from `types/common.ts` → `types/tts.ts` | Lives next to its data types | +| Create | `src/lib/types/stt.ts` | STT types: `STTOptions`, `STTResult`, `STTHandler`, `TranscriptionSegment`, `WordTiming`, `STTLanguage`, `STT_ERROR_CODES`, `STTQuality` | +| Create | `src/lib/types/realtime.ts` | Realtime types: `RealtimeConfig`, `RealtimeSession`, `RealtimeHandler`, `RealtimeAudioChunk`, `RealtimeMessage`, `RealtimeEventHandlers`, `REALTIME_ERROR_CODES`, `DEFAULT_REALTIME_CONFIG` | +| Delete | `src/lib/types/voice.ts` | All exports redistributed to `tts.ts`, `stt.ts`, `realtime.ts` | +| Update | `src/lib/types/index.ts` | Add `export * from "./stt.js"; export * from "./realtime.js";` | + +**Resolution of duplicates:** Pick one variant per type. For `DeepgramSTTOptions`/`VoiceDeepgramSTTOptions` etc., keep the richer adapter version (it's more thorough). Merge missing fields from voice/provider versions. + +**Naming:** All types globally unique per Rule 9. Adapter-specific options: `DeepgramSTTOptions`, `WhisperSTTOptions`, etc. (no `Voice*` prefix needed once duplicates removed). + +--- + +### Phase 1 — TTS Provider Bridging + +**Goal:** New TTS providers register with `TTSProcessor` and become available via `generate({ tts: ... })`. + +**Files:** + +| Action | File | Purpose | +|--------|------|---------| +| Move | `voice/providers/OpenAITTS.ts` → `adapters/tts/openaiTTSHandler.ts` | Match existing convention | +| Move | `voice/providers/ElevenLabsTTS.ts` → `adapters/tts/elevenLabsTTSHandler.ts` | Same | +| Move | `voice/providers/AzureTTS.ts` → `adapters/tts/azureTTSHandler.ts` | Same | +| Delete | `voice/providers/GoogleTTS.ts` | Duplicate of existing `googleTTSHandler.ts` | +| Refactor | All TTS handlers | Implement `TTSHandler` from `types/tts.ts` | +| Update | `factories/providerRegistry.ts` | Register all TTS handlers with TTSProcessor | + +**Provider names registered:** + +```typescript +TTSProcessor.registerHandler("google-ai", googleHandler); // existing +TTSProcessor.registerHandler("vertex", googleHandler); // existing +TTSProcessor.registerHandler("openai", openaiHandler); // NEW +TTSProcessor.registerHandler("elevenlabs", elevenlabsHandler); // NEW +TTSProcessor.registerHandler("azure", azureHandler); // NEW +``` + +**CLI updates (`commandFactory.ts`):** +- Add `--tts-provider` flag to override (e.g., `--provider vertex --tts-provider elevenlabs`) +- Update `TTSOptions` to add optional `provider?: string` field +- Update `BaseProvider.handleDirectTTSSynthesis` and `synthesizeAIResponseIfNeeded` to use `options.tts?.provider ?? options.provider` + +**Streaming TTS:** +- Add `TTSProcessor.synthesizeStream(text, provider, options)` that yields `TTSChunk` +- Stream merges TTS chunks into `StreamChunk { type: "audio", audioChunk: TTSChunk }` + +**Observability** (already covered by `TTSProcessor.synthesize`): +- Span: `tts.synthesize` with `SpanType.TTS` +- Attributes: `provider`, `text_length`, `voice`, `format`, `latency` + +--- + +### Phase 2 — STT Processor (mirror TTSProcessor) + +**Goal:** Create `STTProcessor` as the central STT orchestrator. + +**Files:** + +| Action | File | Purpose | +|--------|------|---------| +| Create | `src/lib/utils/sttProcessor.ts` | Mirror `ttsProcessor.ts`: static class, handler registry, observability | +| Merge | voice providers into adapter versions | Keep adapter versions, port missing features from voice providers | +| Delete | All `src/lib/voice/providers/*STT.ts` | After merge | +| Update | `factories/providerRegistry.ts` | Register STT handlers | +| Delete | `src/lib/voice/STTProvider.ts` | Replaced by `utils/sttProcessor.ts` | + +**`STTProcessor` API:** + +```typescript +export class STTProcessor { + private static readonly handlers = new Map(); + + static registerHandler(providerName: string, handler: STTHandler): void; + static supports(providerName: string): boolean; + static async transcribe(audio: Buffer | ArrayBuffer, provider: string, options: STTOptions): Promise; + static async *transcribeStream(audioStream: AsyncIterable, provider: string, options: STTOptions): AsyncIterable; +} +``` + +**Observability:** +- Add `SpanType.STT` to `spanTypes.ts` +- Span: `stt.transcribe` with attributes: `provider`, `audio_size_bytes`, `format`, `language`, `confidence`, `duration`, `latency` +- Metrics: `recordSTTTranscription(provider, latency, audioSize, success)` in `metricsAggregator.ts` + +**Integration into `BaseProvider`:** + +```typescript +// In BaseProvider.generate(): +if (options.stt?.enabled && options.input?.audio) { + const transcription = await STTProcessor.transcribe( + options.input.audio, + options.stt.provider ?? options.provider ?? this.providerName, + options.stt + ); + options.prompt = options.prompt ?? transcription.text; + enhancedResult.transcription = transcription; +} +``` + +**Type updates:** + +```typescript +type TextGenerationOptions = { + // existing... + input?: { text?: string; audio?: Buffer | ArrayBuffer }; + stt?: STTOptions & { provider?: string }; // NEW + tts?: TTSOptions & { provider?: string }; +}; + +type EnhancedGenerateResult = { + // existing... + audio?: TTSResult; + transcription?: STTResult; // NEW +}; +``` + +--- + +### Phase 3 — Realtime Processor (mirror pattern) + +**Goal:** `RealtimeProcessor` becomes the single registry for bidirectional providers. + +**Files:** + +| Action | File | Purpose | +|--------|------|---------| +| Move | `voice/RealtimeVoiceAPI.ts` → `utils/realtimeProcessor.ts` | Match location pattern | +| Create | `adapters/realtime/` directory | New canonical home | +| Move | `voice/providers/OpenAIRealtime.ts` → `adapters/realtime/openaiRealtimeHandler.ts` | | +| Move | `voice/providers/GeminiLive.ts` → `adapters/realtime/geminiLiveHandler.ts` | | +| Update | `factories/providerRegistry.ts` | Register realtime handlers | + +**SDK surface:** + +```typescript +async startRealtimeSession( + options: RealtimeConfig & { provider: string } +): Promise +``` + +**Observability:** +- Add `SpanType.REALTIME` to `spanTypes.ts` +- Spans: `realtime.connect`, `realtime.audio.in`, `realtime.audio.out`, `realtime.turn` + +--- + +### Phase 4 — SDK Surface Cleanup + +**Goal:** Minimize SDK API. Delete parallel infrastructure. + +**Files:** + +| Action | File | Purpose | +|--------|------|---------| +| Update | `neurolink.ts` | Replace voice methods with minimal set (synthesize, transcribe, startRealtimeSession) | +| Delete | `voice/voiceAgent.ts` | Replaced by `generate({ stt, tts })` | +| Delete | `voice/compositeVoice.ts` | Replaced by Processors | +| Delete | `voice/voiceFactory.ts` | Replaced by Processors | +| Delete | `voice/voiceRegistry.ts` | Replaced by Processors | +| Delete | `voice/index.ts` | No longer needed | +| Move | `voice/audio-utils.ts` → `utils/audioUtils.ts` | General utility | +| Move | `voice/stream-handler.ts` → `utils/audioStreamHandler.ts` | General utility | +| Move | `voice/errors.ts` → `utils/voiceErrors.ts` | Error utilities | +| Delete | `src/lib/voice/` directory | After all moves | + +**Final SDK voice surface (3 methods + options on existing):** + +```typescript +class NeuroLink { + // EXISTING (unchanged): + async generate(options): Promise + async stream(options): Promise + + // NEW thin wrappers: + async transcribe(audio, options?): Promise + async synthesize(text, options?): Promise + async startRealtimeSession(options): Promise +} +``` + +--- + +### Phase 5 — CLI Cleanup + +**Goal:** CLI follows SDK surface. Add `--stt*` flags symmetrically with `--tts*`. + +**Files:** + +| Action | File | Purpose | +|--------|------|---------| +| Delete | `cli/commands/voice.ts` | Use `generate` with flags instead | +| Update | `cli/parser.ts` | Remove `createVoiceCommands` import | +| Update | `cli/factories/commandFactory.ts` | Add `--stt*` flags | + +**New CLI flags:** + +``` +--stt Enable STT +--stt-provider STT provider (whisper, deepgram, google, azure, assemblyai, gladia) +--stt-language Audio language code +--stt-format Audio input format +--stt-diarization Enable speaker diarization +--stt-word-timestamps Enable word-level timestamps +--input-audio Path to audio file for STT input +--tts-provider TTS provider (overrides --provider) +``` + +--- + +### Phase 6 — Voice Server Pluggability + +**Goal:** Swap hardcoded Soniox/Cartesia for Processor calls. + +**Files:** + +| Action | File | Changes | +|--------|------|---------| +| Update | `server/voice/voiceWebSocketHandler.ts` | Replace Soniox WS with `STTProcessor.transcribeStream`; replace `CartesiaStream` with `TTSProcessor.synthesizeStream` | +| Refactor | `adapters/tts/cartesiaHandler.ts` | Register with `TTSProcessor` as `"cartesia"` | +| Create | `adapters/stt/sonioxSTTHandler.ts` | Wrap existing Soniox logic; register as `"soniox"` | +| Update | `server/voice/voiceServerApp.ts` | Read `VOICE_STT_PROVIDER` / `VOICE_TTS_PROVIDER` env vars | + +**Backwards compatibility:** Default to Soniox + Cartesia (current behavior). + +--- + +## 5. Observability Coverage + +### New Span Types + +```typescript +export enum SpanType { + // existing... + TTS = "tts", + STT = "stt", // NEW + REALTIME = "realtime", // NEW +} +``` + +### Span Coverage Matrix + +| Operation | Span Name | Type | Attributes | +|-----------|-----------|------|------------| +| TTS synthesis (existing) | `tts.synthesize` | TTS | provider, text_length, voice, format, latency | +| TTS streaming (new) | `tts.synthesizeStream` | TTS | provider, text_length, voice, format, chunks, total_latency | +| STT transcription | `stt.transcribe` | STT | provider, audio_size_bytes, format, language, latency, confidence, duration, word_count | +| STT streaming | `stt.transcribeStream` | STT | provider, format, language, segments, total_latency | +| Realtime connect | `realtime.connect` | REALTIME | provider, model, voice | +| Realtime audio in | `realtime.audio.in` | REALTIME | session_id, bytes | +| Realtime audio out | `realtime.audio.out` | REALTIME | session_id, bytes | +| Realtime turn | `realtime.turn` | REALTIME | session_id, duration_ms | + +### Metrics Aggregator Additions + +```typescript +recordSTTTranscription(provider, latency, audioSize, success): void +recordSTTStream(provider, latency, segments, success): void +recordTTSStream(provider, latency, chunks, totalSize, success): void +recordRealtimeSession(provider, durationMs, audioBytesIn, audioBytesOut): void +recordRealtimeTurn(provider, latencyMs): void +``` + +--- + +## 6. Test Suite Plan + +### Existing — Extend + +**`test/continuous-test-suite-tts.ts`:** +- Add tests for OpenAI, ElevenLabs, Azure TTS providers +- Add streaming TTS tests +- Add `--tts-provider` CLI flag tests +- Add cross-provider tests: provider=vertex, tts-provider=elevenlabs + +### New Suites + +**`test/continuous-test-suite-stt.ts`:** +- STTProcessor handler registration +- Per-provider transcription with sample WAV +- Format detection and validation +- Language code handling +- Word timestamps and diarization +- Streaming STT +- SDK integration: `generate({ input: { audio }, stt: { enabled: true } })` +- CLI flag tests: `--input-audio`, `--stt`, `--stt-provider` +- Error cases + +**`test/continuous-test-suite-realtime.ts`:** +- RealtimeProcessor handler registration +- Session lifecycle +- Event handler registration and emission +- SDK integration: `neurolink.startRealtimeSession()` +- Error cases + +**`test/continuous-test-suite-voice-server.ts`:** +- Voice server starts with default/custom providers +- WebSocket connection +- Health endpoint +- Frame bus pub/sub +- Turn manager state transitions + +### Replace + +**`test/continuous-test-suite-voice.ts`:** +- Becomes thin smoke test for all Processors +- Defers detailed coverage to specific suites + +--- + +## 7. PR Strategy + +| PR | Phase | Scope | Risk | +|----|-------|-------|------| +| #1 | Phase 0 | Type consolidation in `types/` | Low | +| #2 | Phase 1 | Register new TTS providers + `--tts-provider` flag | Low | +| #3 | Phase 2 | STTProcessor + STT integration into generate | Medium | +| #4 | Phase 3 | RealtimeProcessor + `startRealtimeSession` | Medium | +| #5 | Phase 4+5 | SDK + CLI cleanup, delete `src/lib/voice/` | Low | +| #6 | Phase 6 | Voice server pluggability | Low | + +--- + +## 8. Risks & Mitigations + +| Risk | Mitigation | +|------|------------| +| TTSProcessor registration changes break existing Google flow | Keep existing registration line untouched; new registrations only add | +| Different providers have different `TTSResult.format` defaults | Normalize in `TTSProcessor.synthesize` | +| Voice server regression | Default env values preserve current behavior | +| STT option fragmentation | Use `STTOptions` for common + `providerOptions?: unknown` for specific | +| Streaming TTS chunking inconsistency | All `synthesizeStream` yield `TTSChunk` with consistent shape | + +--- + +## 9. Documentation Updates + +| File | Update | +|------|--------| +| `docs/features/tts.md` | Add new providers + `--tts-provider` flag | +| `docs/features/stt.md` (NEW) | Providers, options, examples for SDK + CLI | +| `docs/features/realtime-voice.md` (NEW) | OpenAI Realtime + Gemini Live usage | +| `docs/features/voice-agent.md` | Replace with `generate({ stt, tts })` examples | +| `docs/real-time-speech-agents.md` | Mark proposal as implemented | diff --git a/memory-bank/voice-cleanup-plan.md b/memory-bank/voice-cleanup-plan.md new file mode 100644 index 000000000..c04142b45 --- /dev/null +++ b/memory-bank/voice-cleanup-plan.md @@ -0,0 +1,251 @@ +# Voice/Speech Integration Cleanup Plan + +**Branch:** `feat/voice-speech-integration` +**Date:** 2026-04-26 +**Status:** Approved, executing + +--- + +## Problem Statement + +The voice/speech integration was built without properly understanding the existing system. The result is: + +1. **Unnecessary SDK methods** (`synthesize()`, `transcribe()`, `startRealtimeVoice()`) that bypass NeuroLink's core pattern of `generate()` + `stream()` with JSON options +2. **Duplicate implementations** (two `STTProcessor` classes, two `GoogleTTS` implementations, two STT interface hierarchies) +3. **Dead code** (6 adapter/stt files implementing wrong interface, 13 zombie tracked files, zombie types) +4. **Missing production hardening** (no fetch timeouts on new providers) +5. **Stale documentation** (planned features shown as planned when they're implemented) +6. **Fake test suite** (tests standalone methods instead of consumer-facing generate/stream/CLI) + +## What Already Exists on `origin/release` + +| Component | Location | Status | +|-----------|----------|--------| +| `TTSProcessor` | `src/lib/utils/ttsProcessor.ts` | Shipped, static handler registry | +| `GoogleTTSHandler` | `src/lib/adapters/tts/googleTTSHandler.ts` | Shipped, Google Cloud TTS with gRPC SDK | +| `CartesiaHandler` | `src/lib/adapters/tts/cartesiaHandler.ts` | Shipped, WebSocket streaming for voice server | +| `generate({ tts })` | `src/lib/core/baseProvider.ts` | Shipped, Mode 1 (direct) + Mode 2 (AI response) | +| CLI `--tts*` flags | `src/cli/factories/commandFactory.ts` | Shipped | +| Voice server | `src/lib/server/voice/` + `src/cli/commands/voiceServer.ts` | Shipped, Cobra+Soniox+Cartesia real-time loop | +| Gemini Live audio | Google AI provider's stream path | Shipped | +| `TTSHandler` type | `src/lib/types/common.ts` | Shipped | +| TTS types | `src/lib/types/tts.ts` | Shipped | + +## What This Branch Should Deliver + +### New Capabilities + +1. **STT via `generate()` and `stream()`** — `generate({ stt: { enabled: true, audio: buffer, provider: "google-stt" } })` → transcribes audio, uses transcription as LLM prompt, returns `result.transcription` +2. **New TTS providers** — OpenAI TTS, ElevenLabs, Azure TTS — usable via `generate({ tts: { enabled: true, provider: "openai-tts" } })` +3. **New STT providers** — Whisper (OpenAI), Google STT, Deepgram, Azure STT +4. **Realtime providers** — OpenAI Realtime, Gemini Live handlers (registered for future use) +5. **CLI `--stt` flags** — `--stt --stt-provider --input-audio --stt-language` +6. **Audio utilities** — format detection, WAV creation, PCM manipulation, chunked streaming +7. **STTProcessor** — mirrors TTSProcessor, central STT orchestrator with observability + +### New Supporting Infrastructure + +- `src/lib/utils/sttProcessor.ts` — static handler registry (mirrors TTSProcessor) +- `src/lib/voice/providers/` — 4 STT + 3 TTS + 2 Realtime provider implementations +- `src/lib/voice/audio-utils.ts` — audio format utilities +- `src/lib/voice/stream-handler.ts` — chunked audio streaming +- `src/lib/voice/errors.ts` — VoiceError, STTError, RealtimeError +- `src/lib/voice/RealtimeVoiceAPI.ts` — RealtimeProcessor + BaseRealtimeHandler +- `src/lib/types/stt.ts`, `realtime.ts`, `voice.ts` — type system + +--- + +## Execution Plan + +### Phase 1: Delete Dead Code + +**Files to delete from disk:** + +| File | Why | +|------|-----| +| `src/lib/adapters/stt/assemblyaiSTTHandler.ts` | Implements `STTProvider` (wrong interface), never registered | +| `src/lib/adapters/stt/azureSTTHandler.ts` | Same | +| `src/lib/adapters/stt/deepgramSTTHandler.ts` | Same | +| `src/lib/adapters/stt/gladiaSTTHandler.ts` | Same | +| `src/lib/adapters/stt/googleSTTHandler.ts` | Same | +| `src/lib/adapters/stt/whisperSTTHandler.ts` | Same | +| `src/lib/voice/providers/GoogleTTS.ts` | Duplicates existing `adapters/tts/googleTTSHandler.ts` | +| `src/lib/voice/STTProvider.ts` | Duplicate STTProcessor with empty Map at runtime | + +**Files to `git rm` (tracked but deleted from disk — zombie files):** + +| File | Why | +|------|-----| +| `src/lib/voice/voiceFactory.ts` | Deleted earlier, still in git index | +| `src/lib/voice/voiceRegistry.ts` | Same | +| `src/lib/voice/compositeVoice.ts` | Same | +| `src/lib/voice/voiceAgent.ts` | Same | +| `src/cli/commands/voice.ts` | Same | +| `src/lib/types/ttsTypes.ts` | Renamed to `tts.ts`, zombie in index | +| `src/lib/voice/types/voiceTypes.ts` | Moved to `src/lib/types/`, zombie | +| `test/voice/RealtimeVoiceAPI.test.ts` | Deleted, zombie | +| `test/voice/STTProvider.test.ts` | Same | +| `test/voice/VoiceFactory.test.ts` | Same | +| `test/voice/VoiceRegistry.test.ts` | Same | +| `test/voice/audio-utils.test.ts` | Same | +| `test/voice/integration.test.ts` | Same | +| `test/voice/integration/voice.integration.test.ts` | Same | + +### Phase 2: Remove Unnecessary SDK Methods from `neurolink.ts` + +Remove these methods from the `NeuroLink` class (lines ~11980-12058): +- `synthesize(text, options?)` — `generate({ tts: { enabled: true } })` already does this +- `transcribe(audio, options?)` — `generate({ stt: { enabled: true, audio } })` already does this +- `startRealtimeVoice(config, provider?)` — realtime voice is CLI-only via `voice-server` + +Remove corresponding imports that become unused after method removal. + +**Keep intact:** +- STT preprocessing in `runStandardGenerateRequest()` (the `stt` option flow) +- `stt` forwarding in `buildGenerateTextOptions()` +- `transcription` in `finalizeGenerateRequestResult()` +- All TTS integration in `baseProvider.ts` (untouched) + +### Phase 3: Clean Types + +**`src/lib/types/voice.ts` — remove zombie types:** +- `CompositeVoiceConfig` — for deleted compositeVoice.ts +- `CompositeVoiceSession` — same +- `VoiceAgentConfig` — for deleted voiceAgent.ts +- `VoiceProcessingResult` — same +- `VoiceAgentEvent` — same +- `VoiceAgentEventData` — same +- `VoiceProviderEntry` — for deleted voiceRegistry.ts +- `VoiceProviderMetadata` — same +- `VoiceProvider` (the full interface with `getCapabilities`, `validateConfig`) — only used by dead adapter layer + +**`src/lib/types/stt.ts` — remove dead types:** +- `STTProvider` type — only implemented by deleted adapter files +- `AssemblyAISTTOptions` — only used by deleted assemblyaiSTTHandler +- `GladiaSTTOptions` — only used by deleted gladiaSTTHandler +- All AssemblyAI/Gladia response types (`AssemblyAITranscriptResponse`, `AssemblyAIUploadResponse`, `GladiaUploadResponse`, `GladiaTranscriptionResponse`, `GladiaResultResponse`, `GladiaTranscriptionResult`) +- Keep: `WhisperSTTOptions`, `DeepgramSTTOptions`, `AzureSTTOptions`, `GoogleSTTOptions` (used by live providers) +- Keep: `WhisperVerboseResponse`, `DeepgramResponse`, `AzureRecognitionResult`, `GoogleRecognizeResponse` etc. (used by live providers) + +**`src/lib/types/index.ts` — fix double-export:** +- Currently exports `tts.js`, `stt.js`, `realtime.js` directly AND via `voice.js` (which re-exports them) +- Remove the direct exports; keep only `export * from "./voice.js"` (which re-exports all three) + +### Phase 4: Clean `voice/index.ts` Barrel + +Remove exports for deleted files: +- Remove `STTProcessor` re-export from `./STTProvider.js` (deleted) +- Remove `GoogleTTS` export from `./providers/GoogleTTS.js` (deleted) +- Remove any references to voiceFactory, voiceRegistry, compositeVoice, voiceAgent +- Keep all other exports (live providers, audio-utils, stream-handler, errors, RealtimeProcessor) + +### Phase 5: Add Fetch Timeouts to Providers + +Add `AbortController` + 30-second timeout to all `fetch()` calls in: +- `src/lib/voice/providers/OpenAITTS.ts` +- `src/lib/voice/providers/ElevenLabsTTS.ts` +- `src/lib/voice/providers/AzureTTS.ts` +- `src/lib/voice/providers/OpenAISTT.ts` +- `src/lib/voice/providers/DeepgramSTT.ts` +- `src/lib/voice/providers/GoogleSTT.ts` +- `src/lib/voice/providers/AzureSTT.ts` + +Pattern: +```typescript +const controller = new AbortController(); +const timeoutId = setTimeout(() => controller.abort(), 30000); +try { + const response = await fetch(url, { ...options, signal: controller.signal }); + // ... +} finally { + clearTimeout(timeoutId); +} +``` + +### Phase 6: Rewrite Test Suite + +`test/continuous-test-suite-voice.ts` — test ONLY through consumer APIs (`generate()`, `stream()`, CLI): + +| # | Test | What It Proves | +|---|------|---------------| +| 1 | `generate({ tts: { enabled: true } })` → content + audio MP3 | Existing TTS pipeline works | +| 2 | `generate({ tts: { enabled: true, format: "wav" } })` → RIFF header | WAV format works | +| 3 | `generate({ tts: { enabled: true, provider: "elevenlabs" } })` → correct error | Unconfigured provider error | +| 4 | `generate({ stt: { enabled: true, audio: wav, provider: "google-stt" } })` → transcription + content | STT→LLM pipeline (core new feature) | +| 5 | `generate({ stt: { enabled: true, audio: empty } })` → correct error | Empty audio guard | +| 6 | `generate({ stt + tts both })` → transcription + content + audio | Full round-trip | +| 7 | `stream({ tts: { enabled: true } })` → chunks | Streaming + TTS | +| 8 | CLI `--tts --tts-output` → audio file on disk | CLI TTS | +| 9 | CLI `--stt --stt-provider --input-audio` → output | CLI STT | +| 10 | Handler registration (5 TTS + 5 STT + 2 Realtime) | providerRegistry wires all handlers | +| 11 | Audio utils (detectFormat, createWavHeader, splitIntoChunks guards) | Utilities work | +| 12 | ChunkedAudioStream (sampleRate=0 throws, etc.) | Stream handler safety | +| 13 | Barrel exports (STT_ERROR_CODES, SpanType.STT, AUDIO_FORMAT_DETAILS) | Types exported | +| 14 | `sdk.synthesize` does NOT exist, `sdk.transcribe` does NOT exist | Removed methods confirmed gone | +| 15 | Audio file save + read-back | File I/O round-trip | + +### Phase 7: Update Documentation + +**`docs/features/audio-input.md`:** +- Move `sdk.transcribe()` from "Planned" → clarify: STT is available via `generate({ stt: ... })` +- Move Whisper, Google STT from "Planned" → "Available" +- Add Deepgram, Azure STT to provider matrix as "Available" +- Add usage examples for `generate({ stt: ... })` and CLI `--stt` +- Update provider support matrix + +**`docs/features/tts.md`:** +- Add OpenAI TTS, ElevenLabs, Azure TTS to "Supported Providers" (no longer "Planned") +- Add `--tts-provider` CLI flag to docs +- Add provider-specific env var requirements + +--- + +## Parallel Execution Map + +``` +Agent 1: File Deletion + Git Cleanup + ├── Delete 8 files from disk + ├── git rm 14 zombie files + └── Touches: filesystem only, no source edits + +Agent 2: neurolink.ts Method Removal + ├── Remove synthesize(), transcribe(), startRealtimeVoice() + ├── Remove unused imports + └── Touches: src/lib/neurolink.ts ONLY + +Agent 3: Types Cleanup + ├── Clean voice.ts (remove zombie types) + ├── Clean stt.ts (remove dead STTProvider + adapter types) + ├── Fix index.ts double-export + └── Touches: src/lib/types/voice.ts, stt.ts, index.ts + +Agent 4: voice/index.ts Barrel Cleanup + ├── Remove exports for deleted files + ├── Remove STTProcessor from STTProvider + └── Touches: src/lib/voice/index.ts ONLY + +Agent 5: Provider Timeouts + ├── Add AbortController + 30s timeout to all fetch() calls + └── Touches: 7 provider files in src/lib/voice/providers/ + +Agent 6: Test Suite Rewrite + ├── Rewrite continuous-test-suite-voice.ts + ├── Test through generate()/stream()/CLI only + └── Touches: test/continuous-test-suite-voice.ts ONLY + +Agent 7: Documentation Updates + ├── Update audio-input.md + ├── Update tts.md + └── Touches: docs/ ONLY + +All 7 agents touch DIFFERENT files — full parallel execution. +``` + +## Post-Execution Verification + +After all agents complete: +1. `pnpm run build` — must succeed +2. `pnpm run lint` — must pass +3. `pnpm run check` — must pass +4. `npx tsx test/continuous-test-suite-voice.ts --provider=vertex` — all tests pass +5. Squash into single commit diff --git a/package.json b/package.json index ee5eff216..7be2f7c8b 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@juspay/neurolink", - "version": "9.60.1", + "version": "9.55.2", "packageManager": "pnpm@10.15.1", "description": "Universal AI Development Platform with working MCP integration, multi-provider support, and professional CLI. Built-in tools operational, 58+ external MCP servers discoverable. Connect to filesystem, GitHub, database operations, and more. Build, test, and deploy AI applications with 13 providers: OpenAI, Anthropic, Google AI, AWS Bedrock, Azure, Hugging Face, Ollama, and Mistral AI.", "author": { @@ -73,21 +73,15 @@ "test:mcp": "npx tsx test/continuous-test-suite-mcp-http.ts", "test:media": "npx tsx test/continuous-test-suite-media-gen.ts", "test:memory": "npx tsx test/continuous-test-suite-memory.ts", - "test:session-memory-bugs": "npx tsx test/continuous-test-suite-session-memory-bugs.ts", "test:middleware": "npx tsx test/continuous-test-suite-middleware.ts", "test:observability": "npx tsx test/continuous-test-suite-observability.ts", "test:ppt": "npx tsx test/continuous-test-suite-ppt.ts", "test:providers": "npx tsx test/continuous-test-suite-providers.ts", - "test:new-providers": "npx tsx test/continuous-test-suite-new-providers.ts", "test:rag": "npx tsx test/continuous-test-suite-rag.ts", "test:servers": "npx tsx test/continuous-test-suite-servers.ts", - "test:tool-reliability": "npx tsx test/continuous-test-suite-tool-reliability.ts", "test:tracing": "npx tsx test/continuous-test-suite-tracing.ts", "test:tts": "npx tsx test/continuous-test-suite-tts.ts", "test:credentials": "tsx test/continuous-test-suite-credentials.ts", - "test:dynamic": "npx tsx test/continuous-test-suite-dynamic.ts", - "test:proxy": "npx tsx test/continuous-test-suite-proxy.ts", - "test:bugfixes": "npx tsx test/continuous-test-suite-bugfixes.ts", "test:workflow": "npx tsx test/continuous-test-suite-workflow.ts", "test:ci": "pnpm run test && pnpm run test:client", "test:performance": "tsx tools/testing/performanceMonitor.ts", @@ -211,6 +205,7 @@ "@ai-sdk/provider": "^3.0.8", "@aws-sdk/client-bedrock": "^3.1000.0", "@aws-sdk/client-bedrock-runtime": "^3.1000.0", + "@aws-sdk/client-sagemaker": "^3.1000.0", "@aws-sdk/client-sagemaker-runtime": "^3.1000.0", "@google-cloud/text-to-speech": "^6.4.0", "@google-cloud/vertexai": "^1.10.0", @@ -230,22 +225,29 @@ "@opentelemetry/sdk-metrics": "^2.6.1", "@opentelemetry/sdk-trace-base": "^2.6.0", "@opentelemetry/semantic-conventions": "^1.40.0", + "@picovoice/cobra-node": "^3.0.2", "adm-zip": "^0.5.16", "ai": "^6.0.134", "chalk": "^5.6.2", "croner": "^9.1.0", "csv-parser": "^3.2.0", "dotenv": "^17.3.1", + "exceljs": "^4.4.0", + "fluent-ffmpeg": "^2.1.3", "google-auth-library": "^10.6.1", "hono": "^4.12.3", "inquirer": "^13.3.0", "jose": "^6.1.3", "json-schema-to-zod": "^2.7.0", + "mammoth": "^1.11.0", + "mediabunny": "^1.40.1", + "music-metadata": "^11.11.2", "nanoid": "^5.1.5", "ollama-ai-provider": "^1.2.0", "open": "^11.0.0", "ora": "^9.3.0", "p-limit": "^7.3.0", + "pptxgenjs": "^4.0.1", "redis": "^5.11.0", "tar-stream": "^3.1.8", "undici": ">=7.22.0", @@ -269,30 +271,22 @@ } }, "optionalDependencies": { - "@aws-sdk/client-sagemaker": "^3.1000.0", + "@langfuse/otel": "^5.0.1", + "bullmq": "^5.52.2", + "pdf-parse": "^2.4.5", + "pdf-to-img": "^5.0.0", "@fastify/cors": "^11.2.0", "@fastify/rate-limit": "^10.3.0", "@hono/node-server": "^1.19.9", "@koa/cors": "^5.0.0", "@koa/router": "^15.3.1", - "@langfuse/otel": "^5.0.1", - "@picovoice/cobra-node": "^3.0.2", - "bullmq": "^5.52.2", "cors": "^2.8.5", - "exceljs": "^4.4.0", "express": "^5.1.0", "express-rate-limit": "^8.2.1", "fastify": "^5.7.2", "ffmpeg-static": "^5.3.0", - "fluent-ffmpeg": "^2.1.3", "koa": "^3.1.1", "koa-bodyparser": "^4.4.1", - "mammoth": "^1.11.0", - "mediabunny": "^1.40.1", - "music-metadata": "^11.11.2", - "pdf-parse": "^2.4.5", - "pdf-to-img": "^5.0.0", - "pptxgenjs": "^4.0.1", "sharp": "^0.34.5" }, "devDependencies": { @@ -322,7 +316,6 @@ "@types/express": "^5.0.6", "@types/fluent-ffmpeg": "^2.1.28", "@types/inquirer": "^9.0.9", - "@types/js-yaml": "^4.0.9", "@types/koa": "^3.0.1", "@types/koa-bodyparser": "^4.3.13", "@types/koa__cors": "^5.0.1", @@ -340,7 +333,6 @@ "esbuild": "^0.27.4", "eslint": "^10.0.2", "husky": "^9.1.7", - "js-yaml": "^4.1.1", "lint-staged": "^16.3.0", "playwright": "^1.58.2", "prettier": "^3.8.1", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 4d1083c30..b643e55fe 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -68,6 +68,9 @@ importers: '@aws-sdk/client-bedrock-runtime': specifier: ^3.1000.0 version: 3.1019.0 + '@aws-sdk/client-sagemaker': + specifier: ^3.1000.0 + version: 3.1019.0 '@aws-sdk/client-sagemaker-runtime': specifier: ^3.1000.0 version: 3.1019.0 @@ -125,6 +128,9 @@ importers: '@opentelemetry/semantic-conventions': specifier: ^1.40.0 version: 1.40.0 + '@picovoice/cobra-node': + specifier: ^3.0.2 + version: 3.0.2 adm-zip: specifier: ^0.5.16 version: 0.5.16 @@ -143,6 +149,12 @@ importers: dotenv: specifier: ^17.3.1 version: 17.3.1 + exceljs: + specifier: ^4.4.0 + version: 4.4.0 + fluent-ffmpeg: + specifier: ^2.1.3 + version: 2.1.3 google-auth-library: specifier: ^10.6.1 version: 10.6.2 @@ -158,6 +170,15 @@ importers: json-schema-to-zod: specifier: ^2.7.0 version: 2.8.0 + mammoth: + specifier: ^1.11.0 + version: 1.12.0 + mediabunny: + specifier: ^1.40.1 + version: 1.40.1 + music-metadata: + specifier: ^11.11.2 + version: 11.12.3 nanoid: specifier: ^5.1.5 version: 5.1.7 @@ -173,6 +194,9 @@ importers: p-limit: specifier: ^7.3.0 version: 7.3.0 + pptxgenjs: + specifier: ^4.0.1 + version: 4.0.1 redis: specifier: ^5.11.0 version: 5.11.0 @@ -270,9 +294,6 @@ importers: '@types/inquirer': specifier: ^9.0.9 version: 9.0.9 - '@types/js-yaml': - specifier: ^4.0.9 - version: 4.0.9 '@types/koa': specifier: ^3.0.1 version: 3.0.2 @@ -324,9 +345,6 @@ importers: husky: specifier: ^9.1.7 version: 9.1.7 - js-yaml: - specifier: ^4.1.1 - version: 4.1.1 lint-staged: specifier: ^16.3.0 version: 16.4.0 @@ -388,9 +406,6 @@ importers: specifier: ^3.2.2 version: 3.2.2 optionalDependencies: - '@aws-sdk/client-sagemaker': - specifier: ^3.1000.0 - version: 3.1019.0 '@fastify/cors': specifier: ^11.2.0 version: 11.2.0 @@ -409,18 +424,12 @@ importers: '@langfuse/otel': specifier: ^5.0.1 version: 5.0.1(@opentelemetry/api@1.9.1)(@opentelemetry/core@2.6.1(@opentelemetry/api@1.9.1))(@opentelemetry/exporter-trace-otlp-http@0.214.0(@opentelemetry/api@1.9.1))(@opentelemetry/sdk-trace-base@2.6.1(@opentelemetry/api@1.9.1)) - '@picovoice/cobra-node': - specifier: ^3.0.2 - version: 3.0.2 bullmq: specifier: ^5.52.2 version: 5.71.1 cors: specifier: ^2.8.5 version: 2.8.6 - exceljs: - specifier: ^4.4.0 - version: 4.4.0 express: specifier: ^5.1.0 version: 5.2.1 @@ -433,33 +442,18 @@ importers: ffmpeg-static: specifier: ^5.3.0 version: 5.3.0 - fluent-ffmpeg: - specifier: ^2.1.3 - version: 2.1.3 koa: specifier: ^3.1.1 version: 3.2.0 koa-bodyparser: specifier: ^4.4.1 version: 4.4.1 - mammoth: - specifier: ^1.11.0 - version: 1.12.0 - mediabunny: - specifier: ^1.40.1 - version: 1.40.1 - music-metadata: - specifier: ^11.11.2 - version: 11.12.3 pdf-parse: specifier: ^2.4.5 version: 2.4.5 pdf-to-img: specifier: ^5.0.0 version: 5.0.0 - pptxgenjs: - specifier: ^4.0.1 - version: 4.0.1 sharp: specifier: ^0.34.5 version: 0.34.5 @@ -3476,9 +3470,6 @@ packages: '@types/inquirer@9.0.9': resolution: {integrity: sha512-/mWx5136gts2Z2e5izdoRCo46lPp5TMs9R15GTSsgg/XnZyxDWVqoVU3R9lWnccKpqwsJLvRoxbCjoJtZB7DSw==} - '@types/js-yaml@4.0.9': - resolution: {integrity: sha512-k4MGaQl5TGo/iipqb2UDG2UwjXziSWkh0uysQelTlJpX1qGlpUZYm8PnO4DxG1qBomtJUdYJ6qR6xdIah10JLg==} - '@types/json-schema@7.0.15': resolution: {integrity: sha512-5+fP8P8MFNC+AyZCDxrB2pkZFPGzqQWUzpSeuuVLvm8VMcorNYavBqoFcxK8bQz4Qsbn4oUEEem4wDLfcysGHA==} @@ -8012,7 +8003,6 @@ snapshots: tslib: 2.8.1 transitivePeerDependencies: - aws-crt - optional: true '@aws-sdk/client-sagemaker@3.1029.0': dependencies: @@ -10798,8 +10788,7 @@ snapshots: '@oxc-project/types@0.122.0': {} - '@picovoice/cobra-node@3.0.2': - optional: true + '@picovoice/cobra-node@3.0.2': {} '@pinojs/redact@0.4.0': optional: true @@ -11786,10 +11775,8 @@ snapshots: '@types/dom-mediacapture-transform@0.1.11': dependencies: '@types/dom-webcodecs': 0.1.13 - optional: true - '@types/dom-webcodecs@0.1.13': - optional: true + '@types/dom-webcodecs@0.1.13': {} '@types/esrecurse@4.3.1': {} @@ -11825,8 +11812,6 @@ snapshots: '@types/through': 0.0.33 rxjs: 7.8.2 - '@types/js-yaml@4.0.9': {} - '@types/json-schema@7.0.15': {} '@types/keygrip@1.0.6': {} @@ -14202,7 +14187,6 @@ snapshots: dependencies: '@types/dom-mediacapture-transform': 0.1.11 '@types/dom-webcodecs': 0.1.13 - optional: true meow@13.2.0: {} diff --git a/src/cli/factories/commandFactory.ts b/src/cli/factories/commandFactory.ts index 7eb5c559f..545d3ad1f 100644 --- a/src/cli/factories/commandFactory.ts +++ b/src/cli/factories/commandFactory.ts @@ -320,9 +320,25 @@ export class CLICommandFactory { type: "string" as const, description: "TTS voice to use (e.g., 'en-US-Neural2-C')", }, + ttsProvider: { + type: "string" as const, + choices: ["google-ai", "vertex", "openai-tts", "elevenlabs", "azure-tts"], + description: "TTS provider (overrides --provider for speech synthesis)", + }, ttsFormat: { type: "string" as const, - choices: ["mp3", "wav", "ogg", "opus"], + choices: [ + "mp3", + "wav", + "ogg", + "opus", + "m4a", + "flac", + "webm", + "mp4", + "mpeg", + "mpga", + ], default: "mp3", description: "Audio output format", }, @@ -348,6 +364,26 @@ export class CLICommandFactory { description: "Auto-play generated audio", }, + // STT (Speech-to-Text) options + stt: { + type: "boolean" as const, + default: false, + description: "Enable speech-to-text transcription of input audio", + }, + sttProvider: { + type: "string" as const, + choices: ["whisper", "deepgram", "google-stt", "azure-stt"], + description: "STT provider to use", + }, + sttLanguage: { + type: "string" as const, + description: "Audio language code for STT (e.g., en-US)", + }, + inputAudio: { + type: "string" as const, + description: "Path to audio file for STT transcription", + }, + // Video Generation options (Veo 3.1) outputMode: { type: "string" as const, @@ -708,11 +744,19 @@ export class CLICommandFactory { // TTS options tts: argv.tts as boolean | undefined, ttsVoice: argv.ttsVoice as string | undefined, - ttsFormat: argv.ttsFormat as "mp3" | "wav" | "ogg" | "opus" | undefined, + ttsProvider: argv.ttsProvider as string | undefined, + ttsFormat: argv.ttsFormat as + | import("../../lib/types/index.js").AudioFormat + | undefined, ttsSpeed: argv.ttsSpeed as number | undefined, ttsQuality: argv.ttsQuality as "standard" | "hd" | undefined, ttsOutput: argv.ttsOutput as string | undefined, ttsPlay: argv.ttsPlay as boolean | undefined, + // STT options + stt: argv.stt as boolean | undefined, + sttProvider: argv.sttProvider as string | undefined, + sttLanguage: argv.sttLanguage as string | undefined, + inputAudio: argv.inputAudio as string | undefined, // Video generation options (Veo 3.1) outputMode: argv.outputMode as "text" | "video" | "ppt" | undefined, videoOutput: argv.videoOutput as string | undefined, @@ -2614,6 +2658,12 @@ export class CLICommandFactory { enhancedOptions, ); + // Read audio file for STT if --input-audio is provided + const inputAudioPath = enhancedOptions.inputAudio as string | undefined; + const inputAudioBuffer = inputAudioPath + ? fs.readFileSync(inputAudioPath) + : undefined; + const runGenerate = () => sdk.generate({ input: generateInput, @@ -2682,6 +2732,7 @@ export class CLICommandFactory { enabled: true, useAiResponse: true, voice: enhancedOptions.ttsVoice as string | undefined, + provider: enhancedOptions.ttsProvider as string | undefined, format: (enhancedOptions.ttsFormat as | "mp3" @@ -2697,6 +2748,15 @@ export class CLICommandFactory { play: enhancedOptions.ttsPlay as boolean | undefined, } : undefined, + // STT configuration + stt: enhancedOptions.stt + ? { + enabled: true, + provider: enhancedOptions.sttProvider as string | undefined, + language: enhancedOptions.sttLanguage as string | undefined, + ...(inputAudioBuffer && { audio: inputAudioBuffer }), + } + : undefined, }); const result = await runGenerate(); @@ -2962,6 +3022,7 @@ export class CLICommandFactory { enabled: true, useAiResponse: true, voice: enhancedOptions.ttsVoice as string | undefined, + provider: enhancedOptions.ttsProvider as string | undefined, format: (enhancedOptions.ttsFormat as "mp3" | "wav" | "ogg" | "opus") || undefined, @@ -2974,6 +3035,24 @@ export class CLICommandFactory { play: enhancedOptions.ttsPlay as boolean | undefined, } : undefined, + // STT configuration + stt: enhancedOptions.stt + ? (() => { + const streamSttAudioPath = enhancedOptions.inputAudio as + | string + | undefined; + const streamSttAudio = + streamSttAudioPath && fs.existsSync(streamSttAudioPath) + ? fs.readFileSync(streamSttAudioPath) + : undefined; + return { + enabled: true as const, + provider: enhancedOptions.sttProvider as string | undefined, + language: enhancedOptions.sttLanguage as string | undefined, + ...(streamSttAudio && { audio: streamSttAudio }), + }; + })() + : undefined, }); const stream = await runStream(); diff --git a/src/cli/factories/sagemakerCommandFactory.ts b/src/cli/factories/sagemakerCommandFactory.ts index 790309fd8..dd44f3362 100644 --- a/src/cli/factories/sagemakerCommandFactory.ts +++ b/src/cli/factories/sagemakerCommandFactory.ts @@ -2,25 +2,11 @@ import type { Argv, CommandModule } from "yargs"; import chalk from "chalk"; import ora from "ora"; import inquirer from "inquirer"; -import type { EndpointSummary } from "@aws-sdk/client-sagemaker"; - -async function loadSageMakerControl() { - try { - return await import(/* @vite-ignore */ "@aws-sdk/client-sagemaker"); - } catch (err) { - const e = err instanceof Error ? (err as NodeJS.ErrnoException) : null; - if ( - e?.code === "ERR_MODULE_NOT_FOUND" && - e.message.includes("client-sagemaker") - ) { - throw new Error( - 'SageMaker setup requires "@aws-sdk/client-sagemaker". Install it with:\n pnpm add @aws-sdk/client-sagemaker', - { cause: err }, - ); - } - throw err; - } -} +import { + SageMakerClient, + ListEndpointsCommand, + type EndpointSummary, +} from "@aws-sdk/client-sagemaker"; import type { UnknownRecord, DoGenerateModel, @@ -250,11 +236,10 @@ export class SageMakerCommandFactory { /** * Validate secure configuration without exposing credentials */ - private static async validateSecureConfiguration( + private static validateSecureConfiguration( secureConfig: SecureConfiguration, - ): Promise { + ): void { // Create temporary AWS SDK client with secure credentials - const { SageMakerClient } = await loadSageMakerControl(); const tempClient = new SageMakerClient({ region: secureConfig.region, credentials: { @@ -440,8 +425,6 @@ export class SageMakerCommandFactory { // Use AWS SDK directly for better security and error handling try { const config = await getSageMakerConfig(); - const { SageMakerClient, ListEndpointsCommand } = - await loadSageMakerControl(); const sagemakerClient = new SageMakerClient({ region: config.region, credentials: { @@ -715,7 +698,7 @@ export class SageMakerCommandFactory { clearConfigurationCache(); try { - await this.validateSecureConfiguration(secureConfig); // Validate configuration is loadable + this.validateSecureConfiguration(secureConfig); // Validate configuration is loadable spinner.succeed("✅ Configuration validated successfully"); logger.always(chalk.green("\n🎉 SageMaker setup complete!")); diff --git a/src/cli/loop/optionsSchema.ts b/src/cli/loop/optionsSchema.ts index ef6b9c0fd..a7c531b23 100644 --- a/src/cli/loop/optionsSchema.ts +++ b/src/cli/loop/optionsSchema.ts @@ -27,6 +27,7 @@ export const textGenerationOptionsSchema: Record< | "region" | "csvOptions" | "tts" + | "stt" // Complex object, set via --stt* flags | "thinkingConfig" // Complex object, use thinking/thinkingBudget instead | "requestId" // Observability ID, not CLI-settable | "fileRegistry" // Internal: set by SDK, not by CLI diff --git a/src/lib/core/baseProvider.ts b/src/lib/core/baseProvider.ts index 000e501d3..8e7b342c8 100644 --- a/src/lib/core/baseProvider.ts +++ b/src/lib/core/baseProvider.ts @@ -398,6 +398,7 @@ export abstract class BaseProvider implements AIProvider { excludeTools: options.excludeTools, skipToolPromptInjection: options.skipToolPromptInjection, timeout: options.timeout, + stt: options.stt, }; logger.debug(`Calling generate for fake streaming`, { @@ -877,7 +878,7 @@ export abstract class BaseProvider implements AIProvider { } baseResult.audio = await TTSProcessor.synthesize( textToSynthesize, - options.provider ?? this.providerName, + options.tts.provider ?? options.provider ?? this.providerName, options.tts, ); } catch (ttsError) { @@ -1050,7 +1051,12 @@ export abstract class BaseProvider implements AIProvider { options, ); - return this.enhanceResult(enhancedResult, options, startTime); + const finalResult = await this.enhanceResult( + enhancedResult, + options, + startTime, + ); + return finalResult; } private async synthesizeAIResponseIfNeeded( @@ -1062,13 +1068,14 @@ export abstract class BaseProvider implements AIProvider { } const aiResponse = enhancedResult.content; - const provider = options.provider ?? this.providerName; - if (!aiResponse || !provider) { + const ttsProvider = + options.tts?.provider ?? options.provider ?? this.providerName; + if (!aiResponse || !ttsProvider) { logger.warn(`TTS synthesis skipped despite being enabled`, { provider: this.providerName, hasAiResponse: !!aiResponse, aiResponseLength: aiResponse?.length ?? 0, - hasProvider: !!provider, + hasProvider: !!ttsProvider, ttsConfig: { enabled: options.tts?.enabled, useAiResponse: options.tts?.useAiResponse, @@ -1083,7 +1090,7 @@ export abstract class BaseProvider implements AIProvider { try { const ttsResult = await TTSProcessor.synthesize( aiResponse, - provider, + ttsProvider, options.tts, ); return { diff --git a/src/lib/factories/providerRegistry.ts b/src/lib/factories/providerRegistry.ts index e3c8606c8..81a8a4197 100644 --- a/src/lib/factories/providerRegistry.ts +++ b/src/lib/factories/providerRegistry.ts @@ -498,6 +498,121 @@ export class ProviderRegistry { ); // Don't throw - TTS is optional functionality } + + // New TTS providers + try { + const { TTSProcessor } = await import("../utils/ttsProcessor.js"); + const { OpenAITTS } = await import("../voice/providers/OpenAITTS.js"); + TTSProcessor.registerHandler("openai-tts", new OpenAITTS()); + } catch { + /* Optional provider */ + } + + try { + const { TTSProcessor } = await import("../utils/ttsProcessor.js"); + const { ElevenLabsTTS } = + await import("../voice/providers/ElevenLabsTTS.js"); + const elevenLabsHandler = new ElevenLabsTTS(); + TTSProcessor.registerHandler("elevenlabs", elevenLabsHandler); + TTSProcessor.registerHandler("elevenlabs-tts", elevenLabsHandler); + } catch { + /* Optional provider */ + } + + try { + const { TTSProcessor } = await import("../utils/ttsProcessor.js"); + const { AzureTTS } = await import("../voice/providers/AzureTTS.js"); + TTSProcessor.registerHandler("azure-tts", new AzureTTS()); + } catch { + /* Optional provider */ + } + + // ===== STT HANDLER REGISTRATION ===== + try { + const { STTProcessor } = await import("../utils/sttProcessor.js"); + + try { + const { OpenAISTT } = await import("../voice/providers/OpenAISTT.js"); + const openAISTT = new OpenAISTT(); + STTProcessor.registerHandler("whisper", openAISTT); + STTProcessor.registerHandler("openai-stt", openAISTT); + } catch { + /* Optional provider - skip if dependency missing */ + } + + try { + const { DeepgramSTT } = + await import("../voice/providers/DeepgramSTT.js"); + STTProcessor.registerHandler("deepgram", new DeepgramSTT()); + } catch { + /* Optional provider - skip if dependency missing */ + } + + try { + const { GoogleSTT } = await import("../voice/providers/GoogleSTT.js"); + STTProcessor.registerHandler("google-stt", new GoogleSTT()); + } catch { + /* Optional provider - skip if dependency missing */ + } + + try { + const { AzureSTT } = await import("../voice/providers/AzureSTT.js"); + STTProcessor.registerHandler("azure-stt", new AzureSTT()); + } catch { + /* Optional provider - skip if dependency missing */ + } + + logger.debug("STT handlers registered successfully", { + providers: ["whisper", "deepgram", "google-stt", "azure-stt"], + }); + } catch (sttError) { + logger.warn( + "Failed to register STT handlers - STT functionality will be unavailable", + { + error: + sttError instanceof Error ? sttError.message : String(sttError), + }, + ); + } + + // ===== REALTIME HANDLER REGISTRATION ===== + try { + const { RealtimeProcessor } = + await import("../voice/RealtimeVoiceAPI.js"); + + try { + const { OpenAIRealtime } = + await import("../voice/providers/OpenAIRealtime.js"); + RealtimeProcessor.registerHandler( + "openai-realtime", + new OpenAIRealtime(), + ); + } catch { + // Optional provider — skip if unavailable + } + + try { + const { GeminiLive } = + await import("../voice/providers/GeminiLive.js"); + RealtimeProcessor.registerHandler("gemini-live", new GeminiLive()); + } catch { + // Optional provider — skip if unavailable + } + + logger.debug("Realtime handlers registered successfully", { + providers: ["openai-realtime", "gemini-live"], + }); + } catch (realtimeError) { + logger.warn( + "Failed to register Realtime handlers - Realtime functionality will be unavailable", + { + error: + realtimeError instanceof Error + ? realtimeError.message + : String(realtimeError), + }, + ); + } } catch (error) { logger.error("Failed to register providers:", error); throw error; diff --git a/src/lib/neurolink.ts b/src/lib/neurolink.ts index 1f535af01..a6d53fdc4 100644 --- a/src/lib/neurolink.ts +++ b/src/lib/neurolink.ts @@ -170,7 +170,8 @@ import { TaskManager } from "./tasks/taskManager.js"; import { createTaskTools } from "./tasks/tools/taskTools.js"; import { ATTR } from "./telemetry/attributes.js"; import { tracers } from "./telemetry/tracers.js"; -// NEW: Generate function imports +// Voice integration imports +import type { STTResult } from "./types/index.js"; import { getConversationMessages, storeConversationTurn, @@ -224,6 +225,7 @@ import { import { isNonNullObject } from "./utils/typeUtils.js"; import { getWorkflow } from "./workflow/core/workflowRegistry.js"; import { runWorkflow } from "./workflow/core/workflowRunner.js"; +// (voice imports removed — use TTSProcessor/STTProcessor/RealtimeProcessor instead) /** * NL-002: Classify MCP error messages into categories for AI disambiguation. @@ -3892,10 +3894,18 @@ Current user's request: ${currentInput}`; !!(options.tools && Object.keys(options.tools).length > 0), ); - this.assertInputText( - options.input?.text, - "Input text is required and must be a non-empty string", - ); + // When STT audio is provided, ensure options.input exists (the transcription + // will supply the text inside runStandardGenerateRequest) and skip text validation. + const hasSttAudio = !!(options.stt?.enabled && options.stt?.audio); + if (hasSttAudio && !options.input) { + options.input = { text: "" }; + } + if (!hasSttAudio) { + this.assertInputText( + options.input?.text, + "Input text is required and must be a non-empty string", + ); + } this.enforceSessionBudget(options.maxBudgetUsd); this.applyGenerateLifecycleMiddleware(options); await this.applyAuthenticatedRequestContext(options); @@ -3971,9 +3981,53 @@ Current user's request: ${currentInput}`; originalPrompt, factoryResult, ); + // STT preprocessing: transcribe audio input before LLM generation + let sttTranscription: STTResult | undefined; + if (options.stt?.enabled && options.stt.audio) { + try { + // Ensure STT handlers are registered + if (!ProviderRegistry.isRegistered()) { + await ProviderRegistry.registerAllProviders(); + } + const { STTProcessor } = await import("./utils/sttProcessor.js"); + const sttProvider = + options.stt.provider ?? (options.provider as string) ?? "whisper"; + sttTranscription = await STTProcessor.transcribe( + options.stt.audio, + sttProvider, + options.stt, + ); + // Inject transcription into the LLM prompt + if (sttTranscription.text) { + const existingText = + textOptions.prompt || textOptions.input?.text || ""; + if (!existingText) { + // No user text — use transcription directly as the prompt + textOptions.prompt = sttTranscription.text; + if (textOptions.input) { + textOptions.input.text = sttTranscription.text; + } + } else { + // User provided text — prepend transcription as context + const combined = `[Transcribed audio]: ${sttTranscription.text}\n\n${existingText}`; + if (textOptions.prompt) { + textOptions.prompt = combined; + } + if (textOptions.input?.text) { + textOptions.input.text = combined; + } + } + } + } catch (sttError) { + logger.error("[NeuroLink] STT transcription failed:", sttError); + // Don't block generation — STT is optional + } + } + const textResult = await this.generateTextInternal(textOptions); - return this.finalizeGenerateRequestResult({ + // Attach STT transcription to result + const generateResult = this.finalizeGenerateRequestResult({ generateSpan, options, textOptions, @@ -3982,6 +4036,10 @@ Current user's request: ${currentInput}`; originalPrompt, startTime, }); + if (sttTranscription) { + generateResult.transcription = sttTranscription; + } + return generateResult; } private async maybeApplyGenerateOrchestration( @@ -4101,6 +4159,7 @@ Current user's request: ${currentInput}`; input: options.input, region: options.region, tts: options.tts, + stt: options.stt, fileRegistry: this.fileRegistry, timeout: options.timeout, abortSignal: options.abortSignal, @@ -4224,6 +4283,7 @@ Current user's request: ${currentInput}`; } : undefined, audio: textResult.audio, + transcription: textResult.transcription, video: textResult.video, ppt: textResult.ppt, ...(textResult.retries && { retries: textResult.retries }), diff --git a/src/lib/observability/exporters/laminarExporter.ts b/src/lib/observability/exporters/laminarExporter.ts index b2fec9b70..a724d1790 100644 --- a/src/lib/observability/exporters/laminarExporter.ts +++ b/src/lib/observability/exporters/laminarExporter.ts @@ -291,6 +291,7 @@ export class LaminarExporter extends BaseExporter { [SpanType.PPT_GENERATION]: "custom", [SpanType.WORKFLOW]: "workflow", [SpanType.TTS]: "custom", + [SpanType.STT]: "custom", [SpanType.SERVER_REQUEST]: "custom", [SpanType.CUSTOM]: "custom", }; diff --git a/src/lib/observability/exporters/posthogExporter.ts b/src/lib/observability/exporters/posthogExporter.ts index e1a9c6edc..c9a20c154 100644 --- a/src/lib/observability/exporters/posthogExporter.ts +++ b/src/lib/observability/exporters/posthogExporter.ts @@ -273,6 +273,7 @@ export class PostHogExporter extends BaseExporter { [SpanType.PPT_GENERATION]: "ai_ppt_generation", [SpanType.WORKFLOW]: "ai_workflow", [SpanType.TTS]: "ai_tts_synthesis", + [SpanType.STT]: "ai_stt_transcription", [SpanType.SERVER_REQUEST]: "ai_server_request", [SpanType.CUSTOM]: "ai_custom_span", }; diff --git a/src/lib/observability/utils/spanSerializer.ts b/src/lib/observability/utils/spanSerializer.ts index 2ea6f5111..500093e9a 100644 --- a/src/lib/observability/utils/spanSerializer.ts +++ b/src/lib/observability/utils/spanSerializer.ts @@ -287,6 +287,7 @@ export class SpanSerializer { [SpanType.PPT_GENERATION]: "chain", [SpanType.WORKFLOW]: "chain", [SpanType.TTS]: "chain", + [SpanType.STT]: "chain", [SpanType.SERVER_REQUEST]: "chain", [SpanType.CUSTOM]: "chain", }; diff --git a/src/lib/server/voice/voiceWebSocketHandler.ts b/src/lib/server/voice/voiceWebSocketHandler.ts index ac1fd0494..74d9256fd 100644 --- a/src/lib/server/voice/voiceWebSocketHandler.ts +++ b/src/lib/server/voice/voiceWebSocketHandler.ts @@ -1,4 +1,5 @@ import WebSocket, { WebSocketServer } from "ws"; +import { Cobra } from "@picovoice/cobra-node"; import type { Server as HttpServer } from "http"; import { FrameBus } from "./frameBus.js"; import { TurnManager, TurnState } from "./turnManager.js"; @@ -8,7 +9,6 @@ import { logger } from "../../utils/logger.js"; import { withTimeout } from "../../utils/async/withTimeout.js"; import type { ClientControlMessage, - CobraInstance, ConversationMessage, Message, SonioxMessage, @@ -151,539 +151,503 @@ export function setupWebSocket(server: HttpServer) { const neurolink = new NeuroLink(); wss.on("connection", (clientWs) => { - void (async () => { - logger.info("[WS] Client connected"); + logger.info("[WS] Client connected"); + + // --- Per-session Cobra instance --- + let cobra: Cobra | null = null; + let FRAME_LENGTH = 512; + let FRAME_BYTES = FRAME_LENGTH * 2; + try { + cobra = new Cobra(accessKey); + FRAME_LENGTH = cobra.frameLength; + FRAME_BYTES = FRAME_LENGTH * 2; + logger.info(`[VAD] Cobra ready (frameLength=${FRAME_LENGTH})`); + } catch (err) { + logger.error("[VAD] Cobra init failed:", err); + clientWs.close(); + return; + } + + // --- Per-session state --- + const bus = new FrameBus(); + const turnManager = new TurnManager(bus); + + let sonioxWs: WebSocket | null = null; + let keepAliveTimer: NodeJS.Timeout | null = null; + + let sessionClosed = false; + let transcriptBuffer = ""; + let activeTTS: CartesiaStream | null = null; + const conversation: ConversationMessage[] = []; + let currentTurnId = 0; + let activePipelineTurnId: number | null = null; + // Safety fallback: if the client never sends playback_done (crash, network drop), + // auto-reset the turn state after this many ms so the assistant isn't stuck. + let playbackResetTimer: NodeJS.Timeout | null = null; + // Timestamp (ms) before which barge-in via Soniox is suppressed. + // Set when TTS starts playing to prevent TTS echo from triggering immediate re-interrupt. + // AEC on the browser needs ~300-400ms to characterise the echo signal before suppressing it. + let bargeInLockedUntil = 0; + + // Cobra VAD state + let isSpeaking = false; + let silenceFrameCount = 0; + let voiceFrameCount = 0; + let frameRemainder = Buffer.alloc(0); + + /* ======= INTERRUPT ======= */ + + function closeTts(stream: CartesiaStream | null, reason: string) { + if (!stream) { + return; + } - // --- Per-session Cobra instance --- - let cobra: CobraInstance | null = null; - let FRAME_LENGTH = 512; - let FRAME_BYTES = FRAME_LENGTH * 2; try { - let mod: { Cobra: new (key: string) => CobraInstance }; - try { - mod = (await import( - /* @vite-ignore */ "@picovoice/cobra-node" - )) as typeof mod; - } catch (err) { - const e = - err instanceof Error ? (err as NodeJS.ErrnoException) : null; - if ( - e?.code === "ERR_MODULE_NOT_FOUND" && - e.message.includes("cobra-node") - ) { - throw new Error( - 'Voice activity detection requires "@picovoice/cobra-node". Install it with:\n pnpm add @picovoice/cobra-node', - { cause: err }, - ); - } - throw err; - } - cobra = new mod.Cobra(accessKey); - FRAME_LENGTH = cobra.frameLength; - FRAME_BYTES = FRAME_LENGTH * 2; - logger.info(`[VAD] Cobra ready (frameLength=${FRAME_LENGTH})`); - } catch (err) { - logger.error("[VAD] Cobra init failed:", err); - clientWs.close(); - return; + // Close the WS first so that any pending done/error/close listeners + // in processTurn() can settle immediately, rather than hanging until + // the withTimeout fires. + stream.close(); + stream.removeAllListeners(); + } catch (error) { + logger.warn(reason, error); } + } - // --- Per-session state --- - const bus = new FrameBus(); - const turnManager = new TurnManager(bus); - - let sonioxWs: WebSocket | null = null; - let keepAliveTimer: NodeJS.Timeout | null = null; - - let sessionClosed = false; - let transcriptBuffer = ""; - let activeTTS: CartesiaStream | null = null; - const conversation: ConversationMessage[] = []; - let currentTurnId = 0; - let activePipelineTurnId: number | null = null; - // Safety fallback: if the client never sends playback_done (crash, network drop), - // auto-reset the turn state after this many ms so the assistant isn't stuck. - let playbackResetTimer: NodeJS.Timeout | null = null; - // Timestamp (ms) before which barge-in via Soniox is suppressed. - // Set when TTS starts playing to prevent TTS echo from triggering immediate re-interrupt. - // AEC on the browser needs ~300-400ms to characterise the echo signal before suppressing it. - let bargeInLockedUntil = 0; - - // Cobra VAD state - let isSpeaking = false; - let silenceFrameCount = 0; - let voiceFrameCount = 0; - let frameRemainder = Buffer.alloc(0); - - /* ======= INTERRUPT ======= */ - - function closeTts(stream: CartesiaStream | null, reason: string) { - if (!stream) { - return; + function doInterrupt() { + logger.info("[INTERRUPT] Cutting TTS"); + if (playbackResetTimer) { + clearTimeout(playbackResetTimer); + playbackResetTimer = null; + } + bargeInLockedUntil = 0; + currentTurnId++; + activePipelineTurnId = null; + transcriptBuffer = ""; + isSpeaking = false; + silenceFrameCount = 0; + voiceFrameCount = 0; + if (activeTTS) { + closeTts(activeTTS, "[INTERRUPT] Failed to close active TTS stream"); + activeTTS = null; + } + turnManager.reset(); + if (clientWs.readyState === WebSocket.OPEN) { + clientWs.send(JSON.stringify({ type: "interrupt" })); + } + } + + /* ======= SONIOX ======= */ + + function connectSoniox() { + const ws = new WebSocket(SONIOX_URL); + sonioxWs = ws; + + ws.on("open", () => { + logger.info("[SONIOX] Connected"); + ws.send( + JSON.stringify({ + api_key: getSonioxApiKey(), + model: "stt-rt-preview", + audio_format: "auto", + language_hints: ["en"], + enable_endpoint_detection: true, + }), + ); + ws.send(makeWavHeader(16000, 1)); + startKeepAlive(); + }); + + ws.on("message", handleSonioxMessage); + ws.on("close", (code, reason) => { + logger.info( + `[SONIOX] Closed: code=${code} reason=${reason.toString() || "(none)"}`, + ); + stopKeepAlive(); + if (!sessionClosed) { + setTimeout(() => { + connectSoniox(); + }, 500); } + }); + ws.on("error", (err) => { + logger.error("[SONIOX] Error:", err.message); + }); + } - try { - // Close the WS first so that any pending done/error/close listeners - // in processTurn() can settle immediately, rather than hanging until - // the withTimeout fires. - stream.close(); - stream.removeAllListeners(); - } catch (error) { - logger.warn(reason, error); + function startKeepAlive() { + keepAliveTimer = setInterval(() => { + if (sonioxWs?.readyState === WebSocket.OPEN) { + sonioxWs.send(JSON.stringify({ type: "keepalive" })); } + }, 8000); + } + + function stopKeepAlive() { + if (keepAliveTimer) { + clearInterval(keepAliveTimer); + keepAliveTimer = null; } + } - function doInterrupt() { - logger.info("[INTERRUPT] Cutting TTS"); - if (playbackResetTimer) { - clearTimeout(playbackResetTimer); - playbackResetTimer = null; - } - bargeInLockedUntil = 0; - currentTurnId++; - activePipelineTurnId = null; - transcriptBuffer = ""; - isSpeaking = false; - silenceFrameCount = 0; - voiceFrameCount = 0; - if (activeTTS) { - closeTts(activeTTS, "[INTERRUPT] Failed to close active TTS stream"); - activeTTS = null; - } - turnManager.reset(); - if (clientWs.readyState === WebSocket.OPEN) { - clientWs.send(JSON.stringify({ type: "interrupt" })); - } + /* ======= STT HANDLER ======= */ + + async function handleSonioxMessage(msg: WebSocket.RawData) { + const data = parseSonioxMessage(msg); + if (!data) { + return; } - /* ======= SONIOX ======= */ - - function connectSoniox() { - const ws = new WebSocket(SONIOX_URL); - sonioxWs = ws; - - ws.on("open", () => { - logger.info("[SONIOX] Connected"); - ws.send( - JSON.stringify({ - api_key: getSonioxApiKey(), - model: "stt-rt-preview", - audio_format: "auto", - language_hints: ["en"], - enable_endpoint_detection: true, - }), - ); - ws.send(makeWavHeader(16000, 1)); - startKeepAlive(); - }); + if (!Array.isArray(data.tokens)) { + if (data.error || data.status || data.type) { + if (logger.shouldLog("debug")) { + logger.info("[SONIOX] msg:", JSON.stringify(data)); + } + } + return; + } - ws.on("message", handleSonioxMessage); - ws.on("close", (code, reason) => { + const tokens = data.tokens; + + // Barge-in detection: + // Soniox non-final tokens = real speech is being recognised right now. + // Browser AEC (echo cancellation) suppresses TTS playback at the mic, so + // non-final tokens can only come from the user's own voice — unlike raw + // Cobra probability which can be fooled by speaker echo. + // We only fire interrupt when the TurnManager confirms TTS is actually + // playing (ASSISTANT_SPEAKING state set by processTurn). + // bargeInLockedUntil suppresses the first ~400ms after TTS starts so that + // TTS audio picked up by the mic (before AEC locks on) can't re-trigger. + if ( + turnManager.state === TurnState.ASSISTANT_SPEAKING && + Date.now() > bargeInLockedUntil + ) { + const speechPartials = tokens.filter( + (token) => + !token.is_final && token.text && token.text.trim().length > 1, + ); + if (speechPartials.length > 0) { logger.info( - `[SONIOX] Closed: code=${code} reason=${reason.toString() || "(none)"}`, + `[BARGE-IN] Detected via Soniox: "${speechPartials.map((token) => token.text).join("")}"`, ); - stopKeepAlive(); - if (!sessionClosed) { - setTimeout(() => { - connectSoniox(); - }, 500); - } - }); - ws.on("error", (err) => { - logger.error("[SONIOX] Error:", err.message); - }); + doInterrupt(); + return; + } } - function startKeepAlive() { - keepAliveTimer = setInterval(() => { - if (sonioxWs?.readyState === WebSocket.OPEN) { - sonioxWs.send(JSON.stringify({ type: "keepalive" })); - } - }, 8000); + const finals = tokens.filter((token) => token.is_final && token.text); + if (!finals.length) { + return; } - function stopKeepAlive() { - if (keepAliveTimer) { - clearInterval(keepAliveTimer); - keepAliveTimer = null; - } + transcriptBuffer += finals.map((token) => token.text).join(""); + + const hasEnd = finals.some((token) => token.text === ""); + if (!hasEnd) { + return; } - /* ======= STT HANDLER ======= */ + const finalText = transcriptBuffer.replace("", "").trim(); + transcriptBuffer = ""; - async function handleSonioxMessage(msg: WebSocket.RawData) { - const data = parseSonioxMessage(msg); - if (!data) { - return; - } + if (!finalText) { + return; + } - if (!Array.isArray(data.tokens)) { - if (data.error || data.status || data.type) { - if (logger.shouldLog("debug")) { - logger.info("[SONIOX] msg:", JSON.stringify(data)); - } - } - return; - } + logger.info("[STT] Final ->", finalText); + try { + await processTurn(finalText); + } catch (err) { + logger.error( + "[PIPELINE] Unhandled error in processTurn:", + (err as Error).message, + ); + turnManager.reset(); + } + } - const tokens = data.tokens; - - // Barge-in detection: - // Soniox non-final tokens = real speech is being recognised right now. - // Browser AEC (echo cancellation) suppresses TTS playback at the mic, so - // non-final tokens can only come from the user's own voice — unlike raw - // Cobra probability which can be fooled by speaker echo. - // We only fire interrupt when the TurnManager confirms TTS is actually - // playing (ASSISTANT_SPEAKING state set by processTurn). - // bargeInLockedUntil suppresses the first ~400ms after TTS starts so that - // TTS audio picked up by the mic (before AEC locks on) can't re-trigger. - if ( - turnManager.state === TurnState.ASSISTANT_SPEAKING && - Date.now() > bargeInLockedUntil - ) { - const speechPartials = tokens.filter( - (token) => - !token.is_final && token.text && token.text.trim().length > 1, - ); - if (speechPartials.length > 0) { - logger.info( - `[BARGE-IN] Detected via Soniox: "${speechPartials.map((token) => token.text).join("")}"`, - ); - doInterrupt(); - return; - } - } + /* ======= TURN PROCESSOR ======= */ - const finals = tokens.filter((token) => token.is_final && token.text); - if (!finals.length) { + async function processTurn(userText: string) { + if (activePipelineTurnId !== null) { + logger.info( + "[PIPELINE] Already running — discarding duplicate STT final", + ); + return; + } + currentTurnId++; + const myTurn = currentTurnId; + activePipelineTurnId = myTurn; + const tSttEnd = now(); + + try { + // Build context without mutating `conversation` — only commit on full completion. + const stream = await streamAnswer(neurolink, [ + ...conversation, + { role: "user", content: userText }, + ]); + if (myTurn !== currentTurnId) { return; } - transcriptBuffer += finals.map((token) => token.text).join(""); + const tts = new CartesiaStream(`turn-${Date.now()}`); + activeTTS = tts; + await tts.ready(); - const hasEnd = finals.some((token) => token.text === ""); - if (!hasEnd) { + if (myTurn !== currentTurnId) { return; } - const finalText = transcriptBuffer.replace("", "").trim(); - transcriptBuffer = ""; + // Register error handler immediately after ready() — before the LLM stream loop — + // so Cartesia errors emitted mid-stream (during token sending) are captured. + // Without this, errors during the for-await loop have no listener and are swallowed. + let ttsError: Error | null = null; + tts.on("error", (err: Error) => { + ttsError = err; + logger.error("[TTS] Mid-stream error:", err.message); + }); - if (!finalText) { - return; - } + // Pre-lock barge-in BEFORE signaling assistant speaking. + // Without this there is a ~700-1000ms gap where TurnState is ASSISTANT_SPEAKING + // but bargeInLockedUntil=0, so Soniox residual tokens from the previous TTS echo + // immediately trigger an interrupt before any audio has even been sent. + bargeInLockedUntil = Date.now() + 1000; - logger.info("[STT] Final ->", finalText); - try { - await processTurn(finalText); - } catch (err) { - logger.error( - "[PIPELINE] Unhandled error in processTurn:", - (err as Error).message, - ); - turnManager.reset(); - } - } + // Signal TurnManager that TTS is about to play — barge-in detection is now live. + turnManager.assistantSpeaking(); - /* ======= TURN PROCESSOR ======= */ + let firstAudioSent = false; + let assistantReply = ""; + let tokenBuffer = ""; - async function processTurn(userText: string) { - if (activePipelineTurnId !== null) { - logger.info( - "[PIPELINE] Already running — discarding duplicate STT final", - ); - return; - } - currentTurnId++; - const myTurn = currentTurnId; - activePipelineTurnId = myTurn; - const tSttEnd = now(); + // Sentence/phrase boundaries to flush on — avoids flooding Cartesia with + // one tiny message per token, which causes "Service unavailable" errors on + // long responses. We flush when we hit natural speech breaks or the buffer + // grows large enough to produce a clean TTS chunk. + const FLUSH_REGEX = /[.!?,;:]\s/; + const FLUSH_MIN_LENGTH = 80; - try { - // Build context without mutating `conversation` — only commit on full completion. - const stream = await streamAnswer(neurolink, [ - ...conversation, - { role: "user", content: userText }, - ]); + tts.on("audio", (audio: Buffer) => { if (myTurn !== currentTurnId) { return; } + if (!firstAudioSent) { + firstAudioSent = true; + // Refresh the lock from when audio ACTUALLY hits the client so it covers + // the AEC lock-on window (~300-400ms for browser echo cancellation). + // This extends the protection past the initial 1000ms pre-lock. + bargeInLockedUntil = Date.now() + 400; + logger.info( + `[LATENCY] STT -> First Audio: ${(now() - tSttEnd).toFixed(0)}ms`, + ); + } + if (clientWs.readyState === WebSocket.OPEN) { + clientWs.send(audio); + } + }); - const tts = new CartesiaStream(`turn-${Date.now()}`); - activeTTS = tts; - await tts.ready(); - + for await (const chunk of stream) { if (myTurn !== currentTurnId) { - return; + logger.info("[PIPELINE] Stale LLM stream — dropping"); + break; } - - // Register error handler immediately after ready() — before the LLM stream loop — - // so Cartesia errors emitted mid-stream (during token sending) are captured. - // Without this, errors during the for-await loop have no listener and are swallowed. - let ttsError: Error | null = null; - tts.on("error", (err: Error) => { - ttsError = err; - logger.error("[TTS] Mid-stream error:", err.message); - }); - - // Pre-lock barge-in BEFORE signaling assistant speaking. - // Without this there is a ~700-1000ms gap where TurnState is ASSISTANT_SPEAKING - // but bargeInLockedUntil=0, so Soniox residual tokens from the previous TTS echo - // immediately trigger an interrupt before any audio has even been sent. - bargeInLockedUntil = Date.now() + 1000; - - // Signal TurnManager that TTS is about to play — barge-in detection is now live. - turnManager.assistantSpeaking(); - - let firstAudioSent = false; - let assistantReply = ""; - let tokenBuffer = ""; - - // Sentence/phrase boundaries to flush on — avoids flooding Cartesia with - // one tiny message per token, which causes "Service unavailable" errors on - // long responses. We flush when we hit natural speech breaks or the buffer - // grows large enough to produce a clean TTS chunk. - const FLUSH_REGEX = /[.!?,;:]\s/; - const FLUSH_MIN_LENGTH = 80; - - tts.on("audio", (audio: Buffer) => { - if (myTurn !== currentTurnId) { - return; - } - if (!firstAudioSent) { - firstAudioSent = true; - // Refresh the lock from when audio ACTUALLY hits the client so it covers - // the AEC lock-on window (~300-400ms for browser echo cancellation). - // This extends the protection past the initial 1000ms pre-lock. - bargeInLockedUntil = Date.now() + 400; - logger.info( - `[LATENCY] STT -> First Audio: ${(now() - tSttEnd).toFixed(0)}ms`, - ); - } - if (clientWs.readyState === WebSocket.OPEN) { - clientWs.send(audio); - } - }); - - for await (const chunk of stream) { - if (myTurn !== currentTurnId) { - logger.info("[PIPELINE] Stale LLM stream — dropping"); - break; - } - // If Cartesia errored mid-stream, abort sending more tokens. - if (ttsError) { - logger.info("[PIPELINE] Aborting LLM stream — Cartesia error"); - break; - } - if (!chunk || typeof chunk !== "object" || !("content" in chunk)) { - continue; - } - if (typeof chunk.content !== "string") { - continue; - } - assistantReply += chunk.content; - tokenBuffer += chunk.content; - - // Flush buffer to Cartesia at sentence/phrase boundaries or when it's - // grown large enough. This batches tokens into meaningful speech chunks - // instead of sending one WebSocket message per token. - if ( - FLUSH_REGEX.test(tokenBuffer) || - tokenBuffer.length >= FLUSH_MIN_LENGTH - ) { - tts.send(tokenBuffer, true); - tokenBuffer = ""; - } + // If Cartesia errored mid-stream, abort sending more tokens. + if (ttsError) { + logger.info("[PIPELINE] Aborting LLM stream — Cartesia error"); + break; } + if (!chunk || typeof chunk !== "object" || !("content" in chunk)) { + continue; + } + if (typeof chunk.content !== "string") { + continue; + } + assistantReply += chunk.content; + tokenBuffer += chunk.content; - // Flush any remaining buffered tokens before the final flush(). - if (tokenBuffer) { + // Flush buffer to Cartesia at sentence/phrase boundaries or when it's + // grown large enough. This batches tokens into meaningful speech chunks + // instead of sending one WebSocket message per token. + if ( + FLUSH_REGEX.test(tokenBuffer) || + tokenBuffer.length >= FLUSH_MIN_LENGTH + ) { tts.send(tokenBuffer, true); tokenBuffer = ""; } + } - // If Cartesia errored during the stream, reset and bail out now. - if (ttsError) { - logger.error( - "[TTS] Error during stream — resetting turn so user can retry:", - String(ttsError), - ); - closeTts( - tts, - "[TTS] Failed to close stream after mid-stream error", - ); - turnManager.reset(); - return; - } + // Flush any remaining buffered tokens before the final flush(). + if (tokenBuffer) { + tts.send(tokenBuffer, true); + tokenBuffer = ""; + } - if (myTurn !== currentTurnId) { - return; - } + // If Cartesia errored during the stream, reset and bail out now. + if (ttsError) { + logger.error( + "[TTS] Error during stream — resetting turn so user can retry:", + String(ttsError), + ); + closeTts(tts, "[TTS] Failed to close stream after mid-stream error"); + turnManager.reset(); + return; + } - let ttsSucceeded = false; - try { - await withTimeout( - new Promise((resolve, reject) => { - tts.once("done", () => { - ttsSucceeded = true; - resolve(); - }); - // Re-use the persistent error handler: if another error arrives during flush, - // the existing "error" listener fires ttsError; reject via a one-time wrapper. - tts.once("error", reject); - // Reject if the socket closes without emitting done or error. - tts.once("close", () => - reject( - new Error("Cartesia WS closed before flush completed"), - ), - ); - tts.flush(); - }), - 10000, - "Cartesia flush timed out", - ); - } catch (err) { - // Cartesia failed (e.g. "Service unavailable"). The user heard nothing. - // Reset state immediately so they can speak and retry — don't commit - // the turn to conversation history since it was never heard. - logger.error( - "[TTS] Error during flush — resetting turn so user can retry:", - (err as Error).message, - ); - closeTts(tts, "[TTS] Failed to close stream after flush error"); - turnManager.reset(); - return; - } + if (myTurn !== currentTurnId) { + return; + } - closeTts( - tts, - "[TTS] Failed to close stream after successful playback", + let ttsSucceeded = false; + try { + await withTimeout( + new Promise((resolve, reject) => { + tts.once("done", () => { + ttsSucceeded = true; + resolve(); + }); + // Re-use the persistent error handler: if another error arrives during flush, + // the existing "error" listener fires ttsError; reject via a one-time wrapper. + tts.once("error", reject); + // Reject if the socket closes without emitting done or error. + tts.once("close", () => + reject(new Error("Cartesia WS closed before flush completed")), + ); + tts.flush(); + }), + 10000, + "Cartesia flush timed out", ); + } catch (err) { + // Cartesia failed (e.g. "Service unavailable"). The user heard nothing. + // Reset state immediately so they can speak and retry — don't commit + // the turn to conversation history since it was never heard. + logger.error( + "[TTS] Error during flush — resetting turn so user can retry:", + (err as Error).message, + ); + closeTts(tts, "[TTS] Failed to close stream after flush error"); + turnManager.reset(); + return; + } - if (!ttsSucceeded || myTurn !== currentTurnId) { - return; - } + closeTts(tts, "[TTS] Failed to close stream after successful playback"); + + if (!ttsSucceeded || myTurn !== currentTurnId) { + return; + } - // Only commit conversation when the turn completed fully and was heard. - conversation.push({ role: "user", content: userText }); - conversation.push({ role: "assistant", content: assistantReply }); - // Do NOT reset state here — the client is still playing buffered audio. - // The client sends playback_done when the last audio chunk finishes playing, - // which is the correct moment to return to IDLE and allow new user speech. - // Safety fallback: if the client never sends playback_done (crash, disconnect), - // auto-reset after 20 seconds so the assistant doesn't stay stuck. + // Only commit conversation when the turn completed fully and was heard. + conversation.push({ role: "user", content: userText }); + conversation.push({ role: "assistant", content: assistantReply }); + // Do NOT reset state here — the client is still playing buffered audio. + // The client sends playback_done when the last audio chunk finishes playing, + // which is the correct moment to return to IDLE and allow new user speech. + // Safety fallback: if the client never sends playback_done (crash, disconnect), + // auto-reset after 20 seconds so the assistant doesn't stay stuck. + if (playbackResetTimer) { + clearTimeout(playbackResetTimer); + } + playbackResetTimer = setTimeout(() => { + playbackResetTimer = null; + turnManager.reset(); + }, 20000); + } finally { + if (activePipelineTurnId === myTurn) { + activePipelineTurnId = null; + } + } + } + + /* ======= CLIENT AUDIO + CONTROL ======= */ + + clientWs.on("message", (data) => { + if (typeof data === "string") { + const msg = parseClientControlMessage(data); + if (msg?.type === "playback_done") { + // Client finished playing all audio — now it's safe to listen again. if (playbackResetTimer) { clearTimeout(playbackResetTimer); - } - playbackResetTimer = setTimeout(() => { playbackResetTimer = null; - turnManager.reset(); - }, 20000); - } finally { - if (activePipelineTurnId === myTurn) { - activePipelineTurnId = null; } + turnManager.reset(); } + return; } - /* ======= CLIENT AUDIO + CONTROL ======= */ + if (!(data instanceof Buffer)) { + return; + } - clientWs.on("message", (data) => { - if (typeof data === "string") { - const msg = parseClientControlMessage(data); - if (msg?.type === "playback_done") { - // Client finished playing all audio — now it's safe to listen again. - if (playbackResetTimer) { - clearTimeout(playbackResetTimer); - playbackResetTimer = null; - } - turnManager.reset(); - } - return; - } + // Reassemble into exact FRAME_BYTES-sized Cobra frames. + const combined = Buffer.concat([frameRemainder, data]); + let pos = 0; - if (!(data instanceof Buffer)) { - return; + while (pos + FRAME_BYTES <= combined.length) { + const frame = new Int16Array(FRAME_LENGTH); + for (let i = 0; i < FRAME_LENGTH; i++) { + frame[i] = combined.readInt16LE(pos + i * 2); } + pos += FRAME_BYTES; - // Reassemble into exact FRAME_BYTES-sized Cobra frames. - const combined = Buffer.concat([frameRemainder, data]); - let pos = 0; - - while (pos + FRAME_BYTES <= combined.length) { - const frame = new Int16Array(FRAME_LENGTH); - for (let i = 0; i < FRAME_LENGTH; i++) { - frame[i] = combined.readInt16LE(pos + i * 2); - } - pos += FRAME_BYTES; - - // Cobra VAD: - // Cobra tracks when the user is speaking vs silent. Its output drives - // TurnManager state (USER_SPEAKING / PROCESSING) but does NOT trigger - // interrupt — that comes from Soniox non-final tokens so echo can't fool it. - let voiceProb = 0; - try { - if (!cobra) { - continue; - } - voiceProb = cobra.process(frame); - } catch (err) { - logger.error("[VAD] Cobra process error:", err); + // Cobra VAD: + // Cobra tracks when the user is speaking vs silent. Its output drives + // TurnManager state (USER_SPEAKING / PROCESSING) but does NOT trigger + // interrupt — that comes from Soniox non-final tokens so echo can't fool it. + let voiceProb = 0; + try { + if (!cobra) { + continue; } + voiceProb = cobra.process(frame); + } catch (err) { + logger.error("[VAD] Cobra process error:", err); + } - const isVoice = voiceProb >= VOICE_THRESHOLD; + const isVoice = voiceProb >= VOICE_THRESHOLD; - if (isVoice) { - voiceFrameCount++; - silenceFrameCount = 0; - if (!isSpeaking && voiceFrameCount >= VOICE_FRAMES_TO_START) { - isSpeaking = true; - logger.info(`[VAD] Speech start (prob=${voiceProb.toFixed(2)})`); - bus.publish({ type: "vad_start" }); - } - } else { - voiceFrameCount = 0; - if (isSpeaking) { - silenceFrameCount++; - if (silenceFrameCount >= SILENCE_FRAMES_TO_STOP) { - isSpeaking = false; - silenceFrameCount = 0; - logger.info("[VAD] Speech stop"); - bus.publish({ type: "vad_stop" }); - } - } + if (isVoice) { + voiceFrameCount++; + silenceFrameCount = 0; + if (!isSpeaking && voiceFrameCount >= VOICE_FRAMES_TO_START) { + isSpeaking = true; + logger.info(`[VAD] Speech start (prob=${voiceProb.toFixed(2)})`); + bus.publish({ type: "vad_start" }); } - - // Always forward every frame to Soniox for continuous transcription. - if (sonioxWs?.readyState === WebSocket.OPEN) { - sonioxWs.send(Buffer.from(frame.buffer)); + } else { + voiceFrameCount = 0; + if (isSpeaking) { + silenceFrameCount++; + if (silenceFrameCount >= SILENCE_FRAMES_TO_STOP) { + isSpeaking = false; + silenceFrameCount = 0; + logger.info("[VAD] Speech stop"); + bus.publish({ type: "vad_stop" }); + } } } - frameRemainder = combined.subarray(pos); - }); - - clientWs.on("close", () => { - logger.info("[WS] Client disconnected"); - sessionClosed = true; - if (cobra) { - cobra.release(); - } - closeTts(activeTTS, "[WS] Failed to close active TTS on disconnect"); - stopKeepAlive(); - if (sonioxWs) { - sonioxWs.close(); + // Always forward every frame to Soniox for continuous transcription. + if (sonioxWs?.readyState === WebSocket.OPEN) { + sonioxWs.send(Buffer.from(frame.buffer)); } - }); + } - connectSoniox(); - })().catch((err) => { - logger.error("[WS] Connection handler failed:", err); - try { - clientWs.close(); - } catch { - /* already closed */ + frameRemainder = combined.subarray(pos); + }); + + clientWs.on("close", () => { + logger.info("[WS] Client disconnected"); + sessionClosed = true; + if (cobra) { + cobra.release(); + } + closeTts(activeTTS, "[WS] Failed to close active TTS on disconnect"); + stopKeepAlive(); + if (sonioxWs) { + sonioxWs.close(); } }); + + connectSoniox(); }); } diff --git a/src/lib/types/generate.ts b/src/lib/types/generate.ts index f14513bd0..d32f42c5b 100644 --- a/src/lib/types/generate.ts +++ b/src/lib/types/generate.ts @@ -19,6 +19,7 @@ import type { } from "./multimodal.js"; import type { PPTGenerationResult, PPTOutputOptions } from "./ppt.js"; import type { TTSOptions, TTSResult } from "./tts.js"; +import type { STTOptions, STTResult } from "./stt.js"; import type { StandardRecord, ValidationSchema, @@ -163,6 +164,25 @@ export type GenerateOptions = { */ tts?: TTSOptions; + /** + * Speech-to-Text (STT) configuration + * + * Enable audio transcription. When enabled, the audio provided via `stt.audio` + * will be transcribed to text and used as the prompt. + * + * @example + * ```typescript + * const neurolink = new NeuroLink(); + * const result = await neurolink.generate({ + * input: { text: "" }, + * provider: "openai", + * stt: { enabled: true, provider: "whisper", language: "en-US", audio: audioBuffer } + * }); + * // STT transcribes the audio, result.transcription contains the transcription + * ``` + */ + stt?: STTOptions & { provider?: string; audio?: Buffer | ArrayBuffer }; + /** * Thinking/reasoning configuration for extended thinking models * @@ -734,6 +754,9 @@ export type GenerateResult = { /** Token count for reasoning content */ reasoningTokens?: number; + /** STT transcription result (present when stt.enabled is true and audio input was provided) */ + transcription?: STTResult; + // NL-007: Retry metadata for observability retries?: { count: number; @@ -957,6 +980,25 @@ export type TextGenerationOptions = { */ tts?: TTSOptions; + /** + * Speech-to-Text (STT) configuration + * + * Enable audio transcription. When enabled, the audio provided via `stt.audio` + * will be transcribed to text and used as the prompt. + * + * @example + * ```typescript + * const neurolink = new NeuroLink(); + * const result = await neurolink.generate({ + * input: { text: "" }, + * provider: "openai", + * stt: { enabled: true, provider: "whisper", language: "en-US", audio: audioBuffer } + * }); + * // STT transcribes the audio, result.transcription contains the transcription + * ``` + */ + stt?: STTOptions & { provider?: string; audio?: Buffer | ArrayBuffer }; + // NEW: Analytics and Evaluation Support enableEvaluation?: boolean; // Default: false - AI quality scoring enableAnalytics?: boolean; // Default: false - Usage tracking @@ -1143,6 +1185,8 @@ export type TextGenerationResult = { analytics?: AnalyticsData; evaluation?: EvaluationData; audio?: TTSResult; + /** STT transcription result (present when stt input was processed) */ + transcription?: STTResult; /** Video generation result */ video?: VideoGenerationResult; /** PowerPoint generation result */ diff --git a/src/lib/types/index.ts b/src/lib/types/index.ts index 1e419be02..509fda3fe 100644 --- a/src/lib/types/index.ts +++ b/src/lib/types/index.ts @@ -50,7 +50,7 @@ export * from "./subscription.js"; export * from "./task.js"; export * from "./taskClassification.js"; export * from "./tools.js"; -export * from "./tts.js"; +export * from "./voice.js"; export * from "./universalProviderOptions.js"; export * from "./utilities.js"; export * from "./workflow.js"; diff --git a/src/lib/types/realtime.ts b/src/lib/types/realtime.ts new file mode 100644 index 000000000..3ec24ab67 --- /dev/null +++ b/src/lib/types/realtime.ts @@ -0,0 +1,322 @@ +/** + * Realtime Voice Type Definitions for NeuroLink + * + * All realtime/bidirectional voice types: session, config, messages, + * event handlers, provider types, handler types, error codes, defaults, + * and type guards. + * + * @module types/realtime + */ + +import type { AudioFormat } from "./tts.js"; +import type { VoiceCapability } from "./voice.js"; + +// ============================================================================ +// REALTIME SESSION TYPES +// ============================================================================ + +/** + * Realtime session state + */ +export type RealtimeSessionState = + | "disconnected" + | "connecting" + | "connected" + | "disconnecting" + | "error"; + +/** + * Realtime voice configuration + */ +export type RealtimeConfig = { + /** Provider to use (openai, gemini) */ + provider: "openai" | "gemini"; + /** API key */ + apiKey?: string; + /** Model to use */ + model?: string; + /** Voice for TTS output */ + voice?: string; + /** Input language */ + inputLanguage?: string; + /** Output language */ + outputLanguage?: string; + /** System prompt for the AI */ + systemPrompt?: string; + /** Session timeout in milliseconds */ + timeout?: number; + /** Audio input format */ + inputFormat?: AudioFormat; + /** Audio output format */ + outputFormat?: AudioFormat; + /** Input sample rate */ + inputSampleRate?: number; + /** Output sample rate */ + outputSampleRate?: number; + /** Enable voice activity detection */ + vadEnabled?: boolean; + /** VAD threshold (0-1) */ + vadThreshold?: number; + /** Turn detection mode */ + turnDetection?: "server_vad" | "manual"; + /** Instructions/system prompt for the session */ + instructions?: string; + /** Temperature for AI responses */ + temperature?: number; + /** Tools/functions available to the model */ + tools?: RealtimeTool[]; +}; + +/** + * Realtime tool definition + */ +export type RealtimeTool = { + /** Tool name */ + name: string; + /** Tool description */ + description: string; + /** JSON schema for parameters */ + parameters: Record; +}; + +/** + * Realtime session information + */ +export type RealtimeSession = { + /** Session ID */ + id: string; + /** Current state */ + state: RealtimeSessionState; + /** Provider name */ + provider: string; + /** Model being used */ + model?: string; + /** Session creation time */ + createdAt: Date; + /** Last activity time */ + lastActivityAt: Date; + /** Session configuration */ + config: RealtimeConfig; + /** Check if session is open */ + isOpen?: () => boolean; + /** Close the session */ + close?: () => Promise; +}; + +/** + * Realtime audio chunk + */ +export type RealtimeAudioChunk = { + /** Audio data */ + data: Buffer; + /** Chunk sequence number */ + index: number; + /** Whether this is the final chunk */ + isFinal: boolean; + /** Audio format */ + format: AudioFormat; + /** Sample rate */ + sampleRate?: number; + /** Duration of this chunk in milliseconds */ + durationMs?: number; +}; + +/** + * Realtime message types + */ +export type RealtimeMessageType = + | "audio" + | "text" + | "transcript" + | "function_call" + | "function_result" + | "error" + | "session_update" + | "turn_start" + | "turn_end"; + +/** + * Realtime message + */ +export type RealtimeMessage = { + /** Message type */ + type: RealtimeMessageType; + /** Message ID */ + id?: string; + /** Audio data (for audio messages) */ + audio?: RealtimeAudioChunk; + /** Text content (for text/transcript messages) */ + text?: string; + /** Whether this is a partial result */ + isPartial?: boolean; + /** Function call data */ + functionCall?: { + name: string; + arguments: Record; + }; + /** Function result data */ + functionResult?: { + name: string; + result: unknown; + }; + /** Error information */ + error?: { + code: string; + message: string; + }; + /** Timestamp */ + timestamp: Date; +}; + +/** + * Realtime event handler callbacks + */ +export type RealtimeEventHandlers = { + /** Called when audio is received */ + onAudio?: (chunk: RealtimeAudioChunk) => void; + /** Called when text/transcript is received */ + onTranscript?: (text: string, isFinal: boolean) => void; + /** Called when the model generates text */ + onText?: (text: string, isFinal: boolean) => void; + /** Called when a function call is requested */ + onFunctionCall?: ( + name: string, + args: Record, + ) => Promise; + /** Called when session state changes */ + onStateChange?: (state: RealtimeSessionState) => void; + /** Called when an error occurs */ + onError?: (error: Error) => void; + /** Called when a turn starts */ + onTurnStart?: () => void; + /** Called when a turn ends */ + onTurnEnd?: () => void; +}; + +// ============================================================================ +// REALTIME PROVIDER TYPE +// ============================================================================ + +/** + * Realtime voice provider type (bidirectional audio) + */ +export type RealtimeVoiceProvider = { + /** Provider name identifier */ + readonly name: string; + /** Get supported capabilities */ + getCapabilities(): VoiceCapability[]; + /** Check if provider is properly configured */ + isConfigured(): boolean; + /** Validate provider configuration */ + validateConfig(): Promise<{ valid: boolean; errors: string[] }>; + /** Get provider-specific options schema */ + getOptionsSchema?(): Record; + /** + * Create a new realtime session + */ + connect(config: RealtimeConfig): Promise; + + /** + * Check if connected + */ + isConnected(): boolean; + + /** + * Disconnect from realtime session + */ + disconnect(): Promise; + + /** + * Get current session configuration + */ + getSessionConfig(): RealtimeConfig | null; +}; + +// ============================================================================ +// REALTIME HANDLER TYPE +// ============================================================================ + +export type RealtimeHandler = { + readonly name: string; + connect(config: RealtimeConfig): Promise; + disconnect(): Promise; + isConnected(): boolean; + getSession(): RealtimeSession | null; + sendAudio(audio: Buffer | RealtimeAudioChunk): Promise; + sendText?(text: string): Promise; + triggerResponse?(): Promise; + cancelResponse?(): Promise; + on(handlers: RealtimeEventHandlers): void; + off(): void; + isConfigured(): boolean; + getSupportedFormats(): AudioFormat[]; +}; + +// ============================================================================ +// REALTIME ERROR CODES +// ============================================================================ + +/** + * Realtime error codes + */ +export const REALTIME_ERROR_CODES = { + CONNECTION_FAILED: "REALTIME_CONNECTION_FAILED", + SESSION_TIMEOUT: "REALTIME_SESSION_TIMEOUT", + PROTOCOL_ERROR: "REALTIME_PROTOCOL_ERROR", + AUDIO_STREAM_ERROR: "REALTIME_AUDIO_STREAM_ERROR", + PROVIDER_NOT_CONFIGURED: "REALTIME_PROVIDER_NOT_CONFIGURED", + PROVIDER_NOT_SUPPORTED: "REALTIME_PROVIDER_NOT_SUPPORTED", + SESSION_ALREADY_ACTIVE: "REALTIME_SESSION_ALREADY_ACTIVE", + SESSION_NOT_ACTIVE: "REALTIME_SESSION_NOT_ACTIVE", + INVALID_MESSAGE: "REALTIME_INVALID_MESSAGE", +} as const; + +// ============================================================================ +// REALTIME DEFAULTS +// ============================================================================ + +/** + * Default realtime configuration + */ +export const DEFAULT_REALTIME_CONFIG: Partial = { + timeout: 30000, + inputSampleRate: 24000, + outputSampleRate: 24000, + vadEnabled: true, + vadThreshold: 0.5, + turnDetection: "server_vad", +}; + +// ============================================================================ +// REALTIME TYPE GUARDS +// ============================================================================ + +/** + * Type guard for valid RealtimeConfig + */ +export function isValidRealtimeConfig( + config: unknown, +): config is RealtimeConfig { + if (!config || typeof config !== "object") { + return false; + } + const conf = config as RealtimeConfig; + if (!conf.provider || !["openai", "gemini"].includes(conf.provider)) { + return false; + } + if (conf.timeout !== undefined) { + if (typeof conf.timeout !== "number" || conf.timeout <= 0) { + return false; + } + } + if (conf.vadThreshold !== undefined) { + if ( + typeof conf.vadThreshold !== "number" || + conf.vadThreshold < 0 || + conf.vadThreshold > 1 + ) { + return false; + } + } + return true; +} diff --git a/src/lib/types/server.ts b/src/lib/types/server.ts index 02459a3bf..201424b0a 100644 --- a/src/lib/types/server.ts +++ b/src/lib/types/server.ts @@ -1465,14 +1465,3 @@ export type SonioxMessage = { export type ClientControlMessage = { type?: string; }; - -/** - * Structural type for Picovoice Cobra VAD instance. - * Defined here so the optional `@picovoice/cobra-node` package - * is not required at typecheck time. - */ -export type CobraInstance = { - frameLength: number; - process: (pcm: Int16Array) => number; - release: () => void; -}; diff --git a/src/lib/types/span.ts b/src/lib/types/span.ts index 04dff8fbe..7684950b7 100644 --- a/src/lib/types/span.ts +++ b/src/lib/types/span.ts @@ -38,6 +38,8 @@ export enum SpanType { WORKFLOW = "workflow", /** TTS synthesis */ TTS = "tts", + /** STT transcription */ + STT = "stt", /** Server adapter request */ SERVER_REQUEST = "server.request", /** Custom span */ diff --git a/src/lib/types/stream.ts b/src/lib/types/stream.ts index 7307c4cda..fc3732caa 100644 --- a/src/lib/types/stream.ts +++ b/src/lib/types/stream.ts @@ -24,6 +24,7 @@ import type { NeurolinkCredentials, } from "./providers.js"; import type { TTSChunk, TTSOptions } from "./tts.js"; +import type { STTOptions } from "./stt.js"; import type { StandardRecord, ValidationSchema } from "./aliases.js"; import type { FileWithMetadata } from "./file.js"; import type { WorkflowConfig } from "./workflow.js"; @@ -299,6 +300,13 @@ export type StreamOptions = { */ tts?: TTSOptions; + /** + * Speech-to-Text (STT) configuration for streaming + * + * When enabled, audio from `stt.audio` is transcribed before streaming begins. + */ + stt?: STTOptions & { provider?: string; audio?: Buffer | ArrayBuffer }; + /** * Thinking/reasoning configuration for extended thinking models * diff --git a/src/lib/types/stt.ts b/src/lib/types/stt.ts new file mode 100644 index 000000000..b1df17d68 --- /dev/null +++ b/src/lib/types/stt.ts @@ -0,0 +1,772 @@ +/** + * Speech-to-Text (STT) Type Definitions for NeuroLink + * + * All STT-specific types: options, results, handlers, + * provider-specific options, error codes, defaults, and type guards. + * + * @module types/stt + */ + +import type { AudioFormat } from "./tts.js"; + +// ============================================================================ +// CORE STT TYPES +// ============================================================================ + +/** + * STT configuration options + */ +export type STTOptions = { + /** Enable STT processing */ + enabled?: boolean; + /** Override STT provider */ + provider?: string; + /** Language code for transcription (e.g., "en-US") */ + language?: string; + /** Audio format of input */ + format?: AudioFormat; + /** Sample rate in Hz */ + sampleRate?: number; + /** Enable punctuation in transcription */ + punctuation?: boolean; + /** Enable punctuation (alias) */ + punctuate?: boolean; + /** Enable profanity filter */ + profanityFilter?: boolean; + /** Enable speaker diarization */ + speakerDiarization?: boolean; + /** Enable speaker diarization (alias) */ + diarization?: boolean; + /** Number of speakers (for diarization) */ + speakerCount?: number; + /** Enable word-level timestamps */ + wordTimestamps?: boolean; + /** Model variant to use */ + model?: string; + /** Custom vocabulary/phrases */ + vocabulary?: string[]; + /** Minimum confidence threshold */ + confidenceThreshold?: number; +}; + +/** + * STT result from transcription + */ +export type STTResult = { + /** Full transcribed text */ + text: string; + /** Confidence score (0-1) */ + confidence: number; + /** Detected language code */ + language?: string; + /** Audio duration in seconds */ + duration?: number; + /** Word-level timings */ + words?: WordTiming[]; + /** Transcription segments */ + segments?: TranscriptionSegment[]; + /** Speaker labels (for diarization) */ + speakers?: string[]; + /** Performance metadata */ + metadata?: { + /** Processing latency in milliseconds */ + latency: number; + /** Provider name */ + provider?: string; + /** Model used */ + model?: string; + /** Additional provider-specific metadata */ + [key: string]: unknown; + }; +}; + +/** + * STT language information + */ +export type STTLanguage = { + /** Language code (e.g., "en-US") */ + code: string; + /** Language name */ + name: string; + /** Whether the language supports speaker diarization */ + supportsDiarization?: boolean; + /** Whether the language supports punctuation */ + supportsPunctuation?: boolean; +}; + +/** + * Word-level timing information + */ +export type WordTiming = { + /** The word */ + word: string; + /** Start time in seconds */ + startTime?: number; + /** Start time alias */ + start?: number; + /** End time in seconds */ + endTime?: number; + /** End time alias */ + end?: number; + /** Confidence score (0-1) */ + confidence?: number; + /** Speaker label (for diarization) */ + speaker?: string; +}; + +/** + * Transcription segment for streaming STT + */ +export type TranscriptionSegment = { + /** Segment index */ + index?: number; + /** Transcribed text */ + text: string; + /** Whether this is a final result */ + isFinal: boolean; + /** Confidence score (0-1) */ + confidence?: number; + /** Start time in audio (seconds) */ + startTime?: number; + /** Start time (alias for startTime) */ + start?: number; + /** End time in audio (seconds) */ + endTime?: number; + /** End time (alias for endTime) */ + end?: number; + /** Word-level timings */ + words?: WordTiming[]; + /** Speaker label */ + speaker?: string; + /** Detected language */ + language?: string; +}; + +// ============================================================================ +// STT HANDLER TYPE +// ============================================================================ + +export type STTHandler = { + transcribe( + audio: Buffer | ArrayBuffer, + options: STTOptions, + ): Promise; + transcribeStream?( + audioStream: AsyncIterable, + options: STTOptions, + ): AsyncIterable; + getSupportedLanguages?(): Promise; + getSupportedFormats(): AudioFormat[]; + isConfigured(): boolean; + maxAudioDuration?: number; + supportsStreaming?: boolean; +}; + +// ============================================================================ +// STT ERROR CODES +// ============================================================================ + +/** + * STT error codes + */ +export const STT_ERROR_CODES = { + AUDIO_EMPTY: "STT_AUDIO_EMPTY", + AUDIO_TOO_LONG: "STT_AUDIO_TOO_LONG", + INVALID_AUDIO_FORMAT: "STT_INVALID_AUDIO_FORMAT", + LANGUAGE_NOT_SUPPORTED: "STT_LANGUAGE_NOT_SUPPORTED", + TRANSCRIPTION_FAILED: "STT_TRANSCRIPTION_FAILED", + PROVIDER_NOT_CONFIGURED: "STT_PROVIDER_NOT_CONFIGURED", + PROVIDER_NOT_SUPPORTED: "STT_PROVIDER_NOT_SUPPORTED", + STREAM_ERROR: "STT_STREAM_ERROR", + STREAMING_NOT_SUPPORTED: "STT_STREAMING_NOT_SUPPORTED", +} as const; + +// ============================================================================ +// STT DEFAULTS +// ============================================================================ + +/** + * Default STT options + */ +export const DEFAULT_STT_OPTIONS: Required< + Pick< + STTOptions, + "language" | "punctuation" | "profanityFilter" | "sampleRate" + > +> = { + language: "en-US", + punctuation: true, + profanityFilter: false, + sampleRate: 16000, +}; + +// ============================================================================ +// STT TYPE GUARDS +// ============================================================================ + +/** + * Type guard for STTResult + */ +export function isSTTResult(value: unknown): value is STTResult { + if (!value || typeof value !== "object") { + return false; + } + const obj = value as Record; + return ( + typeof obj.text === "string" && + typeof obj.confidence === "number" && + obj.confidence >= 0 && + obj.confidence <= 1 + ); +} + +/** + * Type guard for valid STTOptions + */ +export function isValidSTTOptions(options: unknown): options is STTOptions { + if (!options || typeof options !== "object") { + return false; + } + const opts = options as STTOptions; + if (opts.sampleRate !== undefined) { + if (typeof opts.sampleRate !== "number" || opts.sampleRate <= 0) { + return false; + } + } + if (opts.speakerCount !== undefined) { + if ( + typeof opts.speakerCount !== "number" || + opts.speakerCount < 1 || + opts.speakerCount > 10 + ) { + return false; + } + } + return true; +} + +/** + * Type guard for TranscriptionSegment + */ +export function isTranscriptionSegment( + value: unknown, +): value is TranscriptionSegment { + if (!value || typeof value !== "object") { + return false; + } + const obj = value as Record; + return ( + typeof obj.index === "number" && + typeof obj.text === "string" && + typeof obj.isFinal === "boolean" + ); +} + +// ============================================================================ +// PROVIDER-SPECIFIC STT OPTION TYPES +// ============================================================================ + +export type AzureRecognitionMode = "interactive" | "conversation" | "dictation"; + +export type AzureOutputFormat = "simple" | "detailed"; + +export type AzureSTTOptions = STTOptions & { + recognitionMode?: AzureRecognitionMode; + outputFormat?: AzureOutputFormat; + interimResults?: boolean; + endpointId?: string; + /** Custom endpoint ID (alias for endpointId) */ + customEndpointId?: string; + connectionTimeout?: number; + silenceTimeout?: number; + profanityOption?: "masked" | "removed" | "raw"; + /** Profanity mode (alias for profanityOption) */ + profanityMode?: "masked" | "removed" | "raw"; + initialSilenceTimeout?: number; + enableLogging?: boolean; + phraseList?: string[]; + /** Whether to request detailed output format */ + detailed?: boolean; + wordLevelConfidence?: boolean; + initialSilenceTimeoutMs?: number; + endSilenceTimeoutMs?: number; +}; + +export type DeepgramModel = + | "nova-2" + | "nova-2-general" + | "nova-2-meeting" + | "nova-2-phonecall" + | "nova-2-voicemail" + | "nova-2-finance" + | "nova-2-medical" + | "nova" + | "enhanced" + | "base"; + +export type DeepgramSTTOptions = STTOptions & { + model?: DeepgramModel | "nova-3"; + smartFormat?: boolean; + search?: string[]; + replace?: Array<{ find: string; replace: string }>; + utterances?: boolean; + utterSplit?: number; + /** Alias for utterSplit (legacy field name) */ + uttSplit?: number; + paragraphs?: boolean; + keywords?: string[]; + keywordBoost?: "legacy" | "medium" | "high"; + fillerWords?: boolean; + detectTopics?: boolean; + detectEntities?: boolean; + summarize?: boolean; + redact?: ("pci" | "numbers" | "ssn")[]; +}; + +export type GoogleSTTModel = + | "latest_short" + | "latest_long" + | "telephony" + | "medical_conversation" + | "medical_dictation" + | "command_and_search" + | "phone_call" + | "video" + | "default"; + +export type GoogleSTTAudioEncoding = + | "ENCODING_UNSPECIFIED" + | "LINEAR16" + | "FLAC" + | "MULAW" + | "AMR" + | "AMR_WB" + | "OGG_OPUS" + | "SPEEX_WITH_HEADER_BYTE" + | "MP3" + | "WEBM_OPUS"; + +export type GoogleSTTOptions = STTOptions & { + model?: GoogleSTTModel; + encoding?: GoogleSTTAudioEncoding; + sampleRateHertz?: number; + audioChannelCount?: number; + enableSeparateRecognitionPerChannel?: boolean; + alternativeLanguageCodes?: string[]; + maxAlternatives?: number; + enableAutomaticPunctuation?: boolean; + enableSpokenPunctuation?: boolean; + enableSpokenEmojis?: boolean; + speechContexts?: Array<{ + phrases: string[]; + boost?: number; + }>; + adaptation?: { + phraseSets?: string[]; + customClasses?: string[]; + }; + useEnhanced?: boolean; + keywords?: string[]; +}; + +export type WhisperModel = "whisper-1"; + +export type WhisperSTTOptions = STTOptions & { + model?: WhisperModel; + responseFormat?: "json" | "text" | "srt" | "verbose_json" | "vtt"; + temperature?: number; + prompt?: string; + /** Translate audio to English instead of transcribing in original language */ + translate?: boolean; +}; + +// ============================================================================ +// PROVIDER-INTERNAL API RESPONSE TYPES +// (Moved here from individual provider files per Rule 2 / no-local-type-alias) +// ============================================================================ + +// --- Azure STT --- + +export type AzureWord = { + Word: string; + Offset: number; // In 100-nanosecond units + Duration: number; + Confidence?: number; +}; + +export type AzureNBest = { + Confidence: number; + Lexical: string; + ITN: string; + MaskedITN: string; + Display: string; + Words?: AzureWord[]; +}; + +export type AzureRecognitionResult = { + RecognitionStatus: + | "Success" + | "NoMatch" + | "InitialSilenceTimeout" + | "BabbleTimeout" + | "Error" + | string; + Offset?: number; + Duration?: number; + DisplayText?: string; + NBest?: AzureNBest[]; +}; + +export type AzureSpeakerRecognitionResult = AzureRecognitionResult & { + SpeakerId?: string; +}; + +// --- Deepgram --- + +export type DeepgramWord = { + word: string; + start: number; + end: number; + confidence: number; + speaker?: number; + punctuated_word?: string; +}; + +export type DeepgramAlternative = { + transcript: string; + confidence: number; + words: DeepgramWord[]; + paragraphs?: { + transcript: string; + paragraphs: Array<{ + sentences: Array<{ + text: string; + start: number; + end: number; + }>; + }>; + }; +}; + +export type DeepgramChannel = { + alternatives: DeepgramAlternative[]; +}; + +export type DeepgramUtterance = { + start: number; + end: number; + confidence: number; + channel: number; + transcript: string; + words: DeepgramWord[]; + speaker?: number; + id?: string; +}; + +export type DeepgramResult = { + channels: DeepgramChannel[]; + utterances?: DeepgramUtterance[]; +}; + +export type DeepgramResponse = { + metadata: { + request_id: string; + transaction_key?: string; + sha256?: string; + created: string; + duration: number; + channels: number; + models: string[]; + model_info?: Record; + }; + results: DeepgramResult; +}; + +// --- Google STT --- + +export type GoogleWordInfo = { + startTime: string; // Duration format "1.500s" + endTime: string; + word: string; + confidence?: number; + speakerTag?: number; +}; + +export type GoogleSpeechRecognitionAlternative = { + transcript: string; + confidence: number; + words?: GoogleWordInfo[]; +}; + +export type GoogleSpeechRecognitionResult = { + alternatives: GoogleSpeechRecognitionAlternative[]; + channelTag?: number; + languageCode?: string; + resultEndTime?: string; +}; + +export type GoogleLongRunningRecognizeResponse = { + results: GoogleSpeechRecognitionResult[]; + totalBilledTime?: string; +}; + +export type GoogleRecognizeResponse = { + results?: GoogleSpeechRecognitionResult[]; + totalBilledTime?: string; +}; + +export type GoogleOperationResponse = { + name: string; + done: boolean; + metadata?: { + progressPercent?: number; + startTime?: string; + lastUpdateTime?: string; + }; + response?: GoogleLongRunningRecognizeResponse; + error?: { + code: number; + message: string; + }; +}; + +export type GoogleRecognitionConfig = { + encoding: string; + sampleRateHertz?: number; + languageCode: string; + enableAutomaticPunctuation?: boolean; + enableWordTimeOffsets?: boolean; + enableWordConfidence?: boolean; + model?: string; + useEnhanced?: boolean; + maxAlternatives?: number; + profanityFilter?: boolean; + enableSpeakerDiarization?: boolean; + diarizationSpeakerCount?: number; +}; + +export type GoogleRecognitionAudio = { + content: string; +}; + +// --- Whisper --- + +export type WhisperTranscriptionWord = { + word: string; + start: number; + end: number; +}; + +export type WhisperTranscriptionSegment = { + id: number; + seek: number; + start: number; + end: number; + text: string; + tokens: number[]; + temperature: number; + avg_logprob: number; + compression_ratio: number; + no_speech_prob: number; +}; + +export type WhisperVerboseResponse = { + task: string; + language: string; + duration: number; + text: string; + segments?: WhisperTranscriptionSegment[]; + words?: WhisperTranscriptionWord[]; +}; + +export type WhisperSimpleResponse = { + text: string; +}; + +// --- ElevenLabs --- + +export type ElevenLabsVoice = { + voice_id: string; + name: string; + category: string; + labels?: { + accent?: string; + description?: string; + age?: string; + gender?: string; + use_case?: string; + }; + preview_url?: string; +}; + +export type ElevenLabsVoicesResponse = { + voices: ElevenLabsVoice[]; +}; + +// --- Azure TTS --- + +export type AzureVoiceInfo = { + Name: string; + DisplayName: string; + LocalName: string; + ShortName: string; + Gender: string; + Locale: string; + LocaleName: string; + VoiceType: string; + Status: string; + WordsPerMinute?: string; +}; + +// --- Google TTS --- + +export type GoogleAudioConfig = { + audioEncoding: string; + speakingRate?: number; + pitch?: number; + volumeGainDb?: number; + sampleRateHertz?: number; + effectsProfileId?: string[]; +}; + +export type GoogleVoiceSelectionParams = { + languageCode: string; + name?: string; + ssmlGender?: string; +}; + +export type GoogleSynthesisInput = { + text?: string; + ssml?: string; +}; + +export type GoogleSynthesizeRequest = { + input: GoogleSynthesisInput; + voice: GoogleVoiceSelectionParams; + audioConfig: GoogleAudioConfig; +}; + +export type GoogleVoiceInfo = { + languageCodes: string[]; + name: string; + ssmlGender: string; + naturalSampleRateHertz: number; +}; + +export type GoogleListVoicesResponse = { + voices: GoogleVoiceInfo[]; +}; + +export type GoogleSynthesizeResponse = { + audioContent: string; +}; + +// --- OpenAI Realtime --- + +export type OpenAIRealtimeEvent = { + type: string; + event_id?: string; + [key: string]: unknown; +}; + +export type OpenAISessionCreated = OpenAIRealtimeEvent & { + type: "session.created"; + session: { + id: string; + object: string; + model: string; + modalities: string[]; + voice: string; + input_audio_format: string; + output_audio_format: string; + turn_detection: { + type: string; + threshold?: number; + prefix_padding_ms?: number; + silence_duration_ms?: number; + }; + tools: unknown[]; + tool_choice: string; + temperature: number; + max_response_output_tokens: string | number; + }; +}; + +export type OpenAIAudioDelta = OpenAIRealtimeEvent & { + type: "response.audio.delta"; + response_id: string; + item_id: string; + output_index: number; + content_index: number; + delta: string; // base64 audio +}; + +export type OpenAITranscriptDelta = OpenAIRealtimeEvent & { + type: + | "response.audio_transcript.delta" + | "conversation.item.input_audio_transcription.completed"; + delta?: string; + transcript?: string; +}; + +// --- Gemini Live --- + +export type GeminiMessage = { + setup?: { + model: string; + generationConfig?: { + responseModalities?: string[]; + speechConfig?: { + voiceConfig?: { + prebuiltVoiceConfig?: { + voiceName?: string; + }; + }; + }; + }; + systemInstruction?: { + parts: Array<{ text: string }>; + }; + tools?: unknown[]; + }; + realtimeInput?: { + mediaChunks: Array<{ + mimeType: string; + data: string; // base64 + }>; + }; + clientContent?: { + turns: Array<{ + role: string; + parts: Array<{ text: string }>; + }>; + turnComplete: boolean; + }; +}; + +export type GeminiResponse = { + setupComplete?: Record; + serverContent?: { + modelTurn?: { + parts: Array<{ + text?: string; + inlineData?: { + mimeType: string; + data: string; + }; + }>; + }; + turnComplete?: boolean; + interrupted?: boolean; + }; + toolCall?: { + functionCalls: Array<{ + id: string; + name: string; + args: Record; + }>; + }; + toolCallCancellation?: { + ids: string[]; + }; +}; diff --git a/src/lib/types/tts.ts b/src/lib/types/tts.ts index ee4c3eb47..5fe1b45f6 100644 --- a/src/lib/types/tts.ts +++ b/src/lib/types/tts.ts @@ -7,9 +7,19 @@ */ /** - * Supported audio formats for TTS output + * Supported audio formats for TTS output and STT input */ -export type AudioFormat = "mp3" | "wav" | "ogg" | "opus"; +export type AudioFormat = + | "mp3" + | "wav" + | "ogg" + | "opus" + | "m4a" + | "flac" + | "webm" + | "mp4" + | "mpeg" + | "mpga"; /** * TTS quality settings @@ -67,6 +77,8 @@ export type TTSOptions = { output?: string; /** Auto-play audio after generation (default: false) */ play?: boolean; + /** Override TTS provider (e.g., "elevenlabs", "openai-tts", "azure-tts") */ + provider?: string; }; /** @@ -144,6 +156,12 @@ export const VALID_AUDIO_FORMATS: readonly AudioFormat[] = [ "wav", "ogg", "opus", + "m4a", + "flac", + "webm", + "mp4", + "mpeg", + "mpga", ]; /** Valid TTS quality levels as an array for runtime validation */ diff --git a/src/lib/types/voice.ts b/src/lib/types/voice.ts new file mode 100644 index 000000000..5c2ffd458 --- /dev/null +++ b/src/lib/types/voice.ts @@ -0,0 +1,484 @@ +/** + * Voice and Speech Type Definitions for NeuroLink + * + * Core voice types: capabilities, provider config, audio utilities, + * events, and provider abstractions. + * + * STT types are in ./stt.ts + * Realtime types are in ./realtime.ts + * TTS types are in ./tts.ts + * + * @module types/voice + */ + +// Re-export all TTS types +export * from "./tts.js"; +// Re-export all STT types +export * from "./stt.js"; +// Re-export all Realtime types +export * from "./realtime.js"; + +import type { AudioFormat, TTSOptions, TTSResult, TTSVoice } from "./tts.js"; +import type { TTSHandler } from "./common.js"; +import type { STTResult, STTHandler } from "./stt.js"; +import type { RealtimeHandler } from "./realtime.js"; + +// ============================================================================ +// VOICE CAPABILITY / PROVIDER TYPES +// ============================================================================ + +/** + * Voice capability types supported by providers + */ +export type VoiceCapability = "tts" | "stt" | "realtime" | "streaming"; + +/** + * Voice provider types + */ +export type VoiceProviderType = "tts" | "stt" | "realtime"; + +/** + * Voice provider name union type + */ +export type VoiceProviderName = + // TTS providers + | "google-tts" + | "elevenlabs" + | "openai-tts" + | "azure-tts" + | "sarvam" + | "murf" + | "playai" + | "speechify" + | "cartesia" + // STT providers + | "deepgram" + | "gladia" + | "whisper" + | "assemblyai" + | "google-stt" + | "azure-stt" + // Realtime providers + | "openai-realtime" + | "gemini-live"; + +/** + * Base voice provider configuration + */ +export type VoiceProviderConfig = { + /** Provider identifier */ + name: string; + /** API key or credentials */ + apiKey?: string; + /** Custom endpoint URL */ + baseUrl?: string; + /** Request timeout in milliseconds */ + timeout?: number; + /** Maximum retries for failed requests */ + maxRetries?: number; + /** Provider-specific options */ + options?: Record; +}; + +// ============================================================================ +// AUDIO UTILITY TYPES +// ============================================================================ + +/** + * Audio format details + */ +export type AudioFormatDetails = { + /** Format name */ + format: AudioFormat; + /** MIME type */ + mimeType: string; + /** File extension */ + extension: string; + /** Whether format supports streaming */ + supportsStreaming: boolean; + /** Typical sample rates */ + sampleRates: number[]; + /** Bit depths */ + bitDepths: number[]; +}; + +/** + * Audio conversion options + */ +export type AudioConversionOptions = { + /** Target format */ + targetFormat: AudioFormat; + /** Target sample rate */ + sampleRate?: number; + /** Target bit depth */ + bitDepth?: number; + /** Number of channels */ + channels?: number; + /** Normalize audio level */ + normalize?: boolean; +}; + +/** + * Audio stream chunk for streaming operations + */ +export type AudioStreamChunk = { + /** Audio data */ + data: Buffer; + /** Chunk index */ + index: number; + /** Whether this is the final chunk */ + isFinal: boolean; + /** Audio format */ + format: AudioFormat; + /** Sample rate */ + sampleRate: number; + /** Timestamp offset in milliseconds */ + timestampMs: number; + /** Duration of this chunk in milliseconds */ + durationMs: number; +}; + +// ============================================================================ +// VOICE EVENT TYPES +// ============================================================================ + +/** + * Voice event types for event-driven architectures + */ +export type VoiceEventType = + | "synthesis.started" + | "synthesis.progress" + | "synthesis.completed" + | "synthesis.error" + | "transcription.started" + | "transcription.partial" + | "transcription.completed" + | "transcription.error" + | "realtime.connected" + | "realtime.audio.received" + | "realtime.text.received" + | "realtime.disconnected" + | "realtime.error"; + +/** + * Voice event for event-driven operations + */ +export type VoiceEvent = { + type: VoiceEventType; + timestamp: Date; + provider: VoiceProviderName; + data: T; + metadata?: Record; +}; + +/** + * Voice operation result union + */ +export type VoiceResult = TTSResult | STTResult; + +/** + * Voice conversation turn + */ +export type VoiceTurn = { + role: "user" | "assistant"; + text: string; + audio?: Buffer; + timestamp: Date; + metadata?: { + duration?: number; + confidence?: number; + language?: string; + provider?: string; + voice?: string; + [key: string]: unknown; + }; +}; + +// ============================================================================ +// VOICE PROVIDER TYPES +// ============================================================================ + +/** + * TTS-capable voice provider type + */ +export type TTSProvider = { + /** + * Synthesize text to speech + */ + synthesize(text: string, options: TTSOptions): Promise; + + /** + * Stream synthesized audio chunks + */ + synthesizeStream?( + text: string, + options: TTSOptions, + ): AsyncIterable; + + /** + * Get available voices + */ + getVoices(languageCode?: string): Promise; + + /** + * Maximum text length supported + */ + readonly maxTextLength: number; +}; + +// ============================================================================ +// TTS STREAM CHUNK +// ============================================================================ + +/** + * TTS stream chunk for streaming synthesis + */ +export type TTSStreamChunk = { + /** Audio data chunk */ + data: Buffer; + /** Chunk sequence number */ + index: number; + /** Whether this is the final chunk */ + isFinal: boolean; + /** Audio format */ + format: string; + /** Sample rate */ + sampleRate?: number; + /** Timestamp offset in audio (milliseconds) */ + timestampMs?: number; +}; + +// ============================================================================ +// ERROR CODES +// ============================================================================ + +/** + * Voice error codes (general) + */ +export const VOICE_ERROR_CODES = { + PROVIDER_NOT_FOUND: "VOICE_PROVIDER_NOT_FOUND", + INVALID_CONFIGURATION: "VOICE_INVALID_CONFIGURATION", + INITIALIZATION_FAILED: "VOICE_INITIALIZATION_FAILED", + OPERATION_CANCELLED: "VOICE_OPERATION_CANCELLED", + // General + PROVIDER_NOT_CONFIGURED: "VOICE_PROVIDER_NOT_CONFIGURED", + PROVIDER_NOT_SUPPORTED: "VOICE_PROVIDER_NOT_SUPPORTED", + FEATURE_NOT_SUPPORTED: "VOICE_FEATURE_NOT_SUPPORTED", + // TTS specific + TTS_EMPTY_TEXT: "VOICE_TTS_EMPTY_TEXT", + TTS_TEXT_TOO_LONG: "VOICE_TTS_TEXT_TOO_LONG", + TTS_SYNTHESIS_FAILED: "VOICE_TTS_SYNTHESIS_FAILED", + // STT specific + STT_EMPTY_AUDIO: "VOICE_STT_EMPTY_AUDIO", + STT_INVALID_FORMAT: "VOICE_STT_INVALID_FORMAT", + STT_TRANSCRIPTION_FAILED: "VOICE_STT_TRANSCRIPTION_FAILED", + // Realtime specific + REALTIME_CONNECTION_FAILED: "VOICE_REALTIME_CONNECTION_FAILED", + REALTIME_SESSION_ERROR: "VOICE_REALTIME_SESSION_ERROR", + // Network + NETWORK_ERROR: "VOICE_NETWORK_ERROR", + TIMEOUT: "VOICE_TIMEOUT", +} as const; + +// ============================================================================ +// CONSTANTS +// ============================================================================ + +/** + * Supported audio formats with details + */ +export const AUDIO_FORMAT_DETAILS: Partial< + Record +> = { + mp3: { + format: "mp3", + mimeType: "audio/mpeg", + extension: ".mp3", + supportsStreaming: true, + sampleRates: [8000, 16000, 22050, 24000, 44100, 48000], + bitDepths: [16], + }, + wav: { + format: "wav", + mimeType: "audio/wav", + extension: ".wav", + supportsStreaming: false, + sampleRates: [8000, 16000, 22050, 24000, 44100, 48000], + bitDepths: [8, 16, 24, 32], + }, + ogg: { + format: "ogg", + mimeType: "audio/ogg", + extension: ".ogg", + supportsStreaming: true, + sampleRates: [8000, 16000, 22050, 24000, 44100, 48000], + bitDepths: [16], + }, + opus: { + format: "opus", + mimeType: "audio/opus", + extension: ".opus", + supportsStreaming: true, + sampleRates: [8000, 12000, 16000, 24000, 48000], + bitDepths: [16], + }, + m4a: { + format: "m4a", + mimeType: "audio/mp4", + extension: ".m4a", + supportsStreaming: false, + sampleRates: [44100, 48000], + bitDepths: [16], + }, + flac: { + format: "flac", + mimeType: "audio/flac", + extension: ".flac", + supportsStreaming: false, + sampleRates: [44100, 48000, 96000], + bitDepths: [16, 24], + }, + webm: { + format: "webm", + mimeType: "audio/webm", + extension: ".webm", + supportsStreaming: true, + sampleRates: [44100, 48000], + bitDepths: [16], + }, + mp4: { + format: "mp4", + mimeType: "audio/mp4", + extension: ".mp4", + supportsStreaming: false, + sampleRates: [44100, 48000], + bitDepths: [16], + }, + mpeg: { + format: "mpeg", + mimeType: "audio/mpeg", + extension: ".mpeg", + supportsStreaming: true, + sampleRates: [8000, 16000, 22050, 24000, 44100, 48000], + bitDepths: [16], + }, + mpga: { + format: "mpga", + mimeType: "audio/mpeg", + extension: ".mpga", + supportsStreaming: true, + sampleRates: [8000, 16000, 22050, 24000, 44100, 48000], + bitDepths: [16], + }, +}; + +// ============================================================================ +// VOICE ERROR TYPES +// ============================================================================ + +import type { ErrorCategory, ErrorSeverity } from "../constants/enums.js"; + +export type VoiceErrorOptions = { + code: string; + message: string; + category?: ErrorCategory; + severity?: ErrorSeverity; + retriable?: boolean; + context?: Record; + originalError?: Error; + provider?: string; +}; + +// ============================================================================ +// AUDIO UTILITY TYPES (INTERNAL) +// ============================================================================ + +export type AudioMetadata = { + format: AudioFormat; + duration: number; + sampleRate: number; + channels: number; + bitDepth: number; + samples: number; + size: number; +}; + +// ============================================================================ +// STREAM HANDLER TYPES +// ============================================================================ + +export type StreamHandlerConfig = { + chunkDurationMs?: number; + sampleRate?: number; + bytesPerSample?: number; + format?: AudioFormat; + highWaterMark?: number; + bufferTimeoutMs?: number; +}; + +export type StreamEvents = { + chunk: (chunk: AudioStreamChunk) => void; + end: () => void; + error: (error: Error) => void; + drain: () => void; + pause: () => void; + resume: () => void; +}; + +// ============================================================================ +// VOICE HANDLER TYPES +// ============================================================================ + +export type VoiceHandler = TTSHandler | STTHandler | RealtimeHandler; + +// ============================================================================ +// PROVIDER-SPECIFIC TTS OPTION TYPES +// ============================================================================ + +export type AzureTTSOptions = TTSOptions & { + useSSML?: boolean; + ssmlTemplate?: string; + outputFormat?: string; + wordBoundary?: boolean; +}; + +export type ElevenLabsModel = + | "eleven_multilingual_v2" + | "eleven_turbo_v2_5" + | "eleven_turbo_v2" + | "eleven_monolingual_v1"; + +export type ElevenLabsTTSOptions = TTSOptions & { + model?: ElevenLabsModel; + stability?: number; + similarityBoost?: number; + style?: number; + useSpeakerBoost?: boolean; +}; + +export type GoogleVoiceType = + | "Standard" + | "WaveNet" + | "Neural2" + | "Studio" + | "Polyglot"; + +export type GoogleTTSOptions = TTSOptions & { + voiceType?: GoogleVoiceType; + sampleRateHertz?: number; + effectsProfileId?: string[]; +}; + +export type OpenAIVoice = + | "alloy" + | "echo" + | "fable" + | "onyx" + | "nova" + | "shimmer"; + +export type OpenAITTSModel = "tts-1" | "tts-1-hd"; + +export type OpenAITTSOptions = TTSOptions & { + model?: OpenAITTSModel; +}; diff --git a/src/lib/utils/sttProcessor.ts b/src/lib/utils/sttProcessor.ts new file mode 100644 index 000000000..2c759f5c7 --- /dev/null +++ b/src/lib/utils/sttProcessor.ts @@ -0,0 +1,319 @@ +/** + * Speech-to-Text (STT) Processing Utility + * + * Central orchestrator for all STT operations across providers. + * Manages provider-specific STT handlers and audio transcription. + * + * @module utils/sttProcessor + */ + +import { logger } from "./logger.js"; +import type { STTOptions, STTResult, STTHandler } from "../types/index.js"; +import { STT_ERROR_CODES } from "../types/index.js"; +import { ErrorCategory, ErrorSeverity } from "../constants/enums.js"; +import { NeuroLinkError } from "./errorHandling.js"; +import { + SpanSerializer, + SpanType, + SpanStatus, + getMetricsAggregator, +} from "../observability/index.js"; + +/** + * STT Error class for speech-to-text specific errors + */ +export class STTError extends NeuroLinkError { + constructor(options: { + code: string; + message: string; + category?: ErrorCategory; + severity?: ErrorSeverity; + retriable?: boolean; + context?: Record; + originalError?: Error; + }) { + super({ + code: options.code, + message: options.message, + category: options.category ?? ErrorCategory.VALIDATION, + severity: options.severity ?? ErrorSeverity.MEDIUM, + retriable: options.retriable ?? false, + context: options.context, + originalError: options.originalError, + }); + this.name = "STTError"; + } +} + +/** + * STT processor class for orchestrating speech-to-text operations + * + * Follows the same pattern as TTSProcessor, CSVProcessor, ImageProcessor, and PDFProcessor. + * Provides a unified interface for STT transcription across multiple providers. + * + * @example + * ```typescript + * // Register a handler + * STTProcessor.registerHandler('whisper', whisperHandler); + * + * // Check if provider is supported + * if (STTProcessor.supports('whisper')) { + * // Provider is registered + * } + * ``` + */ +export class STTProcessor { + /** + * Handler registry mapping provider names to STT handlers + * Uses Map for O(1) lookups and better type safety + * + * @private + */ + private static readonly handlers = new Map(); + + /** + * Default maximum audio duration for STT transcription (in seconds) + * + * Providers can override this value by specifying the `maxAudioDuration` property + * in their respective `STTHandler` implementation. If not specified, this default + * value will be used (5 minutes). + * + * @private + */ + private static readonly DEFAULT_MAX_AUDIO_DURATION = 300; + + /** + * Register an STT handler for a specific provider + * + * Allows providers to register their STT implementation at runtime. + * + * @param providerName - Provider identifier (e.g., 'whisper', 'deepgram') + * @param handler - STT handler implementation + * + * @example + * ```typescript + * const whisperHandler: STTHandler = { + * transcribe: async (audio, options) => { ... }, + * getSupportedFormats: () => ["mp3", "wav"], + * isConfigured: () => true + * }; + * + * STTProcessor.registerHandler('whisper', whisperHandler); + * ``` + */ + static registerHandler(providerName: string, handler: STTHandler): void { + if (!providerName) { + throw new Error("Provider name is required"); + } + + if (!handler) { + throw new Error("Handler is required"); + } + + const normalizedName = providerName.toLowerCase(); + + if (this.handlers.has(normalizedName)) { + logger.warn( + `[STTProcessor] Overwriting existing handler for provider: ${normalizedName}`, + ); + } + + this.handlers.set(normalizedName, handler); + logger.debug( + `[STTProcessor] Registered STT handler for provider: ${normalizedName}`, + ); + } + + /** + * Get a registered STT handler by provider name + * + * @private + * @param providerName - Provider identifier + * @returns Handler instance or undefined if not registered + */ + private static getHandler(providerName: string): STTHandler | undefined { + const normalizedName = providerName.toLowerCase(); + return this.handlers.get(normalizedName); + } + + /** + * Check if a provider is supported (has a registered STT handler) + * + * @param providerName - Provider identifier + * @returns True if handler is registered + * + * @example + * ```typescript + * if (STTProcessor.supports('whisper')) { + * console.log('Whisper STT is supported'); + * } + * ``` + */ + static supports(providerName: string): boolean { + if (!providerName) { + logger.error( + "[STTProcessor] Provider name is required for supports check", + ); + return false; + } + + const normalizedName = providerName.toLowerCase(); + const isSupported = this.handlers.has(normalizedName); + + if (!isSupported) { + logger.debug(`[STTProcessor] Provider ${providerName} is not supported`); + } + + return isSupported; + } + + /** + * Transcribe audio to text using a registered STT provider + * + * Orchestrates the speech-to-text transcription process: + * 1. Validates audio input (non-empty) + * 2. Looks up the provider handler + * 3. Verifies provider configuration + * 4. Delegates transcription to the provider + * 5. Enriches result with provider metadata + * + * @param audio - Audio data as Buffer or ArrayBuffer + * @param provider - Provider identifier + * @param options - STT configuration options + * @returns Transcription result with text and metadata + * @throws STTError if validation fails or provider not supported/configured + * + * @example + * ```typescript + * const result = await STTProcessor.transcribe(audioBuffer, "whisper", { + * language: "en-US", + * punctuation: true, + * }); + * + * console.log(`Transcription: ${result.text}`); + * console.log(`Confidence: ${result.confidence}`); + * ``` + */ + static async transcribe( + audio: Buffer | ArrayBuffer, + provider: string, + options: STTOptions, + ): Promise { + // Create span early so preflight failures are captured + const span = SpanSerializer.createSpan(SpanType.STT, "stt.transcribe", { + "stt.operation": "transcribe", + "stt.provider": provider, + "stt.language": options.language, + "stt.format": options.format, + }); + + try { + // 1. Audio validation: reject empty audio + const byteLength = + audio instanceof ArrayBuffer ? audio.byteLength : audio.byteLength; + if (!byteLength || byteLength === 0) { + logger.error("[STTProcessor] Audio data is required for transcription"); + throw new STTError({ + code: STT_ERROR_CODES.AUDIO_EMPTY, + message: "Audio data is required for STT transcription", + severity: ErrorSeverity.LOW, + retriable: false, + context: { provider }, + }); + } + + // 2. Handler lookup and error if provider not supported + const handler = this.getHandler(provider); + if (!handler) { + logger.error(`[STTProcessor] Provider "${provider}" is not registered`); + throw new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_SUPPORTED, + message: `STT provider "${provider}" is not supported. Use STTProcessor.registerHandler() to register it.`, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { + provider, + availableProviders: Array.from(this.handlers.keys()), + }, + }); + } + + // 3. Configuration check + if (!handler.isConfigured()) { + logger.warn( + `[STTProcessor] Provider "${provider}" is not properly configured`, + ); + throw new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: `STT provider "${provider}" is not configured. Please set the required API keys.`, + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider }, + }); + } + + logger.debug( + `[STTProcessor] Starting transcription with provider: ${provider}`, + ); + + // 4. Call handler.transcribe() - providers handle their own timeouts + const result = await handler.transcribe(audio, options); + + // 5. Post-processing: enrich result with provider metadata + const enrichedResult: STTResult = { + ...result, + metadata: { + ...result.metadata, + provider, + latency: result.metadata?.latency ?? 0, + }, + }; + + logger.info( + `[STTProcessor] Successfully transcribed audio: "${result.text.substring(0, 80)}${result.text.length > 80 ? "..." : ""}"`, + ); + + // 6. Record successful span + const endedSpan = SpanSerializer.endSpan(span, SpanStatus.OK); + getMetricsAggregator().recordSpan(endedSpan); + + // 7. Return STTResult with text, confidence, metadata + return enrichedResult; + } catch (err: unknown) { + // Record error span + const endedSpan = SpanSerializer.endSpan( + span, + SpanStatus.ERROR, + err instanceof Error ? err.message : String(err), + ); + getMetricsAggregator().recordSpan(endedSpan); + + // Re-throw STTError as-is + if (err instanceof STTError) { + throw err; + } + + // Wrap other errors in STTError + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error( + `[STTProcessor] Transcription failed for provider "${provider}": ${errorMessage}`, + ); + throw new STTError({ + code: STT_ERROR_CODES.TRANSCRIPTION_FAILED, + message: `STT transcription failed for provider "${provider}": ${errorMessage}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { + provider, + audioByteLength: + audio instanceof ArrayBuffer ? audio.byteLength : audio.byteLength, + options, + }, + originalError: err instanceof Error ? err : undefined, + }); + } + } +} diff --git a/src/lib/voice/RealtimeVoiceAPI.ts b/src/lib/voice/RealtimeVoiceAPI.ts new file mode 100644 index 000000000..46b0977c5 --- /dev/null +++ b/src/lib/voice/RealtimeVoiceAPI.ts @@ -0,0 +1,516 @@ +/** + * Realtime Voice API Infrastructure + * + * Base handler and processor for realtime voice communication. + * Supports bidirectional audio streaming with providers like OpenAI and Gemini. + * + * @module voice/RealtimeVoiceAPI + */ + +import { logger } from "../utils/logger.js"; +import type { + AudioFormat, + RealtimeAudioChunk, + RealtimeConfig, + RealtimeEventHandlers, + RealtimeHandler, + RealtimeSession, + RealtimeSessionState, +} from "../types/index.js"; +import { RealtimeError } from "./errors.js"; +import { + DEFAULT_REALTIME_CONFIG, + REALTIME_ERROR_CODES, +} from "../types/index.js"; +import { ErrorCategory, ErrorSeverity } from "../constants/enums.js"; + +/** + * Realtime Processor class for orchestrating realtime voice operations + * + * Provides a unified interface for realtime voice across multiple providers. + * + * @example + * ```typescript + * // Register a handler + * RealtimeProcessor.registerHandler('openai', openaiHandler); + * + * // Connect to a session + * const session = await RealtimeProcessor.connect('openai', { + * provider: 'openai', + * voice: 'alloy', + * systemPrompt: 'You are a helpful assistant.' + * }); + * + * // Send audio + * await RealtimeProcessor.sendAudio('openai', audioBuffer); + * + * // Disconnect + * await RealtimeProcessor.disconnect('openai'); + * ``` + */ +export class RealtimeProcessor { + /** + * Handler registry mapping provider names to Realtime handlers + */ + private static readonly handlers = new Map(); + + /** + * Active sessions by provider + */ + private static readonly sessions = new Map(); + + /** + * Register a Realtime handler for a specific provider + * + * @param providerName - Provider identifier (e.g., 'openai', 'gemini') + * @param handler - Realtime handler implementation + */ + static registerHandler(providerName: string, handler: RealtimeHandler): void { + if (!providerName) { + throw new Error("Provider name is required"); + } + + if (!handler) { + throw new Error("Handler is required"); + } + + const normalizedName = providerName.toLowerCase(); + + if (this.handlers.has(normalizedName)) { + logger.warn( + `[RealtimeProcessor] Overwriting existing handler for provider: ${normalizedName}`, + ); + } + + this.handlers.set(normalizedName, handler); + logger.debug( + `[RealtimeProcessor] Registered Realtime handler for provider: ${normalizedName}`, + ); + } + + /** + * Get a registered Realtime handler by provider name + */ + private static getHandler(providerName: string): RealtimeHandler | undefined { + const normalizedName = providerName.toLowerCase(); + return this.handlers.get(normalizedName); + } + + /** + * Check if a provider is supported + */ + static supports(providerName: string): boolean { + if (!providerName) { + return false; + } + + const normalizedName = providerName.toLowerCase(); + return this.handlers.has(normalizedName); + } + + /** + * Get list of all registered providers + */ + static getProviders(): string[] { + return Array.from(this.handlers.keys()); + } + + /** + * Connect to a realtime session + * + * @param provider - Provider identifier + * @param config - Session configuration + * @param handlers - Event handlers + * @returns Session information + */ + static async connect( + provider: string, + config: RealtimeConfig, + handlers?: RealtimeEventHandlers, + ): Promise { + const handler = this.getHandler(provider); + + if (!handler) { + throw RealtimeError.providerNotSupported( + provider, + Array.from(this.handlers.keys()), + ); + } + + if (!handler.isConfigured()) { + throw RealtimeError.providerNotConfigured(provider); + } + + // Check for existing session + if (handler.isConnected()) { + throw RealtimeError.sessionAlreadyActive(provider); + } + + // Merge with defaults + const mergedConfig: RealtimeConfig = { + ...DEFAULT_REALTIME_CONFIG, + ...config, + }; + + // Register event handlers if provided + if (handlers) { + handler.on(handlers); + } + + try { + logger.debug(`[RealtimeProcessor] Connecting to provider: ${provider}`); + + const session = await handler.connect(mergedConfig); + this.sessions.set(provider.toLowerCase(), session); + + logger.info( + `[RealtimeProcessor] Connected to ${provider} session: ${session.id}`, + ); + + return session; + } catch (err: unknown) { + if (handlers) { + handler.off(); + } + + if (err instanceof RealtimeError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + throw RealtimeError.connectionFailed( + errorMessage, + provider, + err instanceof Error ? err : undefined, + ); + } + } + + /** + * Disconnect from a realtime session + * + * @param provider - Provider identifier + */ + static async disconnect(provider: string): Promise { + const handler = this.getHandler(provider); + + if (!handler) { + throw RealtimeError.providerNotSupported( + provider, + Array.from(this.handlers.keys()), + ); + } + + if (!handler.isConnected()) { + logger.warn( + `[RealtimeProcessor] No active session for provider: ${provider}`, + ); + return; + } + + try { + await handler.disconnect(); + this.sessions.delete(provider.toLowerCase()); + handler.off(); + + logger.info(`[RealtimeProcessor] Disconnected from ${provider}`); + } catch (err: unknown) { + if (err instanceof RealtimeError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + throw RealtimeError.protocolError( + `Disconnect failed: ${errorMessage}`, + provider, + err instanceof Error ? err : undefined, + ); + } + } + + /** + * Send audio to a realtime session + * + * @param provider - Provider identifier + * @param audio - Audio data + */ + static async sendAudio( + provider: string, + audio: Buffer | RealtimeAudioChunk, + ): Promise { + const handler = this.getHandler(provider); + + if (!handler) { + throw RealtimeError.providerNotSupported( + provider, + Array.from(this.handlers.keys()), + ); + } + + if (!handler.isConnected()) { + throw RealtimeError.sessionNotActive(provider); + } + + try { + await handler.sendAudio(audio); + } catch (err: unknown) { + if (err instanceof RealtimeError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + throw RealtimeError.audioStreamError(errorMessage, provider); + } + } + + /** + * Send text to a realtime session + * + * @param provider - Provider identifier + * @param text - Text to send + */ + static async sendText(provider: string, text: string): Promise { + const handler = this.getHandler(provider); + + if (!handler) { + throw RealtimeError.providerNotSupported( + provider, + Array.from(this.handlers.keys()), + ); + } + + if (!handler.isConnected()) { + throw RealtimeError.sessionNotActive(provider); + } + + if (!handler.sendText) { + throw new RealtimeError({ + code: REALTIME_ERROR_CODES.PROTOCOL_ERROR, + message: `Provider "${provider}" does not support text input`, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + context: { provider }, + }); + } + + await handler.sendText(text); + } + + /** + * Trigger a response from the model (manual turn detection) + * + * @param provider - Provider identifier + */ + static async triggerResponse(provider: string): Promise { + const handler = this.getHandler(provider); + + if (!handler) { + throw RealtimeError.providerNotSupported( + provider, + Array.from(this.handlers.keys()), + ); + } + + if (!handler.isConnected()) { + throw RealtimeError.sessionNotActive(provider); + } + + if (handler.triggerResponse) { + await handler.triggerResponse(); + } + } + + /** + * Cancel the current response + * + * @param provider - Provider identifier + */ + static async cancelResponse(provider: string): Promise { + const handler = this.getHandler(provider); + + if (!handler) { + throw RealtimeError.providerNotSupported( + provider, + Array.from(this.handlers.keys()), + ); + } + + if (!handler.isConnected()) { + return; // Nothing to cancel + } + + if (handler.cancelResponse) { + await handler.cancelResponse(); + } + } + + /** + * Get current session for a provider + * + * @param provider - Provider identifier + * @returns Session or null + */ + static getSession(provider: string): RealtimeSession | null { + const handler = this.getHandler(provider); + return handler?.getSession() ?? null; + } + + /** + * Check if a provider has an active session + * + * @param provider - Provider identifier + */ + static isConnected(provider: string): boolean { + const handler = this.getHandler(provider); + return handler?.isConnected() ?? false; + } + + /** + * Get supported formats for a provider + * + * @param provider - Provider identifier + */ + static getSupportedFormats(provider: string): AudioFormat[] { + const handler = this.getHandler(provider); + return handler?.getSupportedFormats() ?? []; + } + + /** + * Clear all handlers and sessions (for testing) + */ + static clearHandlers(): void { + // Disconnect all active sessions + for (const [provider] of this.sessions) { + const handler = this.handlers.get(provider); + if (handler?.isConnected()) { + handler.disconnect().catch(() => { + // Ignore errors during cleanup + }); + } + } + + this.handlers.clear(); + this.sessions.clear(); + logger.debug("[RealtimeProcessor] Cleared all handlers and sessions"); + } +} + +/** + * Base Realtime Handler with common functionality + * + * Providers can extend this class for common behavior. + */ +export abstract class BaseRealtimeHandler implements RealtimeHandler { + abstract readonly name: string; + + protected session: RealtimeSession | null = null; + protected eventHandlers: RealtimeEventHandlers | null = null; + protected state: RealtimeSessionState = "disconnected"; + + abstract connect(config: RealtimeConfig): Promise; + abstract disconnect(): Promise; + abstract sendAudio(audio: Buffer | RealtimeAudioChunk): Promise; + abstract isConfigured(): boolean; + abstract getSupportedFormats(): AudioFormat[]; + + isConnected(): boolean { + return this.state === "connected"; + } + + getSession(): RealtimeSession | null { + return this.session; + } + + on(handlers: RealtimeEventHandlers): void { + this.eventHandlers = handlers; + } + + off(): void { + this.eventHandlers = null; + } + + /** + * Emit state change event + */ + protected emitStateChange(newState: RealtimeSessionState): void { + this.state = newState; + if (this.session) { + this.session.state = newState; + this.session.lastActivityAt = new Date(); + } + this.eventHandlers?.onStateChange?.(newState); + } + + /** + * Emit audio event + */ + protected emitAudio(chunk: RealtimeAudioChunk): void { + this.eventHandlers?.onAudio?.(chunk); + } + + /** + * Emit transcript event + */ + protected emitTranscript(text: string, isFinal: boolean): void { + this.eventHandlers?.onTranscript?.(text, isFinal); + } + + /** + * Emit text event + */ + protected emitText(text: string, isFinal: boolean): void { + this.eventHandlers?.onText?.(text, isFinal); + } + + /** + * Emit function call event + */ + protected async emitFunctionCall( + name: string, + args: Record, + ): Promise { + if (this.eventHandlers?.onFunctionCall) { + return this.eventHandlers.onFunctionCall(name, args); + } + return undefined; + } + + /** + * Emit error event + */ + protected emitError(error: Error): void { + this.eventHandlers?.onError?.(error); + } + + /** + * Emit turn start event + */ + protected emitTurnStart(): void { + this.eventHandlers?.onTurnStart?.(); + } + + /** + * Emit turn end event + */ + protected emitTurnEnd(): void { + this.eventHandlers?.onTurnEnd?.(); + } + + /** + * Create a session object + */ + protected createSession(id: string, config: RealtimeConfig): RealtimeSession { + return { + id, + state: "connected", + provider: this.name, + model: config.model, + createdAt: new Date(), + lastActivityAt: new Date(), + config, + }; + } +} diff --git a/src/lib/voice/audio-utils.ts b/src/lib/voice/audio-utils.ts new file mode 100644 index 000000000..936083d3b --- /dev/null +++ b/src/lib/voice/audio-utils.ts @@ -0,0 +1,552 @@ +/** + * Audio Utilities for Voice Module + * + * Provides audio format conversion, duration calculation, and buffer utilities. + * + * @module voice/audio-utils + */ + +import type { AudioFormat } from "../types/index.js"; +import { AUDIO_FORMAT_DETAILS } from "../types/index.js"; +import { logger } from "../utils/logger.js"; + +/** + * Detect audio format from buffer + * + * @param buffer - Audio data buffer + * @returns Detected audio format or null + */ +export function detectAudioFormat(buffer: Buffer): AudioFormat | null { + if (buffer.length < 12) { + return null; + } + + // Check for WAV (RIFF header) + if ( + buffer[0] === 0x52 && // R + buffer[1] === 0x49 && // I + buffer[2] === 0x46 && // F + buffer[3] === 0x46 && // F + buffer[8] === 0x57 && // W + buffer[9] === 0x41 && // A + buffer[10] === 0x56 && // V + buffer[11] === 0x45 // E + ) { + return "wav"; + } + + // Check for MP3 (ID3 tag or frame sync) + if ( + (buffer[0] === 0x49 && buffer[1] === 0x44 && buffer[2] === 0x33) || // ID3 + (buffer[0] === 0xff && (buffer[1] & 0xe0) === 0xe0) // Frame sync + ) { + return "mp3"; + } + + // Check for OGG (OggS header) + if ( + buffer[0] === 0x4f && // O + buffer[1] === 0x67 && // g + buffer[2] === 0x67 && // g + buffer[3] === 0x53 // S + ) { + // Could be Opus or Vorbis, check for Opus header + // Opus has "OpusHead" in the first page + const opusOffset = buffer.indexOf("OpusHead"); + if (opusOffset !== -1 && opusOffset < 200) { + return "opus"; + } + return "ogg"; + } + + return null; +} + +/** + * Get MIME type for audio format + * + * @param format - Audio format + * @returns MIME type string + */ +export function getMimeType(format: AudioFormat): string { + return AUDIO_FORMAT_DETAILS[format]?.mimeType ?? "application/octet-stream"; +} + +/** + * Get file extension for audio format + * + * @param format - Audio format + * @returns File extension with dot + */ +export function getFileExtension(format: AudioFormat): string { + return AUDIO_FORMAT_DETAILS[format]?.extension ?? ".bin"; +} + +/** + * Calculate audio duration from buffer + * + * @param buffer - Audio data buffer + * @param format - Audio format (optional, will be detected if not provided) + * @param sampleRate - Sample rate in Hz (optional, will be extracted if possible) + * @returns Duration in seconds, or undefined if cannot be calculated + */ +export function calculateDuration( + buffer: Buffer, + format?: AudioFormat, + sampleRate?: number, +): number | undefined { + const detectedFormat = format ?? detectAudioFormat(buffer); + + if (!detectedFormat) { + return undefined; + } + + try { + switch (detectedFormat) { + case "wav": + return calculateWavDuration(buffer); + case "mp3": + return estimateMp3Duration(buffer); + case "ogg": + case "opus": + return estimateOpusDuration(buffer); + default: + // Estimate based on size and assumed bitrate + if (sampleRate) { + // Assume 16-bit mono + return buffer.length / (sampleRate * 2); + } + return undefined; + } + } catch (err) { + logger.debug( + `[audio-utils] Failed to calculate duration: ${err instanceof Error ? err.message : String(err)}`, + ); + return undefined; + } +} + +/** + * Calculate WAV duration from header + */ +function calculateWavDuration(buffer: Buffer): number | undefined { + if (buffer.length < 44) { + return undefined; + } + + // Find data chunk + let offset = 12; + while (offset < buffer.length - 8) { + const chunkId = buffer.toString("ascii", offset, offset + 4); + const chunkSize = buffer.readUInt32LE(offset + 4); + + if (chunkId === "fmt ") { + const channels = buffer.readUInt16LE(offset + 10); + const sampleRate = buffer.readUInt32LE(offset + 12); + const bitsPerSample = buffer.readUInt16LE(offset + 22); + + // Find data chunk size + let dataOffset = offset + 8 + chunkSize; + while (dataOffset < buffer.length - 8) { + const dataChunkId = buffer.toString( + "ascii", + dataOffset, + dataOffset + 4, + ); + const dataChunkSize = buffer.readUInt32LE(dataOffset + 4); + + if (dataChunkId === "data") { + const bytesPerSample = (bitsPerSample / 8) * channels; + const numSamples = dataChunkSize / bytesPerSample; + return numSamples / sampleRate; + } + + dataOffset += 8 + dataChunkSize; + } + } + + offset += 8 + chunkSize; + } + + return undefined; +} + +/** + * Estimate MP3 duration (approximate) + */ +function estimateMp3Duration(buffer: Buffer): number | undefined { + // This is a rough estimate based on file size and assumed bitrate + // For accurate duration, we would need to parse all frames + + // Check for ID3v2 tag and skip it + let offset = 0; + if (buffer[0] === 0x49 && buffer[1] === 0x44 && buffer[2] === 0x33) { + // ID3v2 tag present + const tagSize = + ((buffer[6] & 0x7f) << 21) | + ((buffer[7] & 0x7f) << 14) | + ((buffer[8] & 0x7f) << 7) | + (buffer[9] & 0x7f); + offset = 10 + tagSize; + } + + // Find first MP3 frame header + while (offset < buffer.length - 4) { + if (buffer[offset] === 0xff && (buffer[offset + 1] & 0xe0) === 0xe0) { + // Found frame sync + const version = (buffer[offset + 1] >> 3) & 0x03; + const _layer = (buffer[offset + 1] >> 1) & 0x03; + const bitrateIndex = (buffer[offset + 2] >> 4) & 0x0f; + const sampleRateIndex = (buffer[offset + 2] >> 2) & 0x03; + + // Get sample rate + const sampleRates: Record = { + 3: [44100, 48000, 32000], // MPEG1 + 2: [22050, 24000, 16000], // MPEG2 + 0: [11025, 12000, 8000], // MPEG2.5 + }; + const sampleRate = sampleRates[version]?.[sampleRateIndex]; + + // Get bitrate (MPEG1 Layer III) + const bitrates = [ + 0, 32, 40, 48, 56, 64, 80, 96, 112, 128, 160, 192, 224, 256, 320, 0, + ]; + const bitrate = bitrates[bitrateIndex]; + + if (sampleRate && bitrate) { + // Estimate duration: (file_size_bits) / bitrate + const audioBytes = buffer.length - offset; + return (audioBytes * 8) / (bitrate * 1000); + } + + break; + } + offset++; + } + + // Fallback: assume 128kbps + return (buffer.length * 8) / 128000; +} + +/** + * Estimate Opus/OGG duration (approximate) + */ +function estimateOpusDuration(buffer: Buffer): number | undefined { + // Opus typically uses 48kHz, estimate based on typical bitrate + // For accurate duration, we would need to parse all pages + + // Assume average bitrate of 64kbps for voice + return (buffer.length * 8) / 64000; +} + +/** + * Convert audio format (basic conversion) + * + * Note: For full format conversion, external tools like ffmpeg would be needed. + * This provides basic PCM resampling only. + * + * @param buffer - Input audio buffer + * @param fromFormat - Source format + * @param toFormat - Target format + * @param options - Conversion options + * @returns Converted audio buffer + */ +export async function convertAudioFormat( + buffer: Buffer, + fromFormat: AudioFormat, + toFormat: AudioFormat, + _options: Record = {}, +): Promise { + // If formats are the same, just return the buffer + if (fromFormat === toFormat) { + return buffer; + } + + // For actual conversion, we would need to use external libraries + // This is a placeholder that logs a warning + logger.warn( + `[audio-utils] Audio format conversion from ${fromFormat} to ${toFormat} not fully implemented. ` + + `Consider using ffmpeg or similar tools for production use.`, + ); + + // Return original buffer + return buffer; +} + +/** + * Create PCM audio buffer from raw samples + * + * @param samples - Array of sample values (-1 to 1) + * @param sampleRate - Sample rate in Hz + * @param bitDepth - Bit depth (8, 16, 24, or 32) + * @returns PCM audio buffer + */ +export function createPcmBuffer( + samples: number[], + _sampleRate: number = 16000, + bitDepth: 8 | 16 | 24 | 32 = 16, +): Buffer { + const bytesPerSample = bitDepth / 8; + const buffer = Buffer.alloc(samples.length * bytesPerSample); + + for (let i = 0; i < samples.length; i++) { + const sample = Math.max(-1, Math.min(1, samples[i])); + const offset = i * bytesPerSample; + + switch (bitDepth) { + case 8: + buffer.writeUInt8(Math.round((sample + 1) * 127.5), offset); + break; + case 16: + buffer.writeInt16LE(Math.round(sample * 32767), offset); + break; + case 24: { + const val24 = Math.round(sample * 8388607); + buffer.writeUInt8(val24 & 0xff, offset); + buffer.writeUInt8((val24 >> 8) & 0xff, offset + 1); + buffer.writeUInt8((val24 >> 16) & 0xff, offset + 2); + break; + } + case 32: + buffer.writeInt32LE(Math.round(sample * 2147483647), offset); + break; + } + } + + return buffer; +} + +/** + * Extract PCM samples from buffer + * + * @param buffer - PCM audio buffer + * @param bitDepth - Bit depth (8, 16, 24, or 32) + * @returns Array of sample values (-1 to 1) + */ +export function extractPcmSamples( + buffer: Buffer, + bitDepth: 8 | 16 | 24 | 32 = 16, +): number[] { + const bytesPerSample = bitDepth / 8; + const numSamples = Math.floor(buffer.length / bytesPerSample); + const samples: number[] = []; + + for (let i = 0; i < numSamples; i++) { + const offset = i * bytesPerSample; + + switch (bitDepth) { + case 8: + samples.push(buffer.readUInt8(offset) / 127.5 - 1); + break; + case 16: + samples.push(buffer.readInt16LE(offset) / 32767); + break; + case 24: { + const val24 = + buffer.readUInt8(offset) | + (buffer.readUInt8(offset + 1) << 8) | + (buffer.readUInt8(offset + 2) << 16); + samples.push((val24 > 8388607 ? val24 - 16777216 : val24) / 8388607); + break; + } + case 32: + samples.push(buffer.readInt32LE(offset) / 2147483647); + break; + } + } + + return samples; +} + +/** + * Resample PCM audio + * + * @param samples - Input samples + * @param fromSampleRate - Source sample rate + * @param toSampleRate - Target sample rate + * @returns Resampled samples + */ +export function resamplePcm( + samples: number[], + fromSampleRate: number, + toSampleRate: number, +): number[] { + if (fromSampleRate <= 0 || toSampleRate <= 0) { + return samples; + } + if (fromSampleRate === toSampleRate) { + return samples; + } + + const ratio = fromSampleRate / toSampleRate; + const newLength = Math.round(samples.length / ratio); + const resampled: number[] = []; + + for (let i = 0; i < newLength; i++) { + const srcIndex = i * ratio; + const srcIndexFloor = Math.floor(srcIndex); + const srcIndexCeil = Math.min(srcIndexFloor + 1, samples.length - 1); + const fraction = srcIndex - srcIndexFloor; + + // Linear interpolation + const value = + samples[srcIndexFloor] * (1 - fraction) + + samples[srcIndexCeil] * fraction; + resampled.push(value); + } + + return resampled; +} + +/** + * Normalize audio levels + * + * @param samples - Input samples + * @param targetPeak - Target peak level (0 to 1) + * @returns Normalized samples + */ +export function normalizeAudio( + samples: number[], + targetPeak: number = 0.95, +): number[] { + if (samples.length === 0) { + return samples; + } + + // Find current peak + let peak = 0; + for (const sample of samples) { + peak = Math.max(peak, Math.abs(sample)); + } + + if (peak === 0) { + return samples; + } + + // Calculate gain + const gain = targetPeak / peak; + + // Apply gain + return samples.map((s) => s * gain); +} + +/** + * Create a WAV header + * + * @param dataSize - Size of audio data in bytes + * @param sampleRate - Sample rate in Hz + * @param channels - Number of channels + * @param bitDepth - Bit depth + * @returns WAV header buffer + */ +export function createWavHeader( + dataSize: number, + sampleRate: number = 16000, + channels: number = 1, + bitDepth: number = 16, +): Buffer { + const header = Buffer.alloc(44); + + const byteRate = sampleRate * channels * (bitDepth / 8); + const blockAlign = channels * (bitDepth / 8); + + // RIFF header + header.write("RIFF", 0); + header.writeUInt32LE(36 + dataSize, 4); + header.write("WAVE", 8); + + // fmt chunk + header.write("fmt ", 12); + header.writeUInt32LE(16, 16); // Subchunk1Size (PCM) + header.writeUInt16LE(1, 20); // AudioFormat (PCM) + header.writeUInt16LE(channels, 22); + header.writeUInt32LE(sampleRate, 24); + header.writeUInt32LE(byteRate, 28); + header.writeUInt16LE(blockAlign, 32); + header.writeUInt16LE(bitDepth, 34); + + // data chunk + header.write("data", 36); + header.writeUInt32LE(dataSize, 40); + + return header; +} + +/** + * Create a complete WAV file from PCM data + * + * @param pcmData - PCM audio data + * @param sampleRate - Sample rate in Hz + * @param channels - Number of channels + * @param bitDepth - Bit depth + * @returns Complete WAV file buffer + */ +export function createWavFile( + pcmData: Buffer, + sampleRate: number = 16000, + channels: number = 1, + bitDepth: number = 16, +): Buffer { + const header = createWavHeader( + pcmData.length, + sampleRate, + channels, + bitDepth, + ); + return Buffer.concat([header, pcmData]); +} + +/** + * Split audio buffer into chunks + * + * @param buffer - Audio buffer to split + * @param chunkDurationMs - Duration of each chunk in milliseconds + * @param sampleRate - Sample rate in Hz + * @param bytesPerSample - Bytes per sample (channels * bitDepth / 8) + * @returns Array of audio chunks + */ +export function splitIntoChunks( + buffer: Buffer, + chunkDurationMs: number, + sampleRate: number = 16000, + bytesPerSample: number = 2, +): Buffer[] { + if (chunkDurationMs <= 0 || sampleRate <= 0 || bytesPerSample <= 0) { + return [buffer]; + } + const bytesPerMs = (sampleRate * bytesPerSample) / 1000; + const chunkSize = Math.round(chunkDurationMs * bytesPerMs); + if (chunkSize <= 0) { + return [buffer]; + } + const chunks: Buffer[] = []; + + for (let offset = 0; offset < buffer.length; offset += chunkSize) { + const end = Math.min(offset + chunkSize, buffer.length); + chunks.push(buffer.subarray(offset, end)); + } + + return chunks; +} + +/** + * Audio format signatures for detection + */ +export const AUDIO_SIGNATURES = { + wav: Buffer.from([0x52, 0x49, 0x46, 0x46]), // RIFF + mp3: { + id3: Buffer.from([0x49, 0x44, 0x33]), // ID3 + frameSync: Buffer.from([0xff, 0xe0]), // Frame sync mask + }, + ogg: Buffer.from([0x4f, 0x67, 0x67, 0x53]), // OggS +} as const; + +/** + * MIME types for audio formats + */ +export const MIME_TYPES = { + wav: "audio/wav", + mp3: "audio/mpeg", + ogg: "audio/ogg", + opus: "audio/opus", +} as const; diff --git a/src/lib/voice/errors.ts b/src/lib/voice/errors.ts new file mode 100644 index 000000000..8acf0e965 --- /dev/null +++ b/src/lib/voice/errors.ts @@ -0,0 +1,464 @@ +/** + * Voice Module Error Classes + * + * Comprehensive error handling for TTS, STT, and Realtime Voice operations. + * + * @module voice/errors + */ + +import { ErrorCategory, ErrorSeverity } from "../constants/enums.js"; +import { NeuroLinkError } from "../utils/errorHandling.js"; +import type { VoiceErrorOptions } from "../types/index.js"; +import { + REALTIME_ERROR_CODES, + STT_ERROR_CODES, + VOICE_ERROR_CODES, +} from "../types/index.js"; + +// Re-export error codes for convenience +export { STT_ERROR_CODES, REALTIME_ERROR_CODES, VOICE_ERROR_CODES }; + +/** + * Base Voice Error class for all voice-related errors + */ +export class VoiceError extends NeuroLinkError { + constructor(options: VoiceErrorOptions) { + super({ + code: options.code, + message: options.message, + category: options.category ?? ErrorCategory.EXECUTION, + severity: options.severity ?? ErrorSeverity.MEDIUM, + retriable: options.retriable ?? false, + context: options.context, + originalError: options.originalError, + }); + this.name = "VoiceError"; + } +} + +/** + * STT Error class for speech-to-text specific errors + */ +export class STTError extends NeuroLinkError { + constructor(options: VoiceErrorOptions) { + super({ + code: options.code, + message: options.message, + category: options.category ?? ErrorCategory.VALIDATION, + severity: options.severity ?? ErrorSeverity.MEDIUM, + retriable: options.retriable ?? false, + context: options.context, + originalError: options.originalError, + }); + this.name = "STTError"; + } + + /** + * Create an error for empty audio input + */ + static audioEmpty(provider?: string): STTError { + return new STTError({ + code: STT_ERROR_CODES.AUDIO_EMPTY, + message: "Audio input is empty or invalid", + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.LOW, + retriable: false, + context: { provider }, + }); + } + + /** + * Create an error for audio that exceeds maximum duration + */ + static audioTooLong( + durationSeconds: number, + maxDurationSeconds: number, + provider?: string, + ): STTError { + return new STTError({ + code: STT_ERROR_CODES.AUDIO_TOO_LONG, + message: `Audio duration (${durationSeconds}s) exceeds maximum allowed (${maxDurationSeconds}s)`, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { durationSeconds, maxDurationSeconds, provider }, + }); + } + + /** + * Create an error for invalid audio format + */ + static invalidFormat( + format: string, + supportedFormatsOrProvider?: string[] | string, + provider?: string, + ): STTError { + // Handle overloaded signature: (format, provider) or (format, supportedFormats[], provider?) + let supportedFormats: string[] | undefined; + let actualProvider: string | undefined; + + if (typeof supportedFormatsOrProvider === "string") { + // Called as (format, provider) + actualProvider = supportedFormatsOrProvider; + } else { + // Called as (format, supportedFormats[], provider?) + supportedFormats = supportedFormatsOrProvider; + actualProvider = provider; + } + + const message = supportedFormats + ? `Unsupported audio format: ${format}. Supported formats: ${supportedFormats.join(", ")}` + : `Unsupported audio format: ${format}`; + return new STTError({ + code: STT_ERROR_CODES.INVALID_AUDIO_FORMAT, + message, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { format, supportedFormats, provider: actualProvider }, + }); + } + + /** + * Create an error for unsupported language + */ + static languageNotSupported( + language: string, + supportedLanguages?: string[], + provider?: string, + ): STTError { + const message = supportedLanguages + ? `Language "${language}" is not supported. Supported languages: ${supportedLanguages.slice(0, 10).join(", ")}${supportedLanguages.length > 10 ? "..." : ""}` + : `Language "${language}" is not supported by this provider`; + return new STTError({ + code: STT_ERROR_CODES.LANGUAGE_NOT_SUPPORTED, + message, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { language, supportedLanguages, provider }, + }); + } + + /** + * Create an error for transcription failure + * Supports two signatures: + * - transcriptionFailed(reason, provider?, originalError?) + * - transcriptionFailed(reason, originalError, provider) + */ + static transcriptionFailed( + reason: string, + providerOrError?: string | Error, + originalErrorOrProvider?: Error | string, + ): STTError { + let provider: string | undefined; + let originalError: Error | undefined; + + if (typeof providerOrError === "string") { + // Called as (reason, provider?, originalError?) + provider = providerOrError; + originalError = + originalErrorOrProvider instanceof Error + ? originalErrorOrProvider + : undefined; + } else if (providerOrError instanceof Error) { + // Called as (reason, originalError, provider) + originalError = providerOrError; + provider = + typeof originalErrorOrProvider === "string" + ? originalErrorOrProvider + : undefined; + } + + return new STTError({ + code: STT_ERROR_CODES.TRANSCRIPTION_FAILED, + message: `Transcription failed: ${reason}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { provider }, + originalError, + }); + } + + /** + * Create an error for unconfigured provider + */ + static providerNotConfigured(provider: string): STTError { + return new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: `STT provider "${provider}" is not properly configured. Please set the required API keys.`, + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider }, + }); + } + + /** + * Create an error for unsupported provider + */ + static providerNotSupported( + provider: string, + availableProviders?: string[], + ): STTError { + return new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_SUPPORTED, + message: `STT provider "${provider}" is not supported`, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider, availableProviders }, + }); + } + + /** + * Create an error for stream processing failure + */ + static streamError(reason: string, provider?: string): STTError { + return new STTError({ + code: STT_ERROR_CODES.STREAM_ERROR, + message: `Stream processing error: ${reason}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { provider }, + }); + } + + /** + * Alias for providerNotConfigured + */ + static notConfigured(provider: string): STTError { + return STTError.providerNotConfigured(provider); + } + + /** + * Alias for audioEmpty + */ + static emptyAudio(provider?: string): STTError { + return STTError.audioEmpty(provider); + } +} + +/** + * Realtime Voice Error class for realtime-specific errors + */ +export class RealtimeError extends NeuroLinkError { + constructor(options: VoiceErrorOptions) { + super({ + code: options.code, + message: options.message, + category: options.category ?? ErrorCategory.EXECUTION, + severity: options.severity ?? ErrorSeverity.HIGH, + retriable: options.retriable ?? false, + context: options.context, + originalError: options.originalError, + }); + this.name = "RealtimeError"; + } + + /** + * Create an error for connection failure + * Supports two signatures: + * - connectionFailed(reason, provider?, originalError?) + * - connectionFailed(reason, originalError?, provider?) + */ + static connectionFailed( + reason: string, + providerOrError?: string | Error, + originalErrorOrProvider?: Error | string, + ): RealtimeError { + let provider: string | undefined; + let originalError: Error | undefined; + + if (typeof providerOrError === "string") { + // Called as (reason, provider?, originalError?) + provider = providerOrError; + originalError = + originalErrorOrProvider instanceof Error + ? originalErrorOrProvider + : undefined; + } else if (providerOrError instanceof Error) { + // Called as (reason, originalError, provider) + originalError = providerOrError; + provider = + typeof originalErrorOrProvider === "string" + ? originalErrorOrProvider + : undefined; + } + + return new RealtimeError({ + code: REALTIME_ERROR_CODES.CONNECTION_FAILED, + message: `Failed to connect to realtime service: ${reason}`, + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { provider }, + originalError, + }); + } + + /** + * Create an error for session timeout + */ + static sessionTimeout(timeoutMs: number, provider?: string): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.SESSION_TIMEOUT, + message: `Realtime session timed out after ${timeoutMs}ms`, + category: ErrorCategory.TIMEOUT, + severity: ErrorSeverity.MEDIUM, + retriable: true, + context: { timeoutMs, provider }, + }); + } + + /** + * Create an error for protocol errors + */ + static protocolError( + reason: string, + provider?: string, + originalError?: Error, + ): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.PROTOCOL_ERROR, + message: `Protocol error: ${reason}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider }, + originalError, + }); + } + + /** + * Create an error for audio stream failures + */ + static audioStreamError(reason: string, provider?: string): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.AUDIO_STREAM_ERROR, + message: `Audio stream error: ${reason}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { provider }, + }); + } + + /** + * Create an error for unconfigured provider + */ + static providerNotConfigured(provider: string): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: `Realtime provider "${provider}" is not properly configured. Please set the required API keys.`, + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider }, + }); + } + + /** + * Create an error for unsupported provider + */ + static providerNotSupported( + provider: string, + availableProviders?: string[], + ): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.PROVIDER_NOT_SUPPORTED, + message: `Realtime provider "${provider}" is not supported`, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider, availableProviders }, + }); + } + + /** + * Create an error for duplicate session + */ + static sessionAlreadyActive(provider?: string): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.SESSION_ALREADY_ACTIVE, + message: "A realtime session is already active. Disconnect first.", + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { provider }, + }); + } + + /** + * Create an error for no active session + */ + static sessionNotActive(provider?: string): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.SESSION_NOT_ACTIVE, + message: "No active realtime session. Connect first.", + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { provider }, + }); + } + + /** + * Create an error for invalid messages + */ + static invalidMessage(reason: string, provider?: string): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.INVALID_MESSAGE, + message: `Invalid message: ${reason}`, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { provider }, + }); + } + + /** + * Create an error for connection closed unexpectedly + */ + static connectionClosed( + reason: string, + sessionId?: string, + provider?: string, + ): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.CONNECTION_FAILED, + message: `Connection closed: ${reason}`, + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { sessionId, provider }, + }); + } + + /** + * Create an error for unconfigured provider (alias) + */ + static notConfigured(provider: string): RealtimeError { + return RealtimeError.providerNotConfigured(provider); + } + + /** + * Create an error for operation timeout + */ + static timeout( + operation: string, + timeoutMs: number, + provider?: string, + ): RealtimeError { + return new RealtimeError({ + code: REALTIME_ERROR_CODES.SESSION_TIMEOUT, + message: `Operation "${operation}" timed out after ${timeoutMs}ms`, + category: ErrorCategory.TIMEOUT, + severity: ErrorSeverity.MEDIUM, + retriable: true, + context: { operation, timeoutMs, provider }, + }); + } +} diff --git a/src/lib/voice/index.ts b/src/lib/voice/index.ts new file mode 100644 index 000000000..235b8ef8c --- /dev/null +++ b/src/lib/voice/index.ts @@ -0,0 +1,125 @@ +/** + * Voice Module - Unified Voice/Speech Integration for NeuroLink + * + * Provides TTS (Text-to-Speech), STT (Speech-to-Text), and + * Realtime Voice capabilities across multiple providers. + * + * Use TTSProcessor (src/lib/utils/ttsProcessor.ts) for TTS. + * Use STTProcessor (src/lib/utils/sttProcessor.ts) for STT. + * Use RealtimeProcessor for realtime voice sessions. + * + * @module voice + */ + +// ============================================================================ +// ERROR CODES AND CONSTANTS +// ============================================================================ + +export { + AUDIO_FORMAT_DETAILS, + DEFAULT_REALTIME_CONFIG, + DEFAULT_STT_OPTIONS, + // Type guards + isSTTResult, + isTranscriptionSegment, + isValidRealtimeConfig, + isValidSTTOptions, + REALTIME_ERROR_CODES, + STT_ERROR_CODES, + VOICE_ERROR_CODES, +} from "../types/index.js"; + +// ============================================================================ +// ERRORS +// ============================================================================ + +export { RealtimeError, STTError, VoiceError } from "./errors.js"; + +// ============================================================================ +// REALTIME VOICE API +// ============================================================================ + +export { BaseRealtimeHandler, RealtimeProcessor } from "./RealtimeVoiceAPI.js"; + +// ============================================================================ +// AUDIO UTILITIES +// ============================================================================ + +export { + AUDIO_SIGNATURES, + calculateDuration, + convertAudioFormat, + createPcmBuffer, + createWavFile, + createWavHeader, + detectAudioFormat, + extractPcmSamples, + getFileExtension, + getMimeType, + MIME_TYPES, + normalizeAudio, + resamplePcm, + splitIntoChunks, +} from "./audio-utils.js"; + +// ============================================================================ +// STREAM HANDLER +// ============================================================================ + +export { + asyncIterableToStream, + ChunkedAudioStream, + StreamHandler, + StreamMerger, + StreamSplitter, + streamToAsyncIterable, +} from "./stream-handler.js"; + +// ============================================================================ +// TTS PROVIDERS +// ============================================================================ + +export { AzureTTS, AzureTTS as AzureTTSHandler } from "./providers/AzureTTS.js"; +export { + ElevenLabsTTS, + ElevenLabsTTS as ElevenLabsTTSHandler, +} from "./providers/ElevenLabsTTS.js"; +export { + OpenAITTS, + OpenAITTS as OpenAITTSHandler, +} from "./providers/OpenAITTS.js"; + +// ============================================================================ +// STT PROVIDERS +// ============================================================================ + +export { AzureSTT, AzureSTT as AzureSTTHandler } from "./providers/AzureSTT.js"; +export { + DeepgramSTT, + DeepgramSTT as DeepgramSTTHandler, +} from "./providers/DeepgramSTT.js"; +// Export STT provider classes for direct use +export { + GoogleSTT, + GoogleSTT as GoogleSTTHandler, +} from "./providers/GoogleSTT.js"; +export { + OpenAISTT, + OpenAISTTHandler, + WhisperSTT, + WhisperSTTHandler, +} from "./providers/OpenAISTT.js"; + +// ============================================================================ +// REALTIME PROVIDERS +// ============================================================================ + +export { + GeminiLive, + GeminiLive as GeminiLiveHandler, +} from "./providers/GeminiLive.js"; +// Export Realtime provider classes for direct use +export { + OpenAIRealtime, + OpenAIRealtime as OpenAIRealtimeHandler, +} from "./providers/OpenAIRealtime.js"; diff --git a/src/lib/voice/providers/AzureSTT.ts b/src/lib/voice/providers/AzureSTT.ts new file mode 100644 index 000000000..e27be9e29 --- /dev/null +++ b/src/lib/voice/providers/AzureSTT.ts @@ -0,0 +1,374 @@ +/** + * Azure Cognitive Services Speech-to-Text Handler + * + * Implementation of STT using Azure Speech Services. + * + * @module voice/providers/AzureSTT + */ + +import { logger } from "../../utils/logger.js"; +import { STTError } from "../errors.js"; +import type { + AudioFormat, + AzureRecognitionResult, + AzureSTTOptions, + STTHandler, + STTLanguage, + STTOptions, + STTResult, + TranscriptionSegment, +} from "../../types/index.js"; + +/** + * Azure Cognitive Services Speech-to-Text Handler + * + * Supports speech recognition with custom models and detailed output. + * + * @see https://docs.microsoft.com/azure/cognitive-services/speech-service/ + */ +export class AzureSTT implements STTHandler { + private readonly apiKey: string | null; + private readonly region: string; + + /** + * Maximum audio duration in seconds (4 hours) + */ + public readonly maxAudioDuration = 240 * 60; + + /** + * Azure STT implementation buffers chunks via REST — not true streaming + */ + public readonly supportsStreaming = false; + + constructor(apiKey?: string, region?: string) { + this.apiKey = apiKey ?? process.env.AZURE_SPEECH_KEY ?? null; + this.region = region ?? process.env.AZURE_SPEECH_REGION ?? "eastus"; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + getSupportedFormats(): AudioFormat[] { + return ["mp3", "wav", "ogg"]; + } + + async getSupportedLanguages(): Promise { + // Azure supports 100+ languages + return [ + { + code: "en-US", + name: "English (US)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "en-GB", + name: "English (UK)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "es-ES", + name: "Spanish (Spain)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "es-MX", + name: "Spanish (Mexico)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "fr-FR", + name: "French", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "de-DE", + name: "German", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "it-IT", + name: "Italian", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "pt-BR", + name: "Portuguese (Brazil)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ja-JP", + name: "Japanese", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ko-KR", + name: "Korean", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "zh-CN", + name: "Chinese (Simplified)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "hi-IN", + name: "Hindi", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ar-SA", + name: "Arabic", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ru-RU", + name: "Russian", + supportsDiarization: true, + supportsPunctuation: true, + }, + ]; + } + + async transcribe( + audio: Buffer | ArrayBuffer, + options: STTOptions = {}, + ): Promise { + if (!this.apiKey) { + throw STTError.providerNotConfigured("azure-stt"); + } + + const audioBuffer = Buffer.isBuffer(audio) ? audio : Buffer.from(audio); + + if (audioBuffer.length === 0) { + throw STTError.audioEmpty("azure-stt"); + } + + const azureOptions = options as AzureSTTOptions; + const startTime = Date.now(); + + try { + // Build the URL with query parameters + const params = new URLSearchParams(); + params.set("language", options.language ?? "en-US"); + + // Add detailed output format + if (azureOptions.detailed || options.wordTimestamps) { + params.set("format", "detailed"); + } + + // Add profanity mode + if (azureOptions.profanityMode) { + params.set("profanity", azureOptions.profanityMode); + } else if (options.profanityFilter) { + params.set("profanity", "masked"); + } + + // Add custom endpoint if provided + const baseUrl = `https://${this.region}.stt.speech.microsoft.com`; + if (azureOptions.customEndpointId) { + params.set("cid", azureOptions.customEndpointId); + } + + const url = `${baseUrl}/speech/recognition/conversation/cognitiveservices/v1?${params.toString()}`; + + const controller = new AbortController(); + const timeoutId = setTimeout(() => controller.abort(), 30000); + let response: Response; + try { + response = await fetch(url, { + method: "POST", + headers: { + "Ocp-Apim-Subscription-Key": this.apiKey, + "Content-Type": this.getContentType(options.format ?? "wav"), + Accept: "application/json", + }, + body: new Uint8Array(audioBuffer), + signal: controller.signal, + }); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw STTError.transcriptionFailed( + "Azure STT request timed out after 30 seconds", + "azure-stt", + fetchErr, + ); + } + throw fetchErr; + } finally { + clearTimeout(timeoutId); + } + + if (!response.ok) { + const errorText = await response.text(); + throw STTError.transcriptionFailed( + `HTTP ${response.status}: ${errorText}`, + "azure-stt", + ); + } + + const data = (await response.json()) as AzureRecognitionResult; + const latency = Date.now() - startTime; + + // Check recognition status + if (data.RecognitionStatus !== "Success") { + if (data.RecognitionStatus === "NoMatch") { + return { + text: "", + confidence: 0, + language: options.language, + metadata: { + latency, + provider: "azure-stt", + status: data.RecognitionStatus, + }, + }; + } + + throw STTError.transcriptionFailed( + `Recognition failed: ${data.RecognitionStatus}`, + "azure-stt", + ); + } + + // Build result from NBest or DisplayText + const result: STTResult = { + text: data.DisplayText ?? "", + confidence: 0.9, // Default confidence if not available + language: options.language, + duration: this.ticksToSeconds(data.Duration ?? 0), + metadata: { + latency, + provider: "azure-stt", + status: data.RecognitionStatus, + }, + }; + + // Process NBest results if available + if (data.NBest && data.NBest.length > 0) { + const best = data.NBest[0]; + result.text = best.Display; + result.confidence = best.Confidence; + + // Add word timings + if (best.Words && best.Words.length > 0) { + result.words = best.Words.map((word) => ({ + word: word.Word, + startTime: this.ticksToSeconds(word.Offset), + endTime: this.ticksToSeconds(word.Offset + word.Duration), + confidence: word.Confidence, + })); + } + } + + logger.info(`[AzureSTTHandler] Transcribed audio in ${latency}ms`); + + return result; + } catch (err: unknown) { + if (err instanceof STTError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error(`[AzureSTTHandler] Transcription failed: ${errorMessage}`); + throw STTError.transcriptionFailed( + errorMessage, + "azure-stt", + err instanceof Error ? err : undefined, + ); + } + } + + /** + * Streaming transcription (placeholder - requires SDK) + */ + async *transcribeStream( + audioStream: AsyncIterable, + options: STTOptions, + ): AsyncIterable { + // Azure streaming requires the Microsoft Speech SDK + // For now, buffer and transcribe in chunks + const chunks: Buffer[] = []; + let chunkIndex = 0; + + for await (const chunk of audioStream) { + chunks.push(chunk); + + // Process every ~5 seconds of audio + const bytesPerSecond = (options.sampleRate ?? 16000) * 2; + const totalBytes = chunks.reduce((sum, c) => sum + c.length, 0); + + if (totalBytes >= bytesPerSecond * 5) { + const audio = Buffer.concat(chunks); + chunks.length = 0; + + try { + const result = await this.transcribe(audio, options); + + yield { + index: chunkIndex++, + text: result.text, + isFinal: false, + confidence: result.confidence, + }; + } catch (err) { + logger.warn( + `[AzureSTTHandler] Chunk transcription failed: ${err instanceof Error ? err.message : String(err)}`, + ); + } + } + } + + // Process remaining audio + if (chunks.length > 0) { + const audio = Buffer.concat(chunks); + try { + const result = await this.transcribe(audio, options); + yield { + index: chunkIndex, + text: result.text, + isFinal: true, + confidence: result.confidence, + }; + } catch (err) { + logger.warn( + `[AzureSTTHandler] Final chunk transcription failed: ${err instanceof Error ? err.message : String(err)}`, + ); + } + } + } + + /** + * Get Content-Type header for audio format + */ + private getContentType(format: AudioFormat): string { + const contentTypes: Partial> = { + mp3: "audio/mpeg", + wav: "audio/wav; codecs=audio/pcm; samplerate=16000", + ogg: "audio/ogg; codecs=opus", + opus: "audio/ogg; codecs=opus", + }; + return contentTypes[format] ?? "audio/wav"; + } + + /** + * Convert Azure ticks (100ns units) to seconds + */ + private ticksToSeconds(ticks: number): number { + return ticks / 10000000; + } +} diff --git a/src/lib/voice/providers/AzureTTS.ts b/src/lib/voice/providers/AzureTTS.ts new file mode 100644 index 000000000..c1b828ebd --- /dev/null +++ b/src/lib/voice/providers/AzureTTS.ts @@ -0,0 +1,357 @@ +/** + * Azure Cognitive Services Text-to-Speech Handler + * + * Implementation of TTS using Azure Speech Services. + * + * @module voice/providers/AzureTTS + */ + +import { ErrorCategory, ErrorSeverity } from "../../constants/enums.js"; +import type { + AudioFormat, + AzureTTSOptions, + AzureVoiceInfo, + TTSHandler, + TTSOptions, + TTSResult, + TTSVoice, +} from "../../types/index.js"; +import { logger } from "../../utils/logger.js"; +import { TTS_ERROR_CODES, TTSError } from "../../utils/ttsProcessor.js"; + +/** + * Azure Cognitive Services Text-to-Speech Handler + * + * Supports neural voices with SSML and custom voice styles. + * + * @see https://docs.microsoft.com/azure/cognitive-services/speech-service/ + */ +export class AzureTTS implements TTSHandler { + private readonly apiKey: string | null; + private readonly region: string; + private voicesCache: { voices: TTSVoice[]; timestamp: number } | null = null; + private static readonly CACHE_TTL_MS = 30 * 60 * 1000; // 30 minutes + + /** + * Maximum text length (10000 characters for Azure) + */ + public readonly maxTextLength = 10000; + + constructor(apiKey?: string, region?: string) { + this.apiKey = apiKey ?? process.env.AZURE_SPEECH_KEY ?? null; + this.region = region ?? process.env.AZURE_SPEECH_REGION ?? "eastus"; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + async getVoices(languageCode?: string): Promise { + if (!this.apiKey) { + throw new TTSError({ + code: TTS_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: "Azure Speech key not configured", + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + }); + } + + // Return cached voices if valid + if ( + this.voicesCache && + Date.now() - this.voicesCache.timestamp < AzureTTS.CACHE_TTL_MS && + !languageCode + ) { + return this.voicesCache.voices; + } + + try { + const voicesController = new AbortController(); + const voicesTimeoutId = setTimeout(() => voicesController.abort(), 30000); + let response: Response; + try { + response = await fetch( + `https://${this.region}.tts.speech.microsoft.com/cognitiveservices/voices/list`, + { + method: "GET", + headers: { + "Ocp-Apim-Subscription-Key": this.apiKey, + }, + signal: voicesController.signal, + }, + ); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: "Azure TTS voices request timed out after 30 seconds", + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.MEDIUM, + retriable: true, + originalError: fetchErr, + }); + } + throw fetchErr; + } finally { + clearTimeout(voicesTimeoutId); + } + + if (!response.ok) { + throw new Error(`HTTP ${response.status}`); + } + + const data = (await response.json()) as AzureVoiceInfo[]; + + let voices: TTSVoice[] = data.map((voice) => ({ + id: voice.ShortName, + name: voice.DisplayName, + languageCode: voice.Locale, + languageCodes: [voice.Locale], + gender: this.mapGender(voice.Gender), + type: voice.VoiceType.toLowerCase().includes("neural") + ? "neural" + : "standard", + description: voice.LocaleName, + })); + + // Filter by language if specified + if (languageCode) { + voices = voices.filter( + (v) => + v.languageCode + .toLowerCase() + .startsWith(languageCode.toLowerCase()) || + v.languageCode.toLowerCase() === languageCode.toLowerCase(), + ); + } + + // Cache full list + if (!languageCode) { + this.voicesCache = { voices, timestamp: Date.now() }; + } + + return voices; + } catch (err: unknown) { + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error(`[AzureTTSHandler] Failed to get voices: ${errorMessage}`); + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: `Failed to get voices: ${errorMessage}`, + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.MEDIUM, + retriable: true, + originalError: err instanceof Error ? err : undefined, + }); + } + } + + async synthesize(text: string, options: TTSOptions = {}): Promise { + if (!this.apiKey) { + throw new TTSError({ + code: TTS_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: "Azure Speech key not configured", + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + }); + } + + const startTime = Date.now(); + const azureOptions = options as AzureTTSOptions; + + try { + // Get voice (default to a common neural voice) + const voice = options.voice ?? "en-US-JennyNeural"; + + // Determine output format + const outputFormat = + azureOptions.outputFormat ?? this.mapFormat(options.format ?? "mp3"); + + // Build SSML + const ssml = this.buildSSML(text, voice, options); + + const controller = new AbortController(); + const timeoutId = setTimeout(() => controller.abort(), 30000); + let response: Response; + try { + response = await fetch( + `https://${this.region}.tts.speech.microsoft.com/cognitiveservices/v1`, + { + method: "POST", + headers: { + "Ocp-Apim-Subscription-Key": this.apiKey, + "Content-Type": "application/ssml+xml", + "X-Microsoft-OutputFormat": outputFormat, + }, + body: ssml, + signal: controller.signal, + }, + ); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: "Azure TTS request timed out after 30 seconds", + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.HIGH, + retriable: true, + originalError: fetchErr, + }); + } + throw fetchErr; + } finally { + clearTimeout(timeoutId); + } + + if (!response.ok) { + const errorText = await response.text(); + throw new Error(`HTTP ${response.status}: ${errorText}`); + } + + const latency = Date.now() - startTime; + + // Get audio buffer + const arrayBuffer = await response.arrayBuffer(); + const audioBuffer = Buffer.from(arrayBuffer); + + const result: TTSResult = { + buffer: audioBuffer, + format: options.format ?? "mp3", + size: audioBuffer.length, + voice, + sampleRate: this.getSampleRate(outputFormat), + metadata: { + latency, + provider: "azure-tts", + outputFormat, + region: this.region, + }, + }; + + logger.info( + `[AzureTTSHandler] Synthesized ${audioBuffer.length} bytes in ${latency}ms`, + ); + + return result; + } catch (err: unknown) { + if (err instanceof TTSError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error(`[AzureTTSHandler] Synthesis failed: ${errorMessage}`); + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: `Synthesis failed: ${errorMessage}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { textLength: text.length }, + originalError: err instanceof Error ? err : undefined, + }); + } + } + + /** + * Build SSML from text and options + */ + private buildSSML(text: string, voice: string, options: TTSOptions): string { + const azureOptions = options as AzureTTSOptions; + + // If custom SSML template provided, use it + if (azureOptions.ssmlTemplate) { + return azureOptions.ssmlTemplate + .replace("{text}", this.escapeXml(text)) + .replace("{voice}", this.escapeXml(voice)); + } + + // Check if text is already SSML + if (text.trim().startsWith(" + + + ${this.escapeXml(text)} + + +`; + } + + /** + * Extract language from voice name + */ + private extractLanguage(voice: string): string { + // Voice names are like "en-US-JennyNeural" + const match = voice.match(/^([a-z]{2}-[A-Z]{2})/); + return match ? match[1] : "en-US"; + } + + /** + * Escape XML special characters + */ + private escapeXml(text: string): string { + return text + .replace(/&/g, "&") + .replace(//g, ">") + .replace(/"/g, """) + .replace(/'/g, "'"); + } + + /** + * Map gender string to standard type + */ + private mapGender(gender: string): "male" | "female" | "neutral" { + switch (gender?.toLowerCase()) { + case "male": + return "male"; + case "female": + return "female"; + default: + return "neutral"; + } + } + + /** + * Map AudioFormat to Azure output format + */ + private mapFormat(format: AudioFormat): string { + const formats: Partial> = { + mp3: "audio-24khz-96kbitrate-mono-mp3", + wav: "riff-24khz-16bit-mono-pcm", + ogg: "ogg-24khz-16bit-mono-opus", + opus: "ogg-24khz-16bit-mono-opus", + }; + return formats[format] ?? "audio-24khz-96kbitrate-mono-mp3"; + } + + /** + * Get sample rate from format string + */ + private getSampleRate(format: string): number { + if (format.includes("24khz")) { + return 24000; + } + if (format.includes("16khz")) { + return 16000; + } + if (format.includes("48khz")) { + return 48000; + } + return 24000; + } +} diff --git a/src/lib/voice/providers/DeepgramSTT.ts b/src/lib/voice/providers/DeepgramSTT.ts new file mode 100644 index 000000000..7edb564ce --- /dev/null +++ b/src/lib/voice/providers/DeepgramSTT.ts @@ -0,0 +1,564 @@ +/** + * Deepgram Speech-to-Text Handler + * + * Implementation of STT using Deepgram's Speech Recognition API. + * + * @module voice/providers/DeepgramSTT + */ + +import { logger } from "../../utils/logger.js"; +import { STTError } from "../errors.js"; +import type { + AudioFormat, + DeepgramResponse, + DeepgramSTTOptions, + STTHandler, + STTLanguage, + STTOptions, + STTResult, + TranscriptionSegment, + WordTiming, +} from "../../types/index.js"; + +/** + * Deepgram Speech-to-Text Handler + * + * Supports real-time streaming, speaker diarization, and smart formatting. + * + * @see https://developers.deepgram.com/docs + */ +export class DeepgramSTT implements STTHandler { + private readonly apiKey: string | null; + private readonly baseUrl = "https://api.deepgram.com/v1"; + + /** + * Maximum audio duration in seconds (2 hours) + */ + public readonly maxAudioDuration = 7200; + + /** + * Deepgram supports streaming + */ + public readonly supportsStreaming = true; + + constructor(apiKey?: string) { + this.apiKey = apiKey ?? process.env.DEEPGRAM_API_KEY ?? null; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + getSupportedFormats(): AudioFormat[] { + return ["mp3", "wav", "ogg", "opus"]; + } + + async getSupportedLanguages(): Promise { + // Deepgram supports 40+ languages + return [ + { + code: "en", + name: "English", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "en-US", + name: "English (US)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "en-GB", + name: "English (UK)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "es", + name: "Spanish", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "fr", + name: "French", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "de", + name: "German", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "it", + name: "Italian", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "pt", + name: "Portuguese", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "nl", + name: "Dutch", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ja", + name: "Japanese", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ko", + name: "Korean", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "zh", + name: "Chinese", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "hi", + name: "Hindi", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ru", + name: "Russian", + supportsDiarization: true, + supportsPunctuation: true, + }, + ]; + } + + async transcribe( + audio: Buffer | ArrayBuffer, + options: STTOptions = {}, + ): Promise { + if (!this.apiKey) { + throw STTError.providerNotConfigured("deepgram"); + } + + const audioBuffer = Buffer.isBuffer(audio) ? audio : Buffer.from(audio); + + if (audioBuffer.length === 0) { + throw STTError.audioEmpty("deepgram"); + } + + const deepgramOptions = options as DeepgramSTTOptions; + const startTime = Date.now(); + + try { + // Build query parameters + const params = new URLSearchParams(); + + // Add model + params.set("model", deepgramOptions.model ?? "nova-2"); + + // Add language + if (options.language) { + params.set("language", options.language); + } + + // Add punctuation + if (options.punctuation !== false) { + params.set("punctuate", "true"); + } + + // Add diarization + if (options.speakerDiarization) { + params.set("diarize", "true"); + if (options.speakerCount) { + params.set("diarize_version", "latest"); + } + } + + // Add smart format + if (deepgramOptions.smartFormat) { + params.set("smart_format", "true"); + } + + // Add utterances + if (deepgramOptions.utterances) { + params.set("utterances", "true"); + if (deepgramOptions.uttSplit !== undefined) { + params.set("utt_split", deepgramOptions.uttSplit.toString()); + } + } + + // Add paragraphs + if (deepgramOptions.paragraphs) { + params.set("paragraphs", "true"); + } + + // Add filler words + if (deepgramOptions.fillerWords) { + params.set("filler_words", "true"); + } + + // Add keywords + if (deepgramOptions.keywords && deepgramOptions.keywords.length > 0) { + for (const keyword of deepgramOptions.keywords) { + params.append("keywords", keyword); + } + if (deepgramOptions.keywordBoost) { + params.set("keyword_boost", deepgramOptions.keywordBoost); + } + } + + // Add redaction + if (deepgramOptions.redact && deepgramOptions.redact.length > 0) { + for (const redactType of deepgramOptions.redact) { + params.append("redact", redactType); + } + } + + // Add profanity filter + if (options.profanityFilter) { + params.set("profanity_filter", "true"); + } + + const url = `${this.baseUrl}/listen?${params.toString()}`; + + const controller = new AbortController(); + const timeoutId = setTimeout(() => controller.abort(), 30000); + let response: Response; + try { + response = await fetch(url, { + method: "POST", + headers: { + Authorization: `Token ${this.apiKey}`, + "Content-Type": this.getMimeType(options.format ?? "wav"), + }, + body: new Uint8Array(audioBuffer), + signal: controller.signal, + }); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw STTError.transcriptionFailed( + "Deepgram STT request timed out after 30 seconds", + "deepgram", + fetchErr, + ); + } + throw fetchErr; + } finally { + clearTimeout(timeoutId); + } + + if (!response.ok) { + const errorData = await response + .json() + .catch(() => Object.create(null) as Record); + const errorMessage = + (errorData as { err_msg?: string }).err_msg || + `HTTP ${response.status}`; + throw STTError.transcriptionFailed(errorMessage, "deepgram"); + } + + const data = (await response.json()) as DeepgramResponse; + const latency = Date.now() - startTime; + + // Handle empty results + if ( + !data.results?.channels || + data.results.channels.length === 0 || + !data.results.channels[0].alternatives || + data.results.channels[0].alternatives.length === 0 + ) { + return { + text: "", + confidence: 0, + language: options.language, + duration: data.metadata?.duration, + metadata: { + latency, + provider: "deepgram", + requestId: data.metadata?.request_id, + }, + }; + } + + const firstChannel = data.results.channels[0]; + const firstAlternative = firstChannel.alternatives[0]; + + // Build result + const result: STTResult = { + text: firstAlternative.transcript, + confidence: firstAlternative.confidence, + language: options.language, + duration: data.metadata?.duration, + metadata: { + latency, + provider: "deepgram", + model: deepgramOptions.model ?? "nova-2", + requestId: data.metadata?.request_id, + }, + }; + + // Add word timings + if (firstAlternative.words && firstAlternative.words.length > 0) { + const speakers = new Set(); + + result.words = firstAlternative.words.map((word) => { + const wordTiming: WordTiming = { + word: word.punctuated_word ?? word.word, + startTime: word.start, + endTime: word.end, + confidence: word.confidence, + }; + + if (word.speaker !== undefined) { + wordTiming.speaker = `Speaker ${word.speaker}`; + speakers.add(wordTiming.speaker); + } + + return wordTiming; + }); + + if (speakers.size > 0) { + result.speakers = Array.from(speakers); + } + } + + // Add utterances as segments + if (data.results.utterances && data.results.utterances.length > 0) { + result.segments = data.results.utterances.map((utt, index) => ({ + index, + text: utt.transcript, + isFinal: true, + confidence: utt.confidence, + startTime: utt.start, + endTime: utt.end, + speaker: + utt.speaker !== undefined ? `Speaker ${utt.speaker}` : undefined, + })); + } + + logger.info( + `[DeepgramSTTHandler] Transcribed ${data.metadata?.duration?.toFixed(1) ?? "?"}s audio in ${latency}ms`, + ); + + return result; + } catch (err: unknown) { + if (err instanceof STTError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error( + `[DeepgramSTTHandler] Transcription failed: ${errorMessage}`, + ); + throw STTError.transcriptionFailed( + errorMessage, + "deepgram", + err instanceof Error ? err : undefined, + ); + } + } + + /** + * Streaming transcription using WebSocket + */ + async *transcribeStream( + audioStream: AsyncIterable, + options: STTOptions, + ): AsyncIterable { + if (!this.apiKey) { + throw STTError.providerNotConfigured("deepgram"); + } + + const deepgramOptions = options as DeepgramSTTOptions; + + // Build query parameters + const params = new URLSearchParams(); + params.set("model", deepgramOptions.model ?? "nova-2"); + + if (options.language) { + params.set("language", options.language); + } + if (options.punctuation !== false) { + params.set("punctuate", "true"); + } + if (options.speakerDiarization) { + params.set("diarize", "true"); + } + if (deepgramOptions.smartFormat) { + params.set("smart_format", "true"); + } + + // Indicate interim results + params.set("interim_results", "true"); + + const wsUrl = `wss://api.deepgram.com/v1/listen?${params.toString()}`; + + // Create WebSocket connection + const WebSocket = (await import("ws")).default; + const ws = new WebSocket(wsUrl, { + headers: { + Authorization: `Token ${this.apiKey}`, + }, + }); + + let segmentIndex = 0; + const messageQueue: TranscriptionSegment[] = []; + let resolveNext: + | ((value: IteratorResult) => void) + | null = null; + let done = false; + let error: Error | null = null; + + ws.on("message", (data: Buffer) => { + try { + const response = JSON.parse(data.toString()) as { + type: string; + channel?: { + alternatives?: Array<{ + transcript?: string; + confidence?: number; + }>; + }; + is_final?: boolean; + speech_final?: boolean; + }; + + if (response.type === "Results" && response.channel?.alternatives) { + const alt = response.channel.alternatives[0]; + if (alt && alt.transcript) { + const segment: TranscriptionSegment = { + index: segmentIndex++, + text: alt.transcript, + isFinal: response.is_final ?? false, + confidence: alt.confidence ?? 0, + }; + + if (resolveNext) { + resolveNext({ value: segment, done: false }); + resolveNext = null; + } else { + messageQueue.push(segment); + } + } + } + } catch { + logger.warn(`[DeepgramSTTHandler] Failed to parse WebSocket message`); + } + }); + + ws.on("error", (err: Error) => { + error = err; + if (resolveNext) { + resolveNext({ + value: undefined as unknown as TranscriptionSegment, + done: true, + }); + resolveNext = null; + } + }); + + ws.on("close", () => { + done = true; + if (resolveNext) { + resolveNext({ + value: undefined as unknown as TranscriptionSegment, + done: true, + }); + resolveNext = null; + } + }); + + // Wait for connection (10-second timeout to avoid hanging indefinitely) + await new Promise((resolve, reject) => { + const connectionTimeout = setTimeout(() => { + ws.terminate(); + reject( + STTError.streamError( + "WebSocket connection to Deepgram timed out after 10 seconds", + "deepgram", + ), + ); + }, 10000); + + ws.on("open", () => { + clearTimeout(connectionTimeout); + resolve(); + }); + + ws.on("error", (err) => { + clearTimeout(connectionTimeout); + reject(err); + }); + }); + + // Send audio chunks + const sendAudio = async () => { + try { + for await (const chunk of audioStream) { + if (ws.readyState === WebSocket.OPEN) { + ws.send(chunk); + } + } + // Send close message + if (ws.readyState === WebSocket.OPEN) { + ws.send(JSON.stringify({ type: "CloseStream" })); + } + } catch (sendError) { + logger.error( + `[DeepgramSTTHandler] Error sending audio: ${sendError instanceof Error ? sendError.message : String(sendError)}`, + ); + } + }; + + // Start sending audio in background + sendAudio(); + + // Yield segments + while (!done) { + if (error) { + throw STTError.streamError((error as Error).message, "deepgram"); + } + + if (messageQueue.length > 0) { + yield messageQueue.shift()!; + } else { + // Wait for next message + await new Promise>((resolve) => { + resolveNext = resolve; + }); + } + } + + // Yield remaining messages + while (messageQueue.length > 0) { + yield messageQueue.shift()!; + } + + ws.close(); + } + + /** + * Get MIME type for audio format + */ + private getMimeType(format: AudioFormat): string { + const mimeTypes: Partial> = { + mp3: "audio/mpeg", + wav: "audio/wav", + ogg: "audio/ogg", + opus: "audio/opus", + }; + return mimeTypes[format] ?? "audio/wav"; + } +} diff --git a/src/lib/voice/providers/ElevenLabsTTS.ts b/src/lib/voice/providers/ElevenLabsTTS.ts new file mode 100644 index 000000000..bb29a4778 --- /dev/null +++ b/src/lib/voice/providers/ElevenLabsTTS.ts @@ -0,0 +1,326 @@ +/** + * ElevenLabs Text-to-Speech Handler + * + * Implementation of TTS using ElevenLabs API. + * + * @module voice/providers/ElevenLabsTTS + */ + +import { ErrorCategory, ErrorSeverity } from "../../constants/enums.js"; +import type { + AudioFormat, + ElevenLabsTTSOptions, + ElevenLabsVoicesResponse, + TTSHandler, + TTSOptions, + TTSResult, + TTSVoice, +} from "../../types/index.js"; +import { logger } from "../../utils/logger.js"; +import { TTS_ERROR_CODES, TTSError } from "../../utils/ttsProcessor.js"; + +/** + * ElevenLabs Text-to-Speech Handler + * + * Supports high-quality multilingual TTS with voice cloning. + * + * @see https://elevenlabs.io/docs/api-reference + */ +export class ElevenLabsTTS implements TTSHandler { + private readonly apiKey: string | null; + private readonly baseUrl = "https://api.elevenlabs.io/v1"; + private voicesCache: { voices: TTSVoice[]; timestamp: number } | null = null; + private static readonly CACHE_TTL_MS = 5 * 60 * 1000; // 5 minutes + + /** + * Maximum text length (5000 characters) + */ + public readonly maxTextLength = 5000; + + constructor(apiKey?: string) { + this.apiKey = apiKey ?? process.env.ELEVENLABS_API_KEY ?? null; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + async getVoices(languageCode?: string): Promise { + if (!this.apiKey) { + throw new TTSError({ + code: TTS_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: "ElevenLabs API key not configured", + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + }); + } + + // Return cached voices if valid + if ( + this.voicesCache && + Date.now() - this.voicesCache.timestamp < ElevenLabsTTS.CACHE_TTL_MS && + !languageCode + ) { + return this.voicesCache.voices; + } + + try { + const voicesController = new AbortController(); + const voicesTimeoutId = setTimeout(() => voicesController.abort(), 30000); + let response: Response; + try { + response = await fetch(`${this.baseUrl}/voices`, { + method: "GET", + headers: { + "xi-api-key": this.apiKey, + }, + signal: voicesController.signal, + }); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: "ElevenLabs voices request timed out after 30 seconds", + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.MEDIUM, + retriable: true, + originalError: fetchErr, + }); + } + throw fetchErr; + } finally { + clearTimeout(voicesTimeoutId); + } + + if (!response.ok) { + throw new Error(`HTTP ${response.status}`); + } + + const data = (await response.json()) as ElevenLabsVoicesResponse; + + let voices: TTSVoice[] = data.voices.map((voice) => ({ + id: voice.voice_id, + name: voice.name, + languageCode: "en", // ElevenLabs supports multiple languages per voice + languageCodes: [ + "en", + "es", + "fr", + "de", + "it", + "pt", + "pl", + "hi", + "ar", + "zh", + "ja", + "ko", + ], + gender: this.mapGender(voice.labels?.gender), + type: "neural", + description: voice.labels?.description, + })); + + // Filter by language if specified + if (languageCode) { + const requested = languageCode.toLowerCase(); + voices = voices.filter((v) => + v.languageCodes?.some((code) => + code.toLowerCase().startsWith(requested), + ), + ); + } + + // Cache voices + if (!languageCode) { + this.voicesCache = { voices, timestamp: Date.now() }; + } + + return voices; + } catch (err: unknown) { + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error( + `[ElevenLabsTTSHandler] Failed to get voices: ${errorMessage}`, + ); + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: `Failed to get voices: ${errorMessage}`, + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.MEDIUM, + retriable: true, + originalError: err instanceof Error ? err : undefined, + }); + } + } + + async synthesize(text: string, options: TTSOptions = {}): Promise { + if (!this.apiKey) { + throw new TTSError({ + code: TTS_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: "ElevenLabs API key not configured", + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + }); + } + + const startTime = Date.now(); + const elevenOptions = options as ElevenLabsTTSOptions; + + try { + // Get voice ID (use default if not specified) + const voiceId = options.voice ?? "21m00Tcm4TlvDq8ikWAM"; // Rachel voice as default + + // Determine model + const model = elevenOptions.model ?? "eleven_multilingual_v2"; + + // Build request body + const requestBody = { + text, + model_id: model, + voice_settings: { + stability: elevenOptions.stability ?? 0.5, + similarity_boost: elevenOptions.similarityBoost ?? 0.75, + style: elevenOptions.style ?? 0.0, + use_speaker_boost: elevenOptions.useSpeakerBoost ?? true, + }, + }; + + // Determine output format + const outputFormat = this.mapFormat(options.format ?? "mp3"); + + const controller = new AbortController(); + const timeoutId = setTimeout(() => controller.abort(), 30000); + let response: Response; + try { + response = await fetch( + `${this.baseUrl}/text-to-speech/${voiceId}?output_format=${outputFormat}`, + { + method: "POST", + headers: { + "xi-api-key": this.apiKey, + "Content-Type": "application/json", + }, + body: JSON.stringify(requestBody), + signal: controller.signal, + }, + ); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: "ElevenLabs TTS request timed out after 30 seconds", + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.HIGH, + retriable: true, + originalError: fetchErr, + }); + } + throw fetchErr; + } finally { + clearTimeout(timeoutId); + } + + if (!response.ok) { + const errorData = await response + .json() + .catch(() => Object.create(null) as Record); + const errorMessage = + (errorData as { detail?: { message?: string } }).detail?.message || + `HTTP ${response.status}`; + throw new Error(errorMessage); + } + + const latency = Date.now() - startTime; + + // Get audio buffer + const arrayBuffer = await response.arrayBuffer(); + const audioBuffer = Buffer.from(arrayBuffer); + + const result: TTSResult = { + buffer: audioBuffer, + format: options.format ?? "mp3", + size: audioBuffer.length, + voice: voiceId, + sampleRate: this.getSampleRate(outputFormat), + metadata: { + latency, + provider: "elevenlabs-tts", + model, + outputFormat, + }, + }; + + logger.info( + `[ElevenLabsTTSHandler] Synthesized ${audioBuffer.length} bytes in ${latency}ms`, + ); + + return result; + } catch (err: unknown) { + if (err instanceof TTSError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error(`[ElevenLabsTTSHandler] Synthesis failed: ${errorMessage}`); + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: `Synthesis failed: ${errorMessage}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { textLength: text.length }, + originalError: err instanceof Error ? err : undefined, + }); + } + } + + /** + * Map gender string to standard type + */ + private mapGender(gender?: string): "male" | "female" | "neutral" { + if (!gender) { + return "neutral"; + } + const lower = gender.toLowerCase(); + if (lower.includes("male") && !lower.includes("female")) { + return "male"; + } + if (lower.includes("female")) { + return "female"; + } + return "neutral"; + } + + /** + * Map AudioFormat to ElevenLabs output format + */ + private mapFormat(format: AudioFormat): string { + const formats: Partial> = { + mp3: "mp3_44100_128", + wav: "pcm_44100", + ogg: "ogg_22050", + opus: "ogg_22050", + }; + return formats[format] ?? "mp3_44100_128"; + } + + /** + * Get sample rate from format string + */ + private getSampleRate(format: string): number { + if (format.includes("44100")) { + return 44100; + } + if (format.includes("22050")) { + return 22050; + } + if (format.includes("24000")) { + return 24000; + } + return 44100; + } +} diff --git a/src/lib/voice/providers/GeminiLive.ts b/src/lib/voice/providers/GeminiLive.ts new file mode 100644 index 000000000..cf30201bc --- /dev/null +++ b/src/lib/voice/providers/GeminiLive.ts @@ -0,0 +1,418 @@ +/** + * Google Gemini Live Voice API Handler + * + * Implementation of bidirectional voice communication using Gemini's Live API. + * + * @module voice/providers/GeminiLive + */ + +import type WebSocket from "ws"; +import { logger } from "../../utils/logger.js"; +import { RealtimeError } from "../errors.js"; +import { BaseRealtimeHandler } from "../RealtimeVoiceAPI.js"; +import type { + AudioFormat, + GeminiMessage, + GeminiResponse, + RealtimeAudioChunk, + RealtimeConfig, + RealtimeSession, +} from "../../types/index.js"; + +/** + * Google Gemini Live Voice API Handler + * + * Implements bidirectional voice communication with Gemini's Live API. + * + * @see https://ai.google.dev/gemini-api/docs/live + */ +export class GeminiLive extends BaseRealtimeHandler { + readonly name = "gemini-live"; + + private readonly apiKey: string | null; + private ws: WebSocket | null = null; + private audioChunkIndex = 0; + private pendingFunctionCalls = new Map(); + + constructor(apiKey?: string) { + super(); + this.apiKey = apiKey ?? process.env.GOOGLE_API_KEY ?? null; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + getSupportedFormats(): AudioFormat[] { + return ["opus", "wav"]; + } + + async connect(config: RealtimeConfig): Promise { + if (!this.apiKey) { + throw RealtimeError.providerNotConfigured("gemini-live"); + } + + if (this.isConnected()) { + throw RealtimeError.sessionAlreadyActive("gemini-live"); + } + + this.emitStateChange("connecting"); + + try { + // Import WebSocket + const { default: WebSocket } = await import("ws"); + + // Determine model + const model = + config.model ?? "gemini-2.5-flash-native-audio-preview-09-2025"; + + // Connect to Gemini Live API + const wsUrl = `wss://generativelanguage.googleapis.com/ws/google.ai.generativelanguage.v1alpha.GenerativeService.BidiGenerateContent?key=${this.apiKey}`; + + this.ws = new WebSocket(wsUrl); + + // Wait for connection + await new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error("Connection timeout")); + }, config.timeout ?? 30000); + + this.ws!.on("open", () => { + clearTimeout(timeout); + resolve(); + }); + + this.ws!.on("error", (err) => { + clearTimeout(timeout); + reject(err); + }); + }); + + // Set up message handler + this.ws.on("message", (data: Buffer) => { + this.handleMessage(data); + }); + + this.ws.on("close", () => { + this.emitStateChange("disconnected"); + this.session = null; + }); + + this.ws.on("error", (err) => { + this.emitError(err); + }); + + // Send setup message + await this.sendSetup(config, model); + + // Wait for setup complete + await this.waitForSetupComplete(); + + // Generate session ID + const sessionId = `gemini-${Date.now()}`; + + // Create session object + this.session = this.createSession(sessionId, config); + this.emitStateChange("connected"); + + logger.info(`[GeminiLiveHandler] Connected to session: ${sessionId}`); + + return this.session; + } catch (err: unknown) { + this.emitStateChange("error"); + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + throw RealtimeError.connectionFailed( + errorMessage, + "gemini-live", + err instanceof Error ? err : undefined, + ); + } + } + + async disconnect(): Promise { + if (!this.ws) { + return; + } + + this.emitStateChange("disconnecting"); + + try { + this.ws.close(); + this.ws = null; + this.session = null; + this.audioChunkIndex = 0; + this.pendingFunctionCalls.clear(); + this.emitStateChange("disconnected"); + logger.info("[GeminiLiveHandler] Disconnected"); + } catch (err: unknown) { + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + throw RealtimeError.protocolError( + `Disconnect failed: ${errorMessage}`, + "gemini-live", + err instanceof Error ? err : undefined, + ); + } + } + + async sendAudio(audio: Buffer | RealtimeAudioChunk): Promise { + if (!this.ws || !this.isConnected()) { + throw RealtimeError.sessionNotActive("gemini-live"); + } + + const audioBuffer = Buffer.isBuffer(audio) ? audio : audio.data; + + // Send audio as realtime input + const message: GeminiMessage = { + realtimeInput: { + mediaChunks: [ + { + mimeType: "audio/pcm;rate=16000", + data: audioBuffer.toString("base64"), + }, + ], + }, + }; + + this.ws.send(JSON.stringify(message)); + } + + async sendText(text: string): Promise { + if (!this.ws || !this.isConnected()) { + throw RealtimeError.sessionNotActive("gemini-live"); + } + + // Send text as client content + const message: GeminiMessage = { + clientContent: { + turns: [ + { + role: "user", + parts: [{ text }], + }, + ], + turnComplete: true, + }, + }; + + this.ws.send(JSON.stringify(message)); + } + + async triggerResponse(): Promise { + // Gemini automatically generates responses based on VAD + // This is a no-op for Gemini Live + } + + async cancelResponse(): Promise { + // Gemini doesn't have explicit cancel, but we can send empty content + // to interrupt + if (this.ws && this.isConnected()) { + const message: GeminiMessage = { + clientContent: { + turns: [], + turnComplete: true, + }, + }; + this.ws.send(JSON.stringify(message)); + } + } + + /** + * Send setup message with configuration + */ + private async sendSetup( + config: RealtimeConfig, + model: string, + ): Promise { + if (!this.ws) { + return; + } + + const setupMessage: GeminiMessage = { + setup: { + model: `models/${model}`, + generationConfig: { + responseModalities: ["AUDIO", "TEXT"], + speechConfig: { + voiceConfig: { + prebuiltVoiceConfig: { + voiceName: config.voice ?? "Puck", + }, + }, + }, + }, + }, + }; + + // Add system instruction + if (config.systemPrompt) { + setupMessage.setup!.systemInstruction = { + parts: [{ text: config.systemPrompt }], + }; + } + + // Add tools + if (config.tools && config.tools.length > 0) { + setupMessage.setup!.tools = [ + { + functionDeclarations: config.tools.map((tool) => ({ + name: tool.name, + description: tool.description, + parameters: tool.parameters, + })), + }, + ]; + } + + this.ws.send(JSON.stringify(setupMessage)); + } + + /** + * Wait for setup complete message + */ + private waitForSetupComplete(): Promise { + return new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error("Timeout waiting for setup complete")); + }, 10000); + + const handler = (data: Buffer) => { + try { + const response = JSON.parse(data.toString()) as GeminiResponse; + if (response.setupComplete) { + clearTimeout(timeout); + this.ws?.off("message", handler); + resolve(); + } + } catch { + // Ignore parse errors + } + }; + + this.ws?.on("message", handler); + }); + } + + /** + * Handle incoming WebSocket messages + */ + private handleMessage(data: Buffer): void { + try { + const response = JSON.parse(data.toString()) as GeminiResponse; + + if (response.serverContent) { + const content = response.serverContent; + + // Handle model turn + if (content.modelTurn?.parts) { + for (const part of content.modelTurn.parts) { + // Handle text + if (part.text) { + this.emitText(part.text, content.turnComplete ?? false); + } + + // Handle audio + if (part.inlineData) { + const audioData = Buffer.from(part.inlineData.data, "base64"); + this.emitAudio({ + data: audioData, + index: this.audioChunkIndex++, + isFinal: content.turnComplete ?? false, + format: this.parseAudioFormat(part.inlineData.mimeType), + sampleRate: 24000, + }); + } + } + } + + // Handle turn complete + if (content.turnComplete) { + this.emitTurnEnd(); + this.audioChunkIndex = 0; + } + + // Handle interruption + if (content.interrupted) { + this.emitTurnEnd(); + this.audioChunkIndex = 0; + } + } + + // Handle tool calls + if (response.toolCall?.functionCalls) { + for (const call of response.toolCall.functionCalls) { + this.pendingFunctionCalls.set(call.id, call.name); + this.handleFunctionCall(call.id, call.name, call.args); + } + } + + // Handle tool call cancellation + if (response.toolCallCancellation?.ids) { + for (const id of response.toolCallCancellation.ids) { + this.pendingFunctionCalls.delete(id); + } + } + } catch (err: unknown) { + logger.warn( + `[GeminiLiveHandler] Failed to parse message: ${err instanceof Error ? err.message : String(err)}`, + ); + } + } + + /** + * Parse audio format from MIME type + */ + private parseAudioFormat(mimeType: string): AudioFormat { + if (mimeType.includes("opus")) { + return "opus"; + } + if (mimeType.includes("wav") || mimeType.includes("pcm")) { + return "wav"; + } + if (mimeType.includes("mp3") || mimeType.includes("mpeg")) { + return "mp3"; + } + return "opus"; + } + + /** + * Handle function call from model + */ + private async handleFunctionCall( + callId: string, + name: string, + args: Record, + ): Promise { + try { + const result = await this.emitFunctionCall(name, args); + + // Send function response + if (this.ws && this.isConnected()) { + const responseMessage = { + toolResponse: { + functionResponses: [ + { + id: callId, + name, + response: { result }, + }, + ], + }, + }; + + this.ws.send(JSON.stringify(responseMessage)); + this.pendingFunctionCalls.delete(callId); + } + } catch (err: unknown) { + const error = + err instanceof Error + ? err + : new Error(String(err || "Function call failed")); + logger.error( + `[GeminiLiveHandler] Function call failed: ${error.message}`, + ); + this.emitError(error); + } + } +} diff --git a/src/lib/voice/providers/GoogleSTT.ts b/src/lib/voice/providers/GoogleSTT.ts new file mode 100644 index 000000000..fceea0b80 --- /dev/null +++ b/src/lib/voice/providers/GoogleSTT.ts @@ -0,0 +1,482 @@ +/** + * Google Cloud Speech-to-Text Handler + * + * Implementation of STT using Google Cloud Speech-to-Text API. + * + * @module voice/providers/GoogleSTT + */ + +import { logger } from "../../utils/logger.js"; +import { STTError } from "../errors.js"; +import type { + AudioFormat, + GoogleRecognitionAudio, + GoogleRecognitionConfig, + GoogleRecognizeResponse, + GoogleSpeechRecognitionResult, + GoogleSTTOptions, + STTHandler, + STTLanguage, + STTOptions, + STTResult, + TranscriptionSegment, + WordTiming, +} from "../../types/index.js"; + +/** + * Google Cloud Speech-to-Text Handler + * + * Supports transcription with speaker diarization, word timestamps, and punctuation. + * + * @see https://cloud.google.com/speech-to-text/docs + */ +export class GoogleSTT implements STTHandler { + private readonly apiKey: string | null; + private readonly credentialsPath: string | null; + private readonly baseUrl = "https://speech.googleapis.com/v1"; + + /** + * Maximum audio duration in seconds (480 minutes = 8 hours with async) + */ + public readonly maxAudioDuration = 480 * 60; + + /** + * Google STT supports streaming + */ + public readonly supportsStreaming = true; + + constructor(apiKey?: string, credentialsPath?: string) { + this.apiKey = apiKey ?? process.env.GOOGLE_API_KEY ?? null; + this.credentialsPath = + credentialsPath ?? process.env.GOOGLE_APPLICATION_CREDENTIALS ?? null; + } + + isConfigured(): boolean { + return this.apiKey !== null || this.credentialsPath !== null; + } + + getSupportedFormats(): AudioFormat[] { + return ["mp3", "wav", "ogg", "opus"]; + } + + async getSupportedLanguages(): Promise { + // Return common languages supported by Google STT + return [ + { + code: "en-US", + name: "English (US)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "en-GB", + name: "English (UK)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "es-ES", + name: "Spanish (Spain)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "es-US", + name: "Spanish (US)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "fr-FR", + name: "French", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "de-DE", + name: "German", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "it-IT", + name: "Italian", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "pt-BR", + name: "Portuguese (Brazil)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ja-JP", + name: "Japanese", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ko-KR", + name: "Korean", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "zh-CN", + name: "Chinese (Simplified)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "zh-TW", + name: "Chinese (Traditional)", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ar-SA", + name: "Arabic", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "hi-IN", + name: "Hindi", + supportsDiarization: true, + supportsPunctuation: true, + }, + { + code: "ru-RU", + name: "Russian", + supportsDiarization: true, + supportsPunctuation: true, + }, + ]; + } + + async transcribe( + audio: Buffer | ArrayBuffer, + options: STTOptions = {}, + ): Promise { + if (!this.isConfigured()) { + throw STTError.providerNotConfigured("google-stt"); + } + + const audioBuffer = Buffer.isBuffer(audio) ? audio : Buffer.from(audio); + + if (audioBuffer.length === 0) { + throw STTError.audioEmpty("google-stt"); + } + + const googleOptions = options as GoogleSTTOptions; + const startTime = Date.now(); + + try { + // Build recognition config + const detectedFormat = options.format ?? "wav"; + const config: GoogleRecognitionConfig = { + encoding: this.getEncoding(detectedFormat), + // Omit sampleRateHertz for WAV/FLAC — the API reads it from the header. + // Hardcoding a wrong value causes "sample_rate_hertz must match WAV header" errors. + ...(detectedFormat !== "wav" && detectedFormat !== "flac" + ? { sampleRateHertz: options.sampleRate ?? 16000 } + : options.sampleRate + ? { sampleRateHertz: options.sampleRate } + : {}), + languageCode: options.language ?? "en-US", + enableAutomaticPunctuation: options.punctuation ?? true, + enableWordTimeOffsets: options.wordTimestamps ?? false, + enableWordConfidence: true, + profanityFilter: options.profanityFilter ?? false, + }; + + // Add model if specified + if (googleOptions.model) { + config.model = googleOptions.model; + } + + // Add enhanced model option + if (googleOptions.useEnhanced) { + config.useEnhanced = true; + } + + // Add diarization if requested + if (options.speakerDiarization) { + config.enableSpeakerDiarization = true; + if (options.speakerCount) { + config.diarizationSpeakerCount = options.speakerCount; + } + } + + // Add max alternatives + if (googleOptions.maxAlternatives) { + config.maxAlternatives = googleOptions.maxAlternatives; + } + + // Build request + const requestBody = { + config, + audio: { + content: audioBuffer.toString("base64"), + } as GoogleRecognitionAudio, + }; + + // Build URL with API key + const url = this.apiKey + ? `${this.baseUrl}/speech:recognize?key=${this.apiKey}` + : `${this.baseUrl}/speech:recognize`; + + const controller = new AbortController(); + const timeoutId = setTimeout(() => controller.abort(), 30000); + let response: Response; + try { + response = await fetch(url, { + method: "POST", + headers: { + "Content-Type": "application/json", + ...(this.credentialsPath && !this.apiKey + ? { Authorization: `Bearer ${await this.getAccessToken()}` } + : {}), + }, + body: JSON.stringify(requestBody), + signal: controller.signal, + }); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw STTError.transcriptionFailed( + "Google STT request timed out after 30 seconds", + "google-stt", + fetchErr, + ); + } + throw fetchErr; + } finally { + clearTimeout(timeoutId); + } + + if (!response.ok) { + const errorData = await response + .json() + .catch(() => Object.create(null) as Record); + const errorMessage = + (errorData as { error?: { message?: string } }).error?.message || + `HTTP ${response.status}`; + throw STTError.transcriptionFailed(errorMessage, "google-stt"); + } + + const data = (await response.json()) as GoogleRecognizeResponse; + const latency = Date.now() - startTime; + + // Handle empty results + if (!data.results || data.results.length === 0) { + return { + text: "", + confidence: 0, + language: options.language, + metadata: { + latency, + provider: "google-stt", + }, + }; + } + + // Build result from all alternatives + const result: STTResult = { + text: data.results + .map((r) => r.alternatives[0]?.transcript ?? "") + .join(" ") + .trim(), + confidence: this.calculateAverageConfidence(data.results), + language: data.results[0]?.languageCode ?? options.language, + metadata: { + latency, + provider: "google-stt", + billedTime: data.totalBilledTime, + }, + }; + + // Add word timings + const words: WordTiming[] = []; + const speakers = new Set(); + + for (const resultItem of data.results) { + const alternative = resultItem.alternatives[0]; + if (alternative?.words) { + for (const wordInfo of alternative.words) { + const word: WordTiming = { + word: wordInfo.word, + startTime: this.parseDuration(wordInfo.startTime), + endTime: this.parseDuration(wordInfo.endTime), + confidence: wordInfo.confidence, + }; + + if (wordInfo.speakerTag !== undefined) { + word.speaker = `Speaker ${wordInfo.speakerTag}`; + speakers.add(word.speaker); + } + + words.push(word); + } + } + } + + if (words.length > 0) { + result.words = words; + } + + if (speakers.size > 0) { + result.speakers = Array.from(speakers); + } + + // Add segments + result.segments = data.results.map((resultItem, index) => { + const alt = resultItem.alternatives[0]; + return { + index, + text: alt?.transcript ?? "", + isFinal: true, + confidence: alt?.confidence ?? 0, + language: resultItem.languageCode, + }; + }); + + logger.info(`[GoogleSTTHandler] Transcribed audio in ${latency}ms`); + + return result; + } catch (err: unknown) { + if (err instanceof STTError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error(`[GoogleSTTHandler] Transcription failed: ${errorMessage}`); + throw STTError.transcriptionFailed( + errorMessage, + "google-stt", + err instanceof Error ? err : undefined, + ); + } + } + + /** + * Streaming transcription (placeholder - requires WebSocket/gRPC) + */ + async *transcribeStream( + audioStream: AsyncIterable, + options: STTOptions, + ): AsyncIterable { + // Google streaming STT requires gRPC or WebSocket connection + // For now, buffer and transcribe in chunks + const chunks: Buffer[] = []; + let chunkIndex = 0; + + for await (const chunk of audioStream) { + chunks.push(chunk); + + // Process every ~5 seconds of audio (assuming 16kHz, 16-bit) + const bytesPerSecond = 16000 * 2; // 16kHz * 2 bytes + const totalBytes = chunks.reduce((sum, c) => sum + c.length, 0); + + if (totalBytes >= bytesPerSecond * 5) { + const audio = Buffer.concat(chunks); + chunks.length = 0; + + try { + const result = await this.transcribe(audio, options); + + yield { + index: chunkIndex++, + text: result.text, + isFinal: false, + confidence: result.confidence, + }; + } catch (err) { + logger.warn( + `[GoogleSTTHandler] Chunk transcription failed: ${err instanceof Error ? err.message : String(err)}`, + ); + } + } + } + + // Process remaining audio + if (chunks.length > 0) { + const audio = Buffer.concat(chunks); + try { + const result = await this.transcribe(audio, options); + yield { + index: chunkIndex, + text: result.text, + isFinal: true, + confidence: result.confidence, + }; + } catch (err) { + logger.warn( + `[GoogleSTTHandler] Final chunk transcription failed: ${err instanceof Error ? err.message : String(err)}`, + ); + } + } + } + + /** + * Get encoding string for audio format + */ + private getEncoding(format: AudioFormat): string { + const encodings: Partial> = { + mp3: "MP3", + wav: "LINEAR16", + ogg: "OGG_OPUS", + opus: "OGG_OPUS", + }; + return encodings[format] ?? "LINEAR16"; + } + + /** + * Parse duration string (e.g., "1.5s") to seconds + */ + private parseDuration(duration: string): number { + if (!duration) { + return 0; + } + const match = duration.match(/^([\d.]+)s$/); + return match ? parseFloat(match[1]) : 0; + } + + /** + * Calculate average confidence from results + */ + private calculateAverageConfidence( + results: GoogleSpeechRecognitionResult[], + ): number { + const confidences = results + .map((r) => r.alternatives[0]?.confidence) + .filter((c): c is number => typeof c === "number"); + + if (confidences.length === 0) { + return 0; + } + return confidences.reduce((sum, c) => sum + c, 0) / confidences.length; + } + + /** + * Get access token from service account credentials + */ + private async getAccessToken(): Promise { + try { + const { GoogleAuth } = await import("google-auth-library"); + const auth = new GoogleAuth({ + ...(this.credentialsPath ? { keyFilename: this.credentialsPath } : {}), + scopes: ["https://www.googleapis.com/auth/cloud-platform"], + }); + const client = await auth.getClient(); + const tokenResponse = await client.getAccessToken(); + return tokenResponse.token ?? ""; + } catch (err) { + logger.debug( + `[GoogleSTTHandler] Failed to acquire access token: ${err instanceof Error ? err.message : String(err)}`, + ); + return ""; + } + } +} diff --git a/src/lib/voice/providers/OpenAIRealtime.ts b/src/lib/voice/providers/OpenAIRealtime.ts new file mode 100644 index 000000000..c7ed2294d --- /dev/null +++ b/src/lib/voice/providers/OpenAIRealtime.ts @@ -0,0 +1,471 @@ +/** + * OpenAI Realtime Voice API Handler + * + * Implementation of bidirectional voice communication using OpenAI's Realtime API. + * + * @module voice/providers/OpenAIRealtime + */ + +import type WebSocket from "ws"; +import { logger } from "../../utils/logger.js"; +import { RealtimeError } from "../errors.js"; +import { BaseRealtimeHandler } from "../RealtimeVoiceAPI.js"; +import type { + AudioFormat, + OpenAIAudioDelta, + OpenAIRealtimeEvent, + OpenAISessionCreated, + OpenAITranscriptDelta, + RealtimeAudioChunk, + RealtimeConfig, + RealtimeSession, +} from "../../types/index.js"; + +/** + * OpenAI Realtime API Handler + * + * Implements bidirectional voice communication with OpenAI's Realtime API. + * + * @see https://platform.openai.com/docs/api-reference/realtime + */ +export class OpenAIRealtime extends BaseRealtimeHandler { + readonly name = "openai-realtime"; + + private readonly apiKey: string | null; + private ws: WebSocket | null = null; + private audioChunkIndex = 0; + + constructor(apiKey?: string) { + super(); + this.apiKey = apiKey ?? process.env.OPENAI_API_KEY ?? null; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + getSupportedFormats(): AudioFormat[] { + return ["wav", "opus"]; + } + + async connect(config: RealtimeConfig): Promise { + if (!this.apiKey) { + throw RealtimeError.providerNotConfigured("openai-realtime"); + } + + if (this.isConnected()) { + throw RealtimeError.sessionAlreadyActive("openai-realtime"); + } + + this.emitStateChange("connecting"); + + try { + // Import WebSocket + const { default: WebSocket } = await import("ws"); + + // Determine model + const model = config.model ?? "gpt-4o-realtime-preview-2024-12-17"; + + // Connect to OpenAI Realtime API + const wsUrl = `wss://api.openai.com/v1/realtime?model=${model}`; + + this.ws = new WebSocket(wsUrl, { + headers: { + Authorization: `Bearer ${this.apiKey}`, + "OpenAI-Beta": "realtime=v1", + }, + }); + + // Wait for connection + await new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error("Connection timeout")); + }, config.timeout ?? 30000); + + this.ws!.on("open", () => { + clearTimeout(timeout); + resolve(); + }); + + this.ws!.on("error", (err) => { + clearTimeout(timeout); + reject(err); + }); + }); + + // Set up message handler + this.ws.on("message", (data: Buffer) => { + this.handleMessage(data); + }); + + this.ws.on("close", () => { + this.emitStateChange("disconnected"); + this.session = null; + }); + + this.ws.on("error", (err) => { + this.emitError(err); + }); + + // Send session update with configuration + await this.sendSessionUpdate(config); + + // Wait for session.created event + const sessionId = await this.waitForSessionCreated(); + + // Create session object + this.session = this.createSession(sessionId, config); + this.emitStateChange("connected"); + + logger.info(`[OpenAIRealtimeHandler] Connected to session: ${sessionId}`); + + return this.session; + } catch (err: unknown) { + this.emitStateChange("error"); + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + throw RealtimeError.connectionFailed( + errorMessage, + "openai-realtime", + err instanceof Error ? err : undefined, + ); + } + } + + async disconnect(): Promise { + if (!this.ws) { + return; + } + + this.emitStateChange("disconnecting"); + + try { + this.ws.close(); + this.ws = null; + this.session = null; + this.audioChunkIndex = 0; + this.emitStateChange("disconnected"); + logger.info("[OpenAIRealtimeHandler] Disconnected"); + } catch (err: unknown) { + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + throw RealtimeError.protocolError( + `Disconnect failed: ${errorMessage}`, + "openai-realtime", + err instanceof Error ? err : undefined, + ); + } + } + + async sendAudio(audio: Buffer | RealtimeAudioChunk): Promise { + if (!this.ws || !this.isConnected()) { + throw RealtimeError.sessionNotActive("openai-realtime"); + } + + const audioBuffer = Buffer.isBuffer(audio) ? audio : audio.data; + + // Send audio append event + const event = { + type: "input_audio_buffer.append", + audio: audioBuffer.toString("base64"), + }; + + this.ws.send(JSON.stringify(event)); + } + + async sendText(text: string): Promise { + if (!this.ws || !this.isConnected()) { + throw RealtimeError.sessionNotActive("openai-realtime"); + } + + // Send conversation item create event + const event = { + type: "conversation.item.create", + item: { + type: "message", + role: "user", + content: [ + { + type: "input_text", + text, + }, + ], + }, + }; + + this.ws.send(JSON.stringify(event)); + + // Trigger response + await this.triggerResponse(); + } + + async triggerResponse(): Promise { + if (!this.ws || !this.isConnected()) { + throw RealtimeError.sessionNotActive("openai-realtime"); + } + + // Commit audio buffer + this.ws.send( + JSON.stringify({ + type: "input_audio_buffer.commit", + }), + ); + + // Create response + this.ws.send( + JSON.stringify({ + type: "response.create", + }), + ); + } + + async cancelResponse(): Promise { + if (!this.ws || !this.isConnected()) { + return; + } + + this.ws.send( + JSON.stringify({ + type: "response.cancel", + }), + ); + } + + /** + * Send session update with configuration + */ + private async sendSessionUpdate(config: RealtimeConfig): Promise { + if (!this.ws) { + return; + } + + const sessionConfig: Record = { + modalities: ["text", "audio"], + input_audio_format: "pcm16", + output_audio_format: "pcm16", + input_audio_transcription: { + model: "whisper-1", + }, + }; + + // Add voice if specified + if (config.voice) { + sessionConfig.voice = config.voice; + } + + // Add turn detection + if (config.turnDetection) { + sessionConfig.turn_detection = { + type: config.turnDetection, + threshold: config.vadThreshold ?? 0.5, + prefix_padding_ms: 300, + silence_duration_ms: 500, + }; + } + + // Add system prompt + if (config.systemPrompt) { + sessionConfig.instructions = config.systemPrompt; + } + + // Add tools + if (config.tools && config.tools.length > 0) { + sessionConfig.tools = config.tools.map((tool) => ({ + type: "function", + name: tool.name, + description: tool.description, + parameters: tool.parameters, + })); + } + + const event = { + type: "session.update", + session: sessionConfig, + }; + + this.ws.send(JSON.stringify(event)); + } + + /** + * Wait for session.created event + */ + private waitForSessionCreated(): Promise { + return new Promise((resolve, reject) => { + const timeout = setTimeout(() => { + reject(new Error("Timeout waiting for session.created")); + }, 10000); + + const handler = (data: Buffer) => { + try { + const event = JSON.parse(data.toString()) as OpenAIRealtimeEvent; + if (event.type === "session.created") { + clearTimeout(timeout); + this.ws?.off("message", handler); + const sessionEvent = event as OpenAISessionCreated; + resolve(sessionEvent.session.id); + } else if (event.type === "error") { + clearTimeout(timeout); + this.ws?.off("message", handler); + reject( + new Error( + (event as { error?: { message?: string } }).error?.message ?? + "Unknown error", + ), + ); + } + } catch { + // Ignore parse errors + } + }; + + this.ws?.on("message", handler); + }); + } + + /** + * Handle incoming WebSocket messages + */ + private handleMessage(data: Buffer): void { + try { + const event = JSON.parse(data.toString()) as OpenAIRealtimeEvent; + + switch (event.type) { + case "response.audio.delta": { + const audioEvent = event as OpenAIAudioDelta; + const audioData = Buffer.from(audioEvent.delta, "base64"); + this.emitAudio({ + data: audioData, + index: this.audioChunkIndex++, + isFinal: false, + format: "wav", + sampleRate: 24000, + }); + break; + } + + case "response.audio.done": { + // Audio stream complete + this.emitAudio({ + data: Buffer.alloc(0), + index: this.audioChunkIndex++, + isFinal: true, + format: "wav", + sampleRate: 24000, + }); + break; + } + + case "response.audio_transcript.delta": { + const transcriptEvent = event as OpenAITranscriptDelta; + if (transcriptEvent.delta) { + this.emitText(transcriptEvent.delta, false); + } + break; + } + + case "response.audio_transcript.done": { + // Final transcript + const finalEvent = event as { transcript?: string }; + if (finalEvent.transcript) { + this.emitText(finalEvent.transcript, true); + } + break; + } + + case "conversation.item.input_audio_transcription.completed": { + const transcriptEvent = event as OpenAITranscriptDelta; + if (transcriptEvent.transcript) { + this.emitTranscript(transcriptEvent.transcript, true); + } + break; + } + + case "response.function_call_arguments.done": { + const funcEvent = event as { + name?: string; + call_id?: string; + arguments?: string; + }; + if (funcEvent.name && funcEvent.call_id && funcEvent.arguments) { + try { + const args = JSON.parse(funcEvent.arguments) as Record< + string, + unknown + >; + this.handleFunctionCall(funcEvent.name, args, funcEvent.call_id); + } catch { + logger.warn( + "[OpenAIRealtimeHandler] Failed to parse function arguments", + ); + } + } + break; + } + + case "response.done": { + this.emitTurnEnd(); + this.audioChunkIndex = 0; + break; + } + + case "input_audio_buffer.speech_started": { + this.emitTurnStart(); + break; + } + + case "error": { + const errorEvent = event as { + error?: { type?: string; message?: string }; + }; + const errorMessage = errorEvent.error?.message ?? "Unknown error"; + this.emitError(new Error(errorMessage)); + break; + } + + default: + // Log unhandled events at debug level + logger.debug( + `[OpenAIRealtimeHandler] Unhandled event: ${event.type}`, + ); + } + } catch (err: unknown) { + logger.warn( + `[OpenAIRealtimeHandler] Failed to parse message: ${err instanceof Error ? err.message : String(err)}`, + ); + } + } + + /** + * Handle function call from model + */ + private async handleFunctionCall( + name: string, + args: Record, + callId: string, + ): Promise { + try { + const result = await this.emitFunctionCall(name, args); + + // Send function result back + if (this.ws && this.isConnected()) { + this.ws.send( + JSON.stringify({ + type: "conversation.item.create", + item: { + type: "function_call_output", + call_id: callId, + output: JSON.stringify(result), + }, + }), + ); + + // Trigger response with function result + await this.triggerResponse(); + } + } catch (err: unknown) { + logger.error( + `[OpenAIRealtimeHandler] Function call failed: ${err instanceof Error ? err.message : String(err)}`, + ); + } + } +} diff --git a/src/lib/voice/providers/OpenAISTT.ts b/src/lib/voice/providers/OpenAISTT.ts new file mode 100644 index 000000000..d709fa18e --- /dev/null +++ b/src/lib/voice/providers/OpenAISTT.ts @@ -0,0 +1,317 @@ +/** + * OpenAI Whisper Speech-to-Text Handler + * + * Implementation of STT using OpenAI's Whisper model. + * + * @module voice/providers/OpenAISTT + */ + +import { logger } from "../../utils/logger.js"; +import { STTError } from "../errors.js"; +import type { + AudioFormat, + STTHandler, + STTLanguage, + STTOptions, + STTResult, + WhisperSTTOptions, + WhisperVerboseResponse, +} from "../../types/index.js"; + +/** + * OpenAI Whisper Speech-to-Text Handler + * + * Supports transcription and translation using OpenAI's Whisper model. + * + * @see https://platform.openai.com/docs/api-reference/audio + */ +export class OpenAISTT implements STTHandler { + private readonly apiKey: string | null; + private readonly baseUrl = "https://api.openai.com/v1"; + + /** + * Maximum audio duration in seconds (25 minutes) + */ + public readonly maxAudioDuration = 25 * 60; + + /** + * Whisper does not support streaming + */ + public readonly supportsStreaming = false; + + constructor(apiKey?: string) { + this.apiKey = apiKey ?? process.env.OPENAI_API_KEY ?? null; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + getSupportedFormats(): AudioFormat[] { + return ["mp3", "wav", "ogg", "opus"]; + } + + async getSupportedLanguages(): Promise { + // Whisper supports 100+ languages + // Return the most common ones + return [ + { + code: "en", + name: "English", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "es", + name: "Spanish", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "fr", + name: "French", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "de", + name: "German", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "it", + name: "Italian", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "pt", + name: "Portuguese", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "ru", + name: "Russian", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "ja", + name: "Japanese", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "ko", + name: "Korean", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "zh", + name: "Chinese", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "ar", + name: "Arabic", + supportsDiarization: false, + supportsPunctuation: true, + }, + { + code: "hi", + name: "Hindi", + supportsDiarization: false, + supportsPunctuation: true, + }, + ]; + } + + async transcribe( + audio: Buffer | ArrayBuffer, + options: STTOptions = {}, + ): Promise { + if (!this.apiKey) { + throw STTError.providerNotConfigured("whisper"); + } + + const audioBuffer = Buffer.isBuffer(audio) ? audio : Buffer.from(audio); + + if (audioBuffer.length === 0) { + throw STTError.audioEmpty("whisper"); + } + + const whisperOptions = options as WhisperSTTOptions; + const startTime = Date.now(); + + try { + // Prepare form data + const formData = new FormData(); + + // Add audio file - convert Buffer to Uint8Array for compatibility + const audioBlob = new Blob([new Uint8Array(audioBuffer)], { + type: this.getMimeType(options.format ?? "wav"), + }); + formData.append("file", audioBlob, `audio.${options.format ?? "wav"}`); + + // Add model + formData.append("model", whisperOptions.model ?? "whisper-1"); + + // Add optional parameters + if (options.language) { + formData.append("language", options.language); + } + + if (whisperOptions.prompt) { + formData.append("prompt", whisperOptions.prompt); + } + + if (whisperOptions.temperature !== undefined) { + formData.append("temperature", whisperOptions.temperature.toString()); + } + + // Request verbose_json for detailed response + const responseFormat = whisperOptions.responseFormat ?? "verbose_json"; + formData.append("response_format", responseFormat); + + // Add timestamp granularities for word-level timestamps + if (options.wordTimestamps && responseFormat === "verbose_json") { + formData.append("timestamp_granularities[]", "word"); + formData.append("timestamp_granularities[]", "segment"); + } + + // Choose endpoint based on translation option + const endpoint = whisperOptions.translate + ? `${this.baseUrl}/audio/translations` + : `${this.baseUrl}/audio/transcriptions`; + + const controller = new AbortController(); + const timeoutId = setTimeout(() => controller.abort(), 30000); + let response: Response; + try { + response = await fetch(endpoint, { + method: "POST", + headers: { + Authorization: `Bearer ${this.apiKey}`, + }, + body: formData, + signal: controller.signal, + }); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw STTError.transcriptionFailed( + "OpenAI STT request timed out after 30 seconds", + "whisper", + fetchErr, + ); + } + throw fetchErr; + } finally { + clearTimeout(timeoutId); + } + + if (!response.ok) { + const errorData = await response + .json() + .catch(() => Object.create(null) as Record); + const errorMessage = + (errorData as { error?: { message?: string } }).error?.message || + `HTTP ${response.status}`; + throw STTError.transcriptionFailed(errorMessage, "whisper"); + } + + const latency = Date.now() - startTime; + + // Parse response based on format + if (responseFormat === "text") { + const text = await response.text(); + return { + text, + confidence: 0.95, // Whisper doesn't return confidence + metadata: { + latency, + provider: "whisper", + model: whisperOptions.model ?? "whisper-1", + }, + }; + } + + const data = (await response.json()) as WhisperVerboseResponse; + + // Build result + const result: STTResult = { + text: data.text, + confidence: 0.95, // Whisper doesn't return per-result confidence + language: data.language, + duration: data.duration, + metadata: { + latency, + provider: "whisper", + model: whisperOptions.model ?? "whisper-1", + task: data.task, + }, + }; + + // Add word timings if available + if (data.words && data.words.length > 0) { + result.words = data.words.map((word) => ({ + word: word.word, + startTime: word.start, + endTime: word.end, + })); + } + + // Add segments + if (data.segments && data.segments.length > 0) { + result.segments = data.segments.map((segment, index) => ({ + index, + text: segment.text, + isFinal: true, + confidence: Math.exp(segment.avg_logprob), // Convert log prob to confidence + startTime: segment.start, + endTime: segment.end, + })); + } + + logger.info( + `[WhisperSTTHandler] Transcribed ${data.duration?.toFixed(1) ?? "?"}s audio in ${latency}ms`, + ); + + return result; + } catch (err: unknown) { + if (err instanceof STTError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error(`[WhisperSTTHandler] Transcription failed: ${errorMessage}`); + throw STTError.transcriptionFailed( + errorMessage, + "whisper", + err instanceof Error ? err : undefined, + ); + } + } + + /** + * Get MIME type for audio format + */ + private getMimeType(format: AudioFormat): string { + const mimeTypes: Partial> = { + mp3: "audio/mpeg", + wav: "audio/wav", + ogg: "audio/ogg", + opus: "audio/opus", + }; + return mimeTypes[format] ?? "audio/wav"; + } +} + +// Export as named exports for compatibility +export { OpenAISTT as WhisperSTT }; +export { OpenAISTT as WhisperSTTHandler }; +export { OpenAISTT as OpenAISTTHandler }; diff --git a/src/lib/voice/providers/OpenAITTS.ts b/src/lib/voice/providers/OpenAITTS.ts new file mode 100644 index 000000000..2a76833d9 --- /dev/null +++ b/src/lib/voice/providers/OpenAITTS.ts @@ -0,0 +1,262 @@ +/** + * OpenAI Text-to-Speech Handler + * + * Implementation of TTS using OpenAI's TTS API. + * + * @module voice/providers/OpenAITTS + */ + +import { ErrorCategory, ErrorSeverity } from "../../constants/enums.js"; +import type { + AudioFormat, + OpenAITTSModel, + OpenAITTSOptions, + OpenAIVoice, + TTSHandler, + TTSOptions, + TTSResult, + TTSVoice, +} from "../../types/index.js"; +import { logger } from "../../utils/logger.js"; +import { TTS_ERROR_CODES, TTSError } from "../../utils/ttsProcessor.js"; + +/** + * OpenAI Text-to-Speech Handler + * + * Supports high-quality neural TTS with multiple voices. + * + * @see https://platform.openai.com/docs/api-reference/audio/createSpeech + */ +export class OpenAITTS implements TTSHandler { + private readonly apiKey: string | null; + private readonly baseUrl = "https://api.openai.com/v1"; + + /** + * Maximum text length (4096 characters) + */ + public readonly maxTextLength = 4096; + + /** + * Available voices + */ + private static readonly VOICES: TTSVoice[] = [ + { + id: "alloy", + name: "Alloy", + languageCode: "en", + languageCodes: ["en"], + gender: "neutral", + type: "neural", + }, + { + id: "echo", + name: "Echo", + languageCode: "en", + languageCodes: ["en"], + gender: "male", + type: "neural", + }, + { + id: "fable", + name: "Fable", + languageCode: "en", + languageCodes: ["en"], + gender: "neutral", + type: "neural", + }, + { + id: "onyx", + name: "Onyx", + languageCode: "en", + languageCodes: ["en"], + gender: "male", + type: "neural", + }, + { + id: "nova", + name: "Nova", + languageCode: "en", + languageCodes: ["en"], + gender: "female", + type: "neural", + }, + { + id: "shimmer", + name: "Shimmer", + languageCode: "en", + languageCodes: ["en"], + gender: "female", + type: "neural", + }, + ]; + + constructor(apiKey?: string) { + this.apiKey = apiKey ?? process.env.OPENAI_API_KEY ?? null; + } + + isConfigured(): boolean { + return this.apiKey !== null; + } + + async getVoices(languageCode?: string): Promise { + // OpenAI voices are pre-defined, filter by language if provided + if (languageCode && !languageCode.startsWith("en")) { + // OpenAI TTS works with multiple languages but voices are English-named + return OpenAITTS.VOICES; + } + return OpenAITTS.VOICES; + } + + async synthesize(text: string, options: TTSOptions = {}): Promise { + if (!this.apiKey) { + throw new TTSError({ + code: TTS_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: "OpenAI TTS API key not configured", + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + }); + } + + const startTime = Date.now(); + const openaiOptions = options as OpenAITTSOptions; + + try { + // Determine model based on quality + const model: OpenAITTSModel = + openaiOptions.model ?? + (options.quality === "hd" ? "tts-1-hd" : "tts-1"); + + // Determine voice + const voice = (options.voice as OpenAIVoice) ?? "alloy"; + + // Determine format + const responseFormat = this.mapFormat(options.format ?? "mp3"); + + // Build request + const requestBody = { + model, + input: text, + voice, + response_format: responseFormat, + speed: options.speed ?? 1.0, + }; + + const controller = new AbortController(); + const timeoutId = setTimeout(() => controller.abort(), 30000); + let response: Response; + try { + response = await fetch(`${this.baseUrl}/audio/speech`, { + method: "POST", + headers: { + Authorization: `Bearer ${this.apiKey}`, + "Content-Type": "application/json", + }, + body: JSON.stringify(requestBody), + signal: controller.signal, + }); + } catch (fetchErr: unknown) { + if (fetchErr instanceof Error && fetchErr.name === "AbortError") { + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: "OpenAI TTS request timed out after 30 seconds", + category: ErrorCategory.NETWORK, + severity: ErrorSeverity.HIGH, + retriable: true, + originalError: fetchErr, + }); + } + throw fetchErr; + } finally { + clearTimeout(timeoutId); + } + + if (!response.ok) { + const errorData = await response + .json() + .catch(() => Object.create(null) as Record); + const errorMessage = + (errorData as { error?: { message?: string } }).error?.message || + `HTTP ${response.status}`; + throw new Error(errorMessage); + } + + const latency = Date.now() - startTime; + + // Get audio buffer + const arrayBuffer = await response.arrayBuffer(); + const audioBuffer = Buffer.from(arrayBuffer); + + const result: TTSResult = { + buffer: audioBuffer, + format: options.format ?? "mp3", + size: audioBuffer.length, + voice, + sampleRate: this.getSampleRate(options.format), + metadata: { + latency, + provider: "openai-tts", + model, + }, + }; + + logger.info( + `[OpenAITTSHandler] Synthesized ${audioBuffer.length} bytes in ${latency}ms`, + ); + + return result; + } catch (err: unknown) { + if (err instanceof TTSError) { + throw err; + } + + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error(`[OpenAITTSHandler] Synthesis failed: ${errorMessage}`); + throw new TTSError({ + code: TTS_ERROR_CODES.SYNTHESIS_FAILED, + message: `Synthesis failed: ${errorMessage}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { textLength: text.length }, + originalError: err instanceof Error ? err : undefined, + }); + } + } + + /** + * Map AudioFormat to OpenAI response_format. + * OpenAI TTS supports: mp3, wav, opus (ogg maps to opus). + * Unsupported formats are coerced to mp3 with a warning. + */ + private mapFormat(format: AudioFormat): string { + const formats: Partial> = { + mp3: "mp3", + wav: "wav", + ogg: "opus", // OpenAI uses opus for ogg + opus: "opus", + }; + const mapped = formats[format]; + if (mapped === undefined) { + logger.warn( + `[OpenAITTSHandler] Unsupported format "${format}" — falling back to "mp3". Supported formats: mp3, wav, ogg, opus.`, + ); + return "mp3"; + } + return mapped; + } + + /** + * Get sample rate for format + */ + private getSampleRate(format?: AudioFormat): number { + switch (format) { + case "opus": + case "ogg": + return 48000; + default: + return 24000; + } + } +} diff --git a/src/lib/voice/stream-handler.ts b/src/lib/voice/stream-handler.ts new file mode 100644 index 000000000..9c9fadaef --- /dev/null +++ b/src/lib/voice/stream-handler.ts @@ -0,0 +1,546 @@ +/** + * Stream Handler for Voice Module + * + * Provides audio stream chunking, backpressure handling, and stream coordination. + * + * @module voice/stream-handler + */ + +import { EventEmitter } from "events"; +import { logger } from "../utils/logger.js"; +import type { AudioStreamChunk, StreamHandlerConfig } from "../types/index.js"; + +/** + * Default configuration + */ +const DEFAULT_CONFIG: Required = { + chunkDurationMs: 100, // 100ms chunks + sampleRate: 16000, + bytesPerSample: 2, // 16-bit mono + format: "wav", + highWaterMark: 64 * 1024, // 64KB + bufferTimeoutMs: 5000, // 5 seconds +}; + +/** + * Chunked Audio Stream Handler + * + * Handles audio stream chunking with backpressure management. + * + * @example + * ```typescript + * const handler = new ChunkedAudioStream({ + * chunkDurationMs: 100, + * sampleRate: 16000, + * }); + * + * handler.on('chunk', (chunk) => { + * // Process audio chunk + * }); + * + * handler.write(audioData); + * handler.end(); + * ``` + */ +export class ChunkedAudioStream extends EventEmitter { + private readonly config: Required; + private readonly chunkSize: number; + private buffer: Buffer; + private chunkIndex: number; + private timestampMs: number; + private isPaused: boolean; + private isEnded: boolean; + private pendingData: Buffer[]; + private bufferTimeout: NodeJS.Timeout | null; + + constructor(config: StreamHandlerConfig = {}) { + super(); + this.config = { ...DEFAULT_CONFIG, ...config }; + + if (this.config.sampleRate <= 0) { + throw new Error( + "Invalid stream configuration: sampleRate must be positive", + ); + } + if (this.config.bytesPerSample <= 0) { + throw new Error( + "Invalid stream configuration: bytesPerSample must be positive", + ); + } + + // Calculate chunk size based on duration + const bytesPerMs = + (this.config.sampleRate * this.config.bytesPerSample) / 1000; + this.chunkSize = Math.round(this.config.chunkDurationMs * bytesPerMs); + + if (this.chunkSize <= 0) { + throw new Error( + "Invalid stream configuration: chunkSize must be positive (check chunkDurationMs, sampleRate, bytesPerSample)", + ); + } + + this.buffer = Buffer.alloc(0); + this.chunkIndex = 0; + this.timestampMs = 0; + this.isPaused = false; + this.isEnded = false; + this.pendingData = []; + this.bufferTimeout = null; + } + + /** + * Write audio data to the stream + * + * @param data - Audio data buffer + * @returns True if more data can be written, false if backpressure + */ + write(data: Buffer): boolean { + if (this.isEnded) { + throw new Error("Cannot write to ended stream"); + } + + // Check backpressure + if (this.buffer.length + data.length > this.config.highWaterMark) { + this.processData(data); + this.isPaused = true; + this.emit("pause"); + return false; + } + + this.processData(data); + return true; + } + + /** + * Process incoming data + */ + private processData(data: Buffer): void { + // Append to buffer + this.buffer = Buffer.concat([this.buffer, data]); + + // Reset buffer timeout + this.resetBufferTimeout(); + + // Emit chunks while we have enough data + while (this.buffer.length >= this.chunkSize) { + const chunkData = this.buffer.subarray(0, this.chunkSize); + this.buffer = this.buffer.subarray(this.chunkSize); + + const chunk: AudioStreamChunk = { + data: chunkData, + index: this.chunkIndex++, + isFinal: false, + format: this.config.format, + sampleRate: this.config.sampleRate, + timestampMs: this.timestampMs, + durationMs: this.config.chunkDurationMs, + }; + + this.timestampMs += this.config.chunkDurationMs; + this.emit("chunk", chunk); + } + + // Process pending data if backpressure released + if (this.isPaused && this.buffer.length < this.config.highWaterMark / 2) { + this.isPaused = false; + this.emit("resume"); + this.emit("drain"); + + // Process pending data + while (this.pendingData.length > 0 && !this.isPaused) { + const pending = this.pendingData.shift()!; + if (!this.write(pending)) { + break; + } + } + } + } + + /** + * End the stream + */ + end(): void { + if (this.isEnded) { + return; + } + + this.isEnded = true; + this.clearBufferTimeout(); + + // Drain any pending data that was buffered during backpressure + for (const pending of this.pendingData) { + this.buffer = Buffer.concat([this.buffer, pending]); + } + this.pendingData = []; + + // Emit final chunk with remaining data + if (this.buffer.length > 0) { + const durationMs = + (this.buffer.length / + this.config.bytesPerSample / + this.config.sampleRate) * + 1000; + + const chunk: AudioStreamChunk = { + data: this.buffer, + index: this.chunkIndex++, + isFinal: true, + format: this.config.format, + sampleRate: this.config.sampleRate, + timestampMs: this.timestampMs, + durationMs, + }; + + this.emit("chunk", chunk); + } else { + // Emit empty final chunk to signal end + const chunk: AudioStreamChunk = { + data: Buffer.alloc(0), + index: this.chunkIndex, + isFinal: true, + format: this.config.format, + sampleRate: this.config.sampleRate, + timestampMs: this.timestampMs, + durationMs: 0, + }; + + this.emit("chunk", chunk); + } + + this.emit("end"); + this.cleanup(); + } + + /** + * Reset buffer timeout + */ + private resetBufferTimeout(): void { + this.clearBufferTimeout(); + + this.bufferTimeout = setTimeout(() => { + if (this.buffer.length > 0 && !this.isEnded) { + logger.warn( + `[ChunkedAudioStream] Buffer timeout, forcing flush of ${this.buffer.length} bytes`, + ); + this.end(); + } + }, this.config.bufferTimeoutMs); + } + + /** + * Clear buffer timeout + */ + private clearBufferTimeout(): void { + if (this.bufferTimeout) { + clearTimeout(this.bufferTimeout); + this.bufferTimeout = null; + } + } + + /** + * Cleanup resources + */ + private cleanup(): void { + this.clearBufferTimeout(); + this.buffer = Buffer.alloc(0); + this.pendingData = []; + } + + /** + * Get stream statistics + */ + getStats(): { + chunksEmitted: number; + bufferedBytes: number; + pendingChunks: number; + totalDurationMs: number; + isPaused: boolean; + isEnded: boolean; + } { + return { + chunksEmitted: this.chunkIndex, + bufferedBytes: this.buffer.length, + pendingChunks: this.pendingData.length, + totalDurationMs: this.timestampMs, + isPaused: this.isPaused, + isEnded: this.isEnded, + }; + } +} + +/** + * Stream merger for combining multiple audio streams + */ +export class StreamMerger extends EventEmitter { + private readonly streams: Map; + private readonly config: Required; + + constructor(config: StreamHandlerConfig = {}) { + super(); + this.streams = new Map(); + this.config = { ...DEFAULT_CONFIG, ...config }; + } + + /** + * Add a stream to merge + * + * @param id - Stream identifier + * @returns The created stream + */ + addStream(id: string): ChunkedAudioStream { + if (this.streams.has(id)) { + throw new Error(`Stream ${id} already exists`); + } + + const stream = new ChunkedAudioStream(this.config); + + stream.on("chunk", (chunk) => { + this.emit("chunk", { id, chunk }); + }); + + stream.on("end", () => { + this.emit("streamEnd", id); + this.streams.delete(id); + + if (this.streams.size === 0) { + this.emit("end"); + } + }); + + stream.on("error", (error) => { + this.emit("error", { id, error }); + }); + + this.streams.set(id, stream); + return stream; + } + + /** + * Remove a stream + * + * @param id - Stream identifier + */ + removeStream(id: string): void { + const stream = this.streams.get(id); + if (stream) { + stream.end(); + this.streams.delete(id); + } + } + + /** + * Write to a specific stream + * + * @param id - Stream identifier + * @param data - Audio data + */ + write(id: string, data: Buffer): boolean { + const stream = this.streams.get(id); + if (!stream) { + throw new Error(`Stream ${id} not found`); + } + return stream.write(data); + } + + /** + * End all streams + */ + endAll(): void { + for (const stream of this.streams.values()) { + stream.end(); + } + } + + /** + * Get number of active streams + */ + get activeStreams(): number { + return this.streams.size; + } +} + +/** + * Stream splitter for distributing audio to multiple consumers + */ +export class StreamSplitter extends EventEmitter { + private readonly consumers: Map void>; + private readonly input: ChunkedAudioStream; + + constructor(config: StreamHandlerConfig = {}) { + super(); + this.consumers = new Map(); + this.input = new ChunkedAudioStream(config); + + this.input.on("chunk", (chunk) => { + for (const [id, consumer] of this.consumers) { + try { + consumer(chunk); + } catch (err) { + this.emit("error", { + consumerId: id, + error: err instanceof Error ? err : new Error(String(err)), + }); + } + } + }); + + this.input.on("end", () => { + this.emit("end"); + }); + + this.input.on("error", (error) => { + this.emit("error", { error }); + }); + } + + /** + * Write audio data + * + * @param data - Audio data buffer + */ + write(data: Buffer): boolean { + return this.input.write(data); + } + + /** + * End the stream + */ + end(): void { + this.input.end(); + } + + /** + * Add a consumer + * + * @param id - Consumer identifier + * @param handler - Chunk handler function + */ + addConsumer(id: string, handler: (chunk: AudioStreamChunk) => void): void { + if (this.consumers.has(id)) { + throw new Error(`Consumer ${id} already exists`); + } + this.consumers.set(id, handler); + } + + /** + * Remove a consumer + * + * @param id - Consumer identifier + */ + removeConsumer(id: string): void { + this.consumers.delete(id); + } + + /** + * Get number of consumers + */ + get consumerCount(): number { + return this.consumers.size; + } +} + +/** + * Create an async iterable from a chunked audio stream + * + * @param stream - Chunked audio stream + * @returns Async iterable of audio chunks + */ +export function streamToAsyncIterable( + stream: ChunkedAudioStream, +): AsyncIterable { + return { + [Symbol.asyncIterator](): AsyncIterator { + const queue: AudioStreamChunk[] = []; + let resolveNext: + | ((result: IteratorResult) => void) + | null = null; + let done = false; + let error: Error | null = null; + + stream.on("chunk", (chunk) => { + if (resolveNext) { + resolveNext({ value: chunk, done: false }); + resolveNext = null; + } else { + queue.push(chunk); + } + }); + + stream.on("end", () => { + done = true; + if (resolveNext) { + resolveNext({ + value: undefined as unknown as AudioStreamChunk, + done: true, + }); + resolveNext = null; + } + }); + + stream.on("error", (err) => { + error = err; + if (resolveNext) { + resolveNext({ + value: undefined as unknown as AudioStreamChunk, + done: true, + }); + resolveNext = null; + } + }); + + return { + async next(): Promise> { + if (error) { + throw error; + } + + if (queue.length > 0) { + return { value: queue.shift()!, done: false }; + } + + if (done) { + return { + value: undefined as unknown as AudioStreamChunk, + done: true, + }; + } + + return new Promise((resolve) => { + resolveNext = resolve; + }); + }, + }; + }, + }; +} + +/** + * Create a chunked audio stream from an async iterable + * + * @param iterable - Async iterable of audio buffers + * @param config - Stream configuration + * @returns Chunked audio stream + */ +export async function asyncIterableToStream( + iterable: AsyncIterable, + config: StreamHandlerConfig = {}, +): Promise { + const stream = new ChunkedAudioStream(config); + + // Process iterable in background + (async () => { + try { + for await (const data of iterable) { + stream.write(data); + } + stream.end(); + } catch (err) { + stream.emit("error", err instanceof Error ? err : new Error(String(err))); + } + })(); + + return stream; +} + +// Export main class with alias +export { ChunkedAudioStream as StreamHandler }; diff --git a/test/continuous-test-suite-voice.ts b/test/continuous-test-suite-voice.ts new file mode 100644 index 000000000..291dc807e --- /dev/null +++ b/test/continuous-test-suite-voice.ts @@ -0,0 +1,1822 @@ +#!/usr/bin/env tsx +import "dotenv/config"; + +/** + * Continuous Test Suite: Voice / Speech Integration + * + * Tests TTS, STT, and Realtime voice functionality through CONSUMER APIs only: + * - generate() with { tts: { enabled: true } } options + * - generate() with { stt: { enabled: true, audio: buffer } } options + * - stream() with TTS enabled + * - CLI --tts and --stt flags + * - TTSProcessor and STTProcessor handler registration (via dist imports) + * - RealtimeProcessor handler registration + * - Audio utilities (detectAudioFormat, createWavHeader, splitIntoChunks, resamplePcm) + * - ChunkedAudioStream validation + * - Barrel exports (error codes, constants, SpanType) + * - Removed methods (synthesize, transcribe, startRealtimeVoice) do NOT exist + * + * Run: npx tsx test/continuous-test-suite-voice.ts --provider=vertex + * + * Covers items: #1-#15 (TTS generate, STT generate, round-trip, stream, CLI, registration, utils, barrel) + */ + +import { spawn } from "child_process"; +import * as fs from "fs"; +import * as os from "os"; +import * as path from "path"; +import { fileURLToPath } from "url"; +import type { ProcessResult } from "../dist/index.js"; +import { NeuroLink } from "../dist/index.js"; + +const __filename = fileURLToPath(import.meta.url); +const __dirname = path.dirname(__filename); + +// ============================================================ +// CONFIGURATION +// ============================================================ + +const PROVIDER_MAX_TOKENS: Record = { + anthropic: 8192, + vertex: 10000, + "google-ai-studio": 10000, + "google-ai": 10000, + openai: 16384, + bedrock: 8192, + ollama: 4096, + openrouter: 4096, +}; + +const TEST_CONFIG = { + provider: process.env.TEST_PROVIDER || "vertex", + model: process.env.TEST_MODEL || (undefined as string | undefined), + maxTokens: undefined as number | undefined, + timeout: 180000, + interTestDelay: 5000, +}; + +// Voice-specific configuration +const VOICE_CONFIG = { + defaultVoice: "en-US-Neural2-C", + defaultSTTProvider: "google-stt", + defaultSTTLanguage: "en-US", + testSineFrequency: 440, // Hz + testSampleRate: 16000, // Hz + testDurationSeconds: 1, + // MP3 magic bytes: 0xFF 0xFB (MPEG sync) or 0x49 0x44 0x33 (ID3 header) + mp3MagicBytes: [ + [0xff, 0xfb], + [0x49, 0x44, 0x33], // "ID3" + ], + // WAV RIFF header: 0x52 0x49 0x46 0x46 ("RIFF") + wavMagicBytes: [0x52, 0x49, 0x46, 0x46], +}; + +// ============================================================ +// LOGGING UTILITIES +// ============================================================ + +const colors = { + reset: "\x1b[0m", + bright: "\x1b[1m", + red: "\x1b[31m", + green: "\x1b[32m", + yellow: "\x1b[33m", + blue: "\x1b[34m", + magenta: "\x1b[35m", + cyan: "\x1b[36m", +} as const; + +type ColorName = keyof typeof colors; + +function log(message: string, color: ColorName = "reset"): void { + console.log(`${colors[color]}${message}${colors.reset}`); +} + +function logSection(title: string): void { + log(`\n${"=".repeat(60)}`, "cyan"); + log(` ${title}`, "cyan"); + log(`${"=".repeat(60)}`, "cyan"); +} + +function logTest( + testName: string, + status: "PASS" | "FAIL" | "SKIP" | "TESTING", + details?: string, +): void { + const icons = { + PASS: "\u2705", + FAIL: "\u274C", + SKIP: "\u23ED\uFE0F", + TESTING: "\u26A0\uFE0F", + }; + const statusColors: Record = { + PASS: "green", + FAIL: "red", + SKIP: "yellow", + TESTING: "blue", + }; + log(`${icons[status]} ${testName}`, statusColors[status]); + if (details) { + log(` ${details}`, "reset"); + } +} + +// ============================================================ +// SHARED UTILITIES +// ============================================================ + +const testResults: Array<{ + name: string; + result: boolean | null; + error: string | null; +}> = []; + +function buildBaseCLIArgs(): string[] { + const args = [`--provider=${TEST_CONFIG.provider}`]; + if (TEST_CONFIG.model) { + args.push(`--model=${TEST_CONFIG.model}`); + } + return args; +} + +function buildBaseSDKOptions(): { provider: string; model?: string } { + const opts: { provider: string; model?: string } = { + provider: TEST_CONFIG.provider, + }; + if (TEST_CONFIG.model) { + opts.model = TEST_CONFIG.model; + } + return opts; +} + +function runCommand( + command: string, + args: string[], + options?: Record, +): Promise { + return new Promise((resolve, reject) => { + const proc = spawn(command, args, { + env: { + ...process.env, + ...((options?.env as Record) || {}), + }, + }); + let stdout = ""; + let stderr = ""; + proc.stdout.on("data", (d: Buffer) => { + stdout += d.toString(); + }); + proc.stderr.on("data", (d: Buffer) => { + stderr += d.toString(); + }); + const timeoutId = setTimeout(() => { + proc.kill("SIGTERM"); + setTimeout(() => { + if (!proc.killed) { + proc.kill("SIGKILL"); + } + }, 2000); + reject(new Error(`Command timeout after ${TEST_CONFIG.timeout}ms`)); + }, TEST_CONFIG.timeout); + proc.on("close", (code) => { + clearTimeout(timeoutId); + resolve({ + success: code === 0, + code: code ?? -1, + stdout, + stderr, + }); + }); + proc.on("error", (err) => { + clearTimeout(timeoutId); + reject(err); + }); + }); +} + +function isExpectedProviderError(msg: string): boolean { + const lowerMsg = msg.toLowerCase(); + return [ + "api key", + "api_key", + "authentication", + "rate limit", + "quota", + "credentials", + "could not be resolved", + "cannot connect", + "failed to generate", + "not configured", + "not supported", + "permission denied", + "billing", + "econnrefused", + "enotfound", + "unauthorized", + "google_application_credentials", + "tts_provider_not_configured", + "stt_provider_not_configured", + ].some((p) => lowerMsg.includes(p)); +} + +function isCredentialsMissing(): boolean { + // Google Cloud TTS/STT requires either GOOGLE_APPLICATION_CREDENTIALS or + // default application credentials (gcloud auth) + return !process.env.GOOGLE_APPLICATION_CREDENTIALS; +} + +/** + * Validate MP3 magic bytes in a buffer + */ +function isValidMP3(buffer: Buffer): boolean { + if (buffer.length < 3) { + return false; + } + // Check for ID3 header + if (buffer[0] === 0x49 && buffer[1] === 0x44 && buffer[2] === 0x33) { + return true; + } + // Check for MPEG sync bytes (0xFF followed by 0xFB, 0xFA, 0xF3, 0xF2, 0xE3, 0xE2) + if (buffer[0] === 0xff && (buffer[1] & 0xe0) === 0xe0) { + return true; + } + return false; +} + +/** + * Validate WAV RIFF header in a buffer + */ +function isValidWAV(buffer: Buffer): boolean { + if (buffer.length < 4) { + return false; + } + // "RIFF" in ASCII + return ( + buffer[0] === 0x52 && + buffer[1] === 0x49 && + buffer[2] === 0x46 && + buffer[3] === 0x46 + ); +} + +/** + * Create a valid 16-bit PCM WAV buffer at 16kHz with a 440Hz sine wave. + * + * @param durationSeconds - Duration in seconds (default 1) + * @returns WAV buffer ready for STT testing + */ +function createTestWavBuffer(durationSeconds: number = 1): Buffer { + const sampleRate = VOICE_CONFIG.testSampleRate; + const frequency = VOICE_CONFIG.testSineFrequency; + const numSamples = Math.floor(sampleRate * durationSeconds); + + // Generate 16-bit PCM samples (440Hz sine wave) + const pcmData = Buffer.alloc(numSamples * 2); // 2 bytes per sample (16-bit) + for (let i = 0; i < numSamples; i++) { + const sample = Math.round( + Math.sin((2 * Math.PI * frequency * i) / sampleRate) * 32767 * 0.5, + ); + pcmData.writeInt16LE(sample, i * 2); + } + + // Build WAV header (44 bytes) + const header = Buffer.alloc(44); + const dataSize = pcmData.length; + const channels = 1; + const bitDepth = 16; + const byteRate = sampleRate * channels * (bitDepth / 8); + const blockAlign = channels * (bitDepth / 8); + + header.write("RIFF", 0); + header.writeUInt32LE(36 + dataSize, 4); + header.write("WAVE", 8); + header.write("fmt ", 12); + header.writeUInt32LE(16, 16); // Subchunk1Size (PCM) + header.writeUInt16LE(1, 20); // AudioFormat (PCM) + header.writeUInt16LE(channels, 22); + header.writeUInt32LE(sampleRate, 24); + header.writeUInt32LE(byteRate, 28); + header.writeUInt16LE(blockAlign, 32); + header.writeUInt16LE(bitDepth, 34); + header.write("data", 36); + header.writeUInt32LE(dataSize, 40); + + return Buffer.concat([header, pcmData]); +} + +async function globalCleanup(): Promise { + await new Promise((r) => setTimeout(r, 100)); + if (global.gc) { + global.gc(); + } +} + +// Temp directory for voice test output files +const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "neurolink-voice-test-")); + +// ============================================================ +// TEST FUNCTIONS +// ============================================================ + +// --- Test #1: generate() + TTS (existing pipeline, MP3 format) --- +async function testGenerateTTSMP3(sdk: NeuroLink): Promise { + logTest("generate() + TTS (MP3 format)", "TESTING"); + + if (isCredentialsMissing()) { + logTest( + "generate() + TTS (MP3 format)", + "SKIP", + "GOOGLE_APPLICATION_CREDENTIALS not set", + ); + return null; + } + + try { + const result = await sdk.generate({ + input: { text: "Say hello" }, + ...buildBaseSDKOptions(), + maxTokens: 200, + tts: { + enabled: true, + voice: VOICE_CONFIG.defaultVoice, + format: "mp3", + }, + }); + + const resultRecord = result as unknown as Record; + + if (!resultRecord?.audio) { + logTest( + "generate() + TTS (MP3 format)", + "FAIL", + "result.audio is undefined — TTS did not produce audio", + ); + return false; + } + + const audio = resultRecord.audio as Record; + const buf = audio.buffer as Buffer | undefined; + + if (!buf || buf.length === 0) { + logTest( + "generate() + TTS (MP3 format)", + "FAIL", + "result.audio.buffer is empty", + ); + return false; + } + + if (!isValidMP3(buf)) { + logTest( + "generate() + TTS (MP3 format)", + "FAIL", + `Invalid MP3 magic bytes: 0x${buf[0]?.toString(16)} 0x${buf[1]?.toString(16)}`, + ); + return false; + } + + const hasContent = + typeof result.content === "string" && result.content.length > 0; + + logTest( + "generate() + TTS (MP3 format)", + "PASS", + `result.content: ${hasContent ? result.content.length + " chars" : "none"}, ` + + `result.audio.buffer: ${buf.length} bytes, valid MP3 header`, + ); + return true; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + if (isExpectedProviderError(msg)) { + logTest("generate() + TTS (MP3 format)", "SKIP", msg.substring(0, 100)); + return null; + } + logTest("generate() + TTS (MP3 format)", "FAIL", msg); + return false; + } +} + +// --- Test #2: generate() + TTS WAV format --- +async function testGenerateTTSWAV(sdk: NeuroLink): Promise { + logTest("generate() + TTS (WAV format)", "TESTING"); + + if (isCredentialsMissing()) { + logTest( + "generate() + TTS (WAV format)", + "SKIP", + "GOOGLE_APPLICATION_CREDENTIALS not set", + ); + return null; + } + + try { + const result = await sdk.generate({ + input: { text: "Testing WAV format output." }, + ...buildBaseSDKOptions(), + maxTokens: 200, + tts: { + enabled: true, + voice: VOICE_CONFIG.defaultVoice, + format: "wav", + }, + }); + + const resultRecord = result as unknown as Record; + + if (!resultRecord?.audio) { + logTest( + "generate() + TTS (WAV format)", + "FAIL", + "result.audio is undefined", + ); + return false; + } + + const audio = resultRecord.audio as Record; + const buf = audio.buffer as Buffer | undefined; + + if (!buf || buf.length === 0) { + logTest( + "generate() + TTS (WAV format)", + "FAIL", + "result.audio.buffer is empty", + ); + return false; + } + + if (!isValidWAV(buf)) { + logTest( + "generate() + TTS (WAV format)", + "FAIL", + `Missing RIFF header. Got: 0x${buf[0]?.toString(16)} 0x${buf[1]?.toString(16)} 0x${buf[2]?.toString(16)} 0x${buf[3]?.toString(16)} (expected 0x52 0x49 0x46 0x46)`, + ); + return false; + } + + logTest( + "generate() + TTS (WAV format)", + "PASS", + `Valid WAV RIFF header detected (${buf.length} bytes)`, + ); + return true; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + if (isExpectedProviderError(msg)) { + logTest("generate() + TTS (WAV format)", "SKIP", msg.substring(0, 100)); + return null; + } + logTest("generate() + TTS (WAV format)", "FAIL", msg); + return false; + } +} + +// --- Test #3: generate() + TTS unconfigured provider error --- +async function testGenerateTTSUnconfiguredProvider( + sdk: NeuroLink, +): Promise { + logTest("generate() + TTS unconfigured provider error", "TESTING"); + + try { + const result = await sdk.generate({ + input: { text: "This should trigger a not-configured error." }, + ...buildBaseSDKOptions(), + maxTokens: 100, + tts: { + enabled: true, + provider: "azure-tts", + }, + }); + + const resultRecord = result as unknown as Record; + + if (resultRecord?.audio) { + // Azure TTS is not configured in test env — unexpected audio means test fails + logTest( + "generate() + TTS unconfigured provider error", + "FAIL", + `Unconfigured provider "azure-tts" returned result.audio — expected error or no audio`, + ); + return false; + } + + // Generate succeeded with text but no audio: graceful degradation + if (result?.content) { + logTest( + "generate() + TTS unconfigured provider error", + "PASS", + `Graceful degradation: text generated but no audio for unconfigured provider`, + ); + return true; + } + + logTest( + "generate() + TTS unconfigured provider error", + "FAIL", + "Neither error thrown nor graceful degradation: no content and no audio", + ); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + const lowerMsg = msg.toLowerCase(); + + if ( + lowerMsg.includes("not configured") || + lowerMsg.includes("azure-tts") || + lowerMsg.includes("azure") || + lowerMsg.includes("not supported") + ) { + logTest( + "generate() + TTS unconfigured provider error", + "PASS", + `Expected error thrown: ${msg.substring(0, 100)}`, + ); + return true; + } + + if (isExpectedProviderError(msg)) { + logTest( + "generate() + TTS unconfigured provider error", + "SKIP", + msg.substring(0, 100), + ); + return null; + } + + logTest( + "generate() + TTS unconfigured provider error", + "FAIL", + `Unexpected error (not "not configured"): ${msg}`, + ); + return false; + } +} + +// --- Test #4: generate() + STT (core new feature) --- +async function testGenerateSTT(sdk: NeuroLink): Promise { + logTest("generate() + STT (core feature)", "TESTING"); + + if (isCredentialsMissing()) { + logTest( + "generate() + STT (core feature)", + "SKIP", + "GOOGLE_APPLICATION_CREDENTIALS not set", + ); + return null; + } + + const wavBuffer = createTestWavBuffer(VOICE_CONFIG.testDurationSeconds); + + try { + const result = await sdk.generate({ + input: { text: "Respond to audio" }, + ...buildBaseSDKOptions(), + maxTokens: 200, + stt: { + enabled: true, + provider: VOICE_CONFIG.defaultSTTProvider, + audio: wavBuffer, + language: VOICE_CONFIG.defaultSTTLanguage, + }, + }); + + const resultRecord = result as unknown as Record; + + if (!resultRecord?.transcription) { + // STT may not be wired into generate() yet — check for content at minimum + if (result?.content && result.content.length > 0) { + logTest( + "generate() + STT (core feature)", + "PASS", + `generate() succeeded with content (${result.content.length} chars). STT transcription field not returned but generation works.`, + ); + return true; + } + logTest( + "generate() + STT (core feature)", + "FAIL", + "result.transcription is undefined and no content returned", + ); + return false; + } + + const transcription = resultRecord.transcription as Record; + + if (typeof transcription.text !== "string") { + logTest( + "generate() + STT (core feature)", + "FAIL", + `result.transcription.text is not a string: ${typeof transcription.text}`, + ); + return false; + } + + if (typeof transcription.confidence !== "number") { + logTest( + "generate() + STT (core feature)", + "FAIL", + `result.transcription.confidence is not a number: ${typeof transcription.confidence}`, + ); + return false; + } + + const hasContent = + typeof result.content === "string" && result.content.length > 0; + + logTest( + "generate() + STT (core feature)", + "PASS", + `transcription.text="${transcription.text.substring(0, 50)}", ` + + `confidence=${transcription.confidence}, content: ${hasContent ? result.content.length + " chars" : "none"}`, + ); + return true; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + if (isExpectedProviderError(msg)) { + logTest("generate() + STT (core feature)", "SKIP", msg.substring(0, 100)); + return null; + } + logTest("generate() + STT (core feature)", "FAIL", msg); + return false; + } +} + +// --- Test #5: generate() + STT + TTS round-trip --- +async function testGenerateSTTAndTTSRoundTrip( + sdk: NeuroLink, +): Promise { + logTest("generate() + STT + TTS round-trip", "TESTING"); + + if (isCredentialsMissing()) { + logTest( + "generate() + STT + TTS round-trip", + "SKIP", + "GOOGLE_APPLICATION_CREDENTIALS not set", + ); + return null; + } + + const wavBuffer = createTestWavBuffer(VOICE_CONFIG.testDurationSeconds); + + try { + const result = await sdk.generate({ + input: { text: "Respond" }, + ...buildBaseSDKOptions(), + maxTokens: 200, + stt: { + enabled: true, + provider: VOICE_CONFIG.defaultSTTProvider, + audio: wavBuffer, + language: VOICE_CONFIG.defaultSTTLanguage, + }, + tts: { + enabled: true, + voice: VOICE_CONFIG.defaultVoice, + format: "mp3", + }, + }); + + const resultRecord = result as unknown as Record; + const hasContent = + typeof result.content === "string" && result.content.length > 0; + const audioRecord = resultRecord?.audio as + | Record + | undefined; + const hasAudio = Boolean(audioRecord?.buffer); + const hasTranscription = Boolean(resultRecord?.transcription); + + const checks = [ + { label: "result.content", ok: hasContent }, + { label: "result.audio", ok: hasAudio }, + ]; + + for (const c of checks) { + const icon = c.ok ? "\u2705" : "\u274C"; + log(` ${icon} ${c.label}`, c.ok ? "reset" : "red"); + } + + if (hasTranscription) { + log(` \u2705 result.transcription (bonus)`, "reset"); + } + + if (!hasContent) { + logTest( + "generate() + STT + TTS round-trip", + "FAIL", + "generate() returned no content", + ); + return false; + } + + if (!hasAudio) { + logTest( + "generate() + STT + TTS round-trip", + "FAIL", + "TTS did not produce result.audio", + ); + return false; + } + + const buf = audioRecord!.buffer as Buffer | undefined; + + if (!buf || !isValidMP3(buf)) { + logTest( + "generate() + STT + TTS round-trip", + "FAIL", + `result.audio.buffer has invalid MP3 header or is empty`, + ); + return false; + } + + logTest( + "generate() + STT + TTS round-trip", + "PASS", + `content: ${result.content.length} chars, audio: ${buf.length} bytes (valid MP3)` + + (hasTranscription ? ", transcription present" : ""), + ); + return true; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + if (isExpectedProviderError(msg)) { + logTest( + "generate() + STT + TTS round-trip", + "SKIP", + msg.substring(0, 100), + ); + return null; + } + logTest("generate() + STT + TTS round-trip", "FAIL", msg); + return false; + } +} + +// --- Test #6: stream() + TTS --- +async function testStreamTTS(sdk: NeuroLink): Promise { + logTest("stream() + TTS", "TESTING"); + + if (isCredentialsMissing()) { + logTest("stream() + TTS", "SKIP", "GOOGLE_APPLICATION_CREDENTIALS not set"); + return null; + } + + try { + const streamResult = await sdk.stream({ + input: { text: "Count to three" }, + ...buildBaseSDKOptions(), + maxTokens: 200, + tts: { + enabled: true, + voice: VOICE_CONFIG.defaultVoice, + format: "mp3", + }, + }); + + let chunkCount = 0; + let hasAudioChunk = false; + + for await (const chunk of streamResult.stream) { + chunkCount++; + if ("audio" in chunk || "ttsChunk" in chunk) { + hasAudioChunk = true; + } + if (chunkCount >= 100) { + break; + } + } + + if (chunkCount === 0) { + logTest( + "stream() + TTS", + "FAIL", + "No chunks received from stream — chunkCount is 0", + ); + return false; + } + + logTest( + "stream() + TTS", + "PASS", + `Stream completed: ${chunkCount} chunks${hasAudioChunk ? ", audio chunks present" : ""}`, + ); + return true; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + if (isExpectedProviderError(msg)) { + logTest("stream() + TTS", "SKIP", msg.substring(0, 100)); + return null; + } + if ( + msg.includes("tts") || + msg.includes("TTS") || + msg.includes("not supported") + ) { + logTest( + "stream() + TTS", + "SKIP", + `TTS streaming not supported: ${msg.substring(0, 80)}`, + ); + return null; + } + logTest("stream() + TTS", "FAIL", msg); + return false; + } +} + +// --- Test #7: CLI --tts generate --- +async function testCLITTSGenerate(): Promise { + logTest("CLI --tts generate", "TESTING"); + + if (isCredentialsMissing()) { + logTest( + "CLI --tts generate", + "SKIP", + "GOOGLE_APPLICATION_CREDENTIALS not set", + ); + return null; + } + + const ttsOutputPath = path.join( + os.tmpdir(), + `neurolink-voice-tts-${Date.now()}.mp3`, + ); + + try { + const result = await runCommand("node", [ + "dist/cli/index.js", + "generate", + ...buildBaseCLIArgs(), + "--tts", + `--tts-output=${ttsOutputPath}`, + `--max-tokens=${TEST_CONFIG.maxTokens || 200}`, + "Hello", + ]); + + if (!result.success) { + if (isExpectedProviderError(result.stderr)) { + logTest("CLI --tts generate", "SKIP", result.stderr.substring(0, 100)); + return null; + } + if ( + result.stderr.includes("Unknown argument") || + result.stderr.includes("--tts") + ) { + logTest( + "CLI --tts generate", + "SKIP", + "CLI --tts flag not recognized (not implemented yet)", + ); + return null; + } + logTest( + "CLI --tts generate", + "FAIL", + `Exit code: ${result.code}. stderr: ${result.stderr.substring(0, 200)}`, + ); + return false; + } + + const audioFileExists = fs.existsSync(ttsOutputPath); + const combinedOutput = (result.stdout + result.stderr).toLowerCase(); + const hasTTSIndicator = + audioFileExists || + combinedOutput.includes("audio") || + combinedOutput.includes("tts") || + combinedOutput.includes(".mp3") || + combinedOutput.includes("saved"); + + try { + if (audioFileExists) { + fs.unlinkSync(ttsOutputPath); + } + } catch { + /* ignore */ + } + + if (!hasTTSIndicator) { + logTest( + "CLI --tts generate", + "FAIL", + `Exit code 0 but no audio file created and no TTS-related output. stdout: ${result.stdout.substring(0, 100)}`, + ); + return false; + } + + logTest( + "CLI --tts generate", + "PASS", + audioFileExists + ? `Audio file created at ${ttsOutputPath}` + : `TTS indicator found in output (${result.stdout.length} chars)`, + ); + return true; + } catch (error) { + logTest("CLI --tts generate", "FAIL", String(error)); + return false; + } finally { + try { + if (fs.existsSync(ttsOutputPath)) { + fs.unlinkSync(ttsOutputPath); + } + } catch { + /* ignore */ + } + } +} + +// --- Test #8: CLI --stt generate --- +async function testCLISTTGenerate(): Promise { + logTest("CLI --stt generate", "TESTING"); + + if (isCredentialsMissing()) { + logTest( + "CLI --stt generate", + "SKIP", + "GOOGLE_APPLICATION_CREDENTIALS not set", + ); + return null; + } + + const wavPath = path.join( + os.tmpdir(), + `neurolink-voice-stt-${Date.now()}.wav`, + ); + + try { + // Write test WAV file to disk + const wavBuffer = createTestWavBuffer(VOICE_CONFIG.testDurationSeconds); + fs.writeFileSync(wavPath, wavBuffer); + + const result = await runCommand("node", [ + "dist/cli/index.js", + "generate", + ...buildBaseCLIArgs(), + "--stt", + `--stt-provider=${VOICE_CONFIG.defaultSTTProvider}`, + `--input-audio=${wavPath}`, + `--max-tokens=${TEST_CONFIG.maxTokens || 200}`, + "Respond", + ]); + + if (!result.success) { + if (isExpectedProviderError(result.stderr)) { + logTest("CLI --stt generate", "SKIP", result.stderr.substring(0, 100)); + return null; + } + if ( + result.stderr.includes("Unknown argument") || + result.stderr.includes("--stt") || + result.stderr.includes("--input-audio") + ) { + logTest( + "CLI --stt generate", + "SKIP", + "CLI --stt / --input-audio flag not recognized (not implemented yet)", + ); + return null; + } + logTest( + "CLI --stt generate", + "FAIL", + `Exit code: ${result.code}. stderr: ${result.stderr.substring(0, 200)}`, + ); + return false; + } + + const hasOutput = result.stdout.length > 0; + if (!hasOutput) { + logTest( + "CLI --stt generate", + "FAIL", + "CLI produced exit code 0 but stdout is empty", + ); + return false; + } + + logTest( + "CLI --stt generate", + "PASS", + `Exit code 0, stdout: ${result.stdout.length} chars`, + ); + return true; + } catch (error) { + logTest("CLI --stt generate", "FAIL", String(error)); + return false; + } finally { + try { + if (fs.existsSync(wavPath)) { + fs.unlinkSync(wavPath); + } + } catch { + /* ignore */ + } + } +} + +// --- Test #9: Handler registration (TTSProcessor + STTProcessor) --- +async function testHandlerRegistration(): Promise { + logTest("Handler registration (TTSProcessor + STTProcessor)", "TESTING"); + + try { + // Import ProviderRegistry and trigger provider registration + const { ProviderRegistry } = + await import("../dist/factories/providerRegistry.js"); + await ProviderRegistry.registerAllProviders(); + + // Import TTSProcessor from dist + const { TTSProcessor } = await import("../dist/utils/ttsProcessor.js"); + + const ttsProviders = [ + "google-ai", + "vertex", + "openai-tts", + "elevenlabs", + "azure-tts", + ]; + const ttsChecks: Array<{ provider: string; supported: boolean }> = []; + + for (const provider of ttsProviders) { + ttsChecks.push({ + provider, + supported: TTSProcessor.supports(provider), + }); + } + + for (const c of ttsChecks) { + const icon = c.supported ? "\u2705" : "\u274C"; + log( + ` ${icon} TTSProcessor.supports("${c.provider}"): ${c.supported}`, + "reset", + ); + } + + const ttsAllPass = ttsChecks.every((c) => c.supported); + + // Import STTProcessor from dist + const { STTProcessor } = await import("../dist/utils/sttProcessor.js"); + + const sttProviders = [ + "whisper", + "openai-stt", + "deepgram", + "google-stt", + "azure-stt", + ]; + const sttChecks: Array<{ provider: string; supported: boolean }> = []; + + for (const provider of sttProviders) { + sttChecks.push({ + provider, + supported: STTProcessor.supports(provider), + }); + } + + for (const c of sttChecks) { + const icon = c.supported ? "\u2705" : "\u274C"; + log( + ` ${icon} STTProcessor.supports("${c.provider}"): ${c.supported}`, + "reset", + ); + } + + const sttAllPass = sttChecks.every((c) => c.supported); + + if (ttsAllPass && sttAllPass) { + logTest( + "Handler registration (TTSProcessor + STTProcessor)", + "PASS", + `All ${ttsChecks.length} TTS and ${sttChecks.length} STT handlers registered`, + ); + return true; + } + + const failedTTS = ttsChecks + .filter((c) => !c.supported) + .map((c) => c.provider); + const failedSTT = sttChecks + .filter((c) => !c.supported) + .map((c) => c.provider); + const failedAll = [...failedTTS, ...failedSTT]; + + logTest( + "Handler registration (TTSProcessor + STTProcessor)", + "FAIL", + `Missing handlers: ${failedAll.join(", ")}`, + ); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest("Handler registration (TTSProcessor + STTProcessor)", "FAIL", msg); + return false; + } +} + +// --- Test #10: RealtimeProcessor registration --- +async function testRealtimeProcessorRegistration(): Promise { + logTest("RealtimeProcessor registration", "TESTING"); + + try { + // Import ProviderRegistry and trigger provider registration (may already be done) + const { ProviderRegistry } = + await import("../dist/factories/providerRegistry.js"); + await ProviderRegistry.registerAllProviders(); + + // Import RealtimeProcessor from dist + const { RealtimeProcessor } = + await import("../dist/voice/RealtimeVoiceAPI.js"); + + const realtimeProviders = ["openai-realtime", "gemini-live"]; + const checks: Array<{ provider: string; supported: boolean }> = []; + + for (const provider of realtimeProviders) { + checks.push({ + provider, + supported: RealtimeProcessor.supports(provider), + }); + } + + for (const c of checks) { + const icon = c.supported ? "\u2705" : "\u274C"; + log( + ` ${icon} RealtimeProcessor.supports("${c.provider}"): ${c.supported}`, + "reset", + ); + } + + const allPass = checks.every((c) => c.supported); + + if (allPass) { + logTest( + "RealtimeProcessor registration", + "PASS", + `All ${checks.length} realtime handlers registered`, + ); + return true; + } + + const failed = checks.filter((c) => !c.supported).map((c) => c.provider); + logTest( + "RealtimeProcessor registration", + "FAIL", + `Missing handlers: ${failed.join(", ")}`, + ); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest("RealtimeProcessor registration", "FAIL", msg); + return false; + } +} + +// --- Test #11: Audio utils --- +async function testAudioUtils(): Promise { + logTest("Audio utils", "TESTING"); + + try { + const { detectAudioFormat, createWavHeader } = + await import("../dist/voice/audio-utils.js"); + + const checks: Array<{ label: string; ok: boolean; detail: string }> = []; + + // Test detectAudioFormat with a proper WAV buffer (must be 12+ bytes with RIFF+WAVE magic) + const wavBuffer = createTestWavBuffer(0.1); // short but valid WAV + const wavFormat = detectAudioFormat(wavBuffer); + checks.push({ + label: 'detectAudioFormat(wavBuffer) === "wav"', + ok: wavFormat === "wav", + detail: `got: ${JSON.stringify(wavFormat)}`, + }); + + // Test detectAudioFormat with MP3 ID3 header (need 12+ bytes to pass the length guard) + const mp3Buffer = Buffer.alloc(16); + mp3Buffer[0] = 0x49; // I + mp3Buffer[1] = 0x44; // D + mp3Buffer[2] = 0x33; // 3 + const mp3Format = detectAudioFormat(mp3Buffer); + checks.push({ + label: 'detectAudioFormat(mp3ID3Buffer) === "mp3"', + ok: mp3Format === "mp3", + detail: `got: ${JSON.stringify(mp3Format)}`, + }); + + // Test createWavHeader returns 44-byte buffer + const header = createWavHeader(1000); + checks.push({ + label: "createWavHeader() returns 44-byte buffer", + ok: Buffer.isBuffer(header) && header.length === 44, + detail: `got: ${header.length} bytes`, + }); + + for (const c of checks) { + const icon = c.ok ? "\u2705" : "\u274C"; + log(` ${icon} ${c.label}: ${c.detail}`, c.ok ? "reset" : "red"); + } + + const allPass = checks.every((c) => c.ok); + + if (allPass) { + logTest( + "Audio utils", + "PASS", + `All ${checks.length} audio util assertions passed`, + ); + return true; + } + + const failed = checks.filter((c) => !c.ok).map((c) => c.label); + logTest("Audio utils", "FAIL", `Failed: ${failed.join(", ")}`); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest("Audio utils", "FAIL", msg); + return false; + } +} + +// --- Test #12: Audio utils edge guards --- +async function testAudioUtilsEdgeGuards(): Promise { + logTest("Audio utils edge guards", "TESTING"); + + try { + const { splitIntoChunks, resamplePcm } = + await import("../dist/voice/audio-utils.js"); + + const checks: Array<{ label: string; ok: boolean; detail: string }> = []; + + // splitIntoChunks with zero duration → must not loop, returns 1 chunk + const testBuf = Buffer.alloc(100); + let zeroDurationResult: Buffer[] = []; + let zeroDurationError: string | null = null; + try { + zeroDurationResult = splitIntoChunks(testBuf, 0); + } catch (e) { + zeroDurationError = e instanceof Error ? e.message : String(e); + } + + if (zeroDurationError) { + checks.push({ + label: "splitIntoChunks(buf, 0) → must not crash", + ok: false, + detail: `threw: ${zeroDurationError}`, + }); + } else { + checks.push({ + label: "splitIntoChunks(buf, 0) → returns 1 chunk", + ok: zeroDurationResult.length === 1, + detail: `got ${zeroDurationResult.length} chunk(s)`, + }); + } + + // resamplePcm with zero rate → must not crash, returns original samples + const samples = [0.1, 0.2, 0.3]; + let zeroRateResult: number[] = []; + let zeroRateError: string | null = null; + try { + zeroRateResult = resamplePcm(samples, 0, 16000); + } catch (e) { + zeroRateError = e instanceof Error ? e.message : String(e); + } + + if (zeroRateError) { + checks.push({ + label: "resamplePcm(samples, 0, 16000) → must not crash", + ok: false, + detail: `threw: ${zeroRateError}`, + }); + } else { + checks.push({ + label: "resamplePcm(samples, 0, 16000) → returns array (no crash)", + ok: Array.isArray(zeroRateResult), + detail: `got array of ${zeroRateResult.length} values`, + }); + } + + for (const c of checks) { + const icon = c.ok ? "\u2705" : "\u274C"; + log(` ${icon} ${c.label}: ${c.detail}`, c.ok ? "reset" : "red"); + } + + const allPass = checks.every((c) => c.ok); + + if (allPass) { + logTest( + "Audio utils edge guards", + "PASS", + `All ${checks.length} edge guard assertions passed`, + ); + return true; + } + + const failed = checks.filter((c) => !c.ok).map((c) => c.label); + logTest("Audio utils edge guards", "FAIL", `Failed: ${failed.join(", ")}`); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest("Audio utils edge guards", "FAIL", msg); + return false; + } +} + +// --- Test #13: ChunkedAudioStream validation --- +async function testChunkedAudioStream(): Promise { + logTest("ChunkedAudioStream validation", "TESTING"); + + try { + const { ChunkedAudioStream } = + await import("../dist/voice/stream-handler.js"); + + const checks: Array<{ label: string; ok: boolean; detail: string }> = []; + + // new ChunkedAudioStream({ sampleRate: 0 }) → must throw + let sampleRateZeroThrew = false; + try { + new ChunkedAudioStream({ sampleRate: 0 }); + } catch { + sampleRateZeroThrew = true; + } + checks.push({ + label: "new ChunkedAudioStream({ sampleRate: 0 }) throws", + ok: sampleRateZeroThrew, + detail: sampleRateZeroThrew ? "threw as expected" : "did NOT throw", + }); + + // new ChunkedAudioStream({ chunkDurationMs: 0 }) → must throw + let chunkDurationZeroThrew = false; + try { + new ChunkedAudioStream({ sampleRate: 16000, chunkDurationMs: 0 }); + } catch { + chunkDurationZeroThrew = true; + } + checks.push({ + label: "new ChunkedAudioStream({ chunkDurationMs: 0 }) throws", + ok: chunkDurationZeroThrew, + detail: chunkDurationZeroThrew ? "threw as expected" : "did NOT throw", + }); + + // Valid config → must emit chunks + let validConfigOk = false; + let emittedChunkCount = 0; + try { + const stream = new ChunkedAudioStream({ + sampleRate: 16000, + chunkDurationMs: 100, + bytesPerSample: 2, + }); + + await new Promise((resolve, reject) => { + const timeoutId = setTimeout(() => { + reject(new Error("ChunkedAudioStream did not emit chunks within 2s")); + }, 2000); + + stream.on("chunk", () => { + emittedChunkCount++; + if (emittedChunkCount >= 1) { + clearTimeout(timeoutId); + resolve(); + } + }); + + stream.on("error", (err: Error) => { + clearTimeout(timeoutId); + reject(err); + }); + + // Write enough data to trigger a chunk (100ms @ 16kHz 16-bit = 3200 bytes) + const testAudio = Buffer.alloc(4000, 0); + stream.write(testAudio); + stream.end(); + }); + + validConfigOk = emittedChunkCount >= 1; + } catch { + validConfigOk = false; + } + + checks.push({ + label: "Valid ChunkedAudioStream emits chunks", + ok: validConfigOk, + detail: validConfigOk + ? `emitted ${emittedChunkCount} chunk(s)` + : "no chunks emitted", + }); + + for (const c of checks) { + const icon = c.ok ? "\u2705" : "\u274C"; + log(` ${icon} ${c.label}: ${c.detail}`, c.ok ? "reset" : "red"); + } + + const allPass = checks.every((c) => c.ok); + + if (allPass) { + logTest( + "ChunkedAudioStream validation", + "PASS", + `All ${checks.length} assertions passed`, + ); + return true; + } + + const failed = checks.filter((c) => !c.ok).map((c) => c.label); + logTest( + "ChunkedAudioStream validation", + "FAIL", + `Failed: ${failed.join(", ")}`, + ); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest("ChunkedAudioStream validation", "FAIL", msg); + return false; + } +} + +// --- Test #14: Barrel exports --- +async function testBarrelExports(): Promise { + logTest("Barrel exports", "TESTING"); + + try { + const dist = await import("../dist/index.js"); + const checks: Array<{ label: string; ok: boolean; detail: string }> = []; + + // STT_ERROR_CODES is object + const sttErrorCodes = dist.STT_ERROR_CODES; + checks.push({ + label: "STT_ERROR_CODES is object", + ok: + typeof sttErrorCodes === "object" && + sttErrorCodes !== null && + !Array.isArray(sttErrorCodes), + detail: `typeof: ${typeof sttErrorCodes}`, + }); + + // REALTIME_ERROR_CODES is object + const realtimeErrorCodes = dist.REALTIME_ERROR_CODES; + checks.push({ + label: "REALTIME_ERROR_CODES is object", + ok: + typeof realtimeErrorCodes === "object" && + realtimeErrorCodes !== null && + !Array.isArray(realtimeErrorCodes), + detail: `typeof: ${typeof realtimeErrorCodes}`, + }); + + // VOICE_ERROR_CODES is object + const voiceErrorCodes = dist.VOICE_ERROR_CODES; + checks.push({ + label: "VOICE_ERROR_CODES is object", + ok: + typeof voiceErrorCodes === "object" && + voiceErrorCodes !== null && + !Array.isArray(voiceErrorCodes), + detail: `typeof: ${typeof voiceErrorCodes}`, + }); + + // AUDIO_FORMAT_DETAILS is object + const audioFormatDetails = dist.AUDIO_FORMAT_DETAILS; + checks.push({ + label: "AUDIO_FORMAT_DETAILS is object", + ok: + typeof audioFormatDetails === "object" && + audioFormatDetails !== null && + !Array.isArray(audioFormatDetails), + detail: `typeof: ${typeof audioFormatDetails}`, + }); + + // DEFAULT_STT_OPTIONS is object + const defaultSTTOptions = dist.DEFAULT_STT_OPTIONS; + checks.push({ + label: "DEFAULT_STT_OPTIONS is object", + ok: + typeof defaultSTTOptions === "object" && + defaultSTTOptions !== null && + !Array.isArray(defaultSTTOptions), + detail: `typeof: ${typeof defaultSTTOptions}`, + }); + + // VALID_AUDIO_FORMATS includes "mp4", "mpeg", "mpga" + const validAudioFormats = dist.VALID_AUDIO_FORMATS as unknown; + const validArr = Array.isArray(validAudioFormats) + ? (validAudioFormats as string[]) + : []; + const hasMP4 = validArr.includes("mp4"); + const hasMPEG = validArr.includes("mpeg"); + const hasMPGA = validArr.includes("mpga"); + checks.push({ + label: 'VALID_AUDIO_FORMATS includes "mp4"', + ok: hasMP4, + detail: hasMP4 ? "present" : `missing. Got: ${validArr.join(", ")}`, + }); + checks.push({ + label: 'VALID_AUDIO_FORMATS includes "mpeg"', + ok: hasMPEG, + detail: hasMPEG ? "present" : "missing", + }); + checks.push({ + label: 'VALID_AUDIO_FORMATS includes "mpga"', + ok: hasMPGA, + detail: hasMPGA ? "present" : "missing", + }); + + // SpanType.STT === "stt" + const SpanType = dist.SpanType as unknown; + const spanTypeSTT = + SpanType && typeof SpanType === "object" + ? (SpanType as Record).STT + : undefined; + checks.push({ + label: 'SpanType.STT === "stt"', + ok: spanTypeSTT === "stt", + detail: `got: ${JSON.stringify(spanTypeSTT)}`, + }); + + for (const c of checks) { + const icon = c.ok ? "\u2705" : "\u274C"; + log(` ${icon} ${c.label}: ${c.detail}`, c.ok ? "reset" : "red"); + } + + const allPass = checks.every((c) => c.ok); + + if (allPass) { + logTest( + "Barrel exports", + "PASS", + `All ${checks.length} barrel export assertions passed`, + ); + return true; + } + + const failed = checks.filter((c) => !c.ok).map((c) => c.label); + logTest("Barrel exports", "FAIL", `Failed: ${failed.join(", ")}`); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest("Barrel exports", "FAIL", msg); + return false; + } +} + +// --- Test #15: Removed methods do NOT exist on NeuroLink --- +async function testRemovedMethods(): Promise { + logTest("Removed methods do NOT exist on NeuroLink", "TESTING"); + + try { + const sdk = new NeuroLink(); + const sdkRecord = sdk as unknown as Record; + + const checks: Array<{ label: string; ok: boolean; detail: string }> = []; + + // These methods must NOT exist + const mustBeAbsent = ["synthesize", "transcribe", "startRealtimeVoice"]; + for (const method of mustBeAbsent) { + const exists = typeof sdkRecord[method] === "function"; + checks.push({ + label: `typeof sdk.${method} !== "function"`, + ok: !exists, + detail: exists + ? `STILL EXISTS as function (should have been removed)` + : `not present (correctly removed)`, + }); + } + + // These methods MUST still exist + const mustExist = ["generate", "stream"]; + for (const method of mustExist) { + const exists = typeof sdkRecord[method] === "function"; + checks.push({ + label: `typeof sdk.${method} === "function"`, + ok: exists, + detail: exists ? "present (correct)" : "MISSING (should exist)", + }); + } + + for (const c of checks) { + const icon = c.ok ? "\u2705" : "\u274C"; + log(` ${icon} ${c.label}: ${c.detail}`, c.ok ? "reset" : "red"); + } + + try { + await sdk.shutdown?.(); + } catch { + /* ignore */ + } + + const allPass = checks.every((c) => c.ok); + + if (allPass) { + logTest( + "Removed methods do NOT exist on NeuroLink", + "PASS", + `All ${checks.length} method existence checks passed`, + ); + return true; + } + + const failed = checks.filter((c) => !c.ok).map((c) => c.label); + logTest( + "Removed methods do NOT exist on NeuroLink", + "FAIL", + `Failed: ${failed.join(", ")}`, + ); + return false; + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest("Removed methods do NOT exist on NeuroLink", "FAIL", msg); + return false; + } +} + +// ============================================================ +// MAIN RUNNER +// ============================================================ + +async function runAllTests(): Promise { + const startTime = Date.now(); + log( + "\nNeuroLink Continuous Test Suite: Voice / Speech Integration", + "bright", + ); + log( + ` Provider: ${TEST_CONFIG.provider}, Model: ${TEST_CONFIG.model || "default"}`, + "cyan", + ); + log( + ` Google Credentials: ${process.env.GOOGLE_APPLICATION_CREDENTIALS ? "set" : "NOT SET (API tests will skip)"}`, + process.env.GOOGLE_APPLICATION_CREDENTIALS ? "green" : "yellow", + ); + log(` Temp dir: ${tempDir}`, "cyan"); + + // Prerequisite checks + if (!fs.existsSync("dist") || !fs.existsSync("dist/index.js")) { + log("Build not found. Run: pnpm run build", "red"); + process.exit(1); + } + + const sharedSdk = new NeuroLink(); + + const tests: Array<{ name: string; fn: () => Promise }> = [ + // TTS generate (Tests #1-#3) + { + name: "generate() + TTS (MP3 format)", + fn: () => testGenerateTTSMP3(sharedSdk), + }, + { + name: "generate() + TTS (WAV format)", + fn: () => testGenerateTTSWAV(sharedSdk), + }, + { + name: "generate() + TTS unconfigured provider error", + fn: () => testGenerateTTSUnconfiguredProvider(sharedSdk), + }, + + // STT generate (Test #4) + { + name: "generate() + STT (core feature)", + fn: () => testGenerateSTT(sharedSdk), + }, + + // STT + TTS round-trip (Test #5) + { + name: "generate() + STT + TTS round-trip", + fn: () => testGenerateSTTAndTTSRoundTrip(sharedSdk), + }, + + // Stream + TTS (Test #6) + { name: "stream() + TTS", fn: () => testStreamTTS(sharedSdk) }, + + // CLI (Tests #7-#8) + { name: "CLI --tts generate", fn: () => testCLITTSGenerate() }, + { name: "CLI --stt generate", fn: () => testCLISTTGenerate() }, + + // Handler registration (Tests #9-#10) + { + name: "Handler registration (TTSProcessor + STTProcessor)", + fn: () => testHandlerRegistration(), + }, + { + name: "RealtimeProcessor registration", + fn: () => testRealtimeProcessorRegistration(), + }, + + // Audio utils (Tests #11-#12) + { name: "Audio utils", fn: () => testAudioUtils() }, + { name: "Audio utils edge guards", fn: () => testAudioUtilsEdgeGuards() }, + + // ChunkedAudioStream (Test #13) + { + name: "ChunkedAudioStream validation", + fn: () => testChunkedAudioStream(), + }, + + // Barrel exports (Test #14) + { name: "Barrel exports", fn: () => testBarrelExports() }, + + // Removed methods (Test #15) + { + name: "Removed methods do NOT exist on NeuroLink", + fn: () => testRemovedMethods(), + }, + ]; + + for (const test of tests) { + logSection(test.name); + try { + const result = await test.fn(); + testResults.push({ name: test.name, result, error: null }); + } catch (error) { + const msg = error instanceof Error ? error.message : String(error); + logTest(test.name, "FAIL", `Uncaught: ${msg}`); + testResults.push({ name: test.name, result: false, error: msg }); + } + await globalCleanup(); + await new Promise((r) => setTimeout(r, TEST_CONFIG.interTestDelay)); + } + + // Summary + logSection("Test Results Summary"); + const passed = testResults.filter((r) => r.result === true).length; + const failed = testResults.filter((r) => r.result === false).length; + const skipped = testResults.filter((r) => r.result === null).length; + for (const t of testResults) { + logTest( + t.name, + t.result === true ? "PASS" : t.result === false ? "FAIL" : "SKIP", + t.error || "", + ); + } + + const duration = Math.round((Date.now() - startTime) / 1000); + log( + `\nFinal Results: ${passed} passed, ${failed} failed, ${skipped} skipped (${testResults.length} total) in ${duration}s`, + failed === 0 ? "green" : "red", + ); + + // Cleanup temp directory + try { + fs.rmSync(tempDir, { recursive: true, force: true }); + } catch { + /* ignore */ + } + + try { + await sharedSdk.shutdown?.(); + } catch { + /* ignore */ + } + process.exit(failed === 0 ? 0 : 1); +} + +// ============================================================ +// CLI ARGS + EXECUTION +// ============================================================ + +function parseArguments(): { provider?: string; model?: string } { + const args: { provider?: string; model?: string } = {}; + for (const arg of process.argv.slice(2)) { + if (arg.startsWith("--provider=")) { + args.provider = arg.split("=")[1]; + } + if (arg.startsWith("--model=")) { + args.model = arg.split("=")[1]; + } + if (arg === "--help") { + console.log( + "Usage: npx tsx test/continuous-test-suite-voice.ts [--provider=X] [--model=Y]", + ); + console.log( + "\nTests: 15 (TTS generate, WAV format, unconfigured error, STT generate, STT+TTS round-trip,", + ); + console.log( + " stream+TTS, CLI TTS, CLI STT, handler registration, realtime registration,", + ); + console.log( + " audio utils, edge guards, ChunkedAudioStream, barrel exports, removed methods)", + ); + console.log( + "\nRequires: GOOGLE_APPLICATION_CREDENTIALS env var (API tests will SKIP without it)", + ); + process.exit(0); + } + } + return args; +} + +const cliArgs = parseArguments(); +if (cliArgs.provider) { + TEST_CONFIG.provider = cliArgs.provider; +} +if (cliArgs.model) { + TEST_CONFIG.model = cliArgs.model; +} +if (!TEST_CONFIG.maxTokens) { + TEST_CONFIG.maxTokens = PROVIDER_MAX_TOKENS[TEST_CONFIG.provider] || 8192; +} + +if (typeof describe === "undefined") { + runAllTests().catch((e) => { + log(`Suite crashed: ${e instanceof Error ? e.message : String(e)}`, "red"); + process.exit(1); + }); +} else { + describe.skip("Continuous Test Suite: Voice / Speech Integration", () => { + it("runs standalone via npx tsx", () => runAllTests(), 600000); + }); +}