From 11000e27d1a1637a4b4e6e8add57f5ec92ffb4d0 Mon Sep 17 00:00:00 2001 From: semantic-release-bot Date: Tue, 17 Feb 2026 16:07:36 +0000 Subject: [PATCH] feat(stt): add speech-to-text support using Google Cloud --- CHANGELOG.md | 6 + docs/features/stt.md | 673 +++++++++++++++ package.json | 3 +- pnpm-lock.yaml | 143 +++- src/cli/factories/commandFactory.ts | 854 ++++++++++++++------ src/cli/loop/optionsSchema.ts | 1 + src/cli/parser.ts | 3 + src/lib/adapters/stt/googleSTTHandler.ts | 393 +++++++++ src/lib/constants/enums.ts | 1 + src/lib/core/baseProvider.ts | 274 ++++++- src/lib/factories/providerRegistry.ts | 538 ++++++------ src/lib/index.ts | 12 + src/lib/neurolink.ts | 44 +- src/lib/processors/config/fileTypes.ts | 1 + src/lib/providers/googleVertex.ts | 7 +- src/lib/types/generateTypes.ts | 90 ++- src/lib/types/index.ts | 3 + src/lib/types/streamTypes.ts | 12 +- src/lib/types/sttTypes.ts | 315 ++++++++ src/lib/utils/messageBuilder.ts | 9 +- src/lib/utils/sttProcessor.ts | 500 ++++++++++++ test/unit/telemetry-config-metadata.test.ts | 3 +- 22 files changed, 3376 insertions(+), 509 deletions(-) create mode 100644 docs/features/stt.md create mode 100644 src/lib/adapters/stt/googleSTTHandler.ts create mode 100644 src/lib/types/sttTypes.ts create mode 100644 src/lib/utils/sttProcessor.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index d7a4141ed..b2eea31a5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,9 @@ +## [9.9.0](https://github.com/juspay/neurolink/compare/v9.8.0...v9.9.0) (2026-02-17) + +### Features + +- **(video-analysis):** add video-analysis support in neurolink ([c35f8a8](https://github.com/juspay/neurolink/commit/c35f8a8d52cc1366e10b8701285e1bec52e27d98)) + ## [9.8.0](https://github.com/juspay/neurolink/compare/v9.7.0...v9.8.0) (2026-02-17) ### Features diff --git a/docs/features/stt.md b/docs/features/stt.md new file mode 100644 index 000000000..f1003c45b --- /dev/null +++ b/docs/features/stt.md @@ -0,0 +1,673 @@ +--- +title: Speech-to-Text (STT) Integration Guide +description: Complete guide to NeuroLink's STT capabilities for converting audio to text with Google Cloud Speech-to-Text +keywords: stt, speech-to-text, transcription, audio, voice recognition, google cloud stt, audio transcription +--- + +## Overview + +NeuroLink provides integrated Speech-to-Text (STT) capabilities, allowing you to transcribe audio files to text with high accuracy. This feature is perfect for voice assistants, meeting transcription, accessibility features, and more. + +**Key Features:** + +- **High-accuracy transcription** - Powered by Google Cloud Speech-to-Text +- **7 specialized models** - Optimized models for different audio types +- **Word-level timestamps** - Precise timing for each word +- **Multiple audio formats** - WAV, MP3, FLAC, AAC, M4A, OGG/Opus, WebM, WMA support +- **Automatic punctuation** - Context-aware punctuation insertion +- **Alternative transcriptions** - Multiple transcription candidates with confidence scores + +--- + +## Quick Start + +### Installation + +STT support is built into NeuroLink, so no additional installation required. + +### Environment Setup + +STT requires Google Cloud service account credentials: + +```bash +# Set service account credentials path +export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service-account.json" +``` + +**Service Account Setup:** + +1. Navigate to Google Cloud Console > "IAM & Admin" > "Service Accounts" +2. Create a new service account or select an existing one +3. Grant the "Cloud Speech-to-Text User" role +4. Create and download a JSON key file +5. Set the `GOOGLE_APPLICATION_CREDENTIALS` environment variable to the key file path + +### Basic Usage + +**CLI:** + +```bash +# Transcribe an audio file (language auto-detects) +neurolink generate "Transcribe" --file audio.mp3 --stt-language en-US + +# Or with explicit language +neurolink generate "Transcribe this audio" \ + --file audio.mp3 \ + --stt-language en-US + +# Save transcription to file +neurolink generate "Transcribe this meeting" \ + --file meeting.wav \ + --stt-language en-US \ + -o transcript.txt +``` + +> **Note:** STT follows NeuroLink's multimodal input pattern. Use the `--file` flag to specify files. Language is auto-detected when `--stt-language` is not specified. + +**SDK:** + +```typescript +import { NeuroLink } from "@juspay/neurolink"; +import { readFileSync } from "fs"; + +const neurolink = new NeuroLink(); + +const audioBuffer = readFileSync("audio.mp3"); + +const result = await neurolink.generate({ + input: { + text: "Transcribe this audio", + files: [audioBuffer], + }, + provider: "google-ai", + stt: { + languageCode: "en-US", + enableAutomaticPunctuation: true, + }, +}); + +// Access transcription result +console.log("Transcription:", result.transcription?.text); +console.log("Confidence:", result.transcription?.confidence); +``` + +## Supported Providers + +STT is currently available through Google Cloud Speech-to-Text API through Service Account (`GOOGLE_APPLICATION_CREDENTIALS`) + +--- + +## Model Selection + +Google Cloud STT offers 7 specialized models optimized for different audio types: + +### Available Models + +**SDK:** + +```typescript +// Default provider (google-ai) +const models = await neurolink.getSTTModels(); +console.log("Available models:", models); + +// Or specify a provider +const models = await neurolink.getSTTModels("google-ai"); +``` + +**CLI:** + +```bash +# Default provider (google-ai) +neurolink stt models + +# Or specify a provider +neurolink stt models google-ai +``` + +### Model Types + +| Model | Use Case | Audio Type | Accuracy | +| ---------------------- | -------------------------------- | -------------- | ----------- | +| **default** | General-purpose transcription | Any | High | +| **command_and_search** | Short queries and voice commands | < 15 seconds | High | +| **phone_call** | Telephone/VoIP audio | 8kHz/16kHz | Optimized | +| **video** | Video soundtracks | Mixed audio | High | +| **medical_dictation** | Medical terminology | Clinical notes | Specialized | +| **latest_long** | Long-form audio | > 1 minute | Latest | +| **latest_short** | Short-form audio | < 1 minute | Latest | + +### Model Selection Guidelines + +**For Meeting Transcription:** + +```typescript +stt: { + languageCode: "en-US", + model: "video", + enableSpeakerDiarization: true, + diarizationSpeakerCount: 5, +} +``` + +**For Call Center Analytics:** + +```typescript +stt: { + languageCode: "en-US", + model: "phone_call", + enableSpeakerDiarization: true, + diarizationSpeakerCount: 2, +} +``` + +**For Voice Commands:** + +```typescript +stt: { + languageCode: "en-US", + model: "command_and_search", +} +``` + +**For Medical Transcription:** + +```typescript +stt: { + languageCode: "en-US", + model: "medical_dictation", + useEnhanced: true, +} +``` + +--- + +## Audio Format Support + +### Supported Formats + +| Format | Encoding | Quality | File Size | Use Case | +| ------------ | --------- | -------- | --------- | ----------------------------- | +| **WAV** | LINEAR16 | Best | Large | Uncompressed, highest quality | +| **FLAC** | FLAC | Lossless | Medium | Balanced quality/size | +| **MP3** | MP3 | Good | Small | Common format, web-friendly | +| **AAC** | AAC | Good | Small | Apple devices, iTunes | +| **M4A** | AAC | Good | Small | Apple/iTunes audio | +| **OGG/Opus** | OGG_OPUS | Good | Small | Web streaming | +| **WebM** | WEBM_OPUS | Good | Small | Web video audio tracks | +| **WMA** | WMA | Good | Small | Windows Media Audio | + +### Required Audio Specifications for Google Cloud Speech-to-Text + +**File Size Limits:** + +- Maximum size: **10 MB** per audio file +- For larger files, consider chunking or using streaming API (coming soon) + +**Duration Limits:** + +- Maximum duration: **60 seconds** per request (as per google cloud speech-to-test synchronous API) +- For longer audio, split into chunks + +**Sample Rate:** + +- Recommended: **16,000 Hz** (16 kHz) +- Supported: 8,000 Hz - 48,000 Hz +- Telephony: 8,000 Hz +- High-quality: 48,000 Hz + +### Format Detection + +NeuroLink automatically detects audio format from file extension: + +```typescript +// Automatic format detection +const result = await neurolink.generate({ + input: { + text: "Transcribe", + files: ["meeting.mp3"], + }, + provider: "google-ai", + stt: { + languageCode: "en-US", + }, +}); +``` + +Manual encoding specification: + +```typescript +stt: { + languageCode: "en-US", + encoding: "LINEAR16", + sampleRateHertz: 16000, +} +``` + +--- + +## Advanced Features + +### Word-Level Timestamps + +Get precise timing for each word in the transcription: + +```typescript +const result = await neurolink.generate({ + input: { + text: "Transcribe with timestamps", + files: [audioBuffer], + }, + provider: "google-ai", + stt: { + languageCode: "en-US", + enableWordTimeOffsets: true, + enableWordConfidence: true, + }, +}); + +// Access word-level data +result.transcription?.words?.forEach((word) => { + console.log( + `${word.word}: ${word.startTime}s - ${word.endTime}s (confidence: ${word.confidence})`, + ); +}); +``` + +**Example Output:** + +``` +Hello: 0.0s - 0.3s (confidence: 0.98) +world: 0.4s - 0.7s (confidence: 0.95) +this: 0.8s - 1.0s (confidence: 0.97) +is: 1.1s - 1.2s (confidence: 0.99) +a: 1.3s - 1.4s (confidence: 0.96) +test: 1.5s - 1.8s (confidence: 0.98) +``` + +### Speaker Diarization + +Identify different speakers in multi-speaker audio: + +```typescript +const result = await neurolink.generate({ + input: { + text: "Transcribe meeting", + files: ["meeting.wav"], + }, + provider: "google-ai", + stt: { + languageCode: "en-US", + enableSpeakerDiarization: true, + diarizationSpeakerCount: 3, + enableWordTimeOffsets: true, + }, +}); + +// Group words by speaker +const speakers = new Map(); +result.transcription?.words?.forEach((word) => { + const tag = word.speakerTag || 0; + if (!speakers.has(tag)) speakers.set(tag, []); + speakers.get(tag)?.push(word.word); +}); + +speakers.forEach((words, speaker) => { + console.log(`Speaker ${speaker}: ${words.join(" ")}`); +}); +``` + +**CLI Usage:** + +```bash +neurolink generate "Transcribe this meeting" \ + --file meeting.wav \ + --stt-language en-US \ + --stt-enable-timestamps \ + --stt-enable-diarization \ + -o transcript.txt +``` + +### Alternative Transcriptions + +Get multiple transcription candidates with confidence scores: + +```typescript +const result = await neurolink.generate({ + input: { + text: "Transcribe", + files: [audioBuffer], + }, + provider: "google-ai", + stt: { + languageCode: "en-US", + maxAlternatives: 3, + }, +}); + +// Primary transcription +console.log("Primary:", result.transcription?.text); +console.log("Confidence:", result.transcription?.confidence); + +// Alternative transcriptions +result.transcription?.alternatives?.forEach((alt, i) => { + console.log(`Alternative ${i + 1}: ${alt.transcript} (${alt.confidence})`); +}); +``` + +### Profanity Filtering + +Automatically filter profanity from transcriptions: + +```typescript +stt: { + languageCode: "en-US", + profanityFilter: true, +} +``` + +### Speech Contexts + +Improve recognition accuracy by providing context phrases: + +```typescript +stt: { + languageCode: "en-US", + speechContexts: [ + { + phrases: ["NeuroLink", "API", "authentication", "token"], + boost: 10, + }, + { + phrases: ["customer service", "technical support"], + boost: 5, + }, + ], +} +``` + +Use cases: + +- **Product names**: Bias towards your product terminology +- **Technical terms**: Improve accuracy for domain-specific vocabulary +- **Names**: Recognize company/person names correctly + +--- + +## Complete Configuration Reference + +### SDK Configuration + +```typescript +import { NeuroLink } from "@juspay/neurolink"; +import { readFileSync } from "fs"; + +const neurolink = new NeuroLink(); + +const result = await neurolink.generate({ + input: { + text: "Transcribe this audio", + files: [readFileSync("audio.mp3")], + }, + provider: "google-ai", // or "vertex" + stt: { + // Audio configuration + encoding: "MP3", // Audio encoding (auto-detected if not specified) + sampleRateHertz: 16000, // Sample rate in Hz + audioChannelCount: 1, // 1 = mono, 2 = stereo + + // Language configuration + languageCode: "en-US", // Optional: language code (auto-detects if not specified) + alternativeLanguageCodes: ["es-US", "fr-FR"], // Optional: Alternative languages + + // Model selection + model: "default", // Model type + useEnhanced: false, // Use enhanced models (higher cost/accuracy) + + // Transcription features + enableAutomaticPunctuation: true, // Add punctuation automatically + profanityFilter: false, // Filter profanity + maxAlternatives: 1, // Number of alternative transcriptions + + // Word-level features + enableWordTimeOffsets: false, // Word timestamps + enableWordConfidence: false, // Word confidence scores + + // Speaker diarization + enableSpeakerDiarization: false, // Identify speakers + diarizationSpeakerCount: 2, // Expected speaker count + + // Speech contexts (optional) + speechContexts: [ + { + phrases: ["custom", "terms"], + boost: 10, + }, + ], + }, +}); + +// Access results +console.log("Text:", result.transcription?.text); +console.log("Confidence:", result.transcription?.confidence); +console.log("Duration:", result.transcription?.duration, "seconds"); +console.log("Language:", result.transcription?.languageCode); + +// Word-level details +result.transcription?.words?.forEach((word) => { + console.log(`${word.word}: ${word.startTime}s (speaker ${word.speakerTag})`); +}); + +// Metadata +console.log("Latency:", result.transcription?.metadata.latency, "ms"); +console.log("Provider:", result.transcription?.metadata.provider); +console.log("Model:", result.transcription?.metadata.model); +``` + +### CLI Flags + +```bash +neurolink generate "" \ + --file \ + --provider google-ai \ + --stt-model \ + --stt-enable-timestamps \ + --stt-enable-diarization \ + -o +``` + +**Available CLI Flags:** + +- `--file` - Audio file path (required for transcription) +- `--stt-language` (alias: `--stt-lang`) - Language code (e.g., en-US, es-ES) - auto-detects if not specified +- `--stt-model` - Model type (default, phone_call, video, etc.) +- `--stt-enhanced` - Use enhanced model (higher accuracy, higher cost) +- `--stt-enable-timestamps` - Enable word-level timestamps +- `--stt-enable-confidence` - Enable word-level confidence scores +- `--stt-enable-diarization` - Enable speaker diarization +- `--stt-speaker-count` - Expected number of speakers for diarization +- `--stt-enable-punctuation` - Enable automatic punctuation (default: true) +- `--stt-max-alternatives` - Number of alternative transcriptions (1-30) + +--- + +## Error Handling + +### Common Error Patterns + +```typescript +import { STTError, STT_ERROR_CODES } from "@juspay/neurolink"; + +async function transcribeWithErrorHandling(audioFile: string) { + try { + const audioBuffer = readFileSync(audioFile); + + const result = await neurolink.generate({ + input: { + text: "Transcribe", + files: [audioBuffer], + }, + provider: "google-ai", + stt: { + languageCode: "en-US", + enableAutomaticPunctuation: true, + }, + }); + + // Validate transcription result + if (!result.transcription || !result.transcription.text) { + throw new Error("Empty transcription result"); + } + + if (result.transcription.confidence < 0.7) { + console.warn( + "Low confidence transcription:", + result.transcription.confidence, + ); + } + + return result.transcription; + } catch (error) { + if (error instanceof STTError) { + switch (error.code) { + case STT_ERROR_CODES.AUDIO_TOO_LARGE: + console.error("Audio file exceeds 10MB limit"); + break; + case STT_ERROR_CODES.AUDIO_TOO_LONG: + console.error("Audio exceeds 60 second limit"); + break; + case STT_ERROR_CODES.LANGUAGE_NOT_SUPPORTED: + console.error("Unsupported language code"); + break; + case STT_ERROR_CODES.NO_SPEECH_DETECTED: + console.error("No speech found in audio"); + break; + case STT_ERROR_CODES.PROVIDER_NOT_CONFIGURED: + console.error("Google Cloud credentials not configured"); + break; + default: + console.error("STT error:", error.message); + } + } else { + console.error("Unexpected error:", error); + } + throw error; + } +} +``` + +--- + +## Troubleshooting + +### Common Issues + +| Issue | Cause | Solution | +| -------------------------------- | --------------------------- | ------------------------------------------------------------ | +| **"STT client not initialized"** | Missing credentials | Set `GOOGLE_APPLICATION_CREDENTIALS` to service account path | +| **"Audio file too large"** | File exceeds 10MB | Compress audio or split into smaller chunks | +| **"Audio too long"** | Duration exceeds 60 seconds | Split audio into smaller segments | +| **"No speech detected"** | Audio is empty/silent | Verify audio contains speech | +| **"Low confidence result"** | Poor audio quality | Use higher quality audio, add speech contexts | +| **"Speaker diarization failed"** | Wrong speaker count | Adjust `diarizationSpeakerCount` to match actual speakers | + +### Authentication Issues + +**Service Account:** + +```bash +# Verify credentials file exists +ls -la $GOOGLE_APPLICATION_CREDENTIALS + +# Test authentication +gcloud auth application-default login +``` + +### Audio Quality Issues + +**Best Practices:** + +1. **Sample Rate**: Use 16 kHz for best results +2. **Format**: Use FLAC or WAV for highest quality +3. **Channels**: Use mono (1 channel) unless stereo required +4. **Noise**: Reduce background noise before transcription +5. **Volume**: Ensure audio has consistent volume levels + +**Pre-process Audio:** + +```bash +# Convert to optimal format with ffmpeg +ffmpeg -i input.mp3 -ar 16000 -ac 1 -c:a flac output.flac +``` + +--- + +## Best Practices + +### Performance Optimization + +1. **Use appropriate models** - `phone_call` for telephony, `command_and_search` for short audio +2. **Cache language/model lists** - Lists are cached for 5 minutes +3. **Chunk long audio** - Split audio > 60 seconds into smaller segments +4. **Optimize audio format** - Use FLAC for best quality/size balance + +### Production Deployment + +1. **Use service accounts** - More secure than API keys +2. **Implement retry logic** - Handle transient network failures +3. **Monitor quota usage** - Track Google Cloud STT API usage +4. **Set appropriate timeouts** - Default is 60 seconds +5. **Handle errors gracefully** - Provide fallback behavior +6. **Log transcription metadata** - Track latency and confidence + +### Accuracy Improvement + +1. **Use speech contexts** - Add domain-specific terms +2. **Select correct model** - Match model to audio type +3. **Enable enhanced models** - Higher cost but better accuracy +4. **Specify language code** - Auto-detection works well but explicit language may improve accuracy +5. **Pre-process audio** - Clean audio before transcription +6. **Verify confidence scores** - Flag low-confidence results + +### Cost Management + +1. **Use standard models** - Enhanced models cost more +2. **Optimize audio duration** - Shorter audio = lower cost +3. **Cache results** - Avoid re-transcribing same audio +4. **Monitor API usage** - Set budget alerts in Google Cloud Console + +--- + +## Related Features + +**Multimodal Capabilities:** + +- [Text-to-Speech (TTS)](tts.md) - Audio generation from text +- [Multimodal Guide](multimodal.md) - Images, PDFs, CSV inputs +- [PDF Support](pdf-support.md) - Document processing +- [Video Generation](video-generation.md) - AI-powered video creation + +**Advanced Features:** + +- [Streaming](../advanced/streaming.md) - Stream AI responses in real-time +- [Provider Orchestration](provider-orchestration.md) - Multi-provider failover + +**Documentation:** + +- [CLI Commands](../cli/commands.md) - Complete CLI reference +- [SDK API Reference](../sdk/api-reference.md) - Full API documentation +- [Troubleshooting](../troubleshooting.md) - Extended error catalog + +--- + +## Summary + +NeuroLink's STT integration provides: + +✅ **High-accuracy transcription** - Google Cloud Speech-to-Text +✅ **7 specialized models** - Optimized for different audio types +✅ **Word-level timestamps** - Precise timing information +✅ **Speaker diarization** - Multi-speaker identification +✅ **8 audio formats** - WAV, MP3, FLAC, AAC, M4A, OGG/Opus, WebM, WMA +✅ **Advanced features** - Punctuation, profanity filtering, alternatives +✅ **CLI integration** - Simple `--file ` flag pattern +✅ **Auto-detection** - Language auto-detects if not specified diff --git a/package.json b/package.json index 7652a2bf6..a210c95a4 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@juspay/neurolink", - "version": "9.8.0", + "version": "9.9.0", "description": "Universal AI Development Platform with working MCP integration, multi-provider support, and professional CLI. Built-in tools operational, 58+ external MCP servers discoverable. Connect to filesystem, GitHub, database operations, and more. Build, test, and deploy AI applications with 13 providers: OpenAI, Anthropic, Google AI, AWS Bedrock, Azure, Hugging Face, Ollama, and Mistral AI.", "author": { "name": "Juspay Technologies", @@ -177,6 +177,7 @@ "@aws-sdk/client-sagemaker-runtime": "^3.886.0", "@aws-sdk/credential-provider-node": "^3.886.0", "@aws-sdk/types": "^3.862.0", + "@google-cloud/speech": "^7.2.1", "@google-cloud/text-to-speech": "^5.0.0", "@google-cloud/vertexai": "^1.10.0", "@google/genai": "^1.34.0", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 9c5b1bd0e..1ccd12f0d 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -59,6 +59,9 @@ importers: '@aws-sdk/types': specifier: ^3.862.0 version: 3.901.0 + '@google-cloud/speech': + specifier: ^7.2.1 + version: 7.2.1 '@google-cloud/text-to-speech': specifier: ^5.0.0 version: 5.8.1(encoding@0.1.13) @@ -1225,6 +1228,22 @@ packages: '@gerrit0/mini-shiki@3.20.0': resolution: {integrity: sha512-Wa57i+bMpK6PGJZ1f2myxo3iO+K/kZikcyvH8NIqNNZhQUbDav7V9LQmWOXhf946mz5c1NZ19WMsGYiDKTryzQ==} + '@google-cloud/common@6.0.0': + resolution: {integrity: sha512-IXh04DlkLMxWgYLIUYuHHKXKOUwPDzDgke1ykkkJPe48cGIS9kkL2U/o0pm4ankHLlvzLF/ma1eO86n/bkumIA==} + engines: {node: '>=18'} + + '@google-cloud/projectify@4.0.0': + resolution: {integrity: sha512-MmaX6HeSvyPbWGwFq7mXdo0uQZLGBYCwziiLIGq5JVX+/bdI3SAq6bP98trV5eTWfLuvsMcIC1YJOF2vfteLFA==} + engines: {node: '>=14.0.0'} + + '@google-cloud/promisify@4.1.0': + resolution: {integrity: sha512-G/FQx5cE/+DqBbOpA5jKsegGwdPniU6PuIEMt+qxWgFxvxuFOzVmp6zYchtYuwAWV5/8Dgs0yAmjvNZv3uXLQg==} + engines: {node: '>=18'} + + '@google-cloud/speech@7.2.1': + resolution: {integrity: sha512-3eaX4/aT5cJ0JMAiMwg1RLJJD2fxbF3St9x+9QSQskHJGFuMM2RCUmadLjCK3gPsdVk2eVdStiyxxuw+zeIEbw==} + engines: {node: '>=18'} + '@google-cloud/text-to-speech@5.8.1': resolution: {integrity: sha512-HXyZBtfQq+ETSLwWV/k3zFRWSzt+KEfiC5/OqXNNUed+lU/LEyN0CsqqEmkFfkL8BPsVIMAK2xiYCaDsKENukg==} engines: {node: '>=14.0.0'} @@ -2990,6 +3009,9 @@ packages: '@types/diff-match-patch@1.0.36': resolution: {integrity: sha512-xFdR6tkm0MWvBfO8xXCSsinYxHcqkQUlcHeSpMC2ukzOb6lwQAfDmW+Qt0AvlGd8HpsS28qKsB+oPeJn9I39jg==} + '@types/duplexify@3.6.5': + resolution: {integrity: sha512-fB56ACzlW91UdZ5F3VXplVMDngO8QaX5Y2mjvADtN01TT2TMy4WjF0Lg+tFDvt4uMBeTe4SgaD+qCrA7dL5/tA==} + '@types/estree@1.0.8': resolution: {integrity: sha512-dWHzHa2WqEXI/O1E9OjrocMTKJl2mSrEolh1Iomrv6U+JuNwaHXsXx9bLu5gG7BUWFIN0skIQJQ/L1rIex4X6w==} @@ -3095,6 +3117,9 @@ packages: '@types/phoenix@1.6.7': resolution: {integrity: sha512-oN9ive//QSBkf19rfDv45M7eZPi0eEXylht2OLEXicu5b4KoQ1OzXIw+xDSGWxSxe1JmepRR/ZH283vsu518/Q==} + '@types/pumpify@1.4.5': + resolution: {integrity: sha512-BGVAQyK5yJdfIII230fVYGY47V63hUNAhryuuS3b4lEN2LNwxUXFKsEf8QLDCjmZuimlj23BHppJgcrGvNtqKg==} + '@types/qs@6.14.0': resolution: {integrity: sha512-eOunJqu0K1923aExK6y8p6fsihYEn/BYuQ4g0CxAAgFc4b/ZLN4CrsRZ55srTdqoiLzU2B2evC+apEIxprEzkQ==} @@ -3426,6 +3451,10 @@ packages: resolution: {integrity: sha512-HGyxoOTYUyCM6stUe6EJgnd4EoewAI7zMdfqO+kGjnlZmBDz/cR5pf8r/cR4Wq60sL/p0IkcjUEEPwS3GFrIyw==} engines: {node: '>=8'} + arrify@2.0.1: + resolution: {integrity: sha512-3duEwti880xqi4eAMN8AyR4a0ByT90zoYdLlevfrvU43vb0YZwZVfxOgxWrLXXXpyugL0hNZc9G6BiB5B3nUug==} + engines: {node: '>=8'} + assertion-error@2.0.1: resolution: {integrity: sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA==} engines: {node: '>=12'} @@ -3776,8 +3805,8 @@ packages: resolution: {integrity: sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg==} engines: {node: '>= 0.8'} - commander@14.0.1: - resolution: {integrity: sha512-2JkV3gUZUVrbNA+1sjBOYLsMZ5cEEl8GTFP2a4AVz5hvasAMCQ1D2l2le/cX+pV4N6ZU17zjUahLpIXRrnWL8A==} + commander@14.0.2: + resolution: {integrity: sha512-TywoWNNRbhoD0BXs1P3ZEScW8W5iKrnbithIl0YH+uCmBd0QpPOA8yc82DS3BIE5Ma6FnBVUsJ7wVUDz4dvOWQ==} engines: {node: '>=20'} compare-func@2.0.0: @@ -4660,6 +4689,10 @@ packages: resolution: {integrity: sha512-V6eky/xz2mcKfAd1Ioxyd6nmA61gao3n01C+YeuIwu3vzM9EDR6wcVzMSIbLMDXWeoi9SHYctXuKYC5uJUT3eQ==} engines: {node: '>=14'} + google-gax@5.0.6: + resolution: {integrity: sha512-1kGbqVQBZPAAu4+/R1XxPQKP0ydbNYoLAr4l0ZO2bMV0kLyLW4I1gAk++qBLWt7DPORTzmWRMsCZe86gDjShJA==} + engines: {node: '>=18'} + google-logging-utils@0.0.2: resolution: {integrity: sha512-NEgUnEcBiP5HrPzufUkBzJOD/Sxsco3rLNo1F1TNf7ieU8ryUzBhqba8r756CjLX7rn3fHl6iLEwPYuqpoKgQQ==} engines: {node: '>=14'} @@ -4743,6 +4776,9 @@ packages: resolution: {integrity: sha512-M422h7o/BR3rmCQ8UHi7cyyMqKltdP9Uo+J2fXK+RSAY+wTcKOIRyhTuKv4qn+DJf3g+PL890AzId5KZpX+CBg==} engines: {node: ^20.17.0 || >=22.9.0} + html-entities@2.6.0: + resolution: {integrity: sha512-kig+rMn/QOVRvr7c86gQ8lWXq+Hkv6CbAH1hLu+RG338StTpE8Z0b44SDVaqVu7HGKf27frdmUYEs9hTUX/cLQ==} + html-escaper@2.0.2: resolution: {integrity: sha512-H2iMtd0I4Mt5eYiapRdIDjp+XzelXQ0tFE4JS7YFwFevXXMmOp9myNrUvCg0D6ws8iqkRPBfKHgbwig1SmlLfg==} @@ -6248,6 +6284,10 @@ packages: resolution: {integrity: sha512-SAzp/O4Yh02jGdRc+uIrGoe87dkN/XtwxfZ4ZyafJHymd79ozp5VG5nyZ7ygqPM5+cpLDjjGnYFUkngonyDPOQ==} engines: {node: '>=14.0.0'} + proto3-json-serializer@3.0.4: + resolution: {integrity: sha512-E1sbAYg3aEbXrq0n1ojJkRHQJGE1kaE/O6GLA94y8rnJBfgvOPTOd1b9hOceQK1FFZI9qMh1vBERCyO2ifubcw==} + engines: {node: '>=18'} + protobufjs@7.5.4: resolution: {integrity: sha512-CvexbZtbov6jW2eXAvLukXjXUW1TzFaivC46BpWc/3BpcCysb5Vffu+B3XHMm8lVEuy2Mm4XGex8hBSg1yapPg==} engines: {node: '>=12.0.0'} @@ -6271,6 +6311,9 @@ packages: pump@3.0.3: resolution: {integrity: sha512-todwxLMY7/heScKmntwQG8CXVkWUOdYxIvY2s0VWAAMh/nd8SoYiRaKjlr7+iCs984f2P8zvrfWcDDYVb73NfA==} + pumpify@2.0.1: + resolution: {integrity: sha512-m7KOje7jZxrmutanlkS1daj1dS6z6BgslzOXmcSEpIlCxM3VJH7lG5QLeck/6hgF6F4crFf01UtQmNsJfweTAw==} + punycode.js@2.3.1: resolution: {integrity: sha512-uxFIHU0YlHYhDQtV4R9J6a52SLx28BCjT+4ieh7IGbgwVJWO+km431c4yRlREUAsAmt/uMjQUyQHNEPf0M39CA==} engines: {node: '>=6'} @@ -6425,6 +6468,10 @@ packages: resolution: {integrity: sha512-dUOvLMJ0/JJYEn8NrpOaGNE7X3vpI5XlZS/u0ANjqtcZVKnIxP7IgCFwrKTxENw29emmwug53awKtaMm4i9g5w==} engines: {node: '>=14'} + retry-request@8.0.2: + resolution: {integrity: sha512-JzFPAfklk1kjR1w76f0QOIhoDkNkSqW8wYKT08n9yysTmZfB+RQ2QoXoTAeOi1HD9ZipTyTAZg3c4pM/jeqgSw==} + engines: {node: '>=18'} + retry@0.12.0: resolution: {integrity: sha512-9LkiTwjUh6rT555DtE9rTX+BKByPfrMzEAtnlEtdEwr3Nkffwiihqe2bWADg+OQRjt9gl6ICdmB/ZFDCGAtSow==} engines: {node: '>= 4'} @@ -6882,6 +6929,10 @@ packages: engines: {node: '>=10'} deprecated: Old versions of tar are not supported, and contain widely publicized security vulnerabilities, which have been fixed in the current version. Please update. Support for old versions may be purchased (at exorbitant rates) by contacting i@izs.me + teeny-request@10.1.0: + resolution: {integrity: sha512-3ZnLvgWF29jikg1sAQ1g0o+lr5JX6sVgYvfUJazn7ZjJroDBUTWp44/+cFVX0bULjv4vci+rBD+oGVAkWqhUbw==} + engines: {node: '>=18'} + teeny-request@9.0.0: resolution: {integrity: sha512-resvxdc6Mgb7YEThw6G6bExlXKkv6+YbuzGg9xuXxSgxJF7Ozs+o8Y9+2R3sArdWdW8nOokoQb1yrpFB0pQK2g==} engines: {node: '>=14'} @@ -8667,6 +8718,35 @@ snapshots: '@shikijs/types': 3.20.0 '@shikijs/vscode-textmate': 10.0.2 + '@google-cloud/common@6.0.0': + dependencies: + '@google-cloud/projectify': 4.0.0 + '@google-cloud/promisify': 4.1.0 + arrify: 2.0.1 + duplexify: 4.1.3 + extend: 3.0.2 + google-auth-library: 10.5.0 + html-entities: 2.6.0 + retry-request: 8.0.2 + teeny-request: 10.1.0 + transitivePeerDependencies: + - supports-color + + '@google-cloud/projectify@4.0.0': {} + + '@google-cloud/promisify@4.1.0': {} + + '@google-cloud/speech@7.2.1': + dependencies: + '@google-cloud/common': 6.0.0 + '@types/pumpify': 1.4.5 + google-gax: 5.0.6 + pumpify: 2.0.1 + stream-events: 1.0.5 + uuid: 11.1.0 + transitivePeerDependencies: + - supports-color + '@google-cloud/text-to-speech@5.8.1(encoding@0.1.13)': dependencies: google-gax: 4.6.1(encoding@0.1.13) @@ -10874,6 +10954,10 @@ snapshots: '@types/diff-match-patch@1.0.36': {} + '@types/duplexify@3.6.5': + dependencies: + '@types/node': 20.19.31 + '@types/estree@1.0.8': {} '@types/express-serve-static-core@5.1.0': @@ -11007,6 +11091,11 @@ snapshots: '@types/phoenix@1.6.7': {} + '@types/pumpify@1.4.5': + dependencies: + '@types/duplexify': 3.6.5 + '@types/node': 20.19.31 + '@types/qs@6.14.0': {} '@types/range-parser@1.2.7': {} @@ -11409,6 +11498,8 @@ snapshots: array-union@2.1.0: {} + arrify@2.0.1: {} + assertion-error@2.0.1: {} ast-types@0.13.4: @@ -11787,7 +11878,7 @@ snapshots: dependencies: delayed-stream: 1.0.0 - commander@14.0.1: {} + commander@14.0.2: {} compare-func@2.0.0: dependencies: @@ -12890,6 +12981,22 @@ snapshots: - encoding - supports-color + google-gax@5.0.6: + dependencies: + '@grpc/grpc-js': 1.14.0 + '@grpc/proto-loader': 0.8.0 + duplexify: 4.1.3 + google-auth-library: 10.5.0 + google-logging-utils: 1.1.3 + node-fetch: 3.3.2 + object-hash: 3.0.0 + proto3-json-serializer: 3.0.4 + protobufjs: 7.5.4 + retry-request: 8.0.2 + rimraf: 5.0.10 + transitivePeerDependencies: + - supports-color + google-logging-utils@0.0.2: {} google-logging-utils@1.1.3: {} @@ -12975,6 +13082,8 @@ snapshots: dependencies: lru-cache: 11.2.2 + html-entities@2.6.0: {} + html-escaper@2.0.2: {} http-assert@1.5.0: @@ -13460,7 +13569,7 @@ snapshots: lint-staged@16.2.3: dependencies: - commander: 14.0.1 + commander: 14.0.2 listr2: 9.0.4 micromatch: 4.0.8 nano-spawn: 1.0.3 @@ -14439,6 +14548,10 @@ snapshots: dependencies: protobufjs: 7.5.4 + proto3-json-serializer@3.0.4: + dependencies: + protobufjs: 7.5.4 + protobufjs@7.5.4: dependencies: '@protobufjs/aspromise': 1.1.2 @@ -14486,6 +14599,12 @@ snapshots: end-of-stream: 1.4.5 once: 1.4.0 + pumpify@2.0.1: + dependencies: + duplexify: 4.1.3 + inherits: 2.0.4 + pump: 3.0.3 + punycode.js@2.3.1: {} punycode@2.3.1: {} @@ -14684,6 +14803,13 @@ snapshots: - encoding - supports-color + retry-request@8.0.2: + dependencies: + extend: 3.0.2 + teeny-request: 10.1.0 + transitivePeerDependencies: + - supports-color + retry@0.12.0: optional: true @@ -15291,6 +15417,15 @@ snapshots: mkdirp: 1.0.4 yallist: 4.0.0 + teeny-request@10.1.0: + dependencies: + http-proxy-agent: 5.0.0 + https-proxy-agent: 5.0.1 + node-fetch: 3.3.2 + stream-events: 1.0.5 + transitivePeerDependencies: + - supports-color + teeny-request@9.0.0(encoding@0.1.13): dependencies: http-proxy-agent: 5.0.0 diff --git a/src/cli/factories/commandFactory.ts b/src/cli/factories/commandFactory.ts index daeec846f..a0a728987 100644 --- a/src/cli/factories/commandFactory.ts +++ b/src/cli/factories/commandFactory.ts @@ -4,6 +4,10 @@ import chalk from "chalk"; import ora from "ora"; import type { Argv, CommandModule } from "yargs"; import { ModelResolver } from "../../lib/models/modelResolver.js"; +import { + AUDIO_EXTENSIONS, + type AudioExtension, +} from "../../lib/processors/config/fileTypes.js"; import type { ChunkingStrategy } from "../../lib/rag/types.js"; import { globalSession } from "../../lib/session/globalSessionState.js"; import type { @@ -13,7 +17,6 @@ import type { GenerateResult, StreamCommandArgs, } from "../../lib/types/cli.js"; -import type { JsonValue } from "../../lib/types/common.js"; // Use TokenUsage from standard types - no local interface needed import { type BaseContext, @@ -24,11 +27,12 @@ import type { ConversationMemoryConfig, ConversationSummary, } from "../../lib/types/conversation.js"; -import type { AnalyticsData, TokenUsage } from "../../lib/types/index.js"; - +import type { + AnalyticsData, + JsonValue, + TokenUsage, +} from "../../lib/types/index.js"; import { checkRedisAvailability } from "../../lib/utils/conversationMemory.js"; - -import { normalizeEvaluationData } from "../../lib/utils/evaluationUtils.js"; import { logger } from "../../lib/utils/logger.js"; import { createThinkingConfigFromRecord } from "../../lib/utils/thinkingConfig.js"; import { configManager } from "../commands/config.js"; @@ -124,7 +128,7 @@ export class CLICommandFactory { file: { type: "string" as const, description: - "Add file with auto-detection (CSV, image, etc. - can be used multiple times)", + "Add file with auto-detection (CSV, image, audio, etc. - can be used multiple times)", }, csvMaxRows: { type: "number" as const, @@ -309,6 +313,62 @@ export class CLICommandFactory { description: "Auto-play generated audio", }, + // Speech-to-Text (STT) options + sttLanguage: { + type: "string" as const, + alias: "stt-lang", + description: + "STT language code (optional, auto-detects if not specified)", + }, + sttModel: { + type: "string" as const, + default: "default", + choices: [ + "default", + "command_and_search", + "phone_call", + "video", + "medical_dictation", + "latest_long", + "latest_short", + ], + description: "STT model to use", + }, + sttEnhanced: { + type: "boolean" as const, + default: false, + description: "Use enhanced STT model (higher accuracy, higher cost)", + }, + sttMaxAlternatives: { + type: "number" as const, + default: 1, + description: "Maximum number of alternative transcriptions (1-30)", + }, + sttEnablePunctuation: { + type: "boolean" as const, + default: true, + description: "Enable automatic punctuation in transcription", + }, + sttEnableTimestamps: { + type: "boolean" as const, + default: false, + description: "Enable word-level timestamps", + }, + sttEnableConfidence: { + type: "boolean" as const, + default: false, + description: "Enable word-level confidence scores", + }, + sttEnableDiarization: { + type: "boolean" as const, + default: false, + description: "Enable speaker diarization (who spoke when)", + }, + sttSpeakerCount: { + type: "number" as const, + description: "Expected number of speakers for diarization", + }, + // Video Generation options (Veo 3.1) outputMode: { type: "string" as const, @@ -493,6 +553,26 @@ export class CLICommandFactory { return resolveFilePaths(paths); } + // Helper method to detect if files array contains audio files + private static detectAudioFiles(files?: Array): boolean { + if (!files || files.length === 0) { + return false; + } + + for (const file of files) { + if (typeof file === "string") { + const ext = file.toLowerCase().split(".").pop(); + if (ext && AUDIO_EXTENSIONS.includes(`.${ext}` as AudioExtension)) { + return true; + } + } + // Buffers can't be easily detected as audio without inspecting content + // So we rely on file path detection + } + + return false; + } + // Helper method to process common options private static processOptions( argv: BaseCommandArgs & Record, @@ -581,6 +661,16 @@ export class CLICommandFactory { ttsQuality: argv.ttsQuality as "standard" | "hd" | undefined, ttsOutput: argv.ttsOutput as string | undefined, ttsPlay: argv.ttsPlay as boolean | undefined, + // STT options + sttLanguage: argv.sttLanguage as string | undefined, + sttModel: argv.sttModel as string | undefined, + sttEnhanced: argv.sttEnhanced as boolean | undefined, + sttMaxAlternatives: argv.sttMaxAlternatives as number | undefined, + sttEnablePunctuation: argv.sttEnablePunctuation as boolean | undefined, + sttEnableTimestamps: argv.sttEnableTimestamps as boolean | undefined, + sttEnableConfidence: argv.sttEnableConfidence as boolean | undefined, + sttEnableDiarization: argv.sttEnableDiarization as boolean | undefined, + sttSpeakerCount: argv.sttSpeakerCount as number | undefined, // Video generation options (Veo 3.1) outputMode: argv.outputMode as "text" | "video" | undefined, videoOutput: argv.videoOutput as string | undefined, @@ -1071,6 +1161,18 @@ export class CLICommandFactory { .example( '$0 generate "Smooth camera movement" --image ./input.jpg --provider vertex --model veo-3.1-generate-001 --outputMode video --videoResolution 720p --videoLength 6 --videoAspectRatio 16:9 --videoOutput ./output.mp4', "Video generation with full options", + ) + .example( + '$0 generate "Transcribe this audio" --file audio.mp3', + "Transcribe audio to text (auto-detected)", + ) + .example( + '$0 generate "Transcribe this meeting" --file meeting.wav --stt-language en-US --stt-enable-timestamps --stt-enable-diarization', + "Transcribe with timestamps and speaker detection", + ) + .example( + '$0 generate "Transcribe this call" --file call.mp3 --stt-model phone_call -o transcript.txt', + "Transcribe phone call with custom model", ), ); }, @@ -1220,6 +1322,94 @@ export class CLICommandFactory { return MCPCommandFactory.createDiscoverCommand(); } + /** + * Create STT discovery commands + */ + static createSTTCommands(): CommandModule { + return { + command: "stt ", + describe: "Speech-to-Text discovery and information", + builder: (yargs) => { + return yargs + .command( + "models [provider]", + "List available STT models", + (y) => + y + .positional("provider", { + type: "string", + description: "Provider name (e.g., google-ai, vertex)", + default: "google-ai", + }) + .example("$0 stt models", "List models for default provider") + .example( + "$0 stt models google-ai", + "List Google AI STT models", + ), + async (argv) => { + try { + const { NeuroLink } = await import("../../lib/neurolink.js"); + const sdk = new NeuroLink(); + + const spinner = ora( + `Fetching STT models for ${argv.provider}...`, + ).start(); + + const models = await sdk.getSTTModels(argv.provider as string); + + spinner.succeed(`Found ${models.length} available models`); + + logger.always(chalk.cyan("\n🎙️ Available STT Models:\n")); + + models.forEach((model) => { + let description = ""; + switch (model) { + case "default": + description = "General model for most use cases"; + break; + case "command_and_search": + description = "Optimized for short queries and commands"; + break; + case "phone_call": + description = "Optimized for audio from phone calls"; + break; + case "video": + description = "Optimized for audio from video files"; + break; + case "medical_dictation": + description = "Specialized for medical terminology"; + break; + case "latest_long": + description = "Latest model for long-form audio"; + break; + case "latest_short": + description = "Latest model for short audio"; + break; + } + logger.always(chalk.bold(` ${model}`)); + if (description) { + logger.always(chalk.dim(` ${description}`)); + } + }); + + logger.always( + chalk.dim( + "\n💡 Usage: neurolink generate --file audio.mp3 --stt-model ", + ), + ); + } catch (err) { + handleError(err as Error, "STT Models Error"); + } + }, + ) + .demandCommand(1, "You must specify a subcommand"); + }, + handler: () => { + // Parent command handler - subcommands handle execution + }, + }; + } + /** * Create memory commands */ @@ -1638,6 +1828,7 @@ export class CLICommandFactory { } finally { await conversationSelector.close(); } + await CLICommandFactory.flushLangfuseTraces(); return; } @@ -1818,10 +2009,11 @@ export class CLICommandFactory { } /** - * Execute the generate command + * Handle stdin input for generate command */ - private static async executeGenerate(argv: GenerateCommandArgs) { - // Handle stdin input if no input provided + private static async handleGenerateStdinInput( + argv: GenerateCommandArgs, + ): Promise { if (!argv.input && !process.stdin.isTTY) { let stdinData = ""; process.stdin.setEncoding("utf8"); @@ -1837,117 +2029,415 @@ export class CLICommandFactory { 'Input required. Use: neurolink generate "your prompt" or echo "prompt" | neurolink generate', ); } + } - const options = CLICommandFactory.processOptions(argv); + /** + * Process multimodal CLI inputs (images, CSVs, PDFs, videos, STT audio, files) + */ + private static processMultimodalInputs(argv: GenerateCommandArgs): { + imageBuffers?: Array; + csvFiles?: Array; + pdfFiles?: Array; + videoFiles?: Array; + files?: Array; + } { + return { + imageBuffers: CLICommandFactory.processCliImages( + argv.image as string | string[] | undefined, + ), + csvFiles: CLICommandFactory.processCliCSVFiles( + argv.csv as string | string[] | undefined, + ), + pdfFiles: CLICommandFactory.processCliPDFFiles( + argv.pdf as string | string[] | undefined, + ), + videoFiles: CLICommandFactory.processCliVideoFiles( + argv.video as string | string[] | undefined, + ), + files: CLICommandFactory.processCliFiles( + argv.file as string | string[] | undefined, + ), + }; + } + + /** + * Process context for generate command + */ + private static processGenerateContext( + argv: GenerateCommandArgs, + options: BaseCommandArgs & Record, + ): { + inputText: string; + contextMetadata?: Partial; + } { + let inputText = argv.input as string; + let contextMetadata: Partial | undefined; + + if (options.context && options.contextConfig) { + const processedContextResult = ContextFactory.processContext( + options.context as BaseContext, + options.contextConfig, + ); + + if (processedContextResult.processedContext) { + inputText = `${processedContextResult.processedContext}\n\n${inputText}`; + } + + contextMetadata = { + ...ContextFactory.extractAnalyticsContext( + options.context as BaseContext, + ), + contextMode: processedContextResult.config.mode, + contextTruncated: processedContextResult.metadata.truncated, + }; + + if (options.debug) { + logger.debug("Context processed:", { + mode: processedContextResult.config.mode, + truncated: processedContextResult.metadata.truncated, + }); + } + } + + return { inputText, contextMetadata }; + } + + /** + * Execute dry-run for generate command + */ + private static async executeDryRunGenerate( + options: BaseCommandArgs & Record, + contextMetadata?: Partial, + ): Promise { + if (!options.quiet) { + logger.always( + chalk.blue("🧪 Dry-run mode enabled. No API calls will be made."), + ); + } + + const mockResult = { + content: "Mock response for testing purposes", + analytics: { + provider: options.provider || "auto", + model: options.model || "auto-selected", + tokenUsage: { input: 100, output: 50, total: 150 }, + cost: 0.002, + requestDuration: 1500, + }, + ...(contextMetadata && { context: contextMetadata }), + }; + + if (options.debug) { + logger.debug("Dry-run options:", options); + } + + CLICommandFactory.handleOutput(mockResult, options); + + if (options.enableAnalytics && !options.quiet) { + logger.always("\n" + chalk.cyan("📊 Analytics (Dry-run):")); + logger.always(JSON.stringify(mockResult.analytics, null, 2)); + } + + if (options.enableEvaluation && !options.quiet) { + logger.always("\n" + chalk.cyan("✅ Evaluation (Dry-run):")); + logger.always("Mock evaluation data would appear here"); + } + + if (!globalSession.getCurrentSessionId()) { + await CLICommandFactory.flushLangfuseTraces(); + process.exit(0); + } + } + + /** + * Build SDK generate options from command arguments + */ + private static buildGenerateOptions( + argv: GenerateCommandArgs, + options: BaseCommandArgs & Record, + multimodalInputs: { + imageBuffers?: Array; + csvFiles?: Array; + pdfFiles?: Array; + videoFiles?: Array; + files?: Array; + }, + inputText: string, + context?: Partial, + ): { + isVideoMode: boolean; + isSTTMode: boolean; + generateInput: { text: string; [key: string]: unknown }; + sdkOptions: Record; + } { + const sessionVariables = globalSession.getSessionVariables(); + const enhancedOptions = { ...options, ...sessionVariables }; - // Determine if video generation mode is enabled const isVideoMode = (options as Record).outputMode === "video"; - const spinnerMessage = isVideoMode - ? "🎬 Generating video... (this may take 1-2 minutes)" - : "🤖 Generating text..."; - const spinner = argv.quiet ? null : ora(spinnerMessage).start(); - try { - // Add delay if specified - if (options.delay) { - await new Promise((resolve) => setTimeout(resolve, options.delay)); - } + // Detect STT mode by checking for audio files in the files array + const isSTTMode = this.detectAudioFiles(multimodalInputs.files); - // Process context if provided - let inputText = argv.input as string; - let contextMetadata: Partial | undefined; + if (isVideoMode) { + CLICommandFactory.configureVideoMode(enhancedOptions, argv, options); + } - if (options.context && options.contextConfig) { - const processedContextResult = ContextFactory.processContext( - options.context, - options.contextConfig, - ); + const generateInput = { + text: inputText, + ...(multimodalInputs.imageBuffers && { + images: multimodalInputs.imageBuffers, + }), + ...(multimodalInputs.csvFiles && { csvFiles: multimodalInputs.csvFiles }), + ...(multimodalInputs.pdfFiles && { pdfFiles: multimodalInputs.pdfFiles }), + ...(multimodalInputs.videoFiles && { + videoFiles: multimodalInputs.videoFiles, + }), + ...(multimodalInputs.files && { files: multimodalInputs.files }), + }; - // Integrate context into prompt if configured - if (processedContextResult.processedContext) { - inputText = processedContextResult.processedContext + inputText; - } + const sdkOptions = { + csvOptions: { + maxRows: argv.csvMaxRows as number | undefined, + formatStyle: argv.csvFormat as "raw" | "markdown" | "json" | undefined, + }, + videoOptions: { + frames: argv.videoFrames as number | undefined, + quality: argv.videoQuality as number | undefined, + format: argv.videoFormat as "jpeg" | "png" | undefined, + transcribeAudio: argv.transcribeAudio as boolean | undefined, + }, + output: isVideoMode + ? { + mode: "video" as const, + video: { + resolution: enhancedOptions.videoResolution as + | "720p" + | "1080p" + | undefined, + length: enhancedOptions.videoLength as 4 | 6 | 8 | undefined, + aspectRatio: enhancedOptions.videoAspectRatio as + | "9:16" + | "16:9" + | undefined, + audio: enhancedOptions.videoAudio as boolean | undefined, + }, + } + : undefined, + provider: enhancedOptions.provider, + model: enhancedOptions.model, + temperature: enhancedOptions.temperature, + maxTokens: enhancedOptions.maxTokens, + systemPrompt: enhancedOptions.systemPrompt, + timeout: enhancedOptions.timeout + ? (enhancedOptions.timeout as number) * 1000 + : undefined, + disableTools: enhancedOptions.disableTools, + enableAnalytics: enhancedOptions.enableAnalytics, + enableEvaluation: enhancedOptions.enableEvaluation, + evaluationDomain: enhancedOptions.evaluationDomain as string | undefined, + toolUsageContext: enhancedOptions.toolUsageContext as string | undefined, + context, + region: (options as Record).region as string | undefined, + thinkingConfig: createThinkingConfigFromRecord( + options as Record, + ), + factoryConfig: enhancedOptions.domain + ? { + domainType: enhancedOptions.domain, + enhancementType: "domain-configuration", + validateDomainData: true, + } + : undefined, + stt: isSTTMode + ? { + languageCode: enhancedOptions.sttLanguage as string | undefined, + model: enhancedOptions.sttModel as string | undefined, + useEnhanced: enhancedOptions.sttEnhanced as boolean | undefined, + maxAlternatives: enhancedOptions.sttMaxAlternatives as + | number + | undefined, + enableAutomaticPunctuation: enhancedOptions.sttEnablePunctuation as + | boolean + | undefined, + enableWordTimeOffsets: enhancedOptions.sttEnableTimestamps as + | boolean + | undefined, + enableWordConfidence: enhancedOptions.sttEnableConfidence as + | boolean + | undefined, + enableSpeakerDiarization: enhancedOptions.sttEnableDiarization as + | boolean + | undefined, + diarizationSpeakerCount: enhancedOptions.sttSpeakerCount as + | number + | undefined, + } + : undefined, + rag: (argv.ragFiles as string[] | undefined)?.length + ? { + files: argv.ragFiles as string[], + strategy: argv.ragStrategy as ChunkingStrategy | undefined, + chunkSize: argv.ragChunkSize as number | undefined, + chunkOverlap: argv.ragChunkOverlap as number | undefined, + topK: argv.ragTopK as number | undefined, + } + : undefined, + }; - // Add context metadata for analytics - contextMetadata = { - ...ContextFactory.extractAnalyticsContext( - options.context as BaseContext, - ), - contextMode: processedContextResult.config.mode, - contextTruncated: processedContextResult.metadata.truncated, - }; + return { isVideoMode, isSTTMode, generateInput, sdkOptions }; + } - if (options.debug) { - logger.debug("Context processed:", { - mode: processedContextResult.config.mode, - truncated: processedContextResult.metadata.truncated, - processingTime: processedContextResult.metadata.processingTime, - }); - } + /** + * Handle STT transcription output + */ + private static async handleSTTOutput( + result: { transcription?: unknown } & Record, + options: BaseCommandArgs & Record, + ): Promise { + if (!result.transcription) { + return; + } + + const transcription = result.transcription as { + text: string; + confidence: number; + languageCode?: string; + duration?: number; + metadata: { provider: string; latency: number }; + alternatives?: Array<{ confidence: number; transcript: string }>; + words?: Array<{ + startTime: number; + endTime: number; + word: string; + confidence?: number; + speakerTag?: number; + }>; + }; + + // Save to file if output specified + if (options.output) { + if (options.format === "json") { + fs.writeFileSync( + options.output as string, + JSON.stringify(transcription, null, 2), + ); + } else { + fs.writeFileSync(options.output as string, transcription.text); + } + if (!options.quiet) { + logger.always(`\n✓ Saved to: ${options.output}`); } + } else if (options.format === "json") { + // Write JSON to stdout when no output file specified (regardless of quiet mode) + process.stdout.write(JSON.stringify(transcription, null, 2) + "\n"); + } else { + // Always emit plain transcription text when no file and format is not JSON + logger.always(transcription.text); + } - // Handle dry-run mode for testing - if (options.dryRun) { - const mockResult = { - content: "Mock response for testing purposes", - provider: options.provider || "auto", - model: options.model || "test-model", - usage: { - input: 10, - output: 15, - total: 25, + // Decorative metadata output (suppressed when quiet or JSON format) + if (!options.quiet && options.format !== "json" && !options.output) { + logger.always("\n" + "═".repeat(70)); + logger.always("📄 TRANSCRIPTION METADATA"); + logger.always("═".repeat(70)); + logger.always( + `✓ Confidence: ${(transcription.confidence * 100).toFixed(1)}%`, + ); + logger.always(`✓ Language: ${transcription.languageCode}`); + if (transcription.duration) { + logger.always(`✓ Duration: ${transcription.duration.toFixed(2)}s`); + } + logger.always(`✓ Provider: ${transcription.metadata.provider}`); + logger.always(`✓ Latency: ${transcription.metadata.latency}ms`); + logger.always("═".repeat(70)); + + // Show alternatives if available + if (transcription.alternatives && transcription.alternatives.length > 0) { + logger.always("\nALTERNATIVE TRANSCRIPTIONS:"); + transcription.alternatives.forEach( + (alt: { confidence: number; transcript: string }, idx: number) => { + logger.always( + `${idx + 1}. [${(alt.confidence * 100).toFixed(1)}%] ${alt.transcript}`, + ); }, - responseTime: 150, - analytics: options.enableAnalytics - ? { - provider: options.provider || "auto", - model: options.model || "test-model", - tokenUsage: { input: 10, output: 15, total: 25 }, - cost: 0.00025, - requestDuration: 150, - context: contextMetadata, - } - : undefined, - evaluation: options.enableEvaluation - ? normalizeEvaluationData({ - relevance: 8, - accuracy: 9, - completeness: 8, - overall: 8.3, - reasoning: "Test evaluation response", - evaluationModel: "test-evaluator", - evaluationTime: 50, - }) - : undefined, - }; + ); + } - if (spinner) { - spinner.succeed(chalk.green("✅ Dry-run completed successfully!")); - } + // Show word-level details if enabled + if ( + transcription.words && + transcription.words.length > 0 && + options.sttEnableTimestamps + ) { + logger.always("\nWORD TIMESTAMPS:"); + logger.always("─".repeat(70)); + transcription.words.forEach( + (word: { + startTime: number; + endTime: number; + word: string; + confidence?: number; + speakerTag?: number; + }) => { + const conf = word.confidence + ? ` (${(word.confidence * 100).toFixed(0)}%)` + : ""; + const speaker = word.speakerTag + ? ` [Speaker ${word.speakerTag}]` + : ""; + logger.always( + ` ${word.startTime.toFixed(2)}s - ${word.endTime.toFixed(2)}s: ${word.word}${conf}${speaker}`, + ); + }, + ); + logger.always("═".repeat(70)); + } + } + } + + /** + * Execute the generate command + */ + private static async executeGenerate(argv: GenerateCommandArgs) { + // Handle stdin input if no input provided + await this.handleGenerateStdinInput(argv); - CLICommandFactory.handleOutput(mockResult, options); + const options = this.processOptions(argv); - if (options.debug) { - logger.debug("\n" + chalk.yellow("Debug Information (Dry-run):")); - logger.debug("Provider:", mockResult.provider); - logger.debug("Model:", mockResult.model); - logger.debug("Mode: DRY-RUN (no actual API calls made)"); - } + const isVideoMode = + (options as Record).outputMode === "video"; - if (!globalSession.getCurrentSessionId()) { - await CLICommandFactory.flushLangfuseTraces(); - process.exit(0); - } + const spinnerMessage = isVideoMode + ? "🎬 Generating video... (this may take 1-2 minutes)" + : "🤖 Generating text..."; + + const spinner = argv.quiet ? null : ora(spinnerMessage).start(); + + try { + if (options.delay) { + await new Promise((resolve) => + setTimeout(resolve, options.delay as number), + ); + } + + const { inputText, contextMetadata } = this.processGenerateContext( + argv, + options, + ); + + if (options.dryRun) { + await this.executeDryRunGenerate(options, contextMetadata); + spinner?.succeed(chalk.green("✅ Dry-run completed!")); + return; } const sdk = globalSession.getOrCreateNeuroLink(); - const sessionVariables = globalSession.getSessionVariables(); - const enhancedOptions = { ...options, ...sessionVariables }; const sessionId = globalSession.getCurrentSessionId(); + const context = sessionId - ? { ...options.context, sessionId } - : options.context; + ? { ...contextMetadata, sessionId } + : contextMetadata; if (options.debug) { logger.debug("CLI Tools configuration:", { @@ -1956,165 +2446,55 @@ export class CLICommandFactory { }); } - // Video generation doesn't support tools, so auto-disable them - if (isVideoMode) { - CLICommandFactory.configureVideoMode(enhancedOptions, argv, options); - } + const multimodalInputs = this.processMultimodalInputs(argv); - // Process CLI multimodal inputs - const imageBuffers = CLICommandFactory.processCliImages( - argv.image as string | string[] | undefined, - ); - const csvFiles = CLICommandFactory.processCliCSVFiles( - argv.csv as string | string[] | undefined, - ); - const pdfFiles = CLICommandFactory.processCliPDFFiles( - argv.pdf as string | string[] | undefined, - ); - const videoFiles = CLICommandFactory.processCliVideoFiles( - argv.video as string | string[] | undefined, - ); - const files = CLICommandFactory.processCliFiles( - argv.file as string | string[] | undefined, - ); - - const generateInput = { - text: inputText, - ...(imageBuffers && { images: imageBuffers }), - ...(csvFiles && { csvFiles }), - ...(pdfFiles && { pdfFiles }), - ...(videoFiles && { videoFiles }), - ...(files && { files }), - }; + const { isSTTMode, generateInput, sdkOptions } = + this.buildGenerateOptions( + argv, + options, + multimodalInputs, + inputText, + context, + ); const result = await sdk.generate({ input: generateInput, - csvOptions: { - maxRows: argv.csvMaxRows as number | undefined, - formatStyle: argv.csvFormat as - | "raw" - | "markdown" - | "json" - | undefined, - }, - videoOptions: { - frames: argv.videoFrames as number | undefined, - quality: argv.videoQuality as number | undefined, - format: argv.videoFormat as "jpeg" | "png" | undefined, - transcribeAudio: argv.transcribeAudio as boolean | undefined, - }, - // Video generation output configuration - output: isVideoMode - ? { - mode: "video" as const, - video: { - resolution: enhancedOptions.videoResolution as - | "720p" - | "1080p" - | undefined, - length: enhancedOptions.videoLength as 4 | 6 | 8 | undefined, - aspectRatio: enhancedOptions.videoAspectRatio as - | "9:16" - | "16:9" - | undefined, - audio: enhancedOptions.videoAudio as boolean | undefined, - }, - } - : undefined, - provider: enhancedOptions.provider, - model: enhancedOptions.model, - temperature: enhancedOptions.temperature, - maxTokens: enhancedOptions.maxTokens, - systemPrompt: enhancedOptions.systemPrompt, - timeout: enhancedOptions.timeout - ? enhancedOptions.timeout * 1000 - : undefined, - disableTools: enhancedOptions.disableTools, - enableAnalytics: enhancedOptions.enableAnalytics, - enableEvaluation: enhancedOptions.enableEvaluation, - evaluationDomain: enhancedOptions.evaluationDomain as - | string - | undefined, - toolUsageContext: enhancedOptions.toolUsageContext as - | string - | undefined, - context: context, - region: (options as Record).region as - | string - | undefined, - thinkingConfig: createThinkingConfigFromRecord( - options as Record, - ), - factoryConfig: enhancedOptions.domain - ? { - domainType: enhancedOptions.domain, - enhancementType: "domain-configuration", - validateDomainData: true, - } - : undefined, - // RAG configuration - rag: (argv.ragFiles as string[] | undefined)?.length - ? { - files: argv.ragFiles as string[], - strategy: argv.ragStrategy as ChunkingStrategy | undefined, - chunkSize: argv.ragChunkSize as number | undefined, - chunkOverlap: argv.ragChunkOverlap as number | undefined, - topK: argv.ragTopK as number | undefined, - } - : undefined, + ...sdkOptions, }); - if (spinner) { - if (isVideoMode) { - spinner.succeed(chalk.green("✅ Video generated successfully!")); - } else { - spinner.succeed(chalk.green("✅ Text generated successfully!")); - } + if (isVideoMode) { + spinner?.succeed(chalk.green("✅ Video generated successfully!")); + } else if (isSTTMode) { + spinner?.succeed(chalk.green("✅ Audio transcribed successfully!")); + } else { + spinner?.succeed(chalk.green("✅ Text generated successfully!")); } - // Display provider and model info by default (unless quiet mode) if (!options.quiet) { - const providerInfo = result.provider || "auto"; - const modelInfo = result.model || "default"; logger.always( - chalk.gray(`🔧 Provider: ${providerInfo} | Model: ${modelInfo}`), + chalk.gray( + `🔧 Provider: ${result.provider || "auto"} | Model: ${result.model || "default"}`, + ), ); } - // Handle output with universal formatting (for text mode) - if (!isVideoMode) { - CLICommandFactory.handleOutput(result, options); + if (isSTTMode && result.transcription) { + await this.handleSTTOutput(result, options); } - // Handle TTS audio file output if --tts-output is provided - await CLICommandFactory.handleTTSOutput(result, options); - - // Handle video file output if --videoOutput is provided - await CLICommandFactory.handleVideoOutput(result, options); - - if (options.debug) { - logger.debug("\n" + chalk.yellow("Debug Information:")); - logger.debug("Provider:", result.provider); - logger.debug("Model:", result.model); - if (result.analytics) { - logger.debug("Analytics:", JSON.stringify(result.analytics, null, 2)); - } - if (result.evaluation) { - logger.debug( - "Evaluation:", - JSON.stringify(result.evaluation, null, 2), - ); - } + if (!isVideoMode && !isSTTMode) { + this.handleOutput(result, options); } + await this.handleTTSOutput(result, options); + await this.handleVideoOutput(result, options); + if (!globalSession.getCurrentSessionId()) { - await CLICommandFactory.flushLangfuseTraces(); + await this.flushLangfuseTraces(); process.exit(0); } } catch (error) { - if (spinner) { - spinner.fail(); - } + spinner?.fail(); handleError(error as Error, "Generation"); } } @@ -2746,7 +3126,7 @@ export class CLICommandFactory { ); if (processedContextResult.processedContext) { - inputText = processedContextResult.processedContext + inputText; + inputText = `${processedContextResult.processedContext}\n\n${inputText}`; } contextMetadata = { diff --git a/src/cli/loop/optionsSchema.ts b/src/cli/loop/optionsSchema.ts index b999885cd..a087af7e5 100644 --- a/src/cli/loop/optionsSchema.ts +++ b/src/cli/loop/optionsSchema.ts @@ -26,6 +26,7 @@ export const textGenerationOptionsSchema: Record< | "region" | "csvOptions" | "tts" + | "stt" | "thinkingConfig" // Complex object, use thinking/thinkingBudget instead | "fileRegistry" // Internal: set by SDK, not by CLI | "abortSignal" // Runtime object, not CLI-settable diff --git a/src/cli/parser.ts b/src/cli/parser.ts index 3c282e43e..30c2670a6 100644 --- a/src/cli/parser.ts +++ b/src/cli/parser.ts @@ -171,6 +171,9 @@ export function initializeCliParser() { // Discover Command - Using CLICommandFactory .command(CLICommandFactory.createDiscoverCommand()) + // STT Command Group - Using CLICommandFactory + .command(CLICommandFactory.createSTTCommands()) + // Configuration Command Group - Using CLICommandFactory .command(CLICommandFactory.createConfigCommands()) diff --git a/src/lib/adapters/stt/googleSTTHandler.ts b/src/lib/adapters/stt/googleSTTHandler.ts new file mode 100644 index 000000000..2838f8d7f --- /dev/null +++ b/src/lib/adapters/stt/googleSTTHandler.ts @@ -0,0 +1,393 @@ +/** + * Google Cloud Speech-to-Text Handler + * + * Handler for Google Cloud Speech-to-Text API integration. + * Mirrors the architecture of GoogleTTSHandler. + * + * @module adapters/stt/googleSTTHandler + * @see https://cloud.google.com/speech-to-text/docs + */ +import { SpeechClient, protos } from "@google-cloud/speech"; +import { STTError, STT_ERROR_CODES } from "../../utils/sttProcessor.js"; +import type { STTHandler } from "../../utils/sttProcessor.js"; +import type { + STTOptions, + STTResult, + AudioEncoding, + WordInfo, + TranscriptAlternative, +} from "../../types/sttTypes.js"; +import { ErrorCategory, ErrorSeverity } from "../../constants/enums.js"; +import { logger } from "../../utils/logger.js"; +import { withTimeout } from "../../utils/timeout.js"; + +export class GoogleSTTHandler implements STTHandler { + private client: SpeechClient | null = null; + + // Google Cloud Speech-to-Text limits + private static readonly DEFAULT_MAX_AUDIO_SIZE_MB = 10; + private static readonly DEFAULT_MAX_DURATION_SECONDS = 60; // Synchronous API limit + private static readonly DEFAULT_API_TIMEOUT_MS = 60000; // 60 seconds + + public readonly maxAudioSizeMB: number = + GoogleSTTHandler.DEFAULT_MAX_AUDIO_SIZE_MB; + public readonly maxDurationSeconds: number = + GoogleSTTHandler.DEFAULT_MAX_DURATION_SECONDS; + + constructor(credentialsPath?: string) { + try { + const options = credentialsPath ? { keyFilename: credentialsPath } : {}; // Uses GOOGLE_APPLICATION_CREDENTIALS env var + + this.client = new SpeechClient(options); + logger.info( + "[GoogleSTTHandler] Initialized Google Speech-to-Text client", + ); + } catch (err) { + logger.warn( + `[GoogleSTTHandler] Failed to initialize: ${err instanceof Error ? err.message : "Unknown error"}`, + ); + this.client = null; + } + } + + /** + * Check if provider is properly configured + */ + isConfigured(): boolean { + return this.client !== null; + } + + /** + * Get available models for Google Cloud Speech-to-Text + * + * @returns List of available model identifiers + */ + async getModels(): Promise { + return [ + "default", // General model for most use cases + "command_and_search", // Short queries and commands + "phone_call", // Audio from phone calls + "video", // Audio from video files + "medical_dictation", // Medical terminology + "latest_long", // Latest model for long audio + "latest_short", // Latest model for short audio + ]; + } + + /** + * Transcribe audio to text using Google Cloud Speech-to-Text API + * + * @param audio - Audio buffer to transcribe + * @param options - STT configuration options + * @returns Transcription result with metadata + */ + async transcribe(audio: Buffer, options: STTOptions): Promise { + if (!this.client) { + throw new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: + "Google Cloud Speech-to-Text client not initialized. Set GOOGLE_APPLICATION_CREDENTIALS or pass credentials path.", + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + }); + } + + const startTime = Date.now(); + + try { + // Map the encoding first to determine if we need sampleRateHertz + const encoding = options.encoding || "MP3"; + const mappedEncoding = this.mapAudioEncoding(encoding); + + // Self-describing formats auto-detect sample rate, non-self-describing formats require it + const selfDescribingFormats: AudioEncoding[] = [ + "MP3", + "FLAC", + "OGG_OPUS", + "WEBM_OPUS", + ]; + const isSelfDescribing = selfDescribingFormats.includes(encoding); + + // Build the request + const request = { + audio: { + content: audio.toString("base64"), + }, + config: { + encoding: mappedEncoding, + // For MP3 files and not self-describing formats, Google Cloud Speech-to-Text works better with explicit sample rate + ...((!isSelfDescribing || + options.sampleRateHertz || + encoding === "MP3") && { + sampleRateHertz: + options.sampleRateHertz || (encoding === "MP3" ? 44100 : 16000), + }), + languageCode: options.languageCode || "en-US", + alternativeLanguageCodes: options.alternativeLanguageCodes, + maxAlternatives: options.maxAlternatives || 1, + profanityFilter: options.profanityFilter ?? false, + enableAutomaticPunctuation: + options.enableAutomaticPunctuation ?? true, + enableWordTimeOffsets: options.enableWordTimeOffsets ?? false, + enableWordConfidence: options.enableWordConfidence ?? false, + speechContexts: options.speechContexts, + audioChannelCount: options.audioChannelCount || 1, + model: options.model || "default", + useEnhanced: options.useEnhanced ?? false, + ...(options.enableSpeakerDiarization && { + diarizationConfig: { + enableSpeakerDiarization: true, + minSpeakerCount: options.diarizationSpeakerCount || 2, + maxSpeakerCount: options.diarizationSpeakerCount || 6, + }, + }), + }, + }; + + // Call Google Speech-to-Text API + const [response] = await withTimeout( + this.client.recognize(request, { + timeout: GoogleSTTHandler.DEFAULT_API_TIMEOUT_MS, + }), + GoogleSTTHandler.DEFAULT_API_TIMEOUT_MS, + "google-ai", + "generate", + ); + + if (!response.results || response.results.length === 0) { + throw new STTError({ + code: STT_ERROR_CODES.TRANSCRIPTION_FAILED, + message: "Google Speech-to-Text returned no results", + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + }); + } + + // Aggregate all segments from response.results + const aggregatedTranscripts: string[] = []; + const aggregatedWords: WordInfo[] = []; + const aggregatedAlternatives: TranscriptAlternative[] = []; + let aggregatedConfidence = 0; + let confidenceCount = 0; + let detectedLanguageCode: string | undefined; + + // Iterate over all results to aggregate full response + for (const result of response.results) { + const primaryAlternative = result.alternatives?.[0]; + + if (!primaryAlternative?.transcript) { + continue; // Skip empty segments + } + + // Aggregate transcript text + aggregatedTranscripts.push(primaryAlternative.transcript); + + // Aggregate confidence (compute weighted average) + if ( + primaryAlternative.confidence !== null && + primaryAlternative.confidence !== undefined + ) { + aggregatedConfidence += primaryAlternative.confidence; + confidenceCount++; + } + + // Use language code from first result that has it + if (!detectedLanguageCode && result.languageCode) { + detectedLanguageCode = result.languageCode; + } + + // Aggregate word-level information + if (primaryAlternative.words) { + primaryAlternative.words.forEach( + (w: protos.google.cloud.speech.v1.IWordInfo) => { + aggregatedWords.push({ + word: w.word || "", + startTime: this.convertDurationToSeconds(w.startTime), + endTime: this.convertDurationToSeconds(w.endTime), + confidence: + w.confidence !== null && w.confidence !== undefined + ? w.confidence + : undefined, + speakerTag: + w.speakerTag !== null && w.speakerTag !== undefined + ? w.speakerTag + : undefined, + }); + }, + ); + } + + // Aggregate alternatives from each segment + if (result.alternatives && result.alternatives.length > 1) { + result.alternatives + .slice(1) + .forEach( + ( + alt: protos.google.cloud.speech.v1.ISpeechRecognitionAlternative, + ) => { + aggregatedAlternatives.push({ + transcript: alt.transcript || "", + confidence: alt.confidence || 0, + words: alt.words?.map( + (w: protos.google.cloud.speech.v1.IWordInfo) => ({ + word: w.word || "", + startTime: this.convertDurationToSeconds(w.startTime), + endTime: this.convertDurationToSeconds(w.endTime), + confidence: + w.confidence !== null && w.confidence !== undefined + ? w.confidence + : undefined, + speakerTag: + w.speakerTag !== null && w.speakerTag !== undefined + ? w.speakerTag + : undefined, + }), + ), + }); + }, + ); + } + } + + // Validate that we have at least some transcript + if (aggregatedTranscripts.length === 0) { + throw new STTError({ + code: STT_ERROR_CODES.TRANSCRIPTION_FAILED, + message: "Google Speech-to-Text returned empty transcript", + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + }); + } + + const latency = Date.now() - startTime; + + // Combine all transcript segments with spaces + const fullTranscript = aggregatedTranscripts.join(" "); + + // Compute average confidence + const averageConfidence = + confidenceCount > 0 ? aggregatedConfidence / confidenceCount : 0; + + // Calculate duration from aggregated words if available + const duration = + aggregatedWords.length > 0 + ? aggregatedWords[aggregatedWords.length - 1].endTime + : undefined; + + logger.info( + `[GoogleSTTHandler] Transcribed ${audio.length} bytes in ${latency}ms (${response.results.length} segments)`, + ); + + return { + text: fullTranscript, + confidence: averageConfidence, + languageCode: detectedLanguageCode || options.languageCode, + words: aggregatedWords.length > 0 ? aggregatedWords : undefined, + alternatives: + aggregatedAlternatives.length > 0 + ? aggregatedAlternatives + : undefined, + duration, + metadata: { + latency, + provider: "google-ai", + model: options.model || "default", + billedSeconds: Math.ceil(duration || 0), + }, + }; + } catch (err) { + if (err instanceof STTError) { + throw err; + } + + const latency = Date.now() - startTime; + const message = err instanceof Error ? err.message : "Unknown error"; + throw new STTError({ + code: STT_ERROR_CODES.TRANSCRIPTION_FAILED, + message: `Google Speech-to-Text failed after ${latency}ms: ${message}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { latency }, + originalError: err instanceof Error ? err : undefined, + }); + } + } + + /** + * Convert protobuf Duration to seconds + * + * @param duration - Protobuf duration object with seconds and nanos + * @returns Total duration in seconds as a number + */ + private convertDurationToSeconds( + duration?: protos.google.protobuf.IDuration | null, + ): number { + if (!duration) { + return 0; + } + + const seconds = duration.seconds ?? null; + const nanos = duration.nanos ?? 0; + + // Handle seconds as string, number, or Long (from protobuf) + let secondsValue = 0; + if (typeof seconds === "string") { + secondsValue = parseFloat(seconds); + } else if (typeof seconds === "number") { + secondsValue = seconds; + } else if ( + seconds && + typeof seconds === "object" && + "toNumber" in seconds + ) { + // Handle Long type from protobuf + secondsValue = (seconds as { toNumber: () => number }).toNumber(); + } + + return secondsValue + nanos / 1e9; + } + + /** + * Map generic audio encoding to Google Cloud encoding enum + * Uses proto enum values directly to avoid fragile hardcoded integers + */ + private mapAudioEncoding( + encoding: AudioEncoding, + ): protos.google.cloud.speech.v1.RecognitionConfig.AudioEncoding { + const AudioEncodingEnum = + protos.google.cloud.speech.v1.RecognitionConfig.AudioEncoding; + + const encodingMap: Record< + AudioEncoding, + protos.google.cloud.speech.v1.RecognitionConfig.AudioEncoding + > = { + LINEAR16: AudioEncodingEnum.LINEAR16, + FLAC: AudioEncodingEnum.FLAC, + MULAW: AudioEncodingEnum.MULAW, + AMR: AudioEncodingEnum.AMR, + AMR_WB: AudioEncodingEnum.AMR_WB, + OGG_OPUS: AudioEncodingEnum.OGG_OPUS, + SPEEX_WITH_HEADER_BYTE: AudioEncodingEnum.SPEEX_WITH_HEADER_BYTE, + MP3: AudioEncodingEnum.MP3, + WEBM_OPUS: AudioEncodingEnum.WEBM_OPUS, + }; + + const googleEncoding = encodingMap[encoding]; + if (!googleEncoding) { + throw new STTError({ + code: STT_ERROR_CODES.INVALID_ENCODING, + message: `Unsupported audio encoding: ${encoding}`, + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { encoding }, + }); + } + + return googleEncoding; + } +} diff --git a/src/lib/constants/enums.ts b/src/lib/constants/enums.ts index 070f64fc3..8be3bc3fb 100644 --- a/src/lib/constants/enums.ts +++ b/src/lib/constants/enums.ts @@ -767,6 +767,7 @@ export enum ErrorCategory { CONFIGURATION = "configuration", EXECUTION = "execution", SYSTEM = "system", + STT = "stt", } // Error severity levels diff --git a/src/lib/core/baseProvider.ts b/src/lib/core/baseProvider.ts index 941cf3b06..3801e49c7 100644 --- a/src/lib/core/baseProvider.ts +++ b/src/lib/core/baseProvider.ts @@ -5,6 +5,10 @@ import { IMAGE_GENERATION_MODELS } from "../core/constants.js"; import type { EvaluationData } from "../index.js"; import { MiddlewareFactory } from "../middleware/factory.js"; import type { NeuroLink } from "../neurolink.js"; +import { + AUDIO_EXTENSIONS, + type AudioExtension, +} from "../processors/config/fileTypes.js"; import type { JsonValue, UnknownRecord } from "../types/common.js"; import type { AIProvider, @@ -30,6 +34,11 @@ import { import { shouldDisableBuiltinTools } from "../utils/toolUtils.js"; import { getKeyCount, getKeysAsString } from "../utils/transformationUtils.js"; import { TTSProcessor } from "../utils/ttsProcessor.js"; +import { STTProcessor } from "../utils/sttProcessor.js"; +import type { STTOptions, STTResult } from "../types/sttTypes.js"; +import type { FileWithMetadata } from "../types/fileTypes.js"; +import { withTimeout } from "../utils/timeout.js"; +import { ErrorFactory } from "../utils/errorHandling.js"; import { hasVideoFrames, executeVideoAnalysis, @@ -601,6 +610,247 @@ export abstract class BaseProvider implements AIProvider { this.generationHandler.analyzeAIResponse(result); } + /** + * Auto-detect audio files in the files array and enable STT mode. + * Keeps audio files in the generic files array for transcription. + * Also supports backward compatibility with audioFiles array. + * + * @param options - Generate options to modify + * @private + */ + protected async detectAndEnableSTT( + options: TextGenerationOptions, + ): Promise { + // Skip if STT is already explicitly enabled + if ((options as unknown as { stt?: STTOptions }).stt) { + return; + } + + // Check both files and audioFiles arrays + const filesArray = options.input?.files; + const audioFilesArray = ( + options.input as { audioFiles?: unknown[] } | undefined + )?.audioFiles; + + // If neither array has files, skip detection + if ( + (!filesArray || filesArray.length === 0) && + (!audioFilesArray || audioFilesArray.length === 0) + ) { + return; + } + + // Check if any files are audio files + let audioFileCount = 0; + + // Check files array + if (filesArray) { + for (const file of filesArray) { + let isAudio = false; + + if (typeof file === "string") { + // Check file extension + const ext = file.toLowerCase().split(".").pop(); + if (ext && AUDIO_EXTENSIONS.includes(`.${ext}` as AudioExtension)) { + isAudio = true; + } + } else if (typeof file === "object" && "filename" in file) { + // FileWithMetadata - check filename + const ext = file.filename.toLowerCase().split(".").pop(); + if (ext && AUDIO_EXTENSIONS.includes(`.${ext}` as AudioExtension)) { + isAudio = true; + } + } + // For plain buffers, we can't easily detect format, assume other files + // (MessageBuilder will handle buffer detection later) + + if (isAudio) { + audioFileCount++; + } + } + } + + // Check audioFiles array (backward compatibility) + if (audioFilesArray) { + audioFileCount += audioFilesArray.length; + } + + // If audio files detected, enable STT mode automatically + if (audioFileCount > 0) { + logger.info( + `[BaseProvider] Auto-detected ${audioFileCount} audio file(s), enabling STT mode`, + ); + + // Enable STT with default configuration if not already set + (options as unknown as { stt?: STTOptions }).stt = { + languageCode: "en-US", // Default language + enableAutomaticPunctuation: true, + }; + + // Ensure provider is set (defaults to google-ai for STT) + if (!options.provider) { + options.provider = "google-ai" as AIProviderName; + } + } + } + + /** + * Read audio file from path or buffer + * Internal helper for STT processing + * + * @param audioFile - Audio file as Buffer, string path, or FileWithMetadata + * @returns Audio buffer + * @private + */ + protected async readAudioFile( + audioFile: Buffer | string | FileWithMetadata, + ): Promise { + if (Buffer.isBuffer(audioFile)) { + return audioFile; + } else if (typeof audioFile === "string") { + const fs = await import("fs/promises"); + const fileReadTimeoutMs = 30_000; + return await withTimeout( + fs.readFile(audioFile), + fileReadTimeoutMs, + this.providerName, + "generate", + ); + } else if (typeof audioFile === "object" && "buffer" in audioFile) { + // FileWithMetadata + return audioFile.buffer; + } else { + throw ErrorFactory.invalidParameters( + "audioFiles", + new Error("Invalid audio file format"), + audioFile, + ); + } + } + + /** + * Handle STT transcription + * Internal method called when STT options are provided to generate() + * + * @param options - Generate options with STT configuration + * @param startTime - Generation start timestamp + * @returns Enhanced result with transcription + * @private + */ + protected async handleSTTTranscription( + options: TextGenerationOptions, + startTime: number, + ): Promise { + const sttOptions = (options as unknown as { stt: STTOptions }).stt; + if (!sttOptions) { + throw ErrorFactory.invalidParameters( + "stt", + new Error("STT options are required"), + sttOptions, + ); + } + + const provider = options.provider || ("google-ai" as AIProviderName); + + // Get audio buffer from input + let audioBuffer: Buffer; + + // Check both input.files (new way) and input.audioFiles (backward compatibility) + const filesArray = options.input?.files as + | Array + | undefined; + const audioFilesArray = ( + options.input as { audioFiles?: unknown[] } | undefined + )?.audioFiles as Array | undefined; + + // Prefer audioFiles for explicit audio, fallback to files for auto-detection + const sourceArray = audioFilesArray || filesArray; + + if (sourceArray && sourceArray.length > 0) { + // Find the first audio file + let audioFile: Buffer | string | FileWithMetadata | undefined; + + for (const file of sourceArray) { + let isAudio = false; + + if (typeof file === "string") { + const ext = file.toLowerCase().split(".").pop(); + if (ext && AUDIO_EXTENSIONS.includes(`.${ext}` as AudioExtension)) { + isAudio = true; + } + } else if (typeof file === "object" && "filename" in file) { + const ext = file.filename.toLowerCase().split(".").pop(); + if (ext && AUDIO_EXTENSIONS.includes(`.${ext}` as AudioExtension)) { + isAudio = true; + } + } else if (Buffer.isBuffer(file)) { + // For buffers, assume it's audio if STT is enabled + isAudio = true; + } + + if (isAudio) { + audioFile = file; + break; + } + } + + if (!audioFile) { + throw ErrorFactory.invalidParameters( + "files", + new Error("No audio file found in files/audioFiles array for STT"), + sourceArray, + ); + } + + audioBuffer = await this.readAudioFile(audioFile); + } else { + throw ErrorFactory.invalidParameters( + "files", + new Error( + "Audio input is required for STT (provide files or audioFiles array)", + ), + ); + } + + // Call STT processor + const sttTimeoutMs = 60_000; + const sttResult: STTResult = await withTimeout( + STTProcessor.transcribe(audioBuffer, provider, sttOptions), + sttTimeoutMs, + provider, + "generate", + ); + + const responseTime = Date.now() - startTime; + + // Return as EnhancedGenerateResult + const result: EnhancedGenerateResult = { + content: sttResult.text, + provider, + model: sttResult.metadata.model, + usage: { input: 0, output: 0, total: 0 }, + responseTime, + transcription: sttResult, + }; + + return result; + } + + /** + * Get available STT models for this provider + * + * @returns Promise resolving to list of available model identifiers + * + * @example + * ```typescript + * const models = await provider.getSTTModels(); + * console.log(models); // ['default', 'phone_call', 'video', ...] + * ``` + */ + async getSTTModels(): Promise { + return STTProcessor.getModels(this.providerName); + } + /** * Text generation method - implements AIProvider interface * Tools are always available unless explicitly disabled @@ -609,15 +859,19 @@ export abstract class BaseProvider implements AIProvider { * 1. Direct synthesis (default): TTS synthesizes the input text without AI generation * 2. AI response synthesis: TTS synthesizes the AI-generated response after generation * + * Supports Speech-to-Text (STT) transcription when audio files are detected or STT is explicitly enabled. + * * When TTS is enabled with useAiResponse=false (default), the method returns early with * only the audio result, skipping AI generation entirely for optimal performance. * * When TTS is enabled with useAiResponse=true, the method performs full AI generation * and then synthesizes the AI response to audio. * + * When STT is enabled, the method transcribes the audio file and returns the transcription result. + * * @param optionsOrPrompt - Generation options or prompt string * @param _analysisSchema - Optional analysis schema (not used) - * @returns Enhanced result with optional audio field containing TTSResult + * @returns Enhanced result with optional audio field containing TTSResult or transcription field containing STTResult * * IMPLEMENTATION NOTE: Uses streamText() under the hood and accumulates results * for consistency and better performance @@ -630,6 +884,24 @@ export abstract class BaseProvider implements AIProvider { this.validateOptions(options); const startTime = Date.now(); + // Auto-detect audio files and enable STT mode + await this.detectAndEnableSTT(options); + + // Check if STT transcription is requested + const sttOptionsValue = (options as unknown as { stt?: STTOptions }).stt; + logger.debug("[BaseProvider.generate] STT check", { + hasSttOptions: !!sttOptionsValue, + sttOptions: sttOptionsValue, + provider: this.providerName, + }); + + if (sttOptionsValue) { + logger.info("[BaseProvider.generate] Routing to STT transcription", { + provider: this.providerName, + }); + return this.handleSTTTranscription(options, startTime); + } + try { // ===== VIDEO GENERATION MODE ===== // Generate video from image + prompt using Veo 3.1 diff --git a/src/lib/factories/providerRegistry.ts b/src/lib/factories/providerRegistry.ts index dd938e6be..c21cb3eda 100644 --- a/src/lib/factories/providerRegistry.ts +++ b/src/lib/factories/providerRegistry.ts @@ -30,282 +30,318 @@ export class ProviderRegistry { }; /** - * Register all providers with the factory + * Register core language model providers */ - static async registerAllProviders(): Promise { - if (this.registered) { - return; - } + private static async registerCoreProviders(): Promise { + // Register Google AI Studio Provider (our validated baseline) + ProviderFactory.registerProvider( + AIProviderName.GOOGLE_AI, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { GoogleAIStudioProvider } = await import( + "../providers/googleAiStudio.js" + ); + return new GoogleAIStudioProvider( + modelName, + sdk as NeuroLink | undefined, + ); + }, + GoogleAIModels.GEMINI_2_5_FLASH, + ["googleAiStudio", "google", "gemini", "google-ai", "google-ai-studio"], + ); - try { - // Register providers with dynamic import factory functions - const { ProviderFactory } = await import("./providerFactory.js"); + // Register OpenAI provider + ProviderFactory.registerProvider( + AIProviderName.OPENAI, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { OpenAIProvider } = await import("../providers/openAI.js"); + return new OpenAIProvider(modelName, sdk as NeuroLink | undefined); + }, + OpenAIModels.GPT_4O_MINI, + ["gpt", "chatgpt"], + ); - // Register Google AI Studio Provider (our validated baseline) - ProviderFactory.registerProvider( - AIProviderName.GOOGLE_AI, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { GoogleAIStudioProvider } = await import( - "../providers/googleAiStudio.js" - ); - return new GoogleAIStudioProvider( - modelName, - sdk as NeuroLink | undefined, - ); - }, - GoogleAIModels.GEMINI_2_5_FLASH, - ["googleAiStudio", "google", "gemini", "google-ai", "google-ai-studio"], - ); + // Register Anthropic provider + ProviderFactory.registerProvider( + AIProviderName.ANTHROPIC, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { AnthropicProvider } = await import("../providers/anthropic.js"); + return new AnthropicProvider(modelName, sdk as NeuroLink | undefined); + }, + AnthropicModels.CLAUDE_SONNET_4_0, + ["claude", "anthropic"], + ); - // Register OpenAI provider - ProviderFactory.registerProvider( - AIProviderName.OPENAI, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { OpenAIProvider } = await import("../providers/openAI.js"); - return new OpenAIProvider(modelName, sdk as NeuroLink | undefined); - }, - OpenAIModels.GPT_4O_MINI, - ["gpt", "chatgpt"], - ); + // Register Amazon Bedrock provider + ProviderFactory.registerProvider( + AIProviderName.BEDROCK, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + region?: string, + ) => { + const { AmazonBedrockProvider } = await import( + "../providers/amazonBedrock.js" + ); + return new AmazonBedrockProvider( + modelName, + sdk as NeuroLink | undefined, + region, + ); + }, + undefined, // Let provider read BEDROCK_MODEL from .env + ["bedrock", "aws"], + ); - // Register Anthropic provider - ProviderFactory.registerProvider( - AIProviderName.ANTHROPIC, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { AnthropicProvider } = await import( - "../providers/anthropic.js" - ); - return new AnthropicProvider(modelName, sdk as NeuroLink | undefined); - }, - AnthropicModels.CLAUDE_SONNET_4_0, - ["claude", "anthropic"], - ); + // Register Azure OpenAI provider + ProviderFactory.registerProvider( + AIProviderName.AZURE, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { AzureOpenAIProvider } = await import( + "../providers/azureOpenai.js" + ); + return new AzureOpenAIProvider(modelName, sdk as NeuroLink | undefined); + }, + process.env.AZURE_MODEL || + process.env.AZURE_OPENAI_MODEL || + process.env.AZURE_OPENAI_DEPLOYMENT || + process.env.AZURE_OPENAI_DEPLOYMENT_ID || + "gpt-4o-mini", + ["azure", "azureOpenai"], + ); - // Register Amazon Bedrock provider - ProviderFactory.registerProvider( - AIProviderName.BEDROCK, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - region?: string, - ) => { - const { AmazonBedrockProvider } = await import( - "../providers/amazonBedrock.js" - ); - return new AmazonBedrockProvider( - modelName, - sdk as NeuroLink | undefined, - region, - ); - }, - undefined, // Let provider read BEDROCK_MODEL from .env - ["bedrock", "aws"], - ); + // Register Google Vertex AI provider + ProviderFactory.registerProvider( + AIProviderName.VERTEX, + async ( + modelName?: string, + providerName?: string, + sdk?: UnknownRecord, + region?: string, + ) => { + const { GoogleVertexProvider } = await import( + "../providers/googleVertex.js" + ); + return new GoogleVertexProvider( + modelName, + providerName, + sdk as NeuroLink | undefined, + region, + ); + }, + VertexModels.CLAUDE_4_0_SONNET, + ["vertex", "googleVertex"], + ); + } - // Register Azure OpenAI provider - ProviderFactory.registerProvider( - AIProviderName.AZURE, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { AzureOpenAIProvider } = await import( - "../providers/azureOpenai.js" - ); - return new AzureOpenAIProvider( - modelName, - sdk as NeuroLink | undefined, - ); - }, - process.env.AZURE_MODEL || - process.env.AZURE_OPENAI_MODEL || - process.env.AZURE_OPENAI_DEPLOYMENT || - process.env.AZURE_OPENAI_DEPLOYMENT_ID || - "gpt-4o-mini", - ["azure", "azureOpenai"], - ); + /** + * Register specialized and open-source providers + */ + private static async registerSpecializedProviders(): Promise { + // Register Hugging Face provider (Unified Router implementation) + ProviderFactory.registerProvider( + AIProviderName.HUGGINGFACE, + async (modelName?: string) => { + const { HuggingFaceProvider } = await import( + "../providers/huggingFace.js" + ); + return new HuggingFaceProvider(modelName); + }, + process.env.HUGGINGFACE_MODEL || HuggingFaceModels.QWEN_2_5_72B_INSTRUCT, + ["huggingface", "hf"], + ); - // Register Google Vertex AI provider - ProviderFactory.registerProvider( - AIProviderName.VERTEX, - async ( - modelName?: string, - providerName?: string, - sdk?: UnknownRecord, - region?: string, - ) => { - const { GoogleVertexProvider } = await import( - "../providers/googleVertex.js" - ); - return new GoogleVertexProvider( - modelName, - providerName, - sdk as NeuroLink | undefined, - region, - ); - }, - VertexModels.CLAUDE_4_0_SONNET, - ["vertex", "googleVertex"], - ); + // Register Mistral AI provider + ProviderFactory.registerProvider( + AIProviderName.MISTRAL, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { MistralProvider } = await import("../providers/mistral.js"); + return new MistralProvider( + modelName, + sdk as MistralProviderType | undefined, + ); + }, + MistralModels.MISTRAL_LARGE_LATEST, + ["mistral"], + ); - // Register Hugging Face provider (Unified Router implementation) - ProviderFactory.registerProvider( - AIProviderName.HUGGINGFACE, - async (modelName?: string) => { - const { HuggingFaceProvider } = await import( - "../providers/huggingFace.js" - ); - return new HuggingFaceProvider(modelName); - }, - process.env.HUGGINGFACE_MODEL || - HuggingFaceModels.QWEN_2_5_72B_INSTRUCT, - ["huggingface", "hf"], - ); + // Register Ollama provider + ProviderFactory.registerProvider( + AIProviderName.OLLAMA, + async (modelName?: string) => { + const { OllamaProvider } = await import("../providers/ollama.js"); + return new OllamaProvider(modelName); + }, + process.env.OLLAMA_MODEL || OllamaModels.LLAMA3_2_LATEST, + ["ollama", "local"], + ); - // Register Mistral AI provider - ProviderFactory.registerProvider( - AIProviderName.MISTRAL, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { MistralProvider } = await import("../providers/mistral.js"); - return new MistralProvider( - modelName, - sdk as MistralProviderType | undefined, - ); - }, - MistralModels.MISTRAL_LARGE_LATEST, - ["mistral"], - ); + // Register LiteLLM provider + ProviderFactory.registerProvider( + AIProviderName.LITELLM, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { LiteLLMProvider } = await import("../providers/litellm.js"); + return new LiteLLMProvider(modelName, sdk as NeuroLink | undefined); + }, + process.env.LITELLM_MODEL || LiteLLMModels.OPENAI_GPT_4O_MINI, + ["litellm"], + ); - // Register Ollama provider - ProviderFactory.registerProvider( - AIProviderName.OLLAMA, - async (modelName?: string) => { - const { OllamaProvider } = await import("../providers/ollama.js"); - return new OllamaProvider(modelName); - }, - process.env.OLLAMA_MODEL || OllamaModels.LLAMA3_2_LATEST, - ["ollama", "local"], - ); + // Register OpenAI Compatible provider + ProviderFactory.registerProvider( + AIProviderName.OPENAI_COMPATIBLE, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { OpenAICompatibleProvider } = await import( + "../providers/openaiCompatible.js" + ); + return new OpenAICompatibleProvider( + modelName, + sdk as NeuroLink | undefined, + ); + }, + process.env.OPENAI_COMPATIBLE_MODEL || undefined, // Enable auto-discovery when no model specified + ["openai-compatible", "vllm", "compatible"], + ); - // Register LiteLLM provider - ProviderFactory.registerProvider( - AIProviderName.LITELLM, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { LiteLLMProvider } = await import("../providers/litellm.js"); - return new LiteLLMProvider(modelName, sdk as NeuroLink | undefined); - }, - process.env.LITELLM_MODEL || LiteLLMModels.OPENAI_GPT_4O_MINI, - ["litellm"], + // Register OpenRouter provider (300+ models from 60+ providers) + ProviderFactory.registerProvider( + AIProviderName.OPENROUTER, + async ( + modelName?: string, + _providerName?: string, + sdk?: UnknownRecord, + ) => { + const { OpenRouterProvider } = await import( + "../providers/openRouter.js" + ); + return new OpenRouterProvider(modelName, sdk as NeuroLink | undefined); + }, + process.env.OPENROUTER_MODEL || "anthropic/claude-3-5-sonnet", + ["openrouter", "or"], + ); + + // Register Amazon SageMaker provider + ProviderFactory.registerProvider( + AIProviderName.SAGEMAKER, + async ( + modelName?: string, + _providerName?: string, + _sdk?: UnknownRecord, + region?: string, + ) => { + const { AmazonSageMakerProvider } = await import( + "../providers/amazonSagemaker.js" + ); + return new AmazonSageMakerProvider(modelName, undefined, region); + }, + process.env.SAGEMAKER_MODEL || "sagemaker-model", + ["sagemaker", "aws-sagemaker"], + ); + } + + /** + * Register TTS and STT handlers + */ + private static async registerAudioHandlers(): Promise { + // ===== TTS HANDLER REGISTRATION ===== + try { + // Create handler instance and register explicitly + const { GoogleTTSHandler } = await import( + "../adapters/tts/googleTTSHandler.js" ); + const { TTSProcessor } = await import("../utils/ttsProcessor.js"); + + const googleHandler = new GoogleTTSHandler(); + TTSProcessor.registerHandler("google-ai", googleHandler); + TTSProcessor.registerHandler("vertex", googleHandler); - // Register OpenAI Compatible provider - ProviderFactory.registerProvider( - AIProviderName.OPENAI_COMPATIBLE, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { OpenAICompatibleProvider } = await import( - "../providers/openaiCompatible.js" - ); - return new OpenAICompatibleProvider( - modelName, - sdk as NeuroLink | undefined, - ); + logger.debug("TTS handlers registered successfully", { + providers: ["google-ai", "vertex"], + }); + } catch (ttsError) { + logger.warn( + "Failed to register TTS handlers - TTS functionality will be unavailable", + { + error: + ttsError instanceof Error ? ttsError.message : String(ttsError), }, - process.env.OPENAI_COMPATIBLE_MODEL || undefined, // Enable auto-discovery when no model specified - ["openai-compatible", "vllm", "compatible"], ); + // Don't throw - TTS is optional functionality + } - // Register OpenRouter provider (300+ models from 60+ providers) - ProviderFactory.registerProvider( - AIProviderName.OPENROUTER, - async ( - modelName?: string, - _providerName?: string, - sdk?: UnknownRecord, - ) => { - const { OpenRouterProvider } = await import( - "../providers/openRouter.js" - ); - return new OpenRouterProvider( - modelName, - sdk as NeuroLink | undefined, - ); - }, - process.env.OPENROUTER_MODEL || "anthropic/claude-3-5-sonnet", - ["openrouter", "or"], + // ===== STT HANDLER REGISTRATION ===== + try { + // Create handler instance and register explicitly + const { GoogleSTTHandler } = await import( + "../adapters/stt/googleSTTHandler.js" ); + const { STTProcessor } = await import("../utils/sttProcessor.js"); + + const googleSTTHandler = new GoogleSTTHandler(); + STTProcessor.registerHandler("google-ai", googleSTTHandler); + STTProcessor.registerHandler("vertex", googleSTTHandler); - // Register Amazon SageMaker provider - ProviderFactory.registerProvider( - AIProviderName.SAGEMAKER, - async ( - modelName?: string, - _providerName?: string, - _sdk?: UnknownRecord, - region?: string, - ) => { - const { AmazonSageMakerProvider } = await import( - "../providers/amazonSagemaker.js" - ); - return new AmazonSageMakerProvider(modelName, undefined, region); + logger.debug("STT handlers registered successfully", { + providers: ["google-ai", "vertex"], + }); + } catch (sttError) { + logger.warn( + "Failed to register STT handlers - STT functionality will be unavailable", + { + error: + sttError instanceof Error ? sttError.message : String(sttError), }, - process.env.SAGEMAKER_MODEL || "sagemaker-model", - ["sagemaker", "aws-sagemaker"], ); + // Don't throw - STT is optional functionality + } + } - logger.debug("All providers registered successfully"); - this.registered = true; + /** + * Register all providers with the factory + */ + static async registerAllProviders(): Promise { + if (this.registered) { + return; + } - // ===== TTS HANDLER REGISTRATION ===== - try { - // Create handler instance and register explicitly - const { GoogleTTSHandler } = await import( - "../adapters/tts/googleTTSHandler.js" - ); - const { TTSProcessor } = await import("../utils/ttsProcessor.js"); + try { + // Register providers with dynamic import factory functions + await import("./providerFactory.js"); - const googleHandler = new GoogleTTSHandler(); - TTSProcessor.registerHandler("google-ai", googleHandler); - TTSProcessor.registerHandler("vertex", googleHandler); + await this.registerCoreProviders(); + await this.registerSpecializedProviders(); + await this.registerAudioHandlers(); - logger.debug("TTS handlers registered successfully", { - providers: ["google-ai", "vertex"], - }); - } catch (ttsError) { - logger.warn( - "Failed to register TTS handlers - TTS functionality will be unavailable", - { - error: - ttsError instanceof Error ? ttsError.message : String(ttsError), - }, - ); - // Don't throw - TTS is optional functionality - } + logger.debug("All providers registered successfully"); + this.registered = true; } catch (error) { logger.error("Failed to register providers:", error); throw error; diff --git a/src/lib/index.ts b/src/lib/index.ts index 324ef55c5..c68558d25 100644 --- a/src/lib/index.ts +++ b/src/lib/index.ts @@ -135,6 +135,18 @@ export type { NeuroLinkMiddleware, } from "./types/middlewareTypes.js"; +// STT & TTS Processors +export { + STTProcessor, + STTError, + STT_ERROR_CODES, +} from "./utils/sttProcessor.js"; +export { + TTSProcessor, + TTSError, + TTS_ERROR_CODES, +} from "./utils/ttsProcessor.js"; + // Version export const VERSION = "1.0.0"; diff --git a/src/lib/neurolink.ts b/src/lib/neurolink.ts index a8979dbb1..19b981383 100644 --- a/src/lib/neurolink.ts +++ b/src/lib/neurolink.ts @@ -487,6 +487,7 @@ export class NeuroLink { constructorStartTime, constructorHrTimeStart, ); + this.initializeConversationMemory( config, constructorId, @@ -1836,6 +1837,34 @@ Current user's request: ${currentInput}`; } } + /** + * Get available STT models for a provider + * + * @param providerName - Provider name (e.g., 'google-ai', 'vertex'). Defaults to primary provider. + * @returns Promise resolving to list of available model identifiers + * + * @example + * ```typescript + * const models = await neurolink.getSTTModels('google-ai'); + * console.log(models); // ['default', 'phone_call', 'video', ...] + * ``` + * + * @since 1.0.0 + */ + async getSTTModels(providerName?: string): Promise { + // Initialize provider registry if needed (but not full MCP) + if (!this.mcpInitialized) { + await this.initializeProviderRegistryInternal(); + } + + // Use provided provider name or default to google-ai for STT + const provider = providerName || "google-ai"; + + // Use STTProcessor directly to get models + const { STTProcessor } = await import("./utils/sttProcessor.js"); + return STTProcessor.getModels(provider); + } + /** * Generate AI response with comprehensive feature support. * @@ -2109,6 +2138,7 @@ Current user's request: ${currentInput}`; input: options.input, // This includes text, images, and content arrays region: options.region, tts: options.tts, + stt: options.stt, // Pass through STT options fileRegistry: this.fileRegistry, abortSignal: options.abortSignal, skipToolPromptInjection: options.skipToolPromptInjection, @@ -2213,6 +2243,7 @@ Current user's request: ${currentInput}`; } : undefined, audio: textResult.audio, + transcription: textResult.transcription, // CRITICAL FIX: Include STT transcription result video: textResult.video, ppt: textResult.ppt, }; @@ -3204,7 +3235,7 @@ Current user's request: ${currentInput}`; // Return enhanced result with preserved tool information return { - content: result.content || "", // Ensure content is never undefined + content: result.content || "", provider: providerName, model: result.model, usage: result.usage, @@ -3217,8 +3248,10 @@ Current user's request: ${currentInput}`; transformToolsToExpectedFormat(availableTools), ), audio: result.audio, + transcription: result.transcription, video: result.video, ppt: result.ppt, + imageOutput: result.imageOutput, // Include analytics and evaluation from BaseProvider analytics: result.analytics, evaluation: result.evaluation, @@ -3402,6 +3435,7 @@ Current user's request: ${currentInput}`; analytics: result.analytics, evaluation: result.evaluation, audio: result.audio, + transcription: result.transcription, // CRITICAL: Pass through STT result video: result.video, ppt: result.ppt, // CRITICAL FIX: Include imageOutput for image generation models @@ -4644,6 +4678,11 @@ Current user's request: ${currentInput}`; * console.log(`Reason: ${event.reason || 'Unknown'}`); * }); * + * emitter.on('externalMCP:serverFailed', (event) => { + * console.log(`External MCP server failed: ${event.serverId}`); + * console.log(`Reason: ${event.error || 'Unknown'}`); + * }); + * * emitter.on('externalMCP:toolDiscovered', (event) => { * console.log(`New tool discovered: ${event.toolName} from ${event.serverId}`); * }); @@ -5865,7 +5904,6 @@ Current user's request: ${currentInput}`; // ============================================================================ // PROVIDER DIAGNOSTICS - SDK-First Architecture // ============================================================================ - /** * Get comprehensive status of all AI providers * Primary method for provider health checking and diagnostics @@ -6120,7 +6158,6 @@ Current user's request: ${currentInput}`; // ============================================================================ // MCP DIAGNOSTICS - SDK-First Architecture // ============================================================================ - /** * Get comprehensive MCP (Model Context Protocol) status information * @returns Promise resolving to MCP status details @@ -6640,7 +6677,6 @@ Current user's request: ${currentInput}`; // ============================================================================ // CONVERSATION MEMORY PUBLIC API // ============================================================================ - /** * Initialize conversation memory if enabled (public method for explicit initialization) * This is useful for testing or when you want to ensure conversation memory is ready diff --git a/src/lib/processors/config/fileTypes.ts b/src/lib/processors/config/fileTypes.ts index 65f496eb1..7161c672b 100644 --- a/src/lib/processors/config/fileTypes.ts +++ b/src/lib/processors/config/fileTypes.ts @@ -431,6 +431,7 @@ export const AUDIO_EXTENSIONS = [ ".m4a", ".wma", ".opus", + ".webm", ] as const; // ============================================================================= diff --git a/src/lib/providers/googleVertex.ts b/src/lib/providers/googleVertex.ts index 522e79df8..9ad3c99a2 100644 --- a/src/lib/providers/googleVertex.ts +++ b/src/lib/providers/googleVertex.ts @@ -2907,7 +2907,9 @@ export class GoogleVertexProvider extends BaseProvider { */ private buildImageGenerationParts( prompt: string, - pdfFiles: Array, + pdfFiles: Array< + Buffer | string | import("../types/fileTypes.js").FileWithMetadata + >, inputImages: Array, ): Array<{ text?: string; @@ -2928,6 +2930,9 @@ export class GoogleVertexProvider extends BaseProvider { if (Buffer.isBuffer(pdfFile)) { pdfBase64 = pdfFile.toString("base64"); + } else if (typeof pdfFile === "object" && "buffer" in pdfFile) { + // Handle FileWithMetadata + pdfBase64 = pdfFile.buffer.toString("base64"); } else if (typeof pdfFile === "string") { const isFilePath = pdfFile.startsWith("/") || diff --git a/src/lib/types/generateTypes.ts b/src/lib/types/generateTypes.ts index bce2c4a20..76a66f209 100644 --- a/src/lib/types/generateTypes.ts +++ b/src/lib/types/generateTypes.ts @@ -12,6 +12,7 @@ import type { VideoOutputOptions, } from "./multimodal.js"; import type { PPTGenerationResult, PPTOutputOptions } from "./pptTypes.js"; +import type { STTOptions, STTResult } from "./sttTypes.js"; import type { TTSOptions, TTSResult } from "./ttsTypes.js"; import type { StandardRecord, @@ -44,9 +45,18 @@ export type GenerateOptions = { * ``` */ images?: Array; - csvFiles?: Array; // Explicit CSV files - pdfFiles?: Array; // Explicit PDF files - videoFiles?: Array; // Explicit video files + audioFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Audio files for transcription + csvFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Explicit CSV files + pdfFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Explicit PDF files + videoFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Explicit video files files?: Array; // Auto-detect file types content?: Content[]; // Advanced multimodal content }; @@ -108,6 +118,33 @@ export type GenerateOptions = { transcribeAudio?: boolean; // Extract and transcribe audio (default: false) }; + /** + * Speech-to-Text (STT) configuration + * + * Enable transcription of audio files provided in input.files. + * Audio files are auto-detected by extension (.mp3, .wav, .ogg, etc.). + * The transcription result will be returned in the result's `transcription` field. + * + * @example Basic STT with auto-detection + * ```typescript + * const result = await neurolink.generate({ + * input: { text: "Transcribe this audio", files: ["audio.mp3"] }, + * provider: "google-ai" + * }); + * console.log(result.transcription?.text); // Transcribed text + * ``` + * + * @example Explicit STT options + * ```typescript + * const result = await neurolink.generate({ + * input: { text: "Transcribe", files: [audioBuffer] }, + * provider: "google-ai", + * stt: { languageCode: "en-US", enableAutomaticPunctuation: true } + * }); + * ``` + */ + stt?: STTOptions; + /** * Text-to-Speech (TTS) configuration * @@ -360,6 +397,32 @@ export type GenerateResult = { content: string; // Primary output outputs?: { text: string }; // Future extensible for multi-modal + /** + * Speech-to-Text transcription result + * + * Contains the transcribed text and metadata when STT is enabled and audio files are provided in input.files. + * Generated by STTProcessor.transcribeAudio() using the specified provider. + * + * @example Accessing STT transcription + * ```typescript + * const result = await neurolink.generate({ + * input: { text: "Transcribe this", files: ["audio.mp3"] }, + * provider: "google-ai" + * }); + * + * if (result.transcription) { + * console.log(`Transcribed: ${result.transcription.text}`); + * if (result.transcription.confidence) { + * console.log(`Confidence: ${result.transcription.confidence}`); + * } + * if (result.transcription.language) { + * console.log(`Detected language: ${result.transcription.language}`); + * } + * } + * ``` + */ + transcription?: STTResult; + /** * Text-to-Speech audio result * @@ -575,7 +638,12 @@ export type TextGenerationOptions = { * For video generation, the first image is used as the source frame. */ images?: Array; - pdfFiles?: Array; // Support for PDF inputs (for image generation with Vertex AI) + audioFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Audio files for transcription + pdfFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Support for PDF inputs (for image generation with Vertex AI) files?: Array; // Auto-detect file types (including video for analysis) }; provider?: AIProviderName; @@ -658,6 +726,18 @@ export type TextGenerationOptions = { */ tts?: TTSOptions; + /** + * Speech-to-Text (STT) configuration + * + * Enable transcription of audio files provided in input.files. + * Audio files are auto-detected by extension (.mp3, .wav, .ogg, etc.). + * The transcription result will be returned in the result's `transcription` field. + * + * This is only used when STT is explicitly enabled in GenerateOptions and passed through + * to TextGenerationOptions during the conversion process. + */ + stt?: STTOptions; + // NEW: Analytics and Evaluation Support enableEvaluation?: boolean; // Default: false - AI quality scoring enableAnalytics?: boolean; // Default: false - Usage tracking @@ -830,6 +910,8 @@ export type TextGenerationResult = { analytics?: AnalyticsData; evaluation?: EvaluationData; audio?: TTSResult; + /** Speech-to-Text transcription result */ + transcription?: STTResult; /** Video generation result */ video?: VideoGenerationResult; /** PowerPoint generation result */ diff --git a/src/lib/types/index.ts b/src/lib/types/index.ts index 6429afa5b..b1313ad96 100644 --- a/src/lib/types/index.ts +++ b/src/lib/types/index.ts @@ -230,6 +230,9 @@ export * from "./ttsTypes.js"; // Utilities Types - Utility module types (selective export to avoid conflicts) export * from "./utilities.js"; +// STT (Speech-to-Text) types +export * from "./sttTypes.js"; + // Workflow types export * from "./workflowTypes.js"; diff --git a/src/lib/types/streamTypes.ts b/src/lib/types/streamTypes.ts index 81bc9dd93..36f831e0e 100644 --- a/src/lib/types/streamTypes.ts +++ b/src/lib/types/streamTypes.ts @@ -234,9 +234,15 @@ export type StreamOptions = { * ``` */ images?: Array; - csvFiles?: Array; // Explicit CSV files (converted to text) - pdfFiles?: Array; // Explicit PDF files (processed as binary documents, not converted to text) - videoFiles?: Array; // Explicit video files + csvFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Explicit CSV files (converted to text) + pdfFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Explicit PDF files (processed as binary documents, not converted to text) + videoFiles?: Array< + Buffer | string | import("./fileTypes.js").FileWithMetadata + >; // Explicit video files files?: Array; // Auto-detect file types content?: Content[]; // Advanced multimodal content }; diff --git a/src/lib/types/sttTypes.ts b/src/lib/types/sttTypes.ts new file mode 100644 index 000000000..a39a62c44 --- /dev/null +++ b/src/lib/types/sttTypes.ts @@ -0,0 +1,315 @@ +/** + * Speech-to-Text (STT) Types + * + * Type definitions for audio transcription using Google Cloud Speech-to-Text + * + * @module types/sttTypes + */ + +/** + * Audio encoding formats supported by Google Speech-to-Text + */ +export type AudioEncoding = + | "LINEAR16" // Uncompressed 16-bit signed little-endian samples (WAV) + | "FLAC" // Free Lossless Audio Codec + | "MULAW" // 8-bit µ-law encoding + | "AMR" // Adaptive Multi-Rate Narrowband + | "AMR_WB" // Adaptive Multi-Rate Wideband + | "OGG_OPUS" // Opus encoded audio in Ogg container + | "SPEEX_WITH_HEADER_BYTE" + | "MP3" // MP3 audio + | "WEBM_OPUS"; // Opus in WebM container + +/** + * Valid STT models as an array for runtime validation + * Used to derive STTModel type for compile-time safety + */ +export const VALID_STT_MODELS = [ + "default", + "command_and_search", + "phone_call", + "video", + "medical_dictation", + "latest_long", + "latest_short", +] as const; + +/** + * STT model type derived from VALID_STT_MODELS + * Automatically stays in sync with the runtime validation array + */ +export type STTModel = (typeof VALID_STT_MODELS)[number]; + +/** + * Speech-to-Text configuration options + */ +export type STTOptions = { + /** Audio encoding format */ + encoding?: AudioEncoding; + /** Sample rate in Hz (8000, 16000, 48000, etc.) */ + sampleRateHertz?: number; + /** Language code (e.g., 'en-US', 'es-ES') - leave undefined for auto-detection */ + languageCode?: string; + /** Alternative language codes for multi-language audio */ + alternativeLanguageCodes?: string[]; + /** Maximum number of recognition alternatives to return */ + maxAlternatives?: number; + /** Enable profanity filtering */ + profanityFilter?: boolean; + /** Enable automatic punctuation */ + enableAutomaticPunctuation?: boolean; + /** Enable word-level timestamps */ + enableWordTimeOffsets?: boolean; + /** Enable word confidence scores */ + enableWordConfidence?: boolean; + /** Speech contexts (phrases/words to bias recognition) */ + speechContexts?: Array<{ + phrases: string[]; + boost?: number; + }>; + /** Audio channel count (1 for mono, 2 for stereo) */ + audioChannelCount?: number; + /** Enable speaker diarization (who spoke when) */ + enableSpeakerDiarization?: boolean; + /** Min/max number of speakers for diarization */ + diarizationSpeakerCount?: number; + /** Model to use for transcription */ + model?: STTModel; + /** Use enhanced models (higher cost, better accuracy) */ + useEnhanced?: boolean; +}; + +/** + * Word-level timing information + */ +export type WordInfo = { + /** Start time in seconds */ + startTime: number; + /** End time in seconds */ + endTime: number; + /** The word itself */ + word: string; + /** Confidence score 0.0-1.0 */ + confidence?: number; + /** Speaker tag (if diarization enabled) */ + speakerTag?: number; +}; + +/** + * Alternative transcription result + */ +export type TranscriptAlternative = { + /** Transcribed text */ + transcript: string; + /** Confidence score 0.0-1.0 */ + confidence: number; + /** Word-level details */ + words?: WordInfo[]; +}; + +/** + * Speech-to-Text transcription result + */ +export type STTResult = { + /** Primary transcription text */ + text: string; + /** Confidence score for primary result (0.0-1.0) */ + confidence: number; + /** Alternative transcriptions (if requested) */ + alternatives?: TranscriptAlternative[]; + /** Language detected/used */ + languageCode?: string; + /** Word-level timing and confidence */ + words?: WordInfo[]; + /** Audio duration in seconds */ + duration?: number; + /** Provider metadata */ + metadata: { + /** Processing latency in milliseconds */ + latency: number; + /** Provider name */ + provider: string; + /** Model used */ + model?: string; + /** Total billed time in seconds */ + billedSeconds?: number; + }; +}; + +/** + * Supported audio formats (file extensions) + */ +export const SUPPORTED_AUDIO_FORMATS = [ + "wav", + "flac", + "mp3", + "ogg", + "opus", + "webm", + "amr", +] as const; + +export type SupportedAudioFormat = (typeof SUPPORTED_AUDIO_FORMATS)[number]; + +/** + * Type guard to check if format is supported + */ +export function isSupportedAudioFormat( + format: string, +): format is SupportedAudioFormat { + const normalized = format.replace(/^\./, "").toLowerCase(); + return SUPPORTED_AUDIO_FORMATS.includes(normalized as SupportedAudioFormat); +} + +/** + * Valid audio encodings as an array for runtime validation + */ +export const VALID_AUDIO_ENCODINGS: readonly AudioEncoding[] = [ + "LINEAR16", + "FLAC", + "MULAW", + "AMR", + "AMR_WB", + "OGG_OPUS", + "SPEEX_WITH_HEADER_BYTE", + "MP3", + "WEBM_OPUS", +]; + +/** + * Type guard to check if an object is a valid STTOptions + */ +export function isSTTOptions(value: unknown): value is STTOptions { + if (!value || typeof value !== "object") { + return false; + } + + const opts = value as Record; + + // Check encoding if present + if (opts.encoding !== undefined) { + if ( + typeof opts.encoding !== "string" || + !VALID_AUDIO_ENCODINGS.includes(opts.encoding as AudioEncoding) + ) { + return false; + } + } + + // Check sample rate if present + if (opts.sampleRateHertz !== undefined) { + if (typeof opts.sampleRateHertz !== "number" || opts.sampleRateHertz <= 0) { + return false; + } + } + + // Check language code if present + if (opts.languageCode !== undefined) { + if ( + typeof opts.languageCode !== "string" || + opts.languageCode.length === 0 + ) { + return false; + } + } + + // Check maxAlternatives if present + if (opts.maxAlternatives !== undefined) { + if ( + typeof opts.maxAlternatives !== "number" || + opts.maxAlternatives < 1 || + opts.maxAlternatives > 30 + ) { + return false; + } + } + + // Check model if present + if (opts.model !== undefined) { + const validModels: readonly string[] = VALID_STT_MODELS; + if (typeof opts.model !== "string" || !validModels.includes(opts.model)) { + return false; + } + } + + return true; +} + +/** + * Type guard to check if an object is a valid STTResult + */ +export function isSTTResult(value: unknown): value is STTResult { + if (!value || typeof value !== "object") { + return false; + } + + const result = value as Record; + + // Required fields + if (typeof result.text !== "string") { + return false; + } + + if ( + typeof result.confidence !== "number" || + result.confidence < 0 || + result.confidence > 1 + ) { + return false; + } + + // Metadata is required + if (!result.metadata || typeof result.metadata !== "object") { + return false; + } + + const metadata = result.metadata as Record; + if (typeof metadata.latency !== "number" || metadata.latency < 0) { + return false; + } + + if (typeof metadata.provider !== "string") { + return false; + } + + return true; +} + +/** + * Type guard to check if an object is a valid WordInfo + */ +export function isWordInfo(value: unknown): value is WordInfo { + if (!value || typeof value !== "object") { + return false; + } + + const word = value as Record; + + return ( + typeof word.startTime === "number" && + typeof word.endTime === "number" && + typeof word.word === "string" && + word.startTime >= 0 && + word.endTime >= word.startTime + ); +} + +/** + * Type guard to check if an object is a valid TranscriptAlternative + */ +export function isTranscriptAlternative( + value: unknown, +): value is TranscriptAlternative { + if (!value || typeof value !== "object") { + return false; + } + + const alt = value as Record; + + return ( + typeof alt.transcript === "string" && + typeof alt.confidence === "number" && + alt.confidence >= 0 && + alt.confidence <= 1 + ); +} diff --git a/src/lib/utils/messageBuilder.ts b/src/lib/utils/messageBuilder.ts index 1471347b8..76ea361e7 100644 --- a/src/lib/utils/messageBuilder.ts +++ b/src/lib/utils/messageBuilder.ts @@ -847,9 +847,17 @@ function appendDetectedFileResult( } logger.info(`[FileDetector] ✅ Video: ${filename}`); } else if (result.type === "audio") { + // Keep audio in files array for STT transcription (auto-detection will handle it) + logger.info( + `[FileDetector] ✅ Audio: ${filename} (kept in files for STT auto-detection)`, + ); + + // Also add metadata to text for context if (result.content) { options.input.text += `\n\n## Audio File: "${filename}"\n${result.content}\n`; } + + // Add cover art if present if (result.images && result.images.length > 0) { options.input.images = [ ...(options.input.images || []), @@ -857,7 +865,6 @@ function appendDetectedFileResult( ]; logger.info(`[FileDetector] Added audio cover art as image`); } - logger.info(`[FileDetector] ✅ Audio: ${filename}`); } else if (result.type === "archive") { if (result.content) { options.input.text += `\n\n## Archive File: "${filename}"\n${result.content}\n`; diff --git a/src/lib/utils/sttProcessor.ts b/src/lib/utils/sttProcessor.ts new file mode 100644 index 000000000..ad606c312 --- /dev/null +++ b/src/lib/utils/sttProcessor.ts @@ -0,0 +1,500 @@ +/** + * Speech-to-Text (STT) Processing Utility + * + * Central orchestrator for all STT operations across providers. + * Follows the same pattern as TTSProcessor. + * + * @module utils/sttProcessor + */ + +import { logger } from "./logger.js"; +import type { STTOptions, STTResult } from "../types/sttTypes.js"; +import { ErrorCategory, ErrorSeverity } from "../constants/enums.js"; +import { NeuroLinkError } from "./errorHandling.js"; + +/** + * STT-specific error codes + * + * Comprehensive error codes for all STT operations, following TTS pattern + */ +export const STT_ERROR_CODES = { + // Input validation errors + EMPTY_AUDIO: "STT_EMPTY_AUDIO", + AUDIO_TOO_LARGE: "STT_AUDIO_TOO_LARGE", + AUDIO_TOO_LONG: "STT_AUDIO_TOO_LONG", + INVALID_FORMAT: "STT_INVALID_FORMAT", + INVALID_ENCODING: "STT_INVALID_ENCODING", + INVALID_SAMPLE_RATE: "STT_INVALID_SAMPLE_RATE", + INVALID_OPTIONS: "STT_INVALID_OPTIONS", + + // Provider errors + PROVIDER_NOT_SUPPORTED: "STT_PROVIDER_NOT_SUPPORTED", + PROVIDER_NOT_CONFIGURED: "STT_PROVIDER_NOT_CONFIGURED", + PROVIDER_ERROR: "STT_PROVIDER_ERROR", + + // Transcription errors + TRANSCRIPTION_FAILED: "STT_TRANSCRIPTION_FAILED", + TRANSCRIPTION_TIMEOUT: "STT_TRANSCRIPTION_TIMEOUT", + NO_SPEECH_DETECTED: "STT_NO_SPEECH_DETECTED", + LANGUAGE_NOT_SUPPORTED: "STT_LANGUAGE_NOT_SUPPORTED", + MODEL_NOT_AVAILABLE: "STT_MODEL_NOT_AVAILABLE", + + // Network/API errors + NETWORK_ERROR: "STT_NETWORK_ERROR", + API_ERROR: "STT_API_ERROR", + RATE_LIMIT_ERROR: "STT_RATE_LIMIT_ERROR", + QUOTA_EXCEEDED: "STT_QUOTA_EXCEEDED", +} as const; + +export type STTErrorCode = + (typeof STT_ERROR_CODES)[keyof typeof STT_ERROR_CODES]; + +/** + * STT Error class for speech-to-text specific errors + * + * Extends NeuroLinkError with STT-specific defaults and error handling. + * Provides consistent error reporting across all STT operations. + * + * @example + * ```typescript + * throw new STTError({ + * code: STT_ERROR_CODES.AUDIO_TOO_LARGE, + * message: 'Audio file exceeds 10MB limit', + * severity: ErrorSeverity.MEDIUM, + * retriable: false, + * context: { sizeMB: 15, maxSizeMB: 10 } + * }); + * ``` + */ +export class STTError extends NeuroLinkError { + constructor(params: { + code: string; + message: string; + category?: ErrorCategory; + severity?: ErrorSeverity; + retriable?: boolean; + context?: Record; + originalError?: Error; + }) { + super({ + code: params.code, + message: params.message, + category: params.category ?? ErrorCategory.STT, + severity: params.severity ?? ErrorSeverity.MEDIUM, + retriable: params.retriable ?? false, + context: params.context, + originalError: params.originalError, + }); + this.name = "STTError"; + } +} + +/** + * STT Handler interface for provider-specific implementations + * + * Each provider (Google Cloud, AWS Transcribe, etc.) implements this interface + * to provide STT transcription capabilities using their respective APIs. + * + * **Timeout Handling:** + * Implementations MUST handle their own timeouts for the `transcribe()` method. + * Recommended timeout: 60 seconds. Implementations should use `withTimeout()` utility + * or provider-specific timeout mechanisms. + * + * **Error Handling:** + * Implementations should throw STTError for all failures, including timeouts. + * Use appropriate error codes from STT_ERROR_CODES. + * + * @example + * ```typescript + * class MySTTHandler implements STTHandler { + * async transcribe(audio: Buffer, options: STTOptions): Promise { + * // REQUIRED: Implement timeout handling + * return await withTimeout( + * this.actualTranscription(audio, options), + * 60000, // 60 second timeout + * 'STT transcription timed out' + * ); + * } + * + * isConfigured(): boolean { + * return !!process.env.MY_STT_API_KEY; + * } + * } + * ``` + */ +export interface STTHandler { + /** + * Transcribe audio to text using provider-specific STT API + * + * **IMPORTANT: Timeout Responsibility** + * Implementations MUST enforce their own timeouts (recommended: 60 seconds). + * Use the `withTimeout()` utility or provider-specific timeout mechanisms. + * + * @param audio - Audio buffer to transcribe (pre-validated, non-empty, within size limits) + * @param options - STT configuration options (language, model, encoding, etc.) + * @returns Transcription result with metadata + * @throws {STTError} On transcription failure, timeout, or configuration issues + */ + transcribe(audio: Buffer, options: STTOptions): Promise; + + /** + * Get available models for the provider + * + * @returns List of available model identifiers + */ + getModels?(): Promise; + + /** + * Validate that the provider is properly configured + * + * @returns True if provider can transcribe audio + */ + isConfigured(): boolean; + + /** + * Maximum audio file size in MB + * Different providers have different limits + * + * @default 10 if not specified + */ + readonly maxAudioSizeMB: number; + + /** + * Maximum audio duration in seconds + * Different providers have different limits + * + * @default 60 if not specified + */ + readonly maxDurationSeconds: number; +} + +/** + * STT Processor class for orchestrating speech-to-text operations + * + * Follows the same pattern as TTSProcessor. + * Provides a unified interface for STT transcription across multiple providers. + * + * @example + * ```typescript + * // Register a handler + * STTProcessor.registerHandler('google-ai', googleSTTHandler); + * + * // Check if provider is supported + * if (STTProcessor.supports('google-ai')) { + * // Provider is registered + * } + * + * // Transcribe audio + * const result = await STTProcessor.transcribe( + * audioBuffer, + * 'google-ai', + * { languageCode: 'en-US', enableWordTimeOffsets: true } + * ); + * ``` + */ +export class STTProcessor { + /** + * Handler registry mapping provider names to STT handlers + * Uses Map for O(1) lookups and better type safety + * + * @private + */ + private static readonly handlers = new Map(); + + /** + * Default maximum audio size in MB + * + * Providers can override this value by specifying the `maxAudioSizeMB` property + * in their respective `STTHandler` implementation. + * + * @private + */ + private static readonly DEFAULT_MAX_AUDIO_SIZE_MB = 10; + + /** + * Register an STT handler for a specific provider + * + * Allows providers to register their STT implementation at runtime. + * + * @param providerName - Provider identifier (e.g., 'google-ai', 'vertex') + * @param handler - STT handler implementation + * + * @example + * ```typescript + * const googleHandler: STTHandler = { + * transcribe: async (audio, options) => { ... }, + * isConfigured: () => true, + * maxAudioSizeMB: 10, + * maxDurationSeconds: 60 + * }; + * + * STTProcessor.registerHandler('google-ai', googleHandler); + * ``` + */ + static registerHandler(providerName: string, handler: STTHandler): void { + if (!providerName) { + throw new STTError({ + code: STT_ERROR_CODES.INVALID_OPTIONS, + message: "Provider name is required for STT handler registration", + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { + method: "STTProcessor.registerHandler", + }, + }); + } + + if (!handler) { + throw new STTError({ + code: STT_ERROR_CODES.INVALID_OPTIONS, + message: "Handler is required for STT handler registration", + category: ErrorCategory.VALIDATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { + method: "STTProcessor.registerHandler", + providerName, + }, + }); + } + + const normalizedName = providerName.toLowerCase(); + + if (this.handlers.has(normalizedName)) { + logger.warn( + `[STTProcessor] Overwriting existing handler for provider: ${normalizedName}`, + ); + } + + this.handlers.set(normalizedName, handler); + logger.info( + `[STTProcessor] Registered STT handler for provider: ${normalizedName}`, + ); + } + + /** + * Get a registered STT handler by provider name + * + * @private + * @param providerName - Provider identifier + * @returns Handler instance or undefined if not registered + */ + private static getHandler(providerName: string): STTHandler | undefined { + const normalizedName = providerName.toLowerCase(); + return this.handlers.get(normalizedName); + } + + /** + * Check if a provider is supported (has a registered STT handler) + * + * @param providerName - Provider identifier + * @returns True if handler is registered + * + * @example + * ```typescript + * if (STTProcessor.supports('google-ai')) { + * console.log('Google AI STT is supported'); + * } + * ``` + */ + static supports(providerName: string): boolean { + if (!providerName) { + logger.error( + "[STTProcessor] Provider name is required for supports check", + ); + return false; + } + + const normalizedName = providerName.toLowerCase(); + const isSupported = this.handlers.has(normalizedName); + + if (!isSupported) { + logger.debug(`[STTProcessor] Provider ${providerName} is not supported`); + } + + return isSupported; + } + + /** + * Get list of all registered providers + * + * @returns Array of registered provider names + */ + static getRegisteredProviders(): string[] { + return Array.from(this.handlers.keys()); + } + + /** + * Get available models for a specific provider + * + * @param providerName - Provider identifier + * @returns List of available model identifiers + * @throws {STTError} If provider not supported + */ + static async getModels(providerName: string): Promise { + const handler = this.getHandler(providerName); + + if (!handler) { + throw new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_SUPPORTED, + message: `STT provider "${providerName}" is not supported`, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider: providerName }, + }); + } + + if (!handler.getModels) { + logger.warn( + `[STTProcessor] Provider "${providerName}" does not implement getModels()`, + ); + return []; + } + + return handler.getModels(); + } + + /** + * Transcribe audio using specified provider + * + * Orchestrates the speech-to-text transcription process: + * 1. Validates audio buffer (not empty, within size limits) + * 2. Looks up the provider handler + * 3. Verifies provider configuration + * 4. Delegates transcription to the provider (timeout handled by provider) + * 5. Enriches result with metadata + * + * **Timeout Handling:** + * Timeouts are enforced by individual provider implementations (see STTHandler interface). + * Providers typically use 60-second timeouts via `withTimeout()` utility or + * provider-specific timeout mechanisms. + * + * @param audio - Audio buffer to transcribe (validated, non-empty, within size limits) + * @param provider - Provider identifier + * @param options - STT configuration options + * @returns Transcription result with text, confidence, and metadata + * @throws {STTError} If validation fails or provider not supported/configured + * + * @example + * ```typescript + * const result = await STTProcessor.transcribe( + * audioBuffer, + * 'google-ai', + * { + * languageCode: 'en-US', + * enableWordTimeOffsets: true, + * enableAutomaticPunctuation: true + * } + * ); + * + * console.log(`Transcription: ${result.text}`); + * console.log(`Confidence: ${result.confidence}`); + * ``` + */ + static async transcribe( + audio: Buffer, + provider: string, + options: STTOptions, + ): Promise { + // 1. Validate audio buffer + if (!audio || audio.length === 0) { + logger.error("[STTProcessor] Audio buffer is empty"); + throw new STTError({ + code: STT_ERROR_CODES.EMPTY_AUDIO, + message: "Audio buffer is required for transcription", + severity: ErrorSeverity.LOW, + retriable: false, + context: { provider }, + }); + } + + // 2. Get handler + const handler = this.getHandler(provider); + if (!handler) { + logger.error(`[STTProcessor] Provider "${provider}" is not registered`); + throw new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_SUPPORTED, + message: `STT provider "${provider}" is not supported. Available: ${Array.from(this.handlers.keys()).join(", ")}`, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider, available: Array.from(this.handlers.keys()) }, + }); + } + + // 3. Check configuration + if (!handler.isConfigured()) { + logger.error(`[STTProcessor] Provider "${provider}" is not configured`); + throw new STTError({ + code: STT_ERROR_CODES.PROVIDER_NOT_CONFIGURED, + message: `STT provider "${provider}" is not properly configured. Check API credentials.`, + category: ErrorCategory.CONFIGURATION, + severity: ErrorSeverity.HIGH, + retriable: false, + context: { provider }, + }); + } + + // 4. Validate audio size + const sizeMB = audio.length / (1024 * 1024); + if (sizeMB > handler.maxAudioSizeMB) { + logger.error( + `[STTProcessor] Audio size ${sizeMB.toFixed(1)}MB exceeds limit of ${handler.maxAudioSizeMB}MB`, + ); + throw new STTError({ + code: STT_ERROR_CODES.AUDIO_TOO_LARGE, + message: `Audio size ${sizeMB.toFixed(1)}MB exceeds provider limit of ${handler.maxAudioSizeMB}MB`, + severity: ErrorSeverity.MEDIUM, + retriable: false, + context: { sizeMB, maxSizeMB: handler.maxAudioSizeMB }, + }); + } + + try { + // 5. Transcribe + logger.info( + `[STTProcessor] Transcribing ${sizeMB.toFixed(2)}MB audio with provider: ${provider}`, + ); + + const result = await handler.transcribe(audio, options); + + // 6. Post-processing: add metadata + const enrichedResult: STTResult = { + ...result, + languageCode: result.languageCode ?? options.languageCode, + }; + + logger.info( + `[STTProcessor] Successfully transcribed audio (confidence: ${(result.confidence * 100).toFixed(1)}%)`, + ); + + return enrichedResult; + } catch (err: unknown) { + // 7. Comprehensive error handling + // Re-throw STTError as-is + if (err instanceof STTError) { + throw err; + } + + // Wrap other errors in STTError + const errorMessage = + err instanceof Error ? err.message : String(err || "Unknown error"); + logger.error( + `[STTProcessor] Transcription failed for provider "${provider}": ${errorMessage}`, + ); + throw new STTError({ + code: STT_ERROR_CODES.TRANSCRIPTION_FAILED, + message: `STT transcription failed for provider "${provider}": ${errorMessage}`, + category: ErrorCategory.EXECUTION, + severity: ErrorSeverity.HIGH, + retriable: true, + context: { + provider, + audioSizeMB: sizeMB, + options, + }, + originalError: err instanceof Error ? err : undefined, + }); + } + } +} diff --git a/test/unit/telemetry-config-metadata.test.ts b/test/unit/telemetry-config-metadata.test.ts index f68f96115..f77feec93 100644 --- a/test/unit/telemetry-config-metadata.test.ts +++ b/test/unit/telemetry-config-metadata.test.ts @@ -232,7 +232,7 @@ describe("TelemetryHandler - Custom Metadata Support", () => { model: "attempted-override", toolsEnabled: false as unknown as string, neurolink: false as unknown as string, - operationType: "attempted-override", + operationType: "attempted-override" as unknown as string, originalProvider: "attempted-override", }, }, @@ -267,7 +267,6 @@ describe("TelemetryHandler - Custom Metadata Support", () => { const disabledNeurolink: MockNeuroLink = { isTelemetryEnabled: () => false, }; - const disabledHandler = new TelemetryHandler( AIProviderName.ANTHROPIC, "claude-3",