Repository navigation
feat(tui): push-to-talk voice input #102
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from 3 commits
4677b34
9c03e46
f917149
24713e6
7904b53
02fd8e6
4c686ed
766c295
ebd0511
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,44 @@ | ||
| import { Schema } from "effect" | ||
| import { HttpApi, HttpApiEndpoint, HttpApiGroup, OpenApi } from "effect/unstable/httpapi" | ||
| import { Authorization } from "../middleware/authorization" | ||
| import { InstanceContextMiddleware } from "../middleware/instance-context" | ||
| import { WorkspaceRoutingMiddleware, WorkspaceRoutingQuery } from "../middleware/workspace-routing" | ||
| import { InvalidRequestError, UpstreamError } from "../errors" | ||
| import { described } from "./metadata" | ||
|
|
||
| export const TranscribeInput = Schema.Struct({ | ||
| audio: Schema.String.annotate({ description: "Base64-encoded audio data" }), | ||
| mime: Schema.optional(Schema.String), | ||
| language: Schema.optional(Schema.String), | ||
| }).annotate({ identifier: "VoiceTranscribeInput" }) | ||
|
|
||
| export const TranscribeResult = Schema.Struct({ | ||
| text: Schema.String, | ||
| }).annotate({ identifier: "VoiceTranscribeResult" }) | ||
|
|
||
| export const VoiceApi = HttpApi.make("voice").add( | ||
| HttpApiGroup.make("voice") | ||
| .add( | ||
| HttpApiEndpoint.post("transcribe", "/voice/transcribe", { | ||
| query: WorkspaceRoutingQuery, | ||
| payload: TranscribeInput, | ||
| success: described(TranscribeResult, "Transcribed text"), | ||
| error: [InvalidRequestError, UpstreamError], | ||
| }).annotateMerge( | ||
| OpenApi.annotations({ | ||
| identifier: "voice.transcribe", | ||
| summary: "Transcribe audio", | ||
| description: "Transcribe recorded audio to text using the configured speech-to-text provider.", | ||
| }), | ||
| ), | ||
| ) | ||
| .annotateMerge( | ||
| OpenApi.annotations({ | ||
| title: "voice", | ||
| description: "Voice transcription routes.", | ||
| }), | ||
| ) | ||
| .middleware(InstanceContextMiddleware) | ||
| .middleware(WorkspaceRoutingMiddleware) | ||
| .middleware(Authorization), | ||
| ) | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,35 @@ | ||
| import { VoiceTranscription } from "@/voice/transcription" | ||
| import { Effect } from "effect" | ||
| import { HttpApiBuilder } from "effect/unstable/httpapi" | ||
| import { InstanceHttpApi } from "../api" | ||
| import { InvalidRequestError, UpstreamError } from "../errors" | ||
| import { TranscribeInput } from "../groups/voice" | ||
|
|
||
| const transcribe = Effect.fn("VoiceHttpApi.transcribe")(function* (ctx: { | ||
| payload: typeof TranscribeInput.Type | ||
| }) { | ||
| const audio = decodeAudio(ctx.payload.audio) | ||
| if (!audio) return yield* new InvalidRequestError({ message: "audio must be base64-encoded", field: "audio" }) | ||
| return yield* VoiceTranscription.transcribe({ | ||
| audio, | ||
| mime: ctx.payload.mime ?? "audio/wav", | ||
| language: ctx.payload.language, | ||
| }).pipe( | ||
| Effect.mapError((error) => { | ||
| if (error instanceof VoiceTranscription.NoCredentialError) | ||
| return new InvalidRequestError({ message: error.message }) | ||
| return new UpstreamError({ message: error.message, service: "openai", status: error.status }) | ||
| }), | ||
| ) | ||
| }) | ||
|
|
||
| export const voiceHandlers = HttpApiBuilder.group(InstanceHttpApi, "voice", (handlers) => | ||
| Effect.sync(() => handlers.handle("transcribe", transcribe)), | ||
| ) | ||
|
|
||
| function decodeAudio(input: string) { | ||
| if (input.length === 0) return undefined | ||
| const decoded = Buffer.from(input, "base64") | ||
| if (decoded.length === 0) return undefined | ||
| return new Uint8Array(decoded) | ||
| } | ||
|
Comment on lines
+38
to
+46
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,86 @@ | ||
| import { Auth } from "@/auth" | ||
| import { Env } from "@/env" | ||
| import { Effect, Schema } from "effect" | ||
| import { HttpBody, HttpClient, HttpClientRequest } from "effect/unstable/http" | ||
|
|
||
| export class NoCredentialError extends Schema.TaggedErrorClass<NoCredentialError>()("VoiceNoCredentialError", { | ||
| message: Schema.String, | ||
| }) {} | ||
|
|
||
| export class TranscribeError extends Schema.TaggedErrorClass<TranscribeError>()("VoiceTranscribeError", { | ||
| message: Schema.String, | ||
| status: Schema.optional(Schema.Number), | ||
| }) {} | ||
|
|
||
| const Result = Schema.Struct({ text: Schema.String }) | ||
|
|
||
| const EXTENSIONS: Record<string, string> = { | ||
| "audio/wav": "wav", | ||
| "audio/x-wav": "wav", | ||
| "audio/mpeg": "mp3", | ||
| "audio/mp4": "m4a", | ||
| "audio/webm": "webm", | ||
| "audio/ogg": "ogg", | ||
| "audio/flac": "flac", | ||
| } | ||
|
|
||
| // Single speech-to-text entrypoint. Alternative transcription providers can be | ||
| // added by branching here on the resolved credential/provider instead of | ||
| // touching the HTTP surface or the TUI. | ||
| export const transcribe = Effect.fn("VoiceTranscription.transcribe")(function* (input: { | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
|
||
| audio: Uint8Array | ||
| mime: string | ||
| language?: string | ||
| }) { | ||
| const key = yield* resolveOpenaiKey() | ||
| if (!key) | ||
| return yield* new NoCredentialError({ | ||
| message: "Voice input needs an OpenAI credential. Run `opencode auth login` and add an OpenAI API key.", | ||
| }) | ||
| const form = new FormData() | ||
| form.append("model", "whisper-1") | ||
| if (input.language) form.append("language", input.language) | ||
| form.append( | ||
| "file", | ||
| new File([input.audio as BlobPart], `voice.${EXTENSIONS[input.mime] ?? "wav"}`, { type: input.mime }), | ||
| ) | ||
| const client = yield* HttpClient.HttpClient | ||
| const response = yield* client | ||
| .execute( | ||
| HttpClientRequest.post("https://api.openai.com/v1/audio/transcriptions", { | ||
| headers: { authorization: `Bearer ${key}` }, | ||
| body: HttpBody.formData(form), | ||
| }), | ||
| ) | ||
| .pipe(Effect.mapError((error) => new TranscribeError({ message: `Transcription request failed: ${error}` }))) | ||
| if (response.status < 200 || response.status >= 300) { | ||
|
coderabbitai[bot] marked this conversation as resolved.
|
||
| const body = yield* response.text.pipe(Effect.orElseSucceed(() => "")) | ||
| return yield* new TranscribeError({ | ||
| message: `Transcription failed: ${response.status} ${body}`.trim(), | ||
| status: response.status, | ||
| }) | ||
| } | ||
| const json = yield* response.json.pipe( | ||
| Effect.mapError(() => new TranscribeError({ message: "Transcription returned an unreadable response" })), | ||
| ) | ||
| const decoded = Schema.decodeUnknownOption(Result)(json) | ||
| if (decoded._tag === "None") | ||
| return yield* new TranscribeError({ message: "Transcription returned an unexpected response shape" }) | ||
| return decoded.value | ||
| }) | ||
|
|
||
| function resolveOpenaiKey() { | ||
| return Effect.gen(function* () { | ||
| const env = yield* Env.Service | ||
| const fromEnv = yield* env.get("OPENAI_API_KEY") | ||
| if (fromEnv) return fromEnv | ||
| const auth = yield* Auth.Service | ||
| const info = yield* auth.get("openai").pipe(Effect.orElseSucceed(() => undefined)) | ||
| if (info?.type === "api") return info.key | ||
| if (info?.type === "oauth") return info.access | ||
| if (info?.type === "wellknown") return info.token | ||
| return undefined | ||
| }) | ||
| } | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
|
||
|
|
||
| export * as VoiceTranscription from "./transcription" | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
|
||
Uh oh!
There was an error while loading. Please reload this page.