Skip to content
Closed
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions apps/desktop/src/settings/DesktopClientSettings.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,11 @@ const clientSettings: ClientSettings = {
sidebarV2Enabled: false,
sidebarV2ConfiguredByUser: false,
timestampFormat: "24-hour",
voiceTranscriptionEnabled: true,
voiceTranscriptionProvider: "local",
voiceTranscriptionBaseUrl: "http://127.0.0.1:8080/v1",
voiceTranscriptionModel: "whisper-1",
voiceTranscriptionApiKey: "",
wordWrap: true,
};

Expand Down
46 changes: 46 additions & 0 deletions apps/server/src/http.ts
Original file line number Diff line number Diff line change
Expand Up @@ -40,8 +40,10 @@ import {
} from "./auth/http.ts";
import * as ServerEnvironment from "./environment/ServerEnvironment.ts";
import { browserApiCorsAllowedHeaders, browserApiCorsAllowedMethods } from "./httpCors.ts";
import { forwardWhisperTranscription, MAX_TRANSCRIPTION_AUDIO_BYTES } from "./transcription.ts";

const OTLP_TRACES_PROXY_PATH = "/api/observability/v1/traces";
const TRANSCRIPTION_PATH = "/api/transcription";
const LOOPBACK_HOSTNAMES = new Set(["127.0.0.1", "::1", "localhost"]);
const DESKTOP_RENDERER_ORIGINS = ["t3code://app", "t3code-dev://app"];
const GZIP_MIN_BYTES = 1024;
Expand Down Expand Up @@ -247,6 +249,50 @@ export const otlpTracesProxyRouteLayer = HttpRouter.add(
),
);

export const transcriptionRouteLayer = HttpRouter.add(
Comment thread
macroscopeapp[bot] marked this conversation as resolved.
Outdated
"POST",
TRANSCRIPTION_PATH,
Effect.gen(function* () {
yield* authenticateRawRouteWithScope(AuthOrchestrationOperateScope);
const request = yield* HttpServerRequest.HttpServerRequest;
const declaredLength = Number(request.headers["content-length"] ?? "0");
if (Number.isFinite(declaredLength) && declaredLength > MAX_TRANSCRIPTION_AUDIO_BYTES) {
return HttpServerResponse.jsonUnsafe(
{ error: "The recording exceeds the 25 MB limit." },
{ status: 413 },
);
}

const body = yield* request.arrayBuffer;
const audio = new Uint8Array(body);
Comment thread
cursor[bot] marked this conversation as resolved.
Outdated
const result = yield* Effect.tryPromise(() =>
forwardWhisperTranscription({
audio,
audioMimeType: request.headers["content-type"] ?? "audio/webm",
baseUrl: request.headers["x-t3-transcription-base-url"] ?? "",
Comment thread
macroscopeapp[bot] marked this conversation as resolved.
Outdated
model: request.headers["x-t3-transcription-model"] ?? "",
apiKey: request.headers["x-t3-transcription-api-key"] ?? "",
}),
).pipe(
Effect.orElseSucceed(() => ({
ok: false as const,
status: 502,
message: "Transcription failed unexpectedly.",
})),
);
Comment thread
macroscopeapp[bot] marked this conversation as resolved.
Outdated

return result.ok
? HttpServerResponse.jsonUnsafe({ text: result.text })
: HttpServerResponse.jsonUnsafe({ error: result.message }, { status: result.status });
}).pipe(
Effect.catchTags({
EnvironmentAuthInvalidError: HttpServerRespondable.toResponse,
EnvironmentInternalError: HttpServerRespondable.toResponse,
EnvironmentScopeRequiredError: HttpServerRespondable.toResponse,
}),
),
);

export const assetRouteLayer = HttpRouter.add(
"GET",
`${ASSET_ROUTE_PREFIX}/*`,
Expand Down
3 changes: 3 additions & 0 deletions apps/server/src/httpCors.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,9 @@ export const browserApiCorsAllowedHeaders = [
"traceparent",
"content-type",
"dpop",
"x-t3-transcription-api-key",
"x-t3-transcription-base-url",
"x-t3-transcription-model",
] as const;

export const browserApiCorsHeaders = {
Expand Down
2 changes: 2 additions & 0 deletions apps/server/src/server.ts
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@ import * as ServerConfig from "./config.ts";
import * as HttpResponseCompression from "./httpCompression/HttpResponseCompression.ts";
import {
otlpTracesProxyRouteLayer,
transcriptionRouteLayer,
assetRouteLayer,
serverEnvironmentHttpApiLayer,
staticAndDevRouteLayer,
Expand Down Expand Up @@ -417,6 +418,7 @@ export const makeRoutesLayer = Layer.mergeAll(
Layer.provide(environmentAuthenticatedAuthLayer),
),
otlpTracesProxyRouteLayer,
transcriptionRouteLayer,
assetRouteLayer,
staticAndDevRouteLayer,
websocketRpcRouteLayer,
Expand Down
66 changes: 66 additions & 0 deletions apps/server/src/transcription.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
import { expect, it } from "@effect/vitest";
import { describe, vi } from "vite-plus/test";

import { forwardWhisperTranscription, resolveWhisperTranscriptionUrl } from "./transcription.ts";

describe("resolveWhisperTranscriptionUrl", () => {
it("adds the OpenAI-compatible transcription path", () => {
expect(resolveWhisperTranscriptionUrl("http://127.0.0.1:8080/v1/")?.toString()).toBe(
"http://127.0.0.1:8080/v1/audio/transcriptions",
);
expect(
resolveWhisperTranscriptionUrl(
"https://api.groq.com/openai/v1/audio/transcriptions",
)?.toString(),
).toBe("https://api.groq.com/openai/v1/audio/transcriptions");
});

it("rejects non-http and credential-bearing URLs", () => {
expect(resolveWhisperTranscriptionUrl("file:///tmp/whisper")).toBeNull();
expect(resolveWhisperTranscriptionUrl("https://user:pass@example.com/v1")).toBeNull();
});
});

describe("forwardWhisperTranscription", () => {
it("sends OpenAI-compatible multipart audio with BYOK authorization", async () => {
const fetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
expect(init?.headers).toEqual({ authorization: "Bearer groq-key" });
expect(init?.body).toBeInstanceOf(FormData);
const form = init?.body as FormData;
expect(form.get("model")).toBe("whisper-large-v3-turbo");
expect(form.get("file")).toBeInstanceOf(Blob);
return Response.json({ text: "hello from the microphone" });
});

await expect(
forwardWhisperTranscription(
{
audio: new Uint8Array([1, 2, 3]),
audioMimeType: "audio/webm;codecs=opus",
baseUrl: "https://api.groq.com/openai/v1",
model: "whisper-large-v3-turbo",
apiKey: "groq-key",
},
fetchImpl,
),
).resolves.toEqual({ ok: true, text: "hello from the microphone" });
});

it("returns a safe provider error", async () => {
const fetchImpl = vi.fn(async () =>
Response.json({ error: { message: "bad key" } }, { status: 401 }),
);
await expect(
forwardWhisperTranscription(
{
audio: new Uint8Array([1]),
audioMimeType: "audio/webm",
baseUrl: "https://example.com/v1",
model: "whisper-1",
apiKey: "bad",
},
fetchImpl,
),
).resolves.toEqual({ ok: false, status: 502, message: "bad key" });
});
});
113 changes: 113 additions & 0 deletions apps/server/src/transcription.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,113 @@
const TRANSCRIPTION_PATH = "/audio/transcriptions";

export const MAX_TRANSCRIPTION_AUDIO_BYTES = 25 * 1024 * 1024;

export interface WhisperTranscriptionInput {
readonly audio: Uint8Array;
readonly audioMimeType: string;
readonly baseUrl: string;
readonly model: string;
readonly apiKey: string;
}

type TranscriptionFetch = (input: string | URL | Request, init?: RequestInit) => Promise<Response>;

export type WhisperTranscriptionResult =
| { readonly ok: true; readonly text: string }
| { readonly ok: false; readonly status: number; readonly message: string };
Comment thread
macroscopeapp[bot] marked this conversation as resolved.
Outdated

export function resolveWhisperTranscriptionUrl(baseUrl: string): URL | null {
try {
const url = new URL(baseUrl.trim());
if (url.protocol !== "http:" && url.protocol !== "https:") return null;
if (url.username || url.password) return null;
url.search = "";
url.hash = "";
url.pathname = url.pathname.replace(/\/$/, "");
if (!url.pathname.endsWith(TRANSCRIPTION_PATH)) {
url.pathname += TRANSCRIPTION_PATH;
}
return url;
} catch {
return null;
}
}

function audioFileExtension(mimeType: string): string {
if (mimeType.includes("ogg")) return "ogg";
if (mimeType.includes("mp4") || mimeType.includes("m4a")) return "m4a";
if (mimeType.includes("wav")) return "wav";
return "webm";
}

export async function forwardWhisperTranscription(
input: WhisperTranscriptionInput,
fetchImpl: TranscriptionFetch = globalThis.fetch,
Comment thread
macroscopeapp[bot] marked this conversation as resolved.
Outdated
): Promise<WhisperTranscriptionResult> {
const endpoint = resolveWhisperTranscriptionUrl(input.baseUrl);
if (!endpoint) {
return { ok: false, status: 400, message: "Enter a valid HTTP transcription endpoint." };
}
if (!input.model.trim()) {
return { ok: false, status: 400, message: "Enter a transcription model." };
}
if (input.audio.byteLength === 0) {
return { ok: false, status: 400, message: "The recording was empty." };
}
if (input.audio.byteLength > MAX_TRANSCRIPTION_AUDIO_BYTES) {
return { ok: false, status: 413, message: "The recording exceeds the 25 MB limit." };
}

const mimeType = input.audioMimeType.split(";", 1)[0]?.trim() || "audio/webm";
const form = new FormData();
form.set("model", input.model.trim());
form.set(
"file",
new Blob([input.audio], { type: mimeType }),
`recording.${audioFileExtension(mimeType)}`,
);

let response: Response;
try {
response = await fetchImpl(endpoint, {
method: "POST",
headers: input.apiKey ? { authorization: `Bearer ${input.apiKey}` } : undefined,
body: form,
signal: AbortSignal.timeout(120_000),
});
} catch {
return { ok: false, status: 502, message: "Could not reach the transcription provider." };
}

let payload: unknown;
try {
payload = await response.json();
} catch {
return { ok: false, status: 502, message: "The transcription provider returned invalid JSON." };
}

if (!response.ok) {
const providerMessage =
typeof payload === "object" &&
payload !== null &&
"error" in payload &&
typeof payload.error === "object" &&
payload.error !== null &&
"message" in payload.error &&
typeof payload.error.message === "string"
? payload.error.message
: null;
return {
ok: false,
status: 502,
message: providerMessage?.slice(0, 300) || "The transcription provider rejected the request.",
};
Comment thread
macroscopeapp[bot] marked this conversation as resolved.
Outdated
}

const text =
typeof payload === "object" && payload !== null && "text" in payload ? payload.text : undefined;
if (typeof text !== "string") {
return { ok: false, status: 502, message: "The transcription response did not contain text." };
}
return { ok: true, text: text.trim() };
}
63 changes: 63 additions & 0 deletions apps/web/src/components/chat/ChatComposer.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -107,6 +107,8 @@ import { buildExpandedImagePreview, type ExpandedImagePreview } from "./Expanded
import { basenameOfPath } from "../../pierre-icons";
import { cn, randomUUID } from "~/lib/utils";
import { Separator } from "../ui/separator";
import { useVoiceTranscription } from "../../hooks/useVoiceTranscription";
import { VoiceTranscriptionPanel } from "./VoiceTranscriptionPanel";

function ComposerCommandMenuLayer(props: { anchor: HTMLElement | null; children: ReactNode }) {
const [position, setPosition] = useState<{
Expand Down Expand Up @@ -179,6 +181,7 @@ import {
LockOpenIcon,
PenLineIcon,
SparklesIcon,
MicIcon,
XIcon,
} from "lucide-react";
import { proposedPlanTitle } from "../../proposedPlan";
Expand Down Expand Up @@ -1269,6 +1272,30 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps)
[composerDraftTarget, setComposerDraftPrompt],
);

const appendVoiceTranscript = useCallback(
(transcript: string) => {
const currentPrompt = promptRef.current;
const boundary = currentPrompt.length > 0 && !/\s$/.test(currentPrompt) ? " " : "";
const nextPrompt = `${currentPrompt}${boundary}${transcript}`;
promptRef.current = nextPrompt;
setPrompt(nextPrompt);
const nextCursor = collapseExpandedComposerCursor(nextPrompt, nextPrompt.length);
setComposerCursor(nextCursor);
setComposerTrigger(null);
scheduleComposerFocus();
},
[promptRef, scheduleComposerFocus, setPrompt],
);
const voiceTranscription = useVoiceTranscription({
config: {
provider: settings.voiceTranscriptionProvider,
baseUrl: settings.voiceTranscriptionBaseUrl,
model: settings.voiceTranscriptionModel,
apiKey: settings.voiceTranscriptionApiKey,
},
onTranscript: appendVoiceTranscript,
});

const addComposerImage = useCallback(
(image: ComposerImageAttachment) => {
addComposerDraftImage(composerDraftTarget, image);
Expand Down Expand Up @@ -3102,6 +3129,19 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps)
</div>
</div>

{settings.voiceTranscriptionEnabled && voiceTranscription.status !== "idle" ? (
<VoiceTranscriptionPanel
status={voiceTranscription.status}
elapsedMs={voiceTranscription.elapsedMs}
levels={voiceTranscription.levels}
onStop={voiceTranscription.stop}
/>
) : settings.voiceTranscriptionEnabled && voiceTranscription.error ? (
<p className="mx-4 mb-2 text-xs text-destructive" role="alert">
{voiceTranscription.error}
</p>
) : null}
Comment thread
cursor[bot] marked this conversation as resolved.

{/* Bottom toolbar */}
{isComposerCollapsedMobile ? null : activePendingApproval ? (
<div className="flex items-center justify-end gap-2 px-2.5 pb-2.5 sm:px-3 sm:pb-3">
Expand Down Expand Up @@ -3205,6 +3245,29 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps)
}
className="flex shrink-0 flex-nowrap items-center justify-end gap-2"
>
{settings.voiceTranscriptionEnabled &&
voiceTranscription.status === "idle" &&
!isComposerApprovalState &&
pendingUserInputs.length === 0 ? (
<Tooltip>
<TooltipTrigger
render={
<Button
type="button"
size="icon-sm"
variant="ghost"
disabled={isConnecting || projectSelectionRequired}
className="rounded-full text-muted-foreground"
onClick={() => void voiceTranscription.start()}
aria-label="Start dictation"
>
<MicIcon className="size-4" />
</Button>
}
/>
<TooltipPopup side="top">Start dictation</TooltipPopup>
</Tooltip>
) : null}
<ComposerFooterPrimaryActions
compact={isComposerPrimaryActionsCompact}
activeContextWindow={activeContextWindow}
Expand Down
Loading
Loading