Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions apps/desktop/src/settings/DesktopClientSettings.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,10 @@ const clientSettings: ClientSettings = {
sidebarV2Enabled: false,
sidebarV2ConfiguredByUser: false,
timestampFormat: "24-hour",
voiceTranscriptionEnabled: true,
voiceTranscriptionProvider: "openai",
voiceTranscriptionApiKey: "",
voiceTranscriptionModel: "",
wordWrap: true,
};

Expand Down
129 changes: 129 additions & 0 deletions apps/server/src/http.ts
Original file line number Diff line number Diff line change
Expand Up @@ -40,8 +40,19 @@ import {
} from "./auth/http.ts";
import * as ServerEnvironment from "./environment/ServerEnvironment.ts";
import { browserApiCorsAllowedHeaders, browserApiCorsAllowedMethods } from "./httpCors.ts";
import {
forwardVoiceTranscription,
listVoiceTranscriptionModels,
MAX_TRANSCRIPTION_AUDIO_BYTES,
readTranscriptionAudio,
resolveTranscriptionProvider,
transcriptionEnvironmentApiKeyStatus,
TranscriptionProviderUnsupportedError,
} from "./transcription.ts";

const OTLP_TRACES_PROXY_PATH = "/api/observability/v1/traces";
const TRANSCRIPTION_PATH = "/api/transcription";
const TRANSCRIPTION_MODELS_PATH = "/api/transcription/models";
const LOOPBACK_HOSTNAMES = new Set(["127.0.0.1", "::1", "localhost"]);
const DESKTOP_RENDERER_ORIGINS = ["t3code://app", "t3code-dev://app"];
const GZIP_MIN_BYTES = 1024;
Expand Down Expand Up @@ -247,6 +258,124 @@ export const otlpTracesProxyRouteLayer = HttpRouter.add(
),
);

const transcriptionConfigRouteLayer = HttpRouter.add(
"GET",
TRANSCRIPTION_PATH,
Effect.gen(function* () {
yield* authenticateRawRouteWithScope(AuthOrchestrationOperateScope);
const [openai, groq] = yield* Effect.all([
transcriptionEnvironmentApiKeyStatus("openai"),
transcriptionEnvironmentApiKeyStatus("groq"),
]);
return HttpServerResponse.jsonUnsafe({ openai, groq });
}).pipe(
Effect.catchTags({
EnvironmentAuthInvalidError: HttpServerRespondable.toResponse,
EnvironmentInternalError: HttpServerRespondable.toResponse,
EnvironmentScopeRequiredError: HttpServerRespondable.toResponse,
}),
),
);

const transcriptionUploadRouteLayer = HttpRouter.add(
"POST",
TRANSCRIPTION_PATH,
Effect.gen(function* () {
yield* authenticateRawRouteWithScope(AuthOrchestrationOperateScope);
const request = yield* HttpServerRequest.HttpServerRequest;
const declaredLength = Number(request.headers["content-length"] ?? "0");
if (Number.isFinite(declaredLength) && declaredLength > MAX_TRANSCRIPTION_AUDIO_BYTES) {
return HttpServerResponse.jsonUnsafe(
{ error: "The recording exceeds the 25 MB limit." },
{ status: 413 },
);
}

const provider = resolveTranscriptionProvider(
request.headers["x-t3-transcription-provider"] ?? "",
);
if (!provider) {
return yield* new TranscriptionProviderUnsupportedError();
}

const audio = yield* readTranscriptionAudio(request.stream);
const text = yield* forwardVoiceTranscription({
audio,
audioMimeType: request.headers["content-type"] ?? "audio/webm",
provider,
apiKey: request.headers["x-t3-transcription-api-key"] ?? "",
model: request.headers["x-t3-transcription-model"] ?? "",
});
return HttpServerResponse.jsonUnsafe({ text });
}).pipe(
Effect.catchTags({
EnvironmentAuthInvalidError: HttpServerRespondable.toResponse,
EnvironmentInternalError: HttpServerRespondable.toResponse,
EnvironmentScopeRequiredError: HttpServerRespondable.toResponse,
TranscriptionAudioTooLargeError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 413 })),
TranscriptionApiKeyMissingError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })),
TranscriptionBodyReadError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })),
TranscriptionEmptyAudioError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })),
TranscriptionModelMissingError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })),
TranscriptionProviderError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })),
TranscriptionProviderUnsupportedError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })),
TranscriptionRequestError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })),
TranscriptionResponseError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })),
}),
),
);

const transcriptionModelsRouteLayer = HttpRouter.add(
"GET",
TRANSCRIPTION_MODELS_PATH,
Effect.gen(function* () {
yield* authenticateRawRouteWithScope(AuthOrchestrationOperateScope);
const request = yield* HttpServerRequest.HttpServerRequest;
const provider = resolveTranscriptionProvider(
request.headers["x-t3-transcription-provider"] ?? "",
);
if (!provider) {
return yield* new TranscriptionProviderUnsupportedError();
}
const models = yield* listVoiceTranscriptionModels({
provider,
apiKey: request.headers["x-t3-transcription-api-key"] ?? "",
});
return HttpServerResponse.jsonUnsafe({ models });
}).pipe(
Effect.catchTags({
EnvironmentAuthInvalidError: HttpServerRespondable.toResponse,
EnvironmentInternalError: HttpServerRespondable.toResponse,
EnvironmentScopeRequiredError: HttpServerRespondable.toResponse,
TranscriptionApiKeyMissingError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })),
TranscriptionProviderError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })),
TranscriptionProviderUnsupportedError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })),
TranscriptionRequestError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })),
TranscriptionResponseError: (error) =>
Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })),
}),
),
);

export const transcriptionRouteLayer = Layer.mergeAll(
transcriptionConfigRouteLayer,
transcriptionModelsRouteLayer,
transcriptionUploadRouteLayer,
);

export const assetRouteLayer = HttpRouter.add(
"GET",
`${ASSET_ROUTE_PREFIX}/*`,
Expand Down
3 changes: 3 additions & 0 deletions apps/server/src/httpCors.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,9 @@ export const browserApiCorsAllowedHeaders = [
"traceparent",
"content-type",
"dpop",
"x-t3-transcription-api-key",
"x-t3-transcription-model",
"x-t3-transcription-provider",
] as const;

export const browserApiCorsHeaders = {
Expand Down
6 changes: 6 additions & 0 deletions apps/server/src/server.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1300,6 +1300,9 @@ const assertBrowserApiCorsPreflightHeaders = (
"content-type",
"dpop",
"traceparent",
"x-t3-transcription-api-key",
"x-t3-transcription-model",
"x-t3-transcription-provider",
]);
};
const crossOriginClientOrigin = "http://remote-client.test:3773";
Expand Down Expand Up @@ -4209,6 +4212,9 @@ it.layer(NodeServices.layer)("server router seam", (it) => {
"content-type",
"dpop",
"traceparent",
"x-t3-transcription-api-key",
"x-t3-transcription-model",
"x-t3-transcription-provider",
]);
}).pipe(Effect.provide(NodeHttpServer.layerTest)),
);
Expand Down
2 changes: 2 additions & 0 deletions apps/server/src/server.ts
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@ import * as ServerConfig from "./config.ts";
import * as HttpResponseCompression from "./httpCompression/HttpResponseCompression.ts";
import {
otlpTracesProxyRouteLayer,
transcriptionRouteLayer,
assetRouteLayer,
serverEnvironmentHttpApiLayer,
staticAndDevRouteLayer,
Expand Down Expand Up @@ -417,6 +418,7 @@ export const makeRoutesLayer = Layer.mergeAll(
Layer.provide(environmentAuthenticatedAuthLayer),
),
otlpTracesProxyRouteLayer,
transcriptionRouteLayer,
assetRouteLayer,
staticAndDevRouteLayer,
websocketRpcRouteLayer,
Expand Down
120 changes: 120 additions & 0 deletions apps/server/src/transcription.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,120 @@
import { expect, it } from "@effect/vitest";
import * as ConfigProvider from "effect/ConfigProvider";
import * as Effect from "effect/Effect";
import * as Stream from "effect/Stream";
import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http";
import { describe } from "vite-plus/test";

import {
forwardVoiceTranscription,
listVoiceTranscriptionModels,
MAX_TRANSCRIPTION_AUDIO_BYTES,
readTranscriptionAudio,
resolveTranscriptionProvider,
transcriptionEnvironmentApiKeyStatus,
transcriptionProviderConfig,
} from "./transcription.ts";

describe("transcription providers", () => {
it("uses fixed OpenAI and Groq configurations", () => {
expect(resolveTranscriptionProvider("openai")).toBe("openai");
expect(transcriptionProviderConfig("openai")).toEqual({
endpoint: "https://api.openai.com/v1/audio/transcriptions",
modelsEndpoint: "https://api.openai.com/v1/models",
apiKeyEnvironmentVariable: "OPENAI_API_KEY",
});
expect(resolveTranscriptionProvider("groq")).toBe("groq");
expect(transcriptionProviderConfig("groq")).toEqual({
endpoint: "https://api.groq.com/openai/v1/audio/transcriptions",
modelsEndpoint: "https://api.groq.com/openai/v1/models",
apiKeyEnvironmentVariable: "GROQ_API_KEY",
});
expect(resolveTranscriptionProvider("custom")).toBeNull();
});

it.effect("loads accessible transcription models with the provider API key", () =>
Effect.gen(function* () {
let capturedRequest: HttpClientRequest.HttpClientRequest | undefined;
const client = HttpClient.make((request) =>
Effect.sync(() => {
capturedRequest = request;
return HttpClientResponse.fromWeb(
request,
Response.json({
data: [
{ id: "gpt-4o" },
{ id: " whisper-1 " },
{ id: "gpt-4o-mini-transcribe" },
{ id: "gpt-4o-mini-transcribe" },
],
}),
);
}),
);

const models = yield* listVoiceTranscriptionModels({
provider: "openai",
apiKey: "client-openai-key",
}).pipe(Effect.provideService(HttpClient.HttpClient, client));

expect(models).toEqual(["gpt-4o-mini-transcribe", "whisper-1"]);
expect(capturedRequest?.url).toBe("https://api.openai.com/v1/models");
expect(capturedRequest?.headers.authorization).toBe("Bearer client-openai-key");
}),
);

it.effect("uses a provider API key from the server environment", () =>
Effect.gen(function* () {
let capturedRequest: HttpClientRequest.HttpClientRequest | undefined;
const client = HttpClient.make((request) =>
Effect.sync(() => {
capturedRequest = request;
return HttpClientResponse.fromWeb(request, Response.json({ text: " groq transcript " }));
}),
);

const transcript = yield* forwardVoiceTranscription({
audio: new Uint8Array([1, 2, 3]),
audioMimeType: "audio/webm;codecs=opus",
provider: "groq",
apiKey: "",
model: "whisper-large-v3",
}).pipe(Effect.provideService(HttpClient.HttpClient, client));

expect(yield* transcriptionEnvironmentApiKeyStatus("groq")).toBe(true);
expect(yield* transcriptionEnvironmentApiKeyStatus("openai")).toBe(false);
expect(transcript).toBe("groq transcript");
expect(capturedRequest?.url).toBe("https://api.groq.com/openai/v1/audio/transcriptions");
expect(capturedRequest?.headers.authorization).toBe("Bearer env-groq-key");
expect(capturedRequest?.body._tag).toBe("FormData");
if (capturedRequest?.body._tag === "FormData") {
expect(capturedRequest.body.formData.get("model")).toBe("whisper-large-v3");
expect(capturedRequest.body.formData.get("file")).toBeInstanceOf(Blob);
}
}).pipe(
Effect.provide(
ConfigProvider.layer(ConfigProvider.fromEnv({ env: { GROQ_API_KEY: "env-groq-key" } })),
),
),
);
});

describe("readTranscriptionAudio", () => {
it.effect("combines streamed audio chunks", () =>
Effect.gen(function* () {
const audio = yield* readTranscriptionAudio(
Stream.make(new Uint8Array([1, 2]), new Uint8Array([3, 4])),
);
expect(Array.from(audio)).toEqual([1, 2, 3, 4]);
}),
);

it.effect("stops when streamed audio exceeds the limit", () =>
Effect.gen(function* () {
const error = yield* readTranscriptionAudio(
Stream.make(new Uint8Array(MAX_TRANSCRIPTION_AUDIO_BYTES), new Uint8Array([1])),
).pipe(Effect.flip);
expect(error._tag).toBe("TranscriptionAudioTooLargeError");
}),
);
});
Loading
Loading