Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
67 changes: 53 additions & 14 deletions web/app/api/vm/route.ts
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,7 @@ import {
type VmImageKind,
} from "../../../services/vms/images/resolver";
import { reconcileProPlanMetadata } from "../../../services/billing/pro";
import { after } from "next/server";
import { getStackServerApp, isStackConfigured } from "../../lib/stack";
import {
jsonResponse,
Expand Down Expand Up @@ -182,6 +183,13 @@ export async function POST(request: Request): Promise<Response> {
timing.record("auth", authDurationMs);
setResponseFinalizer((response) => {
timing.finish({ status: response.status });
// Per-stage timings travel with the response too, so a client or a
// smoke run sees where a create spent its time without Axiom.
try {
response.headers.set("Server-Timing", timing.serverTimingHeader());
} catch {
// Immutable headers on a passthrough Response: the span still has them.
}
captureVmProvisionOutcome({ userId: initialUser.id, operation: "create", response, span });
});
let user: AuthedUser = initialUser;
Expand Down Expand Up @@ -334,28 +342,43 @@ export async function POST(request: Request): Promise<Response> {
// right before paid limits apply. Best-effort — billing reads must
// not block VM creation, so the whole reconcile races a hard
// deadline and VM create proceeds with current metadata on timeout.
try {
if (isStackConfigured()) {
const changed = await withBillingReconcileDeadline(
measureVmAsync(timing, "billing_reconcile", async () => {
const serverUser = await getStackServerApp().getUser(user.id);
return serverUser ? reconcileProPlanMetadata(serverUser) : false;
})
);
if (changed) {
// The Stripe-to-Stack plan reconcile (a Stack read plus our subscription
// table) used to run before every create and cost 150 to 360 ms. It can
// only change this request's outcome when the cached plan would block
// it, so it runs inline on that path alone. Otherwise it runs after the
// response, so the next request still sees fresh metadata.
const reconcileProPlan = (recordTiming: boolean) =>
withBillingReconcileDeadline(
measureVmAsync(recordTiming ? timing : undefined, "billing_reconcile", async () => {
const serverUser = await getStackServerApp().getUser(user.id);
return serverUser ? reconcileProPlanMetadata(serverUser) : false;
}),
);
let account = await measureVmAsync(timing, "entitlements", () =>
resolveVmProvisioningAccountScope(user, request, { requestedBillingTeamId })
);
let reconcileMode: "off" | "deferred" | "inline" = isStackConfigured() ? "deferred" : "off";
if (!account.ok && reconcileMode === "deferred") {
reconcileMode = "inline";
try {
if (await reconcileProPlan(true)) {
const reconciledUser = await measureVmAsync(timing, "auth", () =>
verifyRequest(request, { requestedTeamId: requestedBillingTeamId })
);
if (reconciledUser) user = reconciledUser;
account = await measureVmAsync(timing, "entitlements", () =>
resolveVmProvisioningAccountScope(user, request, { requestedBillingTeamId })
);
}
} catch (err) {
console.error("[VM] Pro plan reconcile failed", err);
}
} catch (err) {
console.error("[VM] Pro plan reconcile failed", err);
}
const account = await measureVmAsync(timing, "entitlements", () =>
resolveVmProvisioningAccountScope(user, request, { requestedBillingTeamId })
);
setSpanAttributes(span, { "cmux.billing.reconcile_mode": reconcileMode });
if (!account.ok) return account.response;
if (reconcileMode === "deferred") {
runAfterResponse(() => reconcileProPlan(false).then(() => undefined));
}
const entitlements = account.entitlements;
setSpanAttributes(span, {
"cmux.billing.team_id_set": !!entitlements.billingTeamId,
Expand Down Expand Up @@ -436,6 +459,7 @@ export async function POST(request: Request): Promise<Response> {
"cmux.vm.image_set": image.length > 0,
"cmux.vm.image_version": imageSelection.imageVersion,
"cmux.vm.image_manifest": !!imageSelection.manifestEntry,
"cmux.vm.image_size": imageSelection.size?.name ?? "size-less",
"cmux.idempotency_key_set": !!idempotencyKey,
});

Expand Down Expand Up @@ -463,6 +487,7 @@ export async function POST(request: Request): Promise<Response> {
persistentHome: candidate.persistentHome === true,
perMachineHome: candidate.perMachineHome === true,
memoryMb,
imageSize: imageSelection.size ?? undefined,
modelPlane,
timing,
}));
Expand Down Expand Up @@ -536,6 +561,20 @@ export async function POST(request: Request): Promise<Response> {
// awaited) and VM create proceeds with the user's current plan metadata.
const BILLING_RECONCILE_DEADLINE_MS = 5_000;

/**
* Run best-effort work once the response has been sent. Vercel keeps the
* function alive for `after` callbacks; outside a request scope (tests, a
* plain Node server) `after` throws, and the work runs detached instead.
*/
function runAfterResponse(work: () => Promise<void>): void {
const guarded = () => work().catch((err) => console.error("[VM] deferred work failed", err));
try {
after(guarded);
} catch {
void guarded();
}
}

export async function withBillingReconcileDeadline(
reconcile: Promise<boolean>
): Promise<boolean> {
Expand Down
21 changes: 16 additions & 5 deletions web/services/vms/drivers/freestyle.ts
Original file line number Diff line number Diff line change
Expand Up @@ -733,11 +733,22 @@ export class FreestyleProvider implements VMProvider {
"cmux.vm.network.private": !!networkId,
});
try {
// CreateVmOptions has no size: a VM boots at its snapshot's
// resources (the devbox snapshot is 2 vCPU / 4 GB / 16 GB) and
// only a grow-only resize raises them. Size first so the machine
// the daemon comes up on is the one that was sold.
await this.growToRequestedSize(fs, vm, vmId, options.memoryMb, span);
if (options.imageSize) {
// One snapshot per size: the machine already boots at the shape
// that was sold, so nothing is read back and nothing is grown.
setSpanAttributes(span, {
"cmux.vm.image_size": options.imageSize.name,
"cmux.vm.resources.cpu": options.imageSize.cpu,
"cmux.vm.resources.memory_mb": options.imageSize.memoryMb,
"cmux.vm.resources.storage_mb": options.imageSize.storageMb,
"cmux.vm.resize.requested": false,
});
} else {
// A size-less image boots at its snapshot's resources and only a
// grow-only resize raises them. Size first so the machine the
// daemon comes up on is the one that was sold.
await this.growToRequestedSize(fs, vm, vmId, options.memoryMb, span);
}
// The baked supervisor is already bringing the daemon up; the only
// per-machine input it needs is the model-plane env file.
if (options.envs) await this.writeModelPlaneEnv(vm, vmId, options.envs);
Expand Down
7 changes: 7 additions & 0 deletions web/services/vms/drivers/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,13 @@ export type CreateOptions = {
* this way). Providers without sizing ignore it.
*/
memoryMb?: number;
/**
* The snapshot's own shape when the image is a sized ladder entry
* (services/vms/images/sizes.ts): the machine boots at the shape that was
* sold and the driver must not read it back or resize. Absent for size-less
* images, which are grown to `memoryMb`.
*/
imageSize?: { readonly name: string; readonly cpu: number; readonly memoryMb: number; readonly storageMb: number } | null;
/**
* Machine-level environment delivered at create time (the coderouter
* model-plane env: OPENAI_BASE_URL plus placeholder keys). Treat values as
Expand Down
Loading