diff --git a/CLI/cmux.swift b/CLI/cmux.swift index d68f6058bf42..4d180d393f44 100644 --- a/CLI/cmux.swift +++ b/CLI/cmux.swift @@ -4374,7 +4374,8 @@ struct CMUXCLI { // `vm_image_config_error`. /// `--size` spellings → memory in MB. The supported base-image ladder is /// 4 GB, 8 GB, 16 GB, 24 GB, 32 GB, and 64 GB of RAM, with disk sizes - /// following each image. + /// following each image. Pricing separately describes 5 vCPU, 20 GB RAM, + /// and 200 GB disk as one pool shared across the plan's Cloud VMs. private static let cloudVMSizeAliases: [String: Int] = [ "4g": 4096, "4gb": 4096, "8g": 8192, "8gb": 8192, diff --git a/Resources/Localizable.xcstrings b/Resources/Localizable.xcstrings index 97b8b0334472..df51ce07af8b 100644 --- a/Resources/Localizable.xcstrings +++ b/Resources/Localizable.xcstrings @@ -150566,13 +150566,13 @@ "en": { "stringUnit": { "state": "translated", - "value": "Up to 50 Cloud VMs, each with 8 GB RAM and 32 GB disk by default; sizes from 4 to 64 GB RAM are available" + "value": "Up to 50 Cloud VMs, all sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk; each VM starts at 8 GB RAM and 32 GB disk, with sizes from 4 to 64 GB RAM available as capacity allows" } }, "ja": { "stringUnit": { "state": "translated", - "value": "最大 50 台の Cloud VM、各 8 GB RAM / 32 GB ディスク(標準)。4~64 GB RAM のサイズを利用できます" + "value": "最大 50 台の Cloud VM、すべての VM で合計 5 vCPU / 20 GB RAM / 200 GB ディスクを共有。標準サイズは 8 GB RAM / 32 GB ディスクで、容量の範囲内で RAM 4~64 GB のサイズを選べます" } } } @@ -150787,15 +150787,33 @@ "en": { "stringUnit": { "state": "translated", - "value": "Every Cloud VM has a configurable size: 4, 8, 16, 24, 32, or 64 GB RAM with 16, 32, 64, 96, or 128 GB disk. Pro and Team include up to 50 machines with no metering or overages." + "value": "Pro includes up to 50 Cloud VMs sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same shared pool per user. The default VM size is 8 GB RAM and 32 GB disk; sizes from 4 to 64 GB RAM are available as capacity allows. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, "ja": { "stringUnit": { "state": "translated", - "value": "すべての Cloud VM はサイズを選択できます。RAM は 4、8、16、24、32、64 GB、ディスクは 16、32、64、96、128 GB です。Pro と Team には最大 50 台のマシンが含まれ、従量課金や超過料金はありません。" + "value": "Pro では最大 50 台の Cloud VM で、合計 5 vCPU / 20 GB RAM / 200 GB ディスクを共有します。Team では同じ共有プールがユーザーごとに含まれます。標準 VM サイズは 8 GB RAM / 32 GB ディスクで、容量の範囲内で RAM 4~64 GB のサイズを選べます。新しい VM のディスクは 32 GB で開始し、共有プールの範囲内で拡張できます。従量課金や超過料金はありません。" } - } + }, + "ar": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "bs": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "da": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "de": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "es": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "fr": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "it": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "km": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "ko": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "nb": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "pl": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "pt-BR": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "ru": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "th": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "tr": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "uk": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "zh-Hans": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } }, + "zh-Hant": { "stringUnit": { "state": "translated", "value": "Pro includes up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same allowance and shared capacity per user. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." } } } }, "pricing.native.sizes.colRate": { @@ -150855,15 +150873,33 @@ "en": { "stringUnit": { "state": "translated", - "value": "Cloud VM sizes" + "value": "Shared Cloud VM resources" } }, "ja": { "stringUnit": { "state": "translated", - "value": "Cloud VM サイズ" + "value": "Cloud VM 間で共有するリソース" } - } + }, + "ar": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "bs": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "da": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "de": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "es": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "fr": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "it": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "km": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "ko": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "nb": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "pl": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "pt-BR": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "ru": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "th": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "tr": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "uk": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "zh-Hans": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } }, + "zh-Hant": { "stringUnit": { "state": "translated", "value": "Shared Cloud VM resources" } } } }, "pricing.native.status.billingUnavailable": { @@ -150974,13 +151010,13 @@ "en": { "stringUnit": { "state": "translated", - "value": "Up to 50 Cloud VMs per user" + "value": "Up to 50 Cloud VMs per user, all sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk; each VM starts at 8 GB RAM and 32 GB disk, with sizes from 4 to 64 GB RAM available as capacity allows" } }, "ja": { "stringUnit": { "state": "translated", - "value": "ユーザーごとに最大 50 台の Cloud VM" + "value": "ユーザーごとに最大 50 台の Cloud VM、すべての VM で合計 5 vCPU / 20 GB RAM / 200 GB ディスクを共有。標準サイズは 8 GB RAM / 32 GB ディスクで、容量の範囲内で RAM 4~64 GB のサイズを選べます" } } } diff --git a/Sources/Cloud/NewMachineModel.swift b/Sources/Cloud/NewMachineModel.swift index 9446f627e48c..88cd0808c226 100644 --- a/Sources/Cloud/NewMachineModel.swift +++ b/Sources/Cloud/NewMachineModel.swift @@ -82,11 +82,13 @@ final class NewMachineModel { static let legacyPlanMachineMemoryMb = 20480 /// Mirrors `maxMemoryMbForPlan`: development and paid plans may use the /// largest supported base image unless an operator sets a lower ceiling. + /// The pricing page separately describes the 5 vCPU / 20 GB RAM / 200 GB + /// disk pool shared across a paid plan's Cloud VMs. static func maxMemoryMb(planId: String?) -> Int { _ = planId return memoryOptionsMb.max() ?? planMachineMemoryMb } - /// Mirrors `defaultMemoryMbForPlan`: the plan machine, never above the max. + /// Mirrors `defaultMemoryMbForPlan`: the provider sizing profile, never above the max. static func defaultMemoryMb(planId: String?) -> Int { min(planMachineMemoryMb, maxMemoryMb(planId: planId)) } diff --git a/Sources/PricingPlansScreen.swift b/Sources/PricingPlansScreen.swift index 2e3b6b5a883a..8a9fd3f0974c 100644 --- a/Sources/PricingPlansScreen.swift +++ b/Sources/PricingPlansScreen.swift @@ -417,7 +417,7 @@ private struct NativePricingPlansView: View { isProminent: true, features: [ String(localized: "pricing.native.pro.feature.vms", defaultValue: "Cloud agents on isolated Cloud VMs"), - String(localized: "pricing.native.pro.feature.hours", defaultValue: "Up to 50 Cloud VMs, each with 5 vCPU, 20 GB RAM, and 32 GB disk"), + String(localized: "pricing.native.pro.feature.hours", defaultValue: "Up to 50 Cloud VMs, all sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk; each VM starts at 8 GB RAM and 32 GB disk, with sizes from 4 to 64 GB RAM available as capacity allows"), String(localized: "pricing.native.pro.feature.gateway", defaultValue: "Unlimited workspaces"), String(localized: "pricing.native.pro.feature.ios", defaultValue: "cmux iOS app and email support"), ] @@ -432,7 +432,7 @@ private struct NativePricingPlansView: View { features: [ String(localized: "pricing.native.team.feature.billing", defaultValue: "Unified billing for the whole team"), String(localized: "pricing.native.team.feature.seats", defaultValue: "Centralized seat management"), - String(localized: "pricing.native.team.feature.compute", defaultValue: "Up to 50 Cloud VMs per user"), + String(localized: "pricing.native.team.feature.compute", defaultValue: "Up to 50 Cloud VMs per user, all sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk; each VM starts at 8 GB RAM and 32 GB disk, with sizes from 4 to 64 GB RAM available as capacity allows"), String(localized: "pricing.native.team.feature.gateway", defaultValue: "Team-wide model gateway analytics"), String(localized: "pricing.native.team.feature.support", defaultValue: "Priority email support"), ] @@ -762,12 +762,12 @@ private struct NativePricingTableCell: View { private struct NativePricingSizeSection: View { var body: some View { VStack(alignment: .leading, spacing: 10) { - Text(String(localized: "pricing.native.sizes.title", defaultValue: "Cloud VM sizes")) + Text(String(localized: "pricing.native.sizes.title", defaultValue: "Shared Cloud VM resources")) .font(.system(size: 13, weight: .medium)) .foregroundStyle(.secondary) Text(String( localized: "pricing.native.sizes.body", - defaultValue: "Every Cloud VM is 5 vCPU, 20 GB RAM, and 32 GB disk. Disk can grow to 256 GiB. Pro and Team include up to 50 machines with no metering or overages." + defaultValue: "Pro includes up to 50 Cloud VMs sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk. Team includes the same shared pool per user. The default VM size is 8 GB RAM and 32 GB disk; sizes from 4 to 64 GB RAM are available as capacity allows. New VM disks start at 32 GB and can grow within the shared pool. There is no metering or overage billing." )) .font(.system(size: 13)) .foregroundStyle(.secondary) diff --git a/skills/cmux-billing/SKILL.md b/skills/cmux-billing/SKILL.md index 8f3719dc0f52..e97f62e11545 100644 --- a/skills/cmux-billing/SKILL.md +++ b/skills/cmux-billing/SKILL.md @@ -37,7 +37,7 @@ Read before changing billing, pricing, Stripe, Pro entitlement, checkout, webhoo Prices live in `web/services/billing/plans.ts` and are provisioned by `web/scripts/stripe/provision-catalog.sh`. Current checkout keys: `cmux-pro-monthly-50` ($50/mo), `cmux-pro-yearly-480` ($480/yr, $40/mo equivalent), `cmux-team-monthly-60` ($60/user/mo), `cmux-team-yearly-576` ($576/user/yr, $48/mo equivalent). Stripe amounts are immutable, so a price change mints a new lookup key carrying the amount and adds the old key to `LEGACY_PRICE_LOOKUP_KEYS`; grandfathered Prices (`cmux-pro-monthly` $30, `cmux-pro-yearly` $240, `cmux-pro-yearly-288` $288, `cmux-team-monthly` $35, `cmux-team-yearly-336` $336) stay active for the subscriptions on them and `/dashboard/billing` prices every row from its own Stripe amount. Env price-id overrides carry the amount in their name (`STRIPE_PRO_MONTHLY_50_PRICE_ID` and friends); a retired name fails env validation, so delete it from Vercel before deploying a price change. Test-mode Pro product `prod_UyHgRPpmCzrkLJ`, live `prod_Uq4a28vk0fP3E6`. Staging webhook endpoint `we_1Tq1SZGhInAdn3JbWJReKNEN` forwards to `cmux-staging.vercel.app`; its secrets are already in the `cmux-staging` Vercel project. -The plan machine (50 machines per billing team, 20 GB / 5 vCPU / 200 GB disk) lives in `web/services/vms/machineSpec.ts`; `web/tests/pro-pricing.test.ts` pins the pricing copy to those constants. +The paid Cloud VM allowance lives in `web/services/vms/machineSpec.ts`: Pro allows up to 50 machines for one paid account. Team grants 50 machines per paid seat, so each paid seat adds one allowance and one shared pool. Each allowance shares 5 vCPU, 20 GB RAM, and 200 GB disk across its Cloud VMs. New provider machines start with a 32 GB disk and can grow. The VM repository persists reservations and enforces the aggregate pool under the billing-team lock, treating CPU and memory as shared ceilings and adding disk claims; the entitlement module only resolves the allowance and shared-capacity policy. `web/tests/pro-pricing.test.ts` pins the pricing copy and the provider starting disk. ## Feature flags diff --git a/web/app/api/vm/[id]/fork/route.ts b/web/app/api/vm/[id]/fork/route.ts index 15fe367b5424..932f8efe9a82 100644 --- a/web/app/api/vm/[id]/fork/route.ts +++ b/web/app/api/vm/[id]/fork/route.ts @@ -15,6 +15,7 @@ import { import { forkVm, runVmWorkflow } from "../../../../../services/vms/workflows"; import { VmTimingRecorder } from "../../../../../services/vms/timings"; import { authProviderErrorResponse } from "../../../../../services/vms/authErrors"; +import { vmRequestLocale } from "../../../../../services/vms/vmErrorMessages"; import { idempotencyKeyFromRequest, parseOptionalObjectBody, @@ -94,10 +95,11 @@ export async function POST( }); } catch (err) { if (isVmNotFoundError(err)) return notFoundVm(id); - const response = vmCreateLikeErrorResponse(err, { + const response = await vmCreateLikeErrorResponse(err, { operation: "fork", planId: entitlements.planId, retryAction: "Run `cmux vm ls`, then delete an active VM with `cmux vm rm ` before forking another.", + locale: vmRequestLocale(request), }); if (response) return response; throw err; diff --git a/web/app/api/vm/[id]/resize/route.ts b/web/app/api/vm/[id]/resize/route.ts index f7be1e777c8b..5e7a56fc7056 100644 --- a/web/app/api/vm/[id]/resize/route.ts +++ b/web/app/api/vm/[id]/resize/route.ts @@ -49,9 +49,11 @@ export async function POST( const stats = await runVmWorkflow(resizeVm({ userId: user.id, billingTeamId: account.entitlements.billingTeamId, + billingPlanId: account.entitlements.planId, teamIds: user.teamIds, providerVmId: id, storageMb, + maxActiveVms: account.entitlements.maxActiveVms, })); return jsonResponse({ id, diff --git a/web/app/api/vm/base/routeShared.ts b/web/app/api/vm/base/routeShared.ts index a853e774609b..4a68f2325656 100644 --- a/web/app/api/vm/base/routeShared.ts +++ b/web/app/api/vm/base/routeShared.ts @@ -9,6 +9,7 @@ import { isVmCreateInProgressError, isVmImageConfigError, isVmLimitExceededError, + isVmSharedResourceLimitExceededError, } from "../../../../services/vms/errors"; import { inferVmProviderForImage, @@ -25,6 +26,7 @@ import { requestedVmTeamIdFromRequest, vmActiveLimitExceededResponse, vmErrorResponse, + vmSharedResourceLimitExceededResponse, vmWorkflowErrorResponse, resolveVmProvisioningAccountScope, } from "../../../../services/vms/routeHelpers"; @@ -184,6 +186,9 @@ async function baseWorkflowErrorResponse( phase: "create", }); } + if (isVmSharedResourceLimitExceededError(err)) { + return vmSharedResourceLimitExceededResponse(err, "create", locale); + } if (isVmCreateCreditsInsufficientError(err)) { return vmErrorResponse({ error: "vm_create_credits_insufficient", diff --git a/web/app/api/vm/restore/route.ts b/web/app/api/vm/restore/route.ts index d147c9cb5903..b73618b4de38 100644 --- a/web/app/api/vm/restore/route.ts +++ b/web/app/api/vm/restore/route.ts @@ -16,6 +16,7 @@ import { setSpanAttributes } from "../../../../services/telemetry"; import { restoreVm, runVmWorkflow } from "../../../../services/vms/workflows"; import { VmTimingRecorder } from "../../../../services/vms/timings"; import { authProviderErrorResponse } from "../../../../services/vms/authErrors"; +import { vmRequestLocale } from "../../../../services/vms/vmErrorMessages"; import { idempotencyKeyFromRequest, parseRequiredObjectBody, @@ -134,10 +135,11 @@ export async function POST(request: Request): Promise { createdAt: restored.createdAt, }); } catch (err) { - const response = vmCreateLikeErrorResponse(err, { + const response = await vmCreateLikeErrorResponse(err, { operation: "restore", planId: entitlements.planId, retryAction: "Run `cmux vm ls`, then delete an active VM with `cmux vm rm ` before restoring another.", + locale: vmRequestLocale(request), }); if (response) return response; throw err; diff --git a/web/app/api/vm/route.ts b/web/app/api/vm/route.ts index e22716718628..563dd557a3dc 100644 --- a/web/app/api/vm/route.ts +++ b/web/app/api/vm/route.ts @@ -404,7 +404,8 @@ export async function POST(request: Request): Promise { const requestedMemoryMb = candidate.memoryMb as number | undefined; // The server owns the supported size ladder. A stale client request // that is not on the ladder resolves to the plan default instead of - // failing the create. + // failing the create. The paid plan's 5 vCPU / 20 GB RAM / 200 GB + // shared pool is enforced separately by the repository. // Clients ship their own size table and always trail the server: the // 2026-09-02 pricing change (#11610) left every installed nightly // sending its old 24 GB default and the server rejecting each create diff --git a/web/app/components/pricing-shared.tsx b/web/app/components/pricing-shared.tsx index df5b3b9d9f1a..54f1148009b9 100644 --- a/web/app/components/pricing-shared.tsx +++ b/web/app/components/pricing-shared.tsx @@ -271,7 +271,7 @@ export function pricingActionClassName( const sizeClass = size === "compact" ? "px-3 py-1.5 text-xs" - : "px-5 py-2.5 text-[15px]"; + : "min-h-12 px-5 py-3 text-[15px]"; if (variant === "primary") { return `${base} ${sizeClass} bg-foreground transition-opacity hover:opacity-85`; } diff --git a/web/i18n/messages.ts b/web/i18n/messages.ts index dd8b01c4ccfb..96ce38ca4a81 100644 --- a/web/i18n/messages.ts +++ b/web/i18n/messages.ts @@ -35,6 +35,9 @@ export async function loadMessages(locale: Locale): Promise_MAX_ACTIVE_VMS` and `CMUX_VM_PAID_MAX_ACTIVE_VMS` exist only as incident brakes; the product number lives in code. Paid plans only consume Stack Auth create credits when `CMUX_VM_PLAN__CREATE_CREDIT_ITEM_ID` or the global `CMUX_VM_CREATE_CREDIT_ITEM_ID` is configured. +Plan limits are team-based. Stack Auth personal teams should stay enabled for both dev/staging and production projects (`createTeamOnSignUp` / `teams.createPersonalTeamOnSignUp`). New VM rows store `billing_team_id` and `billing_plan_id`; the free plan allows zero active VMs by default and remains at zero regardless of stale free-limit env values while the paid-plan gate is on. A deliberate `CMUX_VM_ALLOW_FREE_PROVISIONING=1` escape hatch re-enables the configured free allowance for local demos or a controlled rollback; paid plans get the allowance sold on /pricing, 50 active machines per billing team, multiplied by the Team subscription's paid seats (`cmuxSeats` in the team's Stack metadata, written from the Stripe quantity) so "50 per user" holds for the whole team (`PAID_MAX_ACTIVE_VMS_DEFAULT`; `maxActiveVms` in entitlements and the list response). New machines use validated Freestyle base snapshots from 4 GiB RAM / 16 GB disk through 64 GiB RAM / 128 GB disk, including the 24 GiB / 96 GB intermediate size. The default is 8 GiB RAM and 32 GB disk. Pricing separately advertises 5 vCPU, 20 GB memory, and 200 GB disk as one pool shared across the plan's active VMs. The repository records each reservation in `provider_metadata` and adds every CPU, memory, and disk claim under the billing-team lock before create, Base open/reset, and disk resize. The 50-machine count is an upper bound because the shared resource pool can fill first. Disk growth is independent, grow-only, and capped at 256 GiB in 4 GiB steps. The Freestyle driver applies the default at create (`CMUX_VM_DISK_MB` overrides it), and the resize API reads provider stats before and after the provider confirms the change. Destroyed VMs do not count against a limit; pausing does not free quota on the production provider. Paid plan activation should write a readable plan id such as `pro` into Stack Auth team read-only metadata (`cmuxVmPlan`) or equivalent billing sync metadata. `CMUX_VM_PLAN__MAX_ACTIVE_VMS` and `CMUX_VM_PAID_MAX_ACTIVE_VMS` exist only as incident brakes; the product number lives in code. Paid plans only consume Stack Auth create credits when `CMUX_VM_PLAN__CREATE_CREDIT_ITEM_ID` or the global `CMUX_VM_CREATE_CREDIT_ITEM_ID` is configured. ### The free limit is the paywall moment @@ -512,4 +512,4 @@ Plan limits are team-based. Stack Auth personal teams should stay enabled for bo ### Pricing is flat -Paid plans include up to 50 active VMs (per paid seat on Team) for a flat subscription price, every one the plan machine. There is no usage metering, no overages, and no per-hour VM size pricing; an earlier GB-RAM-awake-seconds metering design was considered and dropped to keep pricing simple. +Paid plans include up to 50 active VMs (per paid seat on Team) for a flat subscription price, with 5 vCPU, 20 GB memory, and 200 GB disk shared across those VMs. There is no usage metering, no overages, and no per-hour VM size pricing; an earlier GB-RAM-awake-seconds metering design was considered and dropped to keep pricing simple. Legacy VM resource claims are repaired by the status-reconcile cron in batches of 50, so create and resize requests do not fan out provider stats reads. Until a row is repaired, the repository uses a conservative claim and may delay a create until the next cron pass. diff --git a/web/services/vms/drivers/freestyle.ts b/web/services/vms/drivers/freestyle.ts index 2f976a1c8c8b..46c08f4a6e8d 100644 --- a/web/services/vms/drivers/freestyle.ts +++ b/web/services/vms/drivers/freestyle.ts @@ -34,7 +34,12 @@ import { type VMResizeOptions, type VMStatus, } from "./types"; -import { PLAN_MACHINE_MEMORY_MB, vcpusForMemoryMb, vmDiskMb } from "../machineSpec"; +import { + PLAN_MACHINE_MEMORY_MB, + VM_DISK_MB_DEFAULT, + vcpusForMemoryMb, + vmDiskMb, +} from "../machineSpec"; import { DEVBOX_DESKTOP_NOVNC_PORT, DEVBOX_DESKTOP_START_SCRIPT, @@ -833,19 +838,30 @@ export class FreestyleProvider implements VMProvider { }); try { if (options.imageSize) { - // One snapshot per size: the machine already boots at the shape - // that was sold, so nothing is read back and nothing is grown. + // One snapshot per CPU/memory size: preserve that baked shape, + // then grow only storage when the image is below the documented + // 32 GB starting disk. setSpanAttributes(span, { "cmux.vm.image_size": options.imageSize.name, "cmux.vm.resources.cpu": options.imageSize.cpu, "cmux.vm.resources.memory_mb": options.imageSize.memoryMb, - "cmux.vm.resources.storage_mb": options.imageSize.storageMb, - "cmux.vm.resize.requested": false, }); + await this.growToRequestedSize( + fs, + vm, + vmId, + undefined, + span, + { + cpu: options.imageSize.cpu, + memory: options.imageSize.memoryMb, + storage: Math.max(VM_DISK_MB_DEFAULT, options.imageSize.storageMb, vmDiskMb()), + }, + ); } else { // A size-less image boots at its snapshot's resources and only a - // grow-only resize raises them. Size first so the machine the - // daemon comes up on is the one that was sold. + // grow-only resize raises them. Size first so the daemon comes + // up on the provider profile requested by the server. await this.growToRequestedSize(fs, vm, vmId, options.memoryMb, span); } // The baked supervisor is already bringing the daemon up; the only @@ -853,7 +869,7 @@ export class FreestyleProvider implements VMProvider { } catch (err) { // A VM that failed to size or configure must not survive as an // orphan, and an undersized machine must not ship as if it were - // the plan machine. + // the provider sizing profile. await vm.delete().catch((cleanupErr) => { console.error(`[freestyle] create rollback failed; VM ${vmId} may be orphaned`, cleanupErr); }); @@ -884,8 +900,8 @@ export class FreestyleProvider implements VMProvider { } /** - * Grow the VM to the requested memory (the plan machine when the caller - * sent none), the vCPUs that memory implies, and the plan disk. Freestyle + * Grow the VM to the requested memory (the provider profile when the caller + * sent none), the vCPUs that memory implies, and the starting disk. Freestyle * resize is grow-only, so only larger dimensions are sent; a snapshot that * already carries the size is a no-op. */ @@ -895,9 +911,10 @@ export class FreestyleProvider implements VMProvider { vmId: string, memoryMb: number | undefined, span: Parameters[0], + targetResources?: VmResources, ): Promise { const current = (await fs.vms.get(vmId)).resources; - const target = freestyleTargetResources(memoryMb ?? PLAN_MACHINE_MEMORY_MB); + const target = targetResources ?? freestyleTargetResources(memoryMb ?? PLAN_MACHINE_MEMORY_MB); const request = freestyleResizeRequest(current, target); setSpanAttributes(span, { "cmux.vm.resources.cpu": target.cpu, @@ -1363,7 +1380,7 @@ export class FreestyleProvider implements VMProvider { } } -/** The resources a machine of `memoryMb` is sold with (see entitlements.ts). */ +/** The provider resources each machine receives for `memoryMb` (see entitlements.ts). */ export function freestyleTargetResources( memoryMb: number, env: Record = process.env, diff --git a/web/services/vms/entitlements.ts b/web/services/vms/entitlements.ts index baf7ea3d554f..b1e4cf9bee0a 100644 --- a/web/services/vms/entitlements.ts +++ b/web/services/vms/entitlements.ts @@ -15,6 +15,22 @@ export { VM_DISK_MB_MAX, VM_DISK_MB_STEP, VM_MEMORY_MB_PER_VCPU, + PLAN_SHARED_VCPU, + PLAN_SHARED_MEMORY_MB, + PLAN_SHARED_DISK_MB, + PLAN_SHARED_RESOURCE_CAPACITY, + DEFAULT_VM_RESOURCE_RESERVATION, + VM_RESOURCE_RESERVATION_METADATA_KEY, + VM_RESOURCE_FORK_PENDING_METADATA_KEY, + vmResourceForkPendingFromMetadata, + sharedResourceCapacityForMaxActiveVms, + firstExceededSharedResource, + sharedResourceUsage, + vmResourceReservationForCreate, + vmResourceReservationFromMetadata, + vmResourceResizePendingFromMetadata, + hasVmResourceReservationMetadata, + withVmResourceReservationMetadata, vcpusForMemoryMb, vmDiskMb, } from "./machineSpec"; @@ -148,6 +164,8 @@ function resolveBillingContext( * 4/16, 8/32, 16/64, 24/96, 32/128, and 64/128 (memory/disk in GB). vCPUs * follow memory (vcpusForMemoryMb). The server owns this list so clients show * valid sizes. BusyBox's 128 MiB image is a bootstrap image, not a coding VM. + * The paid plan's 5 vCPU, 20 GB RAM, and 200 GB disk entitlement is a shared + * pool enforced by the repository, not a per-VM size profile. */ export const VM_MEMORY_OPTIONS_MB: readonly number[] = [4096, 8192, 16384, 24576, 32768, 65536]; diff --git a/web/services/vms/errors.ts b/web/services/vms/errors.ts index aa1d7bcdc5b2..d428660992bc 100644 --- a/web/services/vms/errors.ts +++ b/web/services/vms/errors.ts @@ -33,6 +33,11 @@ export class VmResizeInvalidError extends Data.TaggedError("VmResizeInvalidError readonly reason: "below_current" | "above_max"; }> {} +/** A grow-only disk resize is already running for this machine. */ +export class VmResizeInProgressError extends Data.TaggedError("VmResizeInProgressError")<{ + readonly vmId: string; +}> {} + /** * A private-network or tunnel operation on a deployment that does not serve * one — the provider has no `privateNetworking`, or @@ -107,6 +112,17 @@ export class VmLimitExceededError extends Data.TaggedError("VmLimitExceededError readonly limit: number; }> {} +/** A create or resize would exceed the plan's aggregate Cloud VM pool. */ +export class VmSharedResourceLimitExceededError extends Data.TaggedError("VmSharedResourceLimitExceededError")<{ + readonly kind: "shared_resources"; + readonly billingTeamId: string; + readonly phase?: "create" | "resize"; + readonly resource: "vcpus" | "memoryMb" | "diskMb"; + readonly used: number; + readonly requested: number; + readonly limit: number; +}> {} + export class VmCreateCreditsInsufficientError extends Data.TaggedError("VmCreateCreditsInsufficientError")<{ readonly itemId: string; readonly billingCustomerId: string; @@ -169,6 +185,7 @@ export type VmWorkflowError = | VmOperationUnsupportedError | VmNotFoundError | VmResizeInvalidError + | VmResizeInProgressError | VmSnapshotNotFoundError | VmFreeAccessExpiredError | VmCreateInProgressError @@ -177,6 +194,7 @@ export type VmWorkflowError = | VmAccountDeletionInProgressError | VmImageConfigError | VmLimitExceededError + | VmSharedResourceLimitExceededError | VmCreateCreditsInsufficientError | VmBillingError | VmAttachTransportUnsupportedError @@ -203,6 +221,10 @@ export function isVmResizeInvalidError(err: unknown): err is VmResizeInvalidErro return (err as { _tag?: string } | null)?._tag === "VmResizeInvalidError"; } +export function isVmResizeInProgressError(err: unknown): err is VmResizeInProgressError { + return (err as { _tag?: string } | null)?._tag === "VmResizeInProgressError"; +} + export function isVmSnapshotNotFoundError(err: unknown): err is VmSnapshotNotFoundError { return (err as { _tag?: string } | null)?._tag === "VmSnapshotNotFoundError"; } @@ -237,6 +259,12 @@ export function isVmLimitExceededError(err: unknown): err is VmLimitExceededErro return (err as { _tag?: string } | null)?._tag === "VmLimitExceededError"; } +export function isVmSharedResourceLimitExceededError( + err: unknown, +): err is VmSharedResourceLimitExceededError { + return (err as { _tag?: string } | null)?._tag === "VmSharedResourceLimitExceededError"; +} + export function isVmCreateCreditsInsufficientError(err: unknown): err is VmCreateCreditsInsufficientError { return (err as { _tag?: string } | null)?._tag === "VmCreateCreditsInsufficientError"; } @@ -282,6 +310,7 @@ const vmWorkflowErrorTagRecord = { VmOperationUnsupportedError: true, VmNotFoundError: true, VmResizeInvalidError: true, + VmResizeInProgressError: true, VmSnapshotNotFoundError: true, VmFreeAccessExpiredError: true, VmCreateInProgressError: true, @@ -290,6 +319,7 @@ const vmWorkflowErrorTagRecord = { VmAccountDeletionInProgressError: true, VmImageConfigError: true, VmLimitExceededError: true, + VmSharedResourceLimitExceededError: true, VmCreateCreditsInsufficientError: true, VmBillingError: true, VmAttachTransportUnsupportedError: true, diff --git a/web/services/vms/machineSpec.ts b/web/services/vms/machineSpec.ts index d6b2204d4bb7..e2bc4d83c963 100644 --- a/web/services/vms/machineSpec.ts +++ b/web/services/vms/machineSpec.ts @@ -1,14 +1,27 @@ /** - * The plan machine, as sold on /pricing: every paid plan gets up to - * PAID_MAX_ACTIVE_VMS_DEFAULT machines. Machine size options are defined in - * `entitlements.ts`; this constant is the default 8 GB / 32 GB tier. + * The plan machine and the plan-wide Cloud VM resource policy. Every paid + * plan gets up to PAID_MAX_ACTIVE_VMS_DEFAULT machines. Machine size options + * are defined in `entitlements.ts`; the default is the 8 GB / 32 GB tier. * - * Kept dependency-free so the provider drivers can size a machine without - * pulling the billing graph into their module. + * The count allowance and the resource pool are separate limits. Postgres + * records each machine's reservation, and the VM repository checks the live + * claims while holding the billing-team lock. CPU and memory are shared + * ceilings, so every live claim adds to the pool. A provider image can be + * overprovisioned to the nearest baked shape; that physical shape is not extra + * plan capacity. Keeping the + * policy here gives pricing tests, workflows, and provider sizing one source of + * truth. + * + * This module stays dependency-free so provider drivers can size a machine + * without pulling the billing graph into their module. */ export const PAID_MAX_ACTIVE_VMS_DEFAULT = 50; export const PLAN_MACHINE_MEMORY_MB = 8192; export const VM_MEMORY_MB_PER_VCPU = 4096; +export const PLAN_SHARED_VCPU = 5; +export const PLAN_SHARED_MEMORY_MB = 20 * 1024; +export const PLAN_SHARED_DISK_MB = 200 * 1024; + /** New machines start with this disk. Freestyle resizes disks grow-only. */ export const VM_DISK_MB_DEFAULT = 32768; /** Freestyle Pro's documented per-VM disk ceiling. */ @@ -16,11 +29,317 @@ export const VM_DISK_MB_MAX = 262144; /** User-facing disk sizes are aligned to whole GiB steps. */ export const VM_DISK_MB_STEP = 4096; +export type VmResourceReservation = { + readonly vcpus: number; + readonly memoryMb: number; + readonly diskMb: number; +}; + +export type VmSharedResourceName = keyof VmResourceReservation; + +/** + * Bounds for provider-reported machine dimensions. CPU and memory match the + * supported image ladder; disk also permits the grow-only resize ceiling. + * Callers must use their conservative fallback when a provider value is + * outside these bounds instead of treating it as a real machine shape. + */ +export const VM_PROVIDER_RESOURCE_BOUNDS = { + vcpus: { min: 1, max: 32 }, + memoryMb: { min: 4 * 1024, max: 64 * 1024 }, + diskMb: { min: 16 * 1024, max: VM_DISK_MB_MAX }, +} as const satisfies Record; + +/** Read one provider dimension only when it is a supported machine shape. */ +export function vmProviderResourceSize( + resource: VmSharedResourceName, + value: unknown, +): number | null { + const bounds = VM_PROVIDER_RESOURCE_BOUNDS[resource]; + return typeof value === "number" && Number.isSafeInteger(value) && + value >= bounds.min && value <= bounds.max + ? value + : null; +} + +export type VmImageResourceShape = { + readonly cpu: number; + readonly memoryMb: number; + readonly storageMb: number; +}; + +export const PLAN_SHARED_RESOURCE_CAPACITY: VmResourceReservation = { + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: PLAN_SHARED_DISK_MB, +}; + +/** The reservation used for rows written before resource metadata existed. */ +export const DEFAULT_VM_RESOURCE_RESERVATION: VmResourceReservation = { + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: VM_DISK_MB_DEFAULT, +}; + +export const VM_RESOURCE_RESERVATION_METADATA_KEY = "cmuxResourceReservation"; +/** Internal marker for a resize that still holds conservative disk headroom. + * The reservation remains valid while this marker is present, so reconciliation + * cannot lower the claim during the provider call. + */ +export const VM_RESOURCE_RESIZE_PENDING_METADATA_KEY = "cmuxResourceResizePending"; +/** Internal marker for a completed resize whose provider size is not confirmed yet. */ +export const VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY = "cmuxResourceResizeUnconfirmed"; +/** Internal marker that postpones a legacy resource read until a later pass. */ +export const VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY = "cmuxResourceReconcileRetry"; +/** Internal marker for a native fork claim awaiting provider-confirmed shape. */ +export const VM_RESOURCE_FORK_PENDING_METADATA_KEY = "cmuxResourceForkPending"; + +/** Read the minimum source shape stored on a pending native fork marker. */ +export function vmResourceForkPendingFromMetadata( + metadata: Record | null | undefined, +): VmResourceReservation | null { + return resourceReservationFromValue(metadata?.[VM_RESOURCE_FORK_PENDING_METADATA_KEY]); +} + /** vCPUs a machine of `memoryMb` gets: one per 4 GB, rounded up. */ export function vcpusForMemoryMb(memoryMb: number): number { return Math.max(1, Math.ceil(memoryMb / VM_MEMORY_MB_PER_VCPU)); } +/** + * The provider shape represented by a create reservation. Callers that enforce + * the paid shared pool intentionally omit `imageSize` and reserve the logical + * plan profile because a baked image may be larger than that entitlement. + */ +export function vmResourceReservationForCreate(input: { + readonly memoryMb?: number; + readonly imageSize?: VmImageResourceShape | null; + readonly env?: Record; +} = {}): VmResourceReservation { + if (input.imageSize) { + const configuredDiskMb = vmDiskMb(input.env); + const imageReservation = normalizeResourceReservation({ + vcpus: input.imageSize.cpu, + memoryMb: input.imageSize.memoryMb, + diskMb: input.imageSize.storageMb, + }); + // A resolver can provide both the caller's requested memory and the baked + // image selected to satisfy it. CPU and memory stay logical entitlement + // claims. The provider grows a small baked image to the documented starting + // disk, or the operator override, so the reservation must include the + // effective provider disk size. + if (input.memoryMb !== undefined) { + return normalizeResourceReservation({ + vcpus: vcpusForMemoryMb(input.memoryMb), + memoryMb: input.memoryMb, + diskMb: Math.max(configuredDiskMb, imageReservation.diskMb), + }); + } + return { + ...imageReservation, + diskMb: Math.max(configuredDiskMb, imageReservation.diskMb), + }; + } + const memoryMb = input.memoryMb ?? PLAN_MACHINE_MEMORY_MB; + return normalizeResourceReservation({ + vcpus: vcpusForMemoryMb(memoryMb), + memoryMb, + diskMb: vmDiskMb(input.env), + }); +} + +/** + * Team plans multiply both the VM allowance and its shared pool by paid seat. + * Operator limits below one base allowance still keep the base pool, while a + * larger allowance gets one pool per 50-machine block. + */ +export function sharedResourceCapacityForMaxActiveVms( + maxActiveVms: number | null | undefined, +): VmResourceReservation { + const blocks = maxActiveVms !== null && maxActiveVms !== undefined && maxActiveVms > 0 + ? Math.max(1, Math.ceil(maxActiveVms / PAID_MAX_ACTIVE_VMS_DEFAULT)) + : 1; + return { + vcpus: PLAN_SHARED_VCPU * blocks, + memoryMb: PLAN_SHARED_MEMORY_MB * blocks, + diskMb: PLAN_SHARED_DISK_MB * blocks, + }; +} + +/** Return the first resource for which a shared claim would exceed the pool. */ +export function firstExceededSharedResource(input: { + readonly used: VmResourceReservation; + readonly requested: VmResourceReservation; + readonly capacity: VmResourceReservation; +}): { + readonly resource: VmSharedResourceName; + readonly used: number; + readonly requested: number; + readonly limit: number; +} | null { + for (const resource of ["vcpus", "memoryMb", "diskMb"] as const) { + const used = input.used[resource]; + const requested = input.requested[resource]; + const limit = input.capacity[resource]; + const projected = sharedResourceUsage(resource, used, requested); + if (projected > limit) return { resource, used, requested, limit }; + } + return null; +} + +/** Every resource claim adds to the account-wide shared pool. */ +export function sharedResourceUsage( + resource: VmSharedResourceName, + used: number, + requested: number, +): number { + return used + requested; +} + +export type VmResourceResizePending = { + /** Unique request generation used to protect confirmation and rollback. */ + readonly operationId: string; + readonly requestedDiskMb: number; + readonly previousDiskMb: number; + /** Millisecond timestamp used to recover a worker that died before provider I/O. */ + readonly createdAtMs?: number; +}; + +/** Read a validated in-flight resize marker from provider metadata. */ +export function vmResourceResizePendingFromMetadata( + metadata: Record | null | undefined, +): VmResourceResizePending | null { + const raw = metadata?.[VM_RESOURCE_RESIZE_PENDING_METADATA_KEY]; + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null; + const candidate = raw as Record; + const operationId = candidate.operationId; + const requestedDiskMb = candidate.requestedDiskMb; + const previousDiskMb = candidate.previousDiskMb; + const createdAtMs = candidate.createdAtMs; + if ( + typeof operationId !== "string" || + operationId.trim().length === 0 || + operationId.length > 200 || + !isPositiveSafeInteger(requestedDiskMb) || + !isPositiveSafeInteger(previousDiskMb) || + (createdAtMs !== undefined && !isPositiveSafeInteger(createdAtMs)) + ) return null; + return { + operationId: operationId.trim(), + requestedDiskMb, + previousDiskMb, + ...(createdAtMs === undefined ? {} : { createdAtMs }), + }; +} + +export type VmResourceResizeUnconfirmed = { + /** Unique request generation used to protect reconciliation from stale reads. */ + readonly operationId: string; + /** Minimum provider size expected after the completed resize. */ + readonly requestedDiskMb: number; + /** Disk claim to restore when the provider never reaches the request. */ + readonly previousDiskMb?: number; + /** Millisecond timestamp at which conservative recovery started. */ + readonly markedAtMs?: number; +}; + +/** Read a validated completed-resize marker awaiting provider stats. */ +export function vmResourceResizeUnconfirmedFromMetadata( + metadata: Record | null | undefined, +): VmResourceResizeUnconfirmed | null { + const raw = metadata?.[VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY]; + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null; + const candidate = raw as Record; + const operationId = candidate.operationId; + const requestedDiskMb = candidate.requestedDiskMb; + const previousDiskMb = candidate.previousDiskMb; + const markedAtMs = candidate.markedAtMs; + if ( + typeof operationId !== "string" || + operationId.trim().length === 0 || + operationId.length > 200 || + !isPositiveSafeInteger(requestedDiskMb) || + (previousDiskMb !== undefined && !isPositiveSafeInteger(previousDiskMb)) || + (markedAtMs !== undefined && !isPositiveSafeInteger(markedAtMs)) + ) return null; + return { + operationId: operationId.trim(), + requestedDiskMb, + ...(previousDiskMb === undefined ? {} : { previousDiskMb }), + ...(markedAtMs === undefined ? {} : { markedAtMs }), + }; +} + +export type VmResourceReconcileRetry = { + /** Unix epoch milliseconds at which the row may be attempted again. */ + readonly nextAttemptAtMs: number; +}; + +/** Read a validated retry marker for background resource reconciliation. */ +export function vmResourceReconcileRetryFromMetadata( + metadata: Record | null | undefined, +): VmResourceReconcileRetry | null { + const raw = metadata?.[VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY]; + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null; + const candidate = raw as Record; + const nextAttemptAtMs = candidate.nextAttemptAtMs; + if (!isPositiveSafeInteger(nextAttemptAtMs)) return null; + return { nextAttemptAtMs }; +} + +function isPositiveSafeInteger(value: unknown): value is number { + return typeof value === "number" && Number.isSafeInteger(value) && value > 0; +} + +/** Read a persisted reservation, falling back safely for legacy VM rows. */ +export function vmResourceReservationFromMetadata( + metadata: Record | null | undefined, + fallback: VmResourceReservation = DEFAULT_VM_RESOURCE_RESERVATION, +): VmResourceReservation { + const raw = metadata?.[VM_RESOURCE_RESERVATION_METADATA_KEY]; + return resourceReservationFromValue(raw) ?? fallback; +} + +/** Whether a row has a complete, validated reservation marker. */ +export function hasVmResourceReservationMetadata( + metadata: Record | null | undefined, +): boolean { + return resourceReservationFromValue(metadata?.[VM_RESOURCE_RESERVATION_METADATA_KEY]) !== null; +} + +/** Merge a reservation into provider metadata without exposing mutable input. */ +export function withVmResourceReservationMetadata( + metadata: Record | null | undefined, + reservation: VmResourceReservation, +): Record { + return { + ...(metadata ?? {}), + [VM_RESOURCE_RESERVATION_METADATA_KEY]: { ...reservation }, + }; +} + +function normalizeResourceReservation(input: VmResourceReservation): VmResourceReservation { + for (const [name, value] of Object.entries(input)) { + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`${name} must be a positive integer`); + } + } + return input; +} + +function resourceReservationFromValue(value: unknown): VmResourceReservation | null { + if (!value || typeof value !== "object" || Array.isArray(value)) return null; + const candidate = value as Record; + const vcpus = candidate.vcpus; + const memoryMb = candidate.memoryMb; + const diskMb = candidate.diskMb; + if ( + typeof vcpus !== "number" || !Number.isSafeInteger(vcpus) || vcpus <= 0 || + typeof memoryMb !== "number" || !Number.isSafeInteger(memoryMb) || memoryMb <= 0 || + typeof diskMb !== "number" || !Number.isSafeInteger(diskMb) || diskMb <= 0 + ) return null; + return { vcpus, memoryMb, diskMb }; +} + /** Disk every machine is grown to at create, in MB. Env-overridable. */ export function vmDiskMb(env: Record = process.env): number { const raw = (env.CMUX_VM_DISK_MB ?? String(VM_DISK_MB_DEFAULT)).trim(); diff --git a/web/services/vms/repository.ts b/web/services/vms/repository.ts index 4ef47f10d562..400b0e0e1430 100644 --- a/web/services/vms/repository.ts +++ b/web/services/vms/repository.ts @@ -1,3 +1,4 @@ +import { randomUUID } from "node:crypto"; import { and, asc, count, desc, eq, gt, inArray, isNotNull, isNull, lt, ne, or, sql } from "drizzle-orm"; import * as Context from "effect/Context"; import * as Effect from "effect/Effect"; @@ -29,12 +30,36 @@ import { VmAccountDeletionInProgressError, VmDatabaseError, VmLimitExceededError, + VmResizeInProgressError, + VmSharedResourceLimitExceededError, LEGACY_MODEL_PLANE_ENTITLEMENT_FAILURE_CODE, VM_MODEL_PLANE_FAILURE_CODES, isVmAccountDeletionInProgressError, isVmCreateDisabledError, isVmLimitExceededError, + isVmResizeInProgressError, + isVmSharedResourceLimitExceededError, } from "./errors"; +import { + DEFAULT_VM_RESOURCE_RESERVATION, + PLAN_SHARED_DISK_MB, + VM_DISK_MB_DEFAULT, + VM_DISK_MB_MAX, + VM_RESOURCE_RESERVATION_METADATA_KEY, + VM_RESOURCE_FORK_PENDING_METADATA_KEY, + VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY, + VM_RESOURCE_RESIZE_PENDING_METADATA_KEY, + VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY, + firstExceededSharedResource, + hasVmResourceReservationMetadata, + sharedResourceCapacityForMaxActiveVms, + vmResourceReservationFromMetadata, + vmResourceReconcileRetryFromMetadata, + vmResourceResizePendingFromMetadata, + vmResourceResizeUnconfirmedFromMetadata, + withVmResourceReservationMetadata, + type VmResourceReservation, +} from "./machineSpec"; export type CloudVmRow = typeof cloudVms.$inferSelect; export type CloudVmBaseRow = typeof cloudVmBases.$inferSelect; @@ -52,11 +77,21 @@ export type CloudVmSessionRow = typeof cloudVmSessions.$inferSelect; export type CloudVmNetworkRow = typeof cloudVmNetworks.$inferSelect; export type CloudVmTunnelRow = typeof cloudVmTunnels.$inferSelect; export type CloudVmLeaseKind = typeof cloudVmLeases.$inferInsert.kind; +export type VmResourceReservationInput = VmResourceReservation; +export type VmResizeReservation = { + readonly previousDiskMb: number; + readonly reservedDiskMb: number; + /** The requested claim, below the temporary headroom hold. */ + readonly requestedDiskMb?: number; + /** Unique resize generation used by confirmation and rollback. */ + readonly operationId: string; +}; export type CloudVmStatus = CloudVmRow["status"]; export type CloudVmSessionStatus = CloudVmSessionRow["status"]; // Reaper batches are capped at 100. Keep repository calls bounded even if a // future caller passes a malformed or oversized name list. const VM_REAPER_REFERENCE_NAME_LIMIT = 100; +const LIVE_VM_RESOURCE_STATUSES = ["provisioning", "running", "paused"] as const; export type BeginCreateResult = | { readonly inserted: true; readonly vm: CloudVmRow } @@ -164,7 +199,14 @@ export type VmRepositoryShape = { readonly imageVersion?: string | null; readonly maxActiveVms: number | null; readonly idempotencyKey?: string; - }) => Effect.Effect; + /** Provider resources reserved against the plan-wide pool. */ + readonly resourceReservation?: VmResourceReservation; + readonly sharedResourceCapacity?: VmResourceReservation; + /** Hold every remaining pool dimension while a provider-side clone runs. */ + readonly reserveSharedResourceHeadroom?: boolean; + /** Minimum source shape retained when a temporary fork claim is recovered. */ + readonly forkMinimumResourceReservation?: VmResourceReservation; + }) => Effect.Effect; readonly beginBaseOpen: (input: { readonly userId: string; readonly billingTeamId: string; @@ -175,7 +217,9 @@ export type VmRepositoryShape = { readonly imageVersion?: string | null; readonly maxActiveVms: number | null; readonly baseName?: string; - }) => Effect.Effect; + readonly resourceReservation?: VmResourceReservation; + readonly sharedResourceCapacity?: VmResourceReservation; + }) => Effect.Effect; readonly beginBaseReset: (input: { readonly userId: string; readonly billingTeamId: string; @@ -187,7 +231,9 @@ export type VmRepositoryShape = { readonly maxActiveVms: number | null; readonly baseName?: string; readonly reason?: string | null; - }) => Effect.Effect, VmCreateDisabledError | VmAccountDeletionInProgressError | VmCreateInProgressError | VmDatabaseError | VmLimitExceededError>; + readonly resourceReservation?: VmResourceReservation; + readonly sharedResourceCapacity?: VmResourceReservation; + }) => Effect.Effect, VmCreateDisabledError | VmAccountDeletionInProgressError | VmCreateInProgressError | VmDatabaseError | VmLimitExceededError | VmSharedResourceLimitExceededError>; readonly markBaseCreateRunning: (input: { readonly baseId: string; readonly generation: number; @@ -212,6 +258,32 @@ export type VmRepositoryShape = { /** Maximum number of rows to inspect in the synchronous limit retry. */ readonly limit: number; }) => Effect.Effect; + /** Live rows whose resource claim predates the shared-pool marker. */ + readonly legacyResourceReservationCandidates?: (input: { + /** Optional owner scope. Omit both fields for the background migration batch. */ + readonly userId?: string; + readonly billingTeamId?: string | null; + /** Keep provider reconciliation bounded. */ + readonly limit: number; + }) => Effect.Effect; + /** Defer a legacy resource read without losing its place in the batch. */ + readonly deferResourceReservation?: (input: { + readonly id: string; + readonly nextAttemptAt: Date; + }) => Effect.Effect; + /** Persist a provider-confirmed claim for a legacy VM row. */ + readonly setResourceReservation?: (input: { + readonly id: string; + readonly reservation: VmResourceReservation; + /** Replace this exact temporary claim, used by native fork finalization. */ + readonly expectedReservation?: VmResourceReservation; + /** Recheck the replacement against the shared pool while holding the team lock. */ + readonly sharedResourceCapacity?: VmResourceReservation; + /** Clear a pending resize only when this generation owns it. */ + readonly expectedResizeOperationId?: string; + /** Clear an unconfirmed resize only when this generation owns it. */ + readonly expectedResizeUnconfirmedOperationId?: string; + }) => Effect.Effect; readonly reservePausedResume: (input: { readonly id: string; readonly userId: string; @@ -219,6 +291,49 @@ export type VmRepositoryShape = { readonly providerVmId: string; readonly maxActiveVms: number | null; }) => Effect.Effect; + /** Reserve a grow-only disk change before provider I/O. Live shape always provides this. */ + readonly reserveVmResize?: (input: { + readonly id: string; + readonly userId: string; + readonly billingTeamId?: string | null; + readonly providerVmId: string; + /** Provider-confirmed current disk, used to repair legacy reservations. */ + readonly currentDiskMb?: number; + readonly storageMb: number; + readonly maxActiveVms?: number | null; + readonly sharedResourceCapacity?: VmResourceReservation; + }) => Effect.Effect; + /** Persist the provider-confirmed disk claim after a successful resize. */ + readonly confirmVmResize?: (input: { + readonly id: string; + /** The claim written before provider I/O. A newer claim wins the race. */ + readonly expectedDiskMb: number; + /** The requested claim; the temporary headroom hold may be larger. */ + readonly minimumDiskMb?: number; + readonly confirmedDiskMb: number; + /** Unique generation returned by reserveVmResize. */ + readonly operationId: string; + }) => Effect.Effect; + /** Persist a conservative claim while a completed resize awaits provider stats. */ + readonly markVmResizeUnconfirmed?: (input: { + readonly id: string; + /** The claim written before provider I/O. A newer claim wins the race. */ + readonly expectedDiskMb: number; + /** The requested claim; the temporary headroom hold may be larger. */ + readonly minimumDiskMb?: number; + /** The claim to restore if the provider never reaches the request. */ + readonly previousDiskMb: number; + /** Unique generation returned by reserveVmResize. */ + readonly operationId: string; + }) => Effect.Effect; + /** Restore a reservation when the provider rejected the resize. */ + readonly restoreVmResize?: (input: { + readonly id: string; + readonly expectedDiskMb: number; + readonly previousDiskMb: number; + /** Unique generation returned by reserveVmResize. */ + readonly operationId: string; + }) => Effect.Effect; readonly reconciliationCandidates: (input: { readonly limit: number; }) => Effect.Effect; @@ -265,6 +380,13 @@ export type VmRepositoryShape = { readonly provider: ProviderId; readonly snapshotId: string; }) => Effect.Effect; + /** Return the durable resource claim for an owned snapshot, or null when absent. */ + readonly ownedSnapshotResourceReservation?: (input: { + readonly userId: string; + readonly billingTeamId?: string | null; + readonly provider: ProviderId; + readonly snapshotId: string; + }) => Effect.Effect; readonly findUserVm: (input: { readonly userId: string; readonly billingTeamId?: string | null; @@ -489,6 +611,290 @@ function accountScopeWhere(input: { return eq(cloudVms.billingTeamId, billingTeamId); } +/** A safe SQL expression for one reservation field, including legacy rows. */ +function reservedResourceField( + key: "vcpus" | "memoryMb" | "diskMb", + fallback: number, +) { + const keySql = sql.raw(`'${key}'`); + const reservationKeySql = sql.raw(`'${VM_RESOURCE_RESERVATION_METADATA_KEY}'`); + const value = sql`${cloudVms.providerMetadata}->${reservationKeySql}->>${keySql}`; + // Provider metadata is not trusted input. Keep malformed or out-of-range + // legacy values from turning a quota read into a database cast failure. + return sql`case + when coalesce(${value}, '') ~ '^[0-9]+$' + and length(${value}) <= 10 + and (${value})::numeric between 1 and 2147483647 + then (${value})::integer + else ${fallback} + end`; +} + +function positiveReservationInteger(value: unknown): number | null { + return typeof value === "number" && Number.isSafeInteger(value) && value > 0 + ? value + : null; +} + +function reservedResourceFields() { + return { + vcpus: reservedResourceField("vcpus", DEFAULT_VM_RESOURCE_RESERVATION.vcpus), + memoryMb: reservedResourceField("memoryMb", DEFAULT_VM_RESOURCE_RESERVATION.memoryMb), + // A legacy row can already have a disk larger than the 32 GB starting + // profile. Until a provider-confirmed claim is recorded, reserve the + // per-VM maximum so a quota read cannot undercount persistent storage. + diskMb: reservedResourceField("diskMb", VM_DISK_MB_MAX), + }; +} + +/** SQL predicate for a complete, bounded reservation marker. */ +function validResourceReservationMarkerSql() { + const markerKey = sql.raw(`'${VM_RESOURCE_RESERVATION_METADATA_KEY}'`); + const marker = sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb)->${markerKey}`; + const field = (key: "vcpus" | "memoryMb" | "diskMb") => + sql`${marker}->>${sql.raw(`'${key}'`)}`; + const boundedPositiveInteger = (value: ReturnType) => sql` + ${value} ~ '^[1-9][0-9]{0,9}$' + and (length(${value}) < 10 or ${value} <= '2147483647')`; + return sql`jsonb_typeof(${marker}) = 'object' + and ${boundedPositiveInteger(field("vcpus"))} + and ${boundedPositiveInteger(field("memoryMb"))} + and ${boundedPositiveInteger(field("diskMb"))}`; +} + +/** Compare a control-plane reservation marker field by field for CAS updates. */ +function resourceReservationMarkerEqualsSql(expected: VmResourceReservation) { + const markerKey = sql.raw(`'${VM_RESOURCE_RESERVATION_METADATA_KEY}'`); + const marker = sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb)->${markerKey}`; + const field = (key: "vcpus" | "memoryMb" | "diskMb") => + sql`${marker}->>${sql.raw(`'${key}'`)}`; + return sql` + ${field("vcpus")} = ${String(expected.vcpus)} + and ${field("memoryMb")} = ${String(expected.memoryMb)} + and ${field("diskMb")} = ${String(expected.diskMb)}`; +} + +async function reservedResourceTotals( + tx: CloudDbTransaction, + input: { + readonly userId: string; + readonly billingTeamId?: string | null; + readonly excludeVmId?: string; + }, +): Promise { + // Every resource is additive across the account's live machines. Personal + // rows may have a NULL billing_team_id, + // so use the same account scope predicate as ownership and list queries + // instead of matching a synthetic user id in the team column. + const fields = reservedResourceFields(); + const predicates = [ + inArray(cloudVms.status, LIVE_VM_RESOURCE_STATUSES), + accountScopeWhere({ userId: input.userId, billingTeamId: input.billingTeamId }), + ]; + if (input.excludeVmId) predicates.push(ne(cloudVms.id, input.excludeVmId)); + const [row] = await tx + .select({ + vcpus: sql`coalesce(sum(${fields.vcpus}), 0)`, + memoryMb: sql`coalesce(sum(${fields.memoryMb}), 0)`, + diskMb: sql`coalesce(sum(${fields.diskMb}), 0)`, + }) + .from(cloudVms) + .where(and(...predicates)); + return { + vcpus: Number(row?.vcpus ?? 0), + memoryMb: Number(row?.memoryMb ?? 0), + diskMb: Number(row?.diskMb ?? 0), + }; +} + +function resourceReservationForInput( + reservation: VmResourceReservation | undefined, +): VmResourceReservation { + return reservation ?? DEFAULT_VM_RESOURCE_RESERVATION; +} + +/** + * Only paid/shared create paths have measured or intentionally logical claims. + * A free-provisioning row has no resource promise, so leave its marker absent + * until a provider read can measure the actual shape after an upgrade. + */ +function reservationMetadataForInput( + reservation: VmResourceReservation | undefined, + sharedResourceCapacity: VmResourceReservation | undefined, + reserveSharedResourceHeadroom = false, + forkMinimumReservation?: VmResourceReservation, +): Record { + if (reservation || sharedResourceCapacity) { + const metadata = reservationMetadata(resourceReservationForInput(reservation)); + return reserveSharedResourceHeadroom + ? { + ...metadata, + [VM_RESOURCE_FORK_PENDING_METADATA_KEY]: forkMinimumReservation ?? resourceReservationForInput(reservation), + } + : metadata; + } + return {}; +} + +function sharedResourceCapacityForInput( + maxActiveVms: number | null, + capacity: VmResourceReservation | undefined, +): VmResourceReservation { + return capacity ?? sharedResourceCapacityForMaxActiveVms(maxActiveVms); +} + +async function checkedSharedResourceReservation( + tx: CloudDbTransaction, + input: { + readonly userId: string; + readonly billingTeamId: string; + readonly maxActiveVms: number | null; + readonly resourceReservation?: VmResourceReservation; + readonly sharedResourceCapacity?: VmResourceReservation; + readonly excludeVmId?: string; + readonly phase?: "create" | "resize"; + readonly reserveSharedResourceHeadroom?: boolean; + }, +): Promise { + // The shared pool is supplied explicitly by paid-plan workflows. Keep the + // repository compatible with the controlled free-provisioning escape hatch + // and with legacy callers that only use the active-count entitlement. + if (input.resourceReservation === undefined && input.sharedResourceCapacity === undefined) return null; + const reservation = resourceReservationForInput(input.resourceReservation); + const capacity = sharedResourceCapacityForInput(input.maxActiveVms, input.sharedResourceCapacity); + const used = await reservedResourceTotals(tx, { + userId: input.userId, + billingTeamId: input.billingTeamId, + excludeVmId: input.excludeVmId, + }); + const exceeded = firstExceededSharedResource({ used, requested: reservation, capacity }); + if (exceeded) { + throw new VmSharedResourceLimitExceededError({ + kind: "shared_resources", + billingTeamId: input.billingTeamId, + phase: input.phase, + resource: exceeded.resource, + used: exceeded.used, + requested: exceeded.requested, + limit: exceeded.limit, + }); + } + if (!input.reserveSharedResourceHeadroom) return reservation; + // A native provider clone runs outside this transaction. Claim the complete + // remaining pool while it copies the source so a concurrent create or resize + // cannot consume capacity needed by the copy's final measured shape. The + // requested shape remains a floor, and the capacity check above guarantees + // every computed headroom value is non-negative. + return { + vcpus: Math.max(reservation.vcpus, capacity.vcpus - used.vcpus), + memoryMb: Math.max(reservation.memoryMb, capacity.memoryMb - used.memoryMb), + diskMb: Math.max(reservation.diskMb, capacity.diskMb - used.diskMb), + }; +} + +async function assertSharedResourceCapacity( + tx: CloudDbTransaction, + input: { + readonly userId: string; + readonly billingTeamId: string; + readonly maxActiveVms: number | null; + readonly resourceReservation?: VmResourceReservation; + readonly sharedResourceCapacity?: VmResourceReservation; + readonly excludeVmId?: string; + readonly phase?: "create" | "resize"; + }, +): Promise { + await checkedSharedResourceReservation(tx, input); +} + +function reservationMetadata(reservation: VmResourceReservation): Record { + return withVmResourceReservationMetadata({}, reservation); +} + +/** Provider responses cannot write control-plane reservation markers. */ +function providerMetadataPatchForPersistence( + metadata: Record | null | undefined, +): Record { + return Object.fromEntries( + Object.entries(metadata ?? {}).filter(([key]) => + key !== VM_RESOURCE_RESERVATION_METADATA_KEY && + key !== VM_RESOURCE_FORK_PENDING_METADATA_KEY && + key !== VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY && + key !== VM_RESOURCE_RESIZE_PENDING_METADATA_KEY && + key !== VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY, + ), + ); +} + +/** Build trusted numeric claim JSON without binding a JSON string as a JSON scalar. */ +function reservationMetadataJsonb(reservation: VmResourceReservation) { + return sql`jsonb_build_object( + ${sql.raw(`'${VM_RESOURCE_RESERVATION_METADATA_KEY}'`)}, + jsonb_build_object( + 'vcpus', ${reservation.vcpus}::integer, + 'memoryMb', ${reservation.memoryMb}::integer, + 'diskMb', ${reservation.diskMb}::integer + ) + )`; +} + +function resizePendingMetadataJsonb(input: { + readonly operationId: string; + readonly requestedDiskMb: number; + readonly previousDiskMb: number; + readonly createdAtMs: number; +}) { + return sql`jsonb_build_object( + 'operationId', ${input.operationId}::text, + 'requestedDiskMb', ${input.requestedDiskMb}::integer, + 'previousDiskMb', ${input.previousDiskMb}::integer, + 'createdAtMs', ${input.createdAtMs}::bigint + )`; +} + +function resizeUnconfirmedMetadataJsonb(input: { + readonly operationId: string; + readonly requestedDiskMb: number; + readonly previousDiskMb: number; + readonly markedAtMs: number; +}) { + return sql`jsonb_build_object( + ${sql.raw(`'${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY}'`)}, + jsonb_build_object( + 'operationId', ${input.operationId}::text, + 'requestedDiskMb', ${input.requestedDiskMb}::integer, + 'previousDiskMb', ${input.previousDiskMb}::integer, + 'markedAtMs', ${input.markedAtMs}::bigint + ) + )`; +} + +function resourceReconcileRetryMetadataJsonb(nextAttemptAtMs: number) { + return sql`jsonb_build_object( + ${sql.raw(`'${VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY}'`)}, + jsonb_build_object( + 'nextAttemptAtMs', ${nextAttemptAtMs}::bigint + ) + )`; +} + +/** SQL predicate that keeps deferred rows out until their retry time. */ +function resourceReconcileRetryEligibleSql(nowMs: number) { + const metadata = sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb)`; + const retryAt = sql`${metadata}->${sql.raw(`'${VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY}'`)}->>'nextAttemptAtMs'`; + // The CASE keeps malformed provider metadata from reaching a numeric cast. + // Invalid markers are eligible immediately so the background pass can heal + // them instead of starving newer rows. + return sql`case + when ${retryAt} ~ '^[0-9]{1,16}$' then + case + when (${retryAt})::numeric <= 9007199254740991 then (${retryAt})::numeric + else 0 + end + else 0 + end <= ${nowMs}`; +} + function accountUsageScopeWhere(input: { readonly userId: string; readonly billingTeamId?: string | null; @@ -663,11 +1069,15 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { mergeProviderMetadata: (input) => dbEffect("mergeProviderMetadata", async () => { const db = cloudDb(); + // Reservation and resize-generation markers are control-plane state. + // Provider metadata patches may add addresses and network ids, but cannot + // overwrite either quota claim or in-flight operation marker. + const patch = providerMetadataPatchForPersistence(input.patch); await db .update(cloudVms) .set({ // jsonb || jsonb merges at the top level: patch keys win, others stay. - providerMetadata: sql`${cloudVms.providerMetadata} || ${JSON.stringify(input.patch)}::jsonb`, + providerMetadata: sql`${cloudVms.providerMetadata} || ${JSON.stringify(patch)}::jsonb`, updatedAt: new Date(), }) .where(eq(cloudVms.id, input.id)); @@ -806,6 +1216,15 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { limit, }); } + const persistedReservation = await checkedSharedResourceReservation(tx, { + userId: input.userId, + billingTeamId: input.billingTeamId, + maxActiveVms: input.maxActiveVms, + resourceReservation: input.resourceReservation, + sharedResourceCapacity: input.sharedResourceCapacity, + phase: "create", + reserveSharedResourceHeadroom: input.reserveSharedResourceHeadroom, + }); const [vm] = await tx .insert(cloudVms) @@ -818,6 +1237,12 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { imageVersion: input.imageVersion ?? null, status: "provisioning", idempotencyKey, + providerMetadata: reservationMetadataForInput( + persistedReservation ?? input.resourceReservation, + input.sharedResourceCapacity, + input.reserveSharedResourceHeadroom, + input.forkMinimumResourceReservation ?? input.resourceReservation, + ), slug: await allocateSlugInTx(tx, input.billingTeamId), }) .returning(); @@ -832,7 +1257,7 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { throw err; } }, - catch: (cause) => isVmCreateDisabledError(cause) || isVmAccountDeletionInProgressError(cause) || isVmLimitExceededError(cause) + catch: (cause) => isVmCreateDisabledError(cause) || isVmAccountDeletionInProgressError(cause) || isVmLimitExceededError(cause) || isVmSharedResourceLimitExceededError(cause) ? cause : new VmDatabaseError({ operation: "beginCreate", cause }), }), @@ -844,12 +1269,14 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { const scope = baseScope(input); const name = baseName(input.baseName); try { + // oxlint-disable-next-line complexity -- This transaction keeps Base locks, idempotency, and generation writes atomic. return await db.transaction(async (tx) => { await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${`${scope.scopeType}:${scope.scopeId}:${name}`}, 0))`); await assertAccountVmCreateAllowed(tx, { userId: input.userId, provider: input.provider, }); + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${input.billingTeamId}, 0))`); const [existing] = await tx .select({ @@ -908,6 +1335,14 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { limit, }); } + await assertSharedResourceCapacity(tx, { + userId: input.userId, + billingTeamId: input.billingTeamId, + maxActiveVms: input.maxActiveVms, + resourceReservation: input.resourceReservation, + sharedResourceCapacity: input.sharedResourceCapacity, + phase: "create", + }); const now = new Date(); const previousGeneration = existing?.generation ?? null; @@ -925,6 +1360,10 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { imageVersion: input.imageVersion ?? null, status: "provisioning", idempotencyKey, + providerMetadata: reservationMetadataForInput( + input.resourceReservation, + input.sharedResourceCapacity, + ), slug: await allocateSlugInTx(tx, input.billingTeamId), }) .returning(); @@ -1039,7 +1478,7 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { throw err; } }, - catch: (cause) => isVmCreateDisabledError(cause) || isVmAccountDeletionInProgressError(cause) || isVmLimitExceededError(cause) + catch: (cause) => isVmCreateDisabledError(cause) || isVmAccountDeletionInProgressError(cause) || isVmLimitExceededError(cause) || isVmSharedResourceLimitExceededError(cause) ? cause : new VmDatabaseError({ operation: "beginBaseOpen", cause }), }), @@ -1050,12 +1489,14 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { const db = cloudDb(); const scope = baseScope(input); const name = baseName(input.baseName); + // oxlint-disable-next-line complexity -- This transaction keeps Base locks, limits, and generation writes atomic. return await db.transaction(async (tx) => { await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${`${scope.scopeType}:${scope.scopeId}:${name}`}, 0))`); await assertAccountVmCreateAllowed(tx, { userId: input.userId, provider: input.provider, }); + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${input.billingTeamId}, 0))`); const [existing] = await tx .select({ base: cloudVmBases, @@ -1113,6 +1554,14 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { limit, }); } + await assertSharedResourceCapacity(tx, { + userId: input.userId, + billingTeamId: input.billingTeamId, + maxActiveVms: input.maxActiveVms, + resourceReservation: input.resourceReservation, + sharedResourceCapacity: input.sharedResourceCapacity, + phase: "create", + }); const [vm] = await tx .insert(cloudVms) @@ -1125,6 +1574,10 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { imageVersion: input.imageVersion ?? null, status: "provisioning", idempotencyKey, + providerMetadata: reservationMetadataForInput( + input.resourceReservation, + input.sharedResourceCapacity, + ), slug: await allocateSlugInTx(tx, input.billingTeamId), }) .returning(); @@ -1206,7 +1659,7 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { }; }); }, - catch: (cause) => isVmCreateDisabledError(cause) || isVmAccountDeletionInProgressError(cause) || isVmLimitExceededError(cause) + catch: (cause) => isVmCreateDisabledError(cause) || isVmAccountDeletionInProgressError(cause) || isVmLimitExceededError(cause) || isVmSharedResourceLimitExceededError(cause) ? cause : new VmDatabaseError({ operation: "beginBaseReset", cause }), }), @@ -1214,6 +1667,7 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { markBaseCreateRunning: (input) => dbEffect("markBaseCreateRunning", async () => { const db = cloudDb(); + const providerMetadata = providerMetadataPatchForPersistence(input.providerMetadata); return await db.transaction(async (tx) => { const now = new Date(); const [vm] = await tx @@ -1222,7 +1676,15 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { providerVmId: input.providerVmId, imageId: input.image, imageVersion: input.imageVersion ?? null, - providerMetadata: input.providerMetadata ?? {}, + // Provider metadata is additive. Keep the reservation written by + // beginBaseOpen/reset even if a driver omits it or returns a stale + // copy in its handle. + providerMetadata: sql`( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) || ${JSON.stringify(providerMetadata)}::jsonb + ) || case + when ${cloudVms.providerMetadata}->'${sql.raw(VM_RESOURCE_RESERVATION_METADATA_KEY)}' is null then '{}'::jsonb + else jsonb_build_object('${sql.raw(VM_RESOURCE_RESERVATION_METADATA_KEY)}', ${cloudVms.providerMetadata}->'${sql.raw(VM_RESOURCE_RESERVATION_METADATA_KEY)}') + end`, status: "running", failureCode: null, failureMessage: null, @@ -1375,6 +1837,184 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { .limit(input.limit); }), + legacyResourceReservationCandidates: (input) => + dbEffect("legacyResourceReservationCandidates", async () => { + const db = cloudDb(); + const nowMs = Date.now(); + const scope = input.billingTeamId?.trim() + ? eq(cloudVms.billingTeamId, input.billingTeamId.trim()) + : input.userId + ? and( + eq(cloudVms.userId, input.userId), + or(isNull(cloudVms.billingTeamId), eq(cloudVms.billingTeamId, input.userId)), + ) + : null; + const predicates = [ + inArray(cloudVms.status, LIVE_VM_RESOURCE_STATUSES), + isNotNull(cloudVms.providerVmId), + ...(scope ? [scope] : []), + ]; + const rows = await db + .select() + .from(cloudVms) + .where( + and( + ...predicates, + resourceReconcileRetryEligibleSql(nowMs), + or( + sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY}`, + sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY}`, + sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_FORK_PENDING_METADATA_KEY}`, + sql`not coalesce((${validResourceReservationMarkerSql()}), false)`, + ), + ), + ) + .orderBy(asc(cloudVms.updatedAt)) + .limit(input.limit); + // Keep a runtime check as a second boundary for adapters that return + // rows from a different SQL dialect or a stale read replica. + return rows.filter((row) => { + const retry = vmResourceReconcileRetryFromMetadata(row.providerMetadata); + if (retry && retry.nextAttemptAtMs > nowMs) return false; + return Object.prototype.hasOwnProperty.call(row.providerMetadata ?? {}, VM_RESOURCE_RESIZE_PENDING_METADATA_KEY) || + Object.prototype.hasOwnProperty.call(row.providerMetadata ?? {}, VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY) || + Object.prototype.hasOwnProperty.call(row.providerMetadata ?? {}, VM_RESOURCE_FORK_PENDING_METADATA_KEY) || + !hasVmResourceReservationMetadata(row.providerMetadata); + }); + }), + + deferResourceReservation: (input) => + dbEffect("deferResourceReservation", async () => { + const nextAttemptAtMs = input.nextAttemptAt.getTime(); + if (!Number.isSafeInteger(nextAttemptAtMs) || nextAttemptAtMs <= 0) { + throw new Error("resource reconciliation retry time must be a positive timestamp"); + } + const db = cloudDb(); + await db.transaction(async (tx) => { + const [initial] = await tx + .select({ userId: cloudVms.userId, billingTeamId: cloudVms.billingTeamId }) + .from(cloudVms) + .where(eq(cloudVms.id, input.id)) + .limit(1); + if (!initial) return; + const lockKey = initial.billingTeamId?.trim() || `user:${initial.userId}`; + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${lockKey}, 0))`); + await tx + .update(cloudVms) + .set({ + // Keep retry state in the row so every worker observes the same + // backoff and a permanently failing provider cannot monopolize a + // bounded oldest-first batch. + providerMetadata: sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) + || ${resourceReconcileRetryMetadataJsonb(nextAttemptAtMs)}`, + updatedAt: new Date(), + }) + .where(and( + eq(cloudVms.id, input.id), + inArray(cloudVms.status, LIVE_VM_RESOURCE_STATUSES), + )); + }); + }), + + setResourceReservation: (input) => + Effect.tryPromise({ + try: async () => { + const db = cloudDb(); + return await db.transaction(async (tx) => { + const [initial] = await tx + .select({ + userId: cloudVms.userId, + billingTeamId: cloudVms.billingTeamId, + status: cloudVms.status, + }) + .from(cloudVms) + .where(eq(cloudVms.id, input.id)) + .limit(1); + if (!initial) return false; + if (!(LIVE_VM_RESOURCE_STATUSES as readonly string[]).includes(initial.status)) return false; + const lockKey = initial.billingTeamId?.trim() || `user:${initial.userId}`; + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${lockKey}, 0))`); + + if (input.expectedResizeOperationId !== undefined && input.expectedResizeUnconfirmedOperationId !== undefined) { + throw new Error("resource repair cannot target two resize generations"); + } + if (input.expectedReservation !== undefined && + (input.expectedResizeOperationId !== undefined || input.expectedResizeUnconfirmedOperationId !== undefined)) { + throw new Error("resource replacement cannot target a resize generation"); + } + // A normal legacy repair must not lower an in-flight resize. A pending + // or unconfirmed repair is a compare-and-set on its generation, so a + // stale provider read cannot clear a newer marker. A native fork uses + // the same compare-and-set shape for its temporary headroom claim. + const resizeMarkerPredicate = input.expectedResizeOperationId !== undefined + ? sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb)->${sql.raw(`'${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY}'`)}->>'operationId' = ${input.expectedResizeOperationId}` + : input.expectedResizeUnconfirmedOperationId !== undefined + ? sql`coalesce(${cloudVms.providerMetadata}, '{}'::jsonb)->${sql.raw(`'${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY}'`)}->>'operationId' = ${input.expectedResizeUnconfirmedOperationId} + and not (coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY})` + : input.expectedReservation !== undefined + ? sql`not (coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY}) + and not (coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY})` + : sql`not (coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY}) + and not (coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY})`; + const markerPredicate = input.expectedReservation !== undefined + ? resourceReservationMarkerEqualsSql(input.expectedReservation) + : input.expectedResizeOperationId === undefined && input.expectedResizeUnconfirmedOperationId === undefined + ? sql`not coalesce((${validResourceReservationMarkerSql()}), false)` + : sql`true`; + + if (input.sharedResourceCapacity) { + const used = await reservedResourceTotals(tx, { + userId: initial.userId, + billingTeamId: initial.billingTeamId, + excludeVmId: input.id, + }); + const exceeded = firstExceededSharedResource({ + used, + requested: input.reservation, + capacity: input.sharedResourceCapacity, + }); + if (exceeded) { + throw new VmSharedResourceLimitExceededError({ + kind: "shared_resources", + billingTeamId: initial.billingTeamId ?? initial.userId, + phase: "create", + resource: exceeded.resource, + used: exceeded.used, + requested: exceeded.requested, + limit: exceeded.limit, + }); + } + } + + const rows = await tx + .update(cloudVms) + .set({ + // Keep provider metadata and the control-plane claim in one JSON + // document without allowing a stale read to drop other fields. + providerMetadata: sql`( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) + || ${reservationMetadataJsonb(input.reservation)} + ) #- '{${sql.raw(VM_RESOURCE_RESIZE_PENDING_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_FORK_PENDING_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY)}}'`, + updatedAt: new Date(), + }) + .where(and( + eq(cloudVms.id, input.id), + inArray(cloudVms.status, LIVE_VM_RESOURCE_STATUSES), + resizeMarkerPredicate, + markerPredicate, + )) + .returning({ id: cloudVms.id }); + return rows.length > 0; + }); + }, + catch: (cause) => isVmSharedResourceLimitExceededError(cause) + ? cause + : new VmDatabaseError({ operation: "setResourceReservation", cause }), + }), + reservePausedResume: (input) => Effect.tryPromise({ try: async () => { @@ -1434,6 +2074,312 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { : new VmDatabaseError({ operation: "reservePausedResume", cause }), }), + reserveVmResize: (input) => + Effect.tryPromise({ + try: async () => { + const db = cloudDb(); + // oxlint-disable-next-line complexity -- The transaction keeps resize generations, headroom, and pool checks atomic. + return await db.transaction(async (tx) => { + const requestedTeamId = input.billingTeamId?.trim(); + const lockKey = requestedTeamId || `user:${input.userId}`; + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${lockKey}, 0))`); + + const [current] = await tx + .select() + .from(cloudVms) + .where(and( + eq(cloudVms.id, input.id), + accountScopeWhere({ userId: input.userId, billingTeamId: requestedTeamId }), + eq(cloudVms.providerVmId, input.providerVmId), + )) + .limit(1); + if (!current || current.status === "destroyed") return null; + + const hasPendingMarker = Object.prototype.hasOwnProperty.call( + current.providerMetadata ?? {}, + VM_RESOURCE_RESIZE_PENDING_METADATA_KEY, + ); + if (hasPendingMarker) { + const pending = vmResourceResizePendingFromMetadata(current.providerMetadata); + // A malformed marker cannot identify an active owner. The locked + // update below replaces it with a fresh generation. + if (pending) throw new VmResizeInProgressError({ vmId: current.id }); + } + + const previous = vmResourceReservationFromMetadata(current.providerMetadata); + const unconfirmed = vmResourceResizeUnconfirmedFromMetadata(current.providerMetadata); + const isNoopResize = input.currentDiskMb !== undefined && input.storageMb === input.currentDiskMb; + // An unconfirmed resize deliberately holds the maximum disk claim. + // A later no-op retry must use the marker's prior/current size, not + // that temporary maximum, or it can clear the marker while keeping a + // permanent 256 GB reservation. + const previousDiskMb = unconfirmed + ? Math.max( + unconfirmed.previousDiskMb ?? VM_DISK_MB_DEFAULT, + input.currentDiskMb ?? 0, + ) + : Math.max(previous.diskMb, input.currentDiskMb ?? 0); + const requestedDiskMb = Math.max(previousDiskMb, input.storageMb); + const requested = { + ...previous, + // A stale provider read must never make the durable reservation + // shrink. The workflow already validates grow-only semantics. + diskMb: requestedDiskMb, + }; + const unconfirmedStillPending = isNoopResize && + unconfirmed !== null && + (input.currentDiskMb ?? 0) < unconfirmed.requestedDiskMb; + if (unconfirmedStillPending) { + // The observed provider size is still below the requested resize. + // Leave the conservative marker for the background recovery pass. + return { + previousDiskMb, + reservedDiskMb: previous.diskMb, + requestedDiskMb: unconfirmed.requestedDiskMb, + operationId: unconfirmed.operationId, + }; + } + const billingTeamId = current.billingTeamId ?? requestedTeamId; + const capacity = sharedResourceCapacityForInput( + input.maxActiveVms ?? null, + input.sharedResourceCapacity, + ); + const used = await reservedResourceTotals(tx, { + userId: current.userId, + billingTeamId, + excludeVmId: current.id, + }); + const exceeded = firstExceededSharedResource({ + used, + requested, + capacity, + }); + if (exceeded) { + throw new VmSharedResourceLimitExceededError({ + kind: "shared_resources", + billingTeamId: billingTeamId ?? current.userId, + phase: "resize", + resource: exceeded.resource, + used: exceeded.used, + requested: exceeded.requested, + limit: exceeded.limit, + }); + } + + // Provider resize is outside this transaction and may round the + // request upward. Hold all remaining disk headroom while it runs so + // a concurrent create cannot consume the bytes needed by the final + // provider-confirmed claim. A no-op only backfills the measured + // provider size and does not need a pending headroom reservation. + const operationId = randomUUID(); + const createdAtMs = Date.now(); + const diskMb = isNoopResize + ? requestedDiskMb + : Math.max(requestedDiskMb, capacity.diskMb - used.diskMb); + const reserved = { ...requested, diskMb }; + + await tx + .update(cloudVms) + .set({ + providerMetadata: isNoopResize + ? sql`( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) + || ${reservationMetadataJsonb(reserved)} + ) #- '{${sql.raw(VM_RESOURCE_RESIZE_PENDING_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY)}}'` + : sql`( + jsonb_set( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) || ${reservationMetadataJsonb(reserved)}, + '{${sql.raw(VM_RESOURCE_RESIZE_PENDING_METADATA_KEY)}}', + ${resizePendingMetadataJsonb({ operationId, requestedDiskMb, previousDiskMb, createdAtMs })}, + true + ) + ) #- '{${sql.raw(VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY)}}'`, + updatedAt: new Date(), + }) + .where(and(eq(cloudVms.id, current.id), ne(cloudVms.status, "destroyed"))); + return { + previousDiskMb, + reservedDiskMb: reserved.diskMb, + requestedDiskMb, + operationId, + }; + }); + }, + catch: (cause) => isVmSharedResourceLimitExceededError(cause) || isVmResizeInProgressError(cause) + ? cause + : new VmDatabaseError({ operation: "reserveVmResize", cause }), + }), + + confirmVmResize: (input) => + dbEffect("confirmVmResize", async () => { + const confirmedDiskMb = positiveReservationInteger(input.confirmedDiskMb); + const expectedDiskMb = positiveReservationInteger(input.expectedDiskMb); + const minimumDiskMb = input.minimumDiskMb === undefined + ? expectedDiskMb + : positiveReservationInteger(input.minimumDiskMb); + if (confirmedDiskMb === null || expectedDiskMb === null || minimumDiskMb === null) { + throw new Error("resize disk claims must be positive integers"); + } + const operationId = typeof input.operationId === "string" ? input.operationId.trim() : ""; + if (operationId.length === 0 || operationId.length > 200) { + throw new Error("resize operation id must be a non-empty string"); + } + // Keep a larger pre-resize claim when a provider returns a stale or + // rounded-down stat. The compare-and-set predicate prevents a late + // response from overwriting a newer concurrent resize reservation. + const diskMb = Math.max(minimumDiskMb, confirmedDiskMb); + const db = cloudDb(); + return await db.transaction(async (tx) => { + // The resize request runs outside SQL, so confirmation must take the + // same team lock as create and reserveVmResize before lowering the + // conservative pending headroom claim. + const [initial] = await tx + .select({ userId: cloudVms.userId, billingTeamId: cloudVms.billingTeamId }) + .from(cloudVms) + .where(eq(cloudVms.id, input.id)) + .limit(1); + if (!initial) return false; + const lockKey = initial.billingTeamId?.trim() || `user:${initial.userId}`; + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${lockKey}, 0))`); + + const rows = await tx + .update(cloudVms) + .set({ + providerMetadata: sql`( + jsonb_set( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb), + '{cmuxResourceReservation,diskMb}', + to_jsonb(${diskMb}::integer), + true + ) + ) #- '{${sql.raw(VM_RESOURCE_RESIZE_PENDING_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY)}}'`, + updatedAt: new Date(), + }) + .where(and( + eq(cloudVms.id, input.id), + inArray(cloudVms.status, LIVE_VM_RESOURCE_STATUSES), + sql`${cloudVms.providerMetadata}->'cmuxResourceReservation'->>'diskMb' = ${String(expectedDiskMb)}`, + sql`${cloudVms.providerMetadata}->${sql.raw(`'${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY}'`)}->>'operationId' = ${operationId}`, + )) + .returning({ id: cloudVms.id }); + return rows.length > 0; + }); + }), + + markVmResizeUnconfirmed: (input) => + dbEffect("markVmResizeUnconfirmed", async () => { + const expectedDiskMb = positiveReservationInteger(input.expectedDiskMb); + const minimumDiskMb = input.minimumDiskMb === undefined + ? expectedDiskMb + : positiveReservationInteger(input.minimumDiskMb); + const previousDiskMb = positiveReservationInteger(input.previousDiskMb); + const operationId = typeof input.operationId === "string" ? input.operationId.trim() : ""; + if ( + expectedDiskMb === null || + minimumDiskMb === null || + previousDiskMb === null || + operationId.length === 0 || + operationId.length > 200 + ) { + throw new Error("invalid unconfirmed resize claim"); + } + const db = cloudDb(); + return await db.transaction(async (tx) => { + const [initial] = await tx + .select({ userId: cloudVms.userId, billingTeamId: cloudVms.billingTeamId }) + .from(cloudVms) + .where(eq(cloudVms.id, input.id)) + .limit(1); + if (!initial) return false; + const lockKey = initial.billingTeamId?.trim() || `user:${initial.userId}`; + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${lockKey}, 0))`); + const rows = await tx + .update(cloudVms) + .set({ + // Keep the conservative headroom claim until a later provider + // read confirms the real size. The generation check prevents a + // late stats failure from replacing a newer resize. + providerMetadata: sql`( + jsonb_set( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb), + '{cmuxResourceReservation,diskMb}', + to_jsonb(${VM_DISK_MB_MAX}::integer), + true + ) + || ${resizeUnconfirmedMetadataJsonb({ + operationId, + requestedDiskMb: minimumDiskMb, + previousDiskMb, + markedAtMs: Date.now(), + })} + ) #- '{${sql.raw(VM_RESOURCE_RESIZE_PENDING_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY)}}'`, + updatedAt: new Date(), + }) + .where(and( + eq(cloudVms.id, input.id), + inArray(cloudVms.status, LIVE_VM_RESOURCE_STATUSES), + sql`${cloudVms.providerMetadata}->'${sql.raw(VM_RESOURCE_RESERVATION_METADATA_KEY)}'->>'diskMb' = ${String(expectedDiskMb)}`, + sql`${cloudVms.providerMetadata}->${sql.raw(`'${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY}'`)}->>'operationId' = ${operationId}`, + )) + .returning({ id: cloudVms.id }); + return rows.length > 0; + }); + }), + + restoreVmResize: (input) => + dbEffect("restoreVmResize", async () => { + const operationId = typeof input.operationId === "string" ? input.operationId.trim() : ""; + const previousDiskMb = positiveReservationInteger(input.previousDiskMb); + const expectedDiskMb = positiveReservationInteger(input.expectedDiskMb); + if (operationId.length === 0 || operationId.length > 200 || previousDiskMb === null || expectedDiskMb === null) { + throw new Error("invalid resize rollback claim"); + } + const db = cloudDb(); + await db.transaction(async (tx) => { + const [initial] = await tx + .select() + .from(cloudVms) + .where(eq(cloudVms.id, input.id)) + .limit(1); + if (!initial || initial.status === "destroyed") return; + const lockKey = initial.billingTeamId?.trim() || `user:${initial.userId}`; + await tx.execute(sql`select pg_advisory_xact_lock(hashtextextended(${lockKey}, 0))`); + const [current] = await tx + .select() + .from(cloudVms) + .where(eq(cloudVms.id, input.id)) + .limit(1); + if (!current || current.status === "destroyed") return; + const reservation = vmResourceReservationFromMetadata(current.providerMetadata); + if (reservation.diskMb !== expectedDiskMb) return; + await tx + .update(cloudVms) + .set({ + providerMetadata: sql`( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) + || ${reservationMetadataJsonb({ + ...reservation, + diskMb: previousDiskMb, + })} + ) #- '{${sql.raw(VM_RESOURCE_RESIZE_PENDING_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY)}}' + #- '{${sql.raw(VM_RESOURCE_RECONCILE_RETRY_METADATA_KEY)}}'`, + updatedAt: new Date(), + }) + .where(and( + eq(cloudVms.id, input.id), + ne(cloudVms.status, "destroyed"), + sql`${cloudVms.providerMetadata}->${sql.raw(`'${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY}'`)}->>'operationId' = ${operationId}`, + )); + }); + }), + reconciliationCandidates: (input) => dbEffect("reconciliationCandidates", async () => { const db = cloudDb(); @@ -1543,13 +2489,21 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { markCreateRunning: (input) => dbEffect("markCreateRunning", async () => { const db = cloudDb(); + const providerMetadata = providerMetadataPatchForPersistence(input.providerMetadata); const [vm] = await db .update(cloudVms) .set({ providerVmId: input.providerVmId, imageId: input.image, imageVersion: input.imageVersion ?? null, - providerMetadata: input.providerMetadata ?? {}, + // Keep the reservation from the transactional create claim. It is + // the control-plane accounting record, not provider metadata. + providerMetadata: sql`( + coalesce(${cloudVms.providerMetadata}, '{}'::jsonb) || ${JSON.stringify(providerMetadata)}::jsonb + ) || case + when ${cloudVms.providerMetadata}->'${sql.raw(VM_RESOURCE_RESERVATION_METADATA_KEY)}' is null then '{}'::jsonb + else jsonb_build_object('${sql.raw(VM_RESOURCE_RESERVATION_METADATA_KEY)}', ${cloudVms.providerMetadata}->'${sql.raw(VM_RESOURCE_RESERVATION_METADATA_KEY)}') + end`, status: "running", failureCode: null, failureMessage: null, @@ -1593,6 +2547,46 @@ export const vmRepositoryLiveShape: VmRepositoryShape = { return !!event; }), + ownedSnapshotResourceReservation: (input) => + dbEffect("ownedSnapshotResourceReservation", async () => { + const db = cloudDb(); + const [event] = await db + .select({ + metadata: cloudVmUsageEvents.metadata, + sourceMetadata: cloudVms.providerMetadata, + }) + .from(cloudVmUsageEvents) + .leftJoin(cloudVms, eq(cloudVmUsageEvents.vmId, cloudVms.id)) + .where( + and( + accountUsageScopeWhere({ userId: input.userId, billingTeamId: input.billingTeamId }), + eq(cloudVmUsageEvents.provider, input.provider), + eq(cloudVmUsageEvents.eventType, "vm.snapshot.created"), + sql`${cloudVmUsageEvents.metadata}->>'snapshotId' = ${input.snapshotId}`, + ), + ) + .orderBy(desc(cloudVmUsageEvents.createdAt), desc(cloudVmUsageEvents.id)) + .limit(1); + if (!event) return null; + + const conservativeFallback = { + ...DEFAULT_VM_RESOURCE_RESERVATION, + diskMb: PLAN_SHARED_DISK_MB, + }; + const source = vmResourceReservationFromMetadata(event.sourceMetadata, conservativeFallback); + const recordedVcpus = positiveReservationInteger(event.metadata?.vcpus); + const recordedMemoryMb = positiveReservationInteger(event.metadata?.memoryMb); + const recordedDiskMb = positiveReservationInteger(event.metadata?.diskMb); + return { + // New snapshot events record the provider-confirmed shape. For a + // legacy source, a recorded dimension is authoritative; an absent + // dimension uses the source claim or the conservative fallback. + vcpus: recordedVcpus ?? source.vcpus, + memoryMb: recordedMemoryMb ?? source.memoryMb, + diskMb: recordedDiskMb ?? source.diskMb, + }; + }), + findUserVm: (input) => dbEffect("findUserVm", async () => { const db = cloudDb(); diff --git a/web/services/vms/routeHelpers.ts b/web/services/vms/routeHelpers.ts index fdf317d05587..df92061f9ec7 100644 --- a/web/services/vms/routeHelpers.ts +++ b/web/services/vms/routeHelpers.ts @@ -39,6 +39,8 @@ import { isVmPrivateNetworkUnavailableError, isVmProviderOperationError, isVmResizeInvalidError, + isVmResizeInProgressError, + isVmSharedResourceLimitExceededError, isVmSnapshotNotFoundError, isVmTunnelNotFoundError, vmWorkflowErrorCause, @@ -60,7 +62,13 @@ import { vmIdFromRequestPath, type VmRequestContext, } from "./requestContext"; -import { vmRequestLocale, vmRequiresProCopy, vmUnsupportedCopy, vmUnsupportedOperationKey } from "./vmErrorMessages"; +import { + vmRequestLocale, + vmRequiresProCopy, + vmSharedResourceCopy, + vmUnsupportedCopy, + vmUnsupportedOperationKey, +} from "./vmErrorMessages"; import type { Locale } from "../../i18n/routing"; /** Bearer + refresh token pair the mac app stashes in keychain. */ @@ -508,6 +516,50 @@ export function vmActiveLimitExceededResponse(input: { }); } +/** Translate a plan-wide resource pool rejection into a stable client error. */ +export async function vmSharedResourceLimitExceededResponse( + err: { + readonly resource: "vcpus" | "memoryMb" | "diskMb"; + readonly used: number; + readonly requested: number; + readonly limit: number; + readonly phase?: VmLifecyclePhase; + }, + phase: VmLifecyclePhase = err.phase ?? "create", + locale: Locale = "en", +): Promise { + const resource = err.resource === "vcpus" + ? "vCPU" + : err.resource === "memoryMb" + ? "memory" + : "disk"; + const unit = err.resource === "vcpus" + ? "vCPU" + : err.resource === "memoryMb" + ? "MB of memory" + : "MB of disk"; + const copy = await vmSharedResourceCopy(locale, { + resource, + oversized: err.requested > err.limit, + }); + return vmErrorResponse({ + error: "vm_shared_resource_limit_exceeded", + status: 409, + message: copy.message, + action: copy.action, + phase, + retryable: false, + details: { + resource: err.resource, + used: err.used, + requested: err.requested, + limit: err.limit, + unit, + shared: true, + }, + }); +} + export type VmCreateLikeOperation = "fork" | "restore"; /** @@ -515,14 +567,15 @@ export type VmCreateLikeOperation = "fork" | "restore"; * Operation-specific retry guidance stays at the route boundary, while the response * shape and billing errors remain centralized here. */ -export function vmCreateLikeErrorResponse( +export async function vmCreateLikeErrorResponse( err: unknown, input: { readonly operation: VmCreateLikeOperation; readonly planId: string; readonly retryAction: string; + readonly locale?: Locale; }, -): Response | null { +): Promise { if (isVmCreateInProgressError(err)) { return vmErrorResponse({ error: "vm_create_in_progress", @@ -548,6 +601,9 @@ export function vmCreateLikeErrorResponse( retryAction: input.retryAction, }); } + if (isVmSharedResourceLimitExceededError(err)) { + return vmSharedResourceLimitExceededResponse(err, input.operation, input.locale ?? "en"); + } if (input.operation === "restore" && isVmSnapshotNotFoundError(err)) { return vmErrorResponse({ error: "vm_snapshot_not_found", @@ -598,6 +654,7 @@ export function vmModelPlaneErrorResponse( } /** Translate a normalized workflow failure into the public VM error contract. */ +// oxlint-disable-next-line complexity -- Centralized dispatch preserves the public error precedence contract. export async function vmWorkflowErrorResponse( err: unknown, options: { readonly locale?: Locale } = {}, @@ -645,6 +702,10 @@ export async function vmWorkflowErrorResponse( return vmModelPlaneErrorResponse(workflowError); } + if (isVmSharedResourceLimitExceededError(workflowError)) { + return vmSharedResourceLimitExceededResponse(workflowError, workflowError.phase ?? "create", options.locale ?? "en"); + } + if (isVmResizeInvalidError(workflowError)) { const requested = Math.round(workflowError.requestedMb / 1024); const current = Math.round(workflowError.currentMb / 1024); @@ -662,6 +723,18 @@ export async function vmWorkflowErrorResponse( }); } + if (isVmResizeInProgressError(workflowError)) { + return vmErrorResponse({ + error: "vm_resize_in_progress", + status: 409, + message: "A disk resize is already running for this Cloud VM.", + action: "Wait for the current resize to finish, then retry.", + phase: "resize", + retryable: true, + retryAfterSeconds: 5, + }); + } + if (isVmPrivateNetworkUnavailableError(workflowError)) { return vmErrorResponse({ error: "vm_private_network_unavailable", diff --git a/web/services/vms/vmErrorMessages.ts b/web/services/vms/vmErrorMessages.ts index efbdde9b140c..353ca873e6c0 100644 --- a/web/services/vms/vmErrorMessages.ts +++ b/web/services/vms/vmErrorMessages.ts @@ -92,6 +92,28 @@ export async function vmRequiresProCopy( }; } +/** Copy returned when an account's shared Cloud VM resource pool is full. */ +export type VmSharedResourceCopy = { + readonly message: string; + readonly action: string; +}; + +/** Load the localized shared-resource rejection copy. */ +export async function vmSharedResourceCopy( + locale: Locale, + values: { readonly resource: string; readonly oversized: boolean }, +): Promise { + const translator = createTranslator({ + locale, + messages: await loadMessages(locale), + namespace: "vmErrors.sharedResource", + }) as unknown as (key: string, values?: Record) => string; + return { + message: translator("message", { resource: values.resource }), + action: translator(values.oversized ? "oversizedAction" : "action", { resource: values.resource }), + }; +} + function localeFromPath(value: string): Locale | null { try { const firstSegment = new URL(value).pathname.split("/").filter(Boolean)[0]; diff --git a/web/services/vms/workflows.ts b/web/services/vms/workflows.ts index 292055f45f6b..564ed5d72729 100644 --- a/web/services/vms/workflows.ts +++ b/web/services/vms/workflows.ts @@ -1,6 +1,7 @@ import { createHash, randomUUID } from "node:crypto"; import * as Effect from "effect/Effect"; import * as Either from "effect/Either"; +import * as Exit from "effect/Exit"; import type { CreateOptions } from "./drivers/types"; import * as Layer from "effect/Layer"; import type { @@ -11,6 +12,7 @@ import type { SSHEndpoint, VmEdgeRule, VMHandle, + VMStats, VMStatus, } from "./drivers"; import { isProviderId, vmCapabilitiesFor } from "./drivers"; @@ -23,7 +25,29 @@ import { type VmBillingGatewayShape, } from "./billingGateway"; import { vmCreateDisabledReason } from "./config"; -import { VM_DISK_MB_MAX, VM_DISK_MB_STEP } from "./machineSpec"; +import { + DEFAULT_VM_RESOURCE_RESERVATION, + PLAN_SHARED_DISK_MB, + PLAN_SHARED_MEMORY_MB, + PLAN_SHARED_VCPU, + VM_DISK_MB_MAX, + VM_DISK_MB_STEP, + VM_RESOURCE_RESIZE_PENDING_METADATA_KEY, + VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY, + VM_RESOURCE_FORK_PENDING_METADATA_KEY, + hasVmResourceReservationMetadata, + sharedResourceCapacityForMaxActiveVms, + vmResourceReconcileRetryFromMetadata, + vmResourceReservationForCreate, + vmResourceReservationFromMetadata, + vmResourceForkPendingFromMetadata, + vmResourceResizePendingFromMetadata, + vmResourceResizeUnconfirmedFromMetadata, + vmProviderResourceSize, + type VmResourceReservation, + type VmResourceResizePending, + type VmResourceResizeUnconfirmed, +} from "./machineSpec"; import { VmBillingError, VmAccountDeletionIdentityRevocationError, @@ -31,6 +55,7 @@ import { VmCreateDisabledError, VmCreateFailedError, VmCreateInProgressError, + VmDatabaseError, VmFreeAccessExpiredError, VmModelPlaneError, VmNotFoundError, @@ -41,12 +66,17 @@ import { VM_MODEL_PLANE_FAILURE_CODES, isVmCreateCreditsInsufficientError, isVmLimitExceededError, + isVmSharedResourceLimitExceededError, isVmModelPlaneError, vmWorkflowErrorCause, - type VmDatabaseError, type VmWorkflowError, } from "./errors"; -import { isVmFreeAccessExpired, maxActiveVmsForPlan, vmFreeAccessWindowDays } from "./entitlements"; +import { + isPaidVmPlan, + isVmFreeAccessExpired, + maxActiveVmsForPlan, + vmFreeAccessWindowDays, +} from "./entitlements"; import { networkSlugForUser, privateNetworkUnavailableReason, resolveOwnerNetwork } from "./privateNetwork"; import { isProviderIdentityNotFoundError, isProviderNotFoundError } from "./providerErrors"; import { VmProviderGateway, VmProviderGatewayLive, type VmProviderGatewayShape } from "./providerGateway"; @@ -65,6 +95,7 @@ import { type CloudVmLeaseKind, type CloudVmRow, type VmRepositoryShape, + type VmResizeReservation, } from "./repository"; import { measureVmEffect, type VmTimingSink } from "./timings"; @@ -158,6 +189,19 @@ const IDENTITY_REVOKE_PROVIDER_TIMEOUT = "5 seconds"; const ACTIVE_IDENTITY_REVOKE_HOT_PATH_LIMIT = 8; const ACCOUNT_DELETION_IDENTITY_REVOKE_BATCH = 8; const VM_STATUS_RECONCILE_BATCH_LIMIT = 200; +const LEGACY_RESOURCE_RECONCILE_BATCH_LIMIT = 50; +const LEGACY_RESOURCE_RECONCILE_REQUEST_LIMIT = 5; +const LEGACY_RESOURCE_RECONCILE_CONCURRENCY = 5; +const LEGACY_RESOURCE_RECONCILE_RETRY_AFTER_MS = 5 * 60 * 1000; +// Ten concurrent waves of this batch must leave time for status reconciliation +// in a short-lived cron invocation, even when a provider is fully hung. +const LEGACY_RESOURCE_RECONCILE_PROVIDER_TIMEOUT = "2 seconds"; +const LEGACY_RESOURCE_RECONCILE_REQUEST_TIMEOUT = "5 seconds"; +// Provider stats are advisory on request paths. A stalled provider must not +// keep a snapshot or fork HTTP request open indefinitely. +const FOREGROUND_PROVIDER_STATS_TIMEOUT = "2 seconds"; +const RESIZE_PENDING_RECOVERY_AFTER_MS = 15 * 60 * 1000; +const RESIZE_UNCONFIRMED_RECOVERY_AFTER_MS = 30 * 60 * 1000; const PREVIEW_ENDPOINT_LEASE_TTL_MS = 12 * 60 * 60 * 1000; type ExistingVmAccessInput = { @@ -270,6 +314,12 @@ export function reconcileVmProviderStatuses(input: { } = {}): Effect.Effect { return Effect.gen(function* () { const providers = yield* VmProviderGateway; + const repo = yield* VmRepository; + // Legacy resource claims are repaired by this background cron. Keeping + // provider fanout here removes migration work from user-facing creates. + yield* reconcileLegacyResourceReservations(repo, providers, { + limit: LEGACY_RESOURCE_RECONCILE_BATCH_LIMIT, + }); const getStatus = providers.getStatus; if (!getStatus) { return { @@ -281,7 +331,6 @@ export function reconcileVmProviderStatuses(input: { }; } - const repo = yield* VmRepository; const candidates = yield* repo.reconciliationCandidates({ limit: boundedVmStatusReconcileLimit(input.limit), }); @@ -409,6 +458,8 @@ export function createVm(input: { readonly memoryMb?: number; /** See CreateOptions.imageSize: CPU and memory are baked; disk can grow later. */ readonly imageSize?: CreateOptions["imageSize"]; + /** Override the reservation when cloning an existing machine shape. */ + readonly resourceReservation?: VmResourceReservation; /** How the machine came to exist; analytics only. Defaults to `create`. */ readonly origin?: VmCreateOrigin; /** @@ -424,6 +475,22 @@ export function createVm(input: { const repo = yield* VmRepository; const providers = yield* VmProviderGateway; const billing = yield* VmBillingGateway; + // The advertised shared pool is a paid-plan entitlement. Free provisioning + // is an operator-only demo escape hatch and has no pricing resource promise. + const beginInput = isPaidVmPlan(input.billingPlanId) + ? { + ...input, + // Reserve the logical CPU and memory profile when memoryMb is present, + // while retaining the baked image's actual disk claim. A direct caller + // may instead provide only imageSize; in that form the image is the + // authoritative request. + resourceReservation: input.resourceReservation ?? vmResourceReservationForCreate({ + memoryMb: input.memoryMb, + imageSize: input.imageSize, + }), + sharedResourceCapacity: sharedResourceCapacityForMaxActiveVms(input.maxActiveVms), + } + : input; // The owner's network row and the create row do not depend on each other, // so the request pays the slower of the two reads, not their sum. A network @@ -439,7 +506,7 @@ export function createVm(input: { resolveOwnerNetwork({ userId: input.userId, provider: input.provider }), ), ), - beginCreateWithLazyProviderRefresh(repo, providers, input), + beginCreateWithLazyProviderRefresh(repo, providers, beginInput), ], { concurrency: 2 }, ); @@ -643,10 +710,17 @@ export function openBaseVm(input: { const repo = yield* VmRepository; const providers = yield* VmProviderGateway; const billing = yield* VmBillingGateway; + const beginInput = isPaidVmPlan(input.billingPlanId) + ? { + ...input, + resourceReservation: vmResourceReservationForCreate(), + sharedResourceCapacity: sharedResourceCapacityForMaxActiveVms(input.maxActiveVms), + } + : input; const create = yield* measureVmEffect( input.timing, "begin_base_open", - repo.beginBaseOpen(input), + repo.beginBaseOpen(beginInput), ); return yield* finishBaseCreate(repo, providers, billing, input, create); }); @@ -669,10 +743,17 @@ export function resetBaseVm(input: { const repo = yield* VmRepository; const providers = yield* VmProviderGateway; const billing = yield* VmBillingGateway; + const beginInput = isPaidVmPlan(input.billingPlanId) + ? { + ...input, + resourceReservation: vmResourceReservationForCreate(), + sharedResourceCapacity: sharedResourceCapacityForMaxActiveVms(input.maxActiveVms), + } + : input; const create = yield* measureVmEffect( input.timing, "begin_base_reset", - repo.beginBaseReset(input), + repo.beginBaseReset(beginInput), ); return yield* finishBaseCreate(repo, providers, billing, input, create); }); @@ -897,7 +978,15 @@ function reopenBaseIfProviderDeleted( return yield* measureVmEffect( input.timing, "begin_base_open", - repo.beginBaseOpen(input), + Effect.suspend(() => repo.beginBaseOpen( + isPaidVmPlan(input.billingPlanId) + ? { + ...input, + resourceReservation: vmResourceReservationForCreate(), + sharedResourceCapacity: sharedResourceCapacityForMaxActiveVms(input.maxActiveVms), + } + : input, + )), ); }) : Effect.succeed(null) @@ -922,6 +1011,24 @@ export function snapshotVm(input: { provider: vm.provider, operation: "snapshot", }))); + // Read after the provider confirms the snapshot. Grow-only resizes that + // finish during snapshot creation are then included in the captured claim; + // a later resize can only make this conservative. + const snapshotStats = providers.getStats + ? yield* providers.getStats(vm.provider, vm.providerVmId ?? input.providerVmId).pipe( + Effect.timeoutFail({ + duration: FOREGROUND_PROVIDER_STATS_TIMEOUT, + onTimeout: () => new Error(`snapshot stats timed out for ${vm.providerVmId ?? input.providerVmId}`), + }), + Effect.map((stats) => ({ + vcpus: vmProviderResourceSize("vcpus", stats.cpus), + memoryMb: vmProviderResourceSize("memoryMb", stats.memoryTotalMb), + diskMb: vmProviderResourceSize("diskMb", stats.diskTotalMb), + })), + Effect.catchAll(() => Effect.succeed(null)), + ) + : null; + const snapshotReservation = snapshotResourceReservation(vm.providerMetadata, snapshotStats); yield* repo.recordUsageEvent({ userId: vm.userId, billingTeamId: vm.billingTeamId, @@ -930,12 +1037,65 @@ export function snapshotVm(input: { eventType: "vm.snapshot.created", provider: vm.provider, imageId: vm.imageId, - metadata: { snapshotId: snapshot.id, named: !!input.name, name: input.name ?? null }, + metadata: { + snapshotId: snapshot.id, + named: !!input.name, + name: input.name ?? null, + // Persist the complete source claim with the snapshot event. Restores + // can then reserve the captured shape after the source VM is gone or + // grows. Unknown dimensions already use a fail-closed pool claim. + vcpus: snapshotReservation.vcpus, + memoryMb: snapshotReservation.memoryMb, + diskMb: snapshotReservation.diskMb, + }, }); return snapshot; }); } +type SnapshotProviderResources = { + readonly vcpus: number | null; + readonly memoryMb: number | null; + readonly diskMb: number | null; +}; + +/** + * Preserve a durable reservation as a floor, while repairing legacy snapshots + * from provider-confirmed dimensions and failing closed for missing fields. + */ +function snapshotResourceReservation( + providerMetadata: Record | null | undefined, + providerResources: SnapshotProviderResources | null, +): VmResourceReservation { + const sourceReservation = vmResourceReservationFromMetadata(providerMetadata); + if (!hasVmResourceReservationMetadata(providerMetadata)) { + return { + vcpus: providerResources?.vcpus ?? PLAN_SHARED_VCPU, + memoryMb: providerResources?.memoryMb ?? PLAN_SHARED_MEMORY_MB, + diskMb: providerResources?.diskMb ?? PLAN_SHARED_DISK_MB, + }; + } + return { + vcpus: Math.max(sourceReservation.vcpus, providerResources?.vcpus ?? sourceReservation.vcpus), + memoryMb: Math.max(sourceReservation.memoryMb, providerResources?.memoryMb ?? sourceReservation.memoryMb), + diskMb: Math.max(sourceReservation.diskMb, providerResources?.diskMb ?? sourceReservation.diskMb), + }; +} + +/** Include the provider's grow-only create target in a captured snapshot claim. */ +function restoreResourceReservation( + snapshotReservation: VmResourceReservation, +): VmResourceReservation { + const createTarget = vmResourceReservationForCreate({ + memoryMb: snapshotReservation.memoryMb, + }); + return { + vcpus: Math.max(snapshotReservation.vcpus, createTarget.vcpus), + memoryMb: Math.max(snapshotReservation.memoryMb, createTarget.memoryMb), + diskMb: Math.max(snapshotReservation.diskMb, createTarget.diskMb), + }; +} + export function restoreVm(input: { readonly userId: string; readonly billingCustomerType: BillingCustomerType; @@ -960,6 +1120,23 @@ export function restoreVm(input: { if (!hasSnapshot) { return yield* Effect.fail(new VmSnapshotNotFoundError({ snapshotId: input.snapshotId })); } + let snapshotReservation: VmResourceReservation | null = null; + if (isPaidVmPlan(input.billingPlanId) && repo.ownedSnapshotResourceReservation) { + snapshotReservation = yield* repo.ownedSnapshotResourceReservation({ + userId: input.userId, + billingTeamId: input.billingTeamId, + provider: input.provider, + snapshotId: input.snapshotId, + }); + } + const resourceReservation = isPaidVmPlan(input.billingPlanId) + ? restoreResourceReservation(snapshotReservation ?? { + ...DEFAULT_VM_RESOURCE_RESERVATION, + // A snapshot event written before resource metadata existed has no + // trustworthy shape. Claim the complete shared pool dimensions. + diskMb: PLAN_SHARED_DISK_MB, + }) + : undefined; return yield* createVm({ userId: input.userId, billingCustomerType: input.billingCustomerType, @@ -969,14 +1146,123 @@ export function restoreVm(input: { provider: input.provider, image: input.snapshotId, imageVersion: null, + ...(resourceReservation ? { memoryMb: resourceReservation.memoryMb } : {}), idempotencyKey: input.idempotencyKey, origin: "restore", + ...(resourceReservation ? { resourceReservation } : {}), modelPlane: input.modelPlane, timing: input.timing, }); }); } +/** + * Resolve the source shape used by a fork. Legacy rows have no durable claim, + * so paid forks read provider stats and fail closed at the shared-pool claim + * when the provider cannot report a dimension. + */ +function resourceReservationForFork( + providers: VmProviderGatewayShape, + source: CloudVmRow, + providerVmId: string, + billingPlanId: string, +): Effect.Effect { + const reservation = vmResourceReservationFromMetadata(source.providerMetadata); + if (!isPaidVmPlan(billingPlanId) || hasVmResourceReservationMetadata(source.providerMetadata)) { + return Effect.succeed(reservation); + } + + // Unknown legacy dimensions claim the complete base pool. This keeps the + // fallback bounded by the entitlement instead of undercounting a large VM. + const unknownShape = { + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: PLAN_SHARED_DISK_MB, + } satisfies VmResourceReservation; + if (!providers.getStats) return Effect.succeed(unknownShape); + return providers.getStats(source.provider, source.providerVmId ?? providerVmId).pipe( + Effect.timeoutFail({ + duration: FOREGROUND_PROVIDER_STATS_TIMEOUT, + onTimeout: () => new Error(`fork source stats timed out for ${source.providerVmId ?? providerVmId}`), + }), + Effect.map((stats) => { + return { + vcpus: vmProviderResourceSize("vcpus", stats.cpus) ?? unknownShape.vcpus, + memoryMb: vmProviderResourceSize("memoryMb", stats.memoryTotalMb) ?? unknownShape.memoryMb, + diskMb: vmProviderResourceSize("diskMb", stats.diskTotalMb) ?? unknownShape.diskMb, + } satisfies VmResourceReservation; + }), + Effect.catchAll(() => Effect.succeed(unknownShape)), + ); +} + +function reservationFromProviderStats( + stats: { readonly cpus?: unknown; readonly memoryTotalMb?: unknown; readonly diskTotalMb?: unknown }, + fallback: VmResourceReservation, + minimum: VmResourceReservation = fallback, +): VmResourceReservation { + return { + vcpus: Math.max(minimum.vcpus, vmProviderResourceSize("vcpus", stats.cpus) ?? fallback.vcpus), + memoryMb: Math.max(minimum.memoryMb, vmProviderResourceSize("memoryMb", stats.memoryTotalMb) ?? fallback.memoryMb), + diskMb: Math.max(minimum.diskMb, vmProviderResourceSize("diskMb", stats.diskTotalMb) ?? fallback.diskMb), + }; +} + +/** Replace a native fork's temporary headroom claim after the copy is measured. */ +function finalizeNativeForkReservation( + repo: VmRepositoryShape, + providers: VmProviderGatewayShape, + input: { + readonly vm: CloudVmRow; + readonly providerVmId: string; + readonly fallbackReservation: VmResourceReservation; + readonly minimumReservation: VmResourceReservation; + readonly sharedResourceCapacity: VmResourceReservation; + }, +): Effect.Effect { + const setReservation = repo.setResourceReservation; + const getStats = providers.getStats; + if (!setReservation || !getStats) return Effect.void; + // beginCreate may expand the requested floor to the remaining headroom. Read + // that exact marker back from the returned row so the replacement is a CAS, + // even when another adapter computes a different temporary claim. + const expectedReservation = vmResourceReservationFromMetadata( + input.vm.providerMetadata, + input.fallbackReservation, + ); + return getStats(input.vm.provider, input.providerVmId).pipe( + Effect.timeoutFail({ + duration: FOREGROUND_PROVIDER_STATS_TIMEOUT, + onTimeout: () => new Error(`fork copy stats timed out for ${input.providerVmId}`), + }), + Effect.map((stats) => reservationFromProviderStats( + stats, + input.fallbackReservation, + input.minimumReservation, + )), + // A successful provider fork is still usable when its first stats read is + // unavailable. Keep the temporary claim and let the bounded reconciler + // replace it later; never release capacity on an unconfirmed shape. + Effect.catchAll(() => Effect.succeed(null)), + Effect.flatMap((reservation) => { + if (!reservation) return Effect.void; + return setReservation({ + id: input.vm.id, + expectedReservation, + reservation, + sharedResourceCapacity: input.sharedResourceCapacity, + }).pipe( + Effect.flatMap((replaced) => replaced + ? Effect.void + : Effect.fail(new VmDatabaseError({ + operation: "replaceForkResourceReservation", + cause: new Error("fork reservation generation changed before finalization"), + }))), + ); + }), + ); +} + export function forkVm(input: { readonly userId: string; readonly billingCustomerType: BillingCustomerType; @@ -1014,15 +1300,57 @@ export function forkVm(input: { { forceProviderProbe: true }, ); - if (source.provider === "freestyle" && providers.fork) { + const nativeFork = source.provider === "freestyle" && providers.fork !== undefined; + // Native forks are serialized with source resizes by the create + // transaction's temporary headroom claim. Do not read source stats first: + // that read would race a resize before the claim is acquired. + const sourceHasReservation = hasVmResourceReservationMetadata(source.providerMetadata); + const nativeForkReservation = isPaidVmPlan(input.billingPlanId) + ? sourceHasReservation + ? vmResourceReservationFromMetadata(source.providerMetadata) + : { + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: PLAN_SHARED_DISK_MB, + } + : undefined; + + if (nativeFork) { + const sourceReservation = nativeForkReservation ?? DEFAULT_VM_RESOURCE_RESERVATION; const create = yield* beginCreateWithLazyProviderRefresh(repo, providers, { userId: input.userId, billingTeamId: input.billingTeamId, - billingPlanId: input.billingPlanId, provider: source.provider, image: source.imageId, imageVersion: source.imageVersion, maxActiveVms: input.maxActiveVms, + ...(isPaidVmPlan(input.billingPlanId) + ? { + resourceReservation: sourceReservation, + sharedResourceCapacity: sharedResourceCapacityForMaxActiveVms(input.maxActiveVms), + reserveSharedResourceHeadroom: true, + forkMinimumResourceReservation: sourceHasReservation + ? sourceReservation + : { vcpus: 1, memoryMb: 4 * 1024, diskMb: 16 * 1024 }, + } + : {}), + ...(!sourceHasReservation + ? { + refreshResourceReservation: () => Effect.suspend(() => repo.findUserVm({ + userId: input.userId, + billingTeamId: input.billingTeamId, + providerVmId: source.providerVmId ?? input.providerVmId, + provider: source.provider, + }).pipe( + Effect.map((row) => row && hasVmResourceReservationMetadata(row.providerMetadata) + ? vmResourceReservationFromMetadata(row.providerMetadata) + : null), + )), + } + : {}), + // The helper uses the plan to reconcile legacy rows before its shared + // resource transaction. Keep this field explicit after all spreads. + billingPlanId: input.billingPlanId, idempotencyKey: input.idempotencyKey, timing: input.timing, }); @@ -1095,17 +1423,30 @@ export function forkVm(input: { ), ); - const running = yield* measureVmEffect( - input.timing, - "mark_running", - repo.markCreateRunning({ - id: create.vm.id, - providerVmId: handle.providerVmId, - image: source.imageId, - imageVersion: source.imageVersion, - providerMetadata: handle.providerMetadata ?? source.providerMetadata, - }), - ).pipe( + const running = yield* Effect.gen(function* () { + if (isPaidVmPlan(input.billingPlanId)) { + yield* finalizeNativeForkReservation(repo, providers, { + vm: create.vm, + providerVmId: handle.providerVmId, + fallbackReservation: sourceReservation, + minimumReservation: sourceHasReservation + ? sourceReservation + : { vcpus: 1, memoryMb: 4 * 1024, diskMb: 16 * 1024 }, + sharedResourceCapacity: sharedResourceCapacityForMaxActiveVms(input.maxActiveVms), + }); + } + return yield* measureVmEffect( + input.timing, + "mark_running", + repo.markCreateRunning({ + id: create.vm.id, + providerVmId: handle.providerVmId, + image: source.imageId, + imageVersion: source.imageVersion, + providerMetadata: handle.providerMetadata ?? source.providerMetadata, + }), + ); + }).pipe( Effect.catchAll((err) => Effect.gen(function* () { yield* rollbackProviderCreate(providers, source.provider, handle); @@ -1160,6 +1501,12 @@ export function forkVm(input: { providerVmId: input.providerVmId, name: input.name, }); + // Snapshotting establishes the copy point for providers without a native + // fork. Read the source shape after that point so a concurrent grow cannot + // understate the copied machine's claim. + const sourceReservation = isPaidVmPlan(input.billingPlanId) + ? yield* resourceReservationForFork(providers, source, input.providerVmId, input.billingPlanId) + : undefined; const fork = yield* createVm({ userId: input.userId, billingCustomerType: input.billingCustomerType, @@ -1169,6 +1516,7 @@ export function forkVm(input: { provider: source.provider, image: snapshot.id, imageVersion: null, + ...(sourceReservation ? { resourceReservation: sourceReservation } : {}), idempotencyKey: input.idempotencyKey, origin: "fork", timing: input.timing, @@ -1199,24 +1547,386 @@ function beginCreateWithLazyProviderRefresh( readonly billingTeamId: string; readonly modelPlane?: VmModelPlaneRevoker; readonly timing?: VmTimingSink; + /** Re-read a source claim after scoped legacy repair before retrying. */ + readonly refreshResourceReservation?: () => Effect.Effect; } & Parameters[0], ): Effect.Effect { - return measureVmEffect(input.timing, "begin_create", repo.beginCreate(input)).pipe( + // Construct the repository effect lazily. A count conflict refreshes provider + // statuses. A shared-resource conflict repairs the scoped legacy claims that + // can otherwise make an account wait for the background batch. An impossible + // requested dimension is returned directly, because no refresh can change it. + const beginCreate = Effect.suspend(() => + measureVmEffect(input.timing, "begin_create", repo.beginCreate(input)) + ); + return beginCreate.pipe( Effect.catchAll((err) => { - if (!isVmLimitExceededError(err)) return Effect.fail(err); - return Effect.gen(function* () { - yield* measureVmEffect( - input.timing, - "limit_reconcile", - refreshActiveLimitProviderStatuses(repo, providers, input), - ).pipe(Effect.catchAll(() => Effect.void)); - return yield* measureVmEffect(input.timing, "begin_create", repo.beginCreate(input)); + if (!isVmLimitExceededError(err) && !isVmSharedResourceLimitExceededError(err)) return Effect.fail(err); + if (isVmSharedResourceLimitExceededError(err) && err.requested > err.limit) return Effect.fail(err); + const reconcile = isVmSharedResourceLimitExceededError(err) + ? reconcileSharedResourceLimit(repo, providers, input) + : refreshActiveLimitProviderStatuses(repo, providers, input); + const retryInput = input.refreshResourceReservation + ? Effect.suspend(() => input.refreshResourceReservation!()).pipe( + Effect.map((reservation) => { + return reservation ? { ...input, resourceReservation: reservation } : input; + }), + // A failed source re-read leaves the conservative original claim in + // place. The retry may fail closed, while the background pass repairs + // the row later. + Effect.catchAll(() => Effect.succeed(input)), + ) + : Effect.succeed(input); + return measureVmEffect( + input.timing, + "limit_reconcile", + reconcile, + ).pipe( + Effect.catchAll(() => Effect.void), + Effect.andThen(retryInput), + Effect.flatMap((nextInput) => Effect.suspend(() => + measureVmEffect(input.timing, "begin_create", repo.beginCreate(nextInput)) + )), + ); + }), + ); +} + +/** Backfill legacy claims in a bounded background batch. */ +function reconcileLegacyResourceReservations( + repo: VmRepositoryShape, + providers: VmProviderGatewayShape, + input: { + readonly userId?: string; + readonly billingTeamId?: string | null; + readonly limit?: number; + }, +): Effect.Effect { + const findCandidates = repo.legacyResourceReservationCandidates; + if (!findCandidates || !repo.setResourceReservation || !providers.getStats) return Effect.void; + + return Effect.gen(function*() { + const candidates = yield* findCandidates({ + userId: input.userId, + billingTeamId: input.billingTeamId, + limit: input.limit ?? LEGACY_RESOURCE_RECONCILE_BATCH_LIMIT, + }).pipe(Effect.catchAll(() => Effect.succeed([]))); + yield* Effect.forEach( + candidates, + (vm) => reconcileLegacyResourceCandidate(repo, providers, vm), + { concurrency: LEGACY_RESOURCE_RECONCILE_CONCURRENCY, discard: true }, + ); + }); +} + +/** Refresh lifecycle state and legacy claims before retrying a full pool. */ +function reconcileSharedResourceLimit( + repo: VmRepositoryShape, + providers: VmProviderGatewayShape, + input: { + readonly userId: string; + readonly billingTeamId: string; + }, +): Effect.Effect { + return Effect.gen(function*() { + // A deleted provider VM can retain a valid reservation marker. Refresh a + // small status set first, then repair rows whose marker is missing. + yield* refreshActiveLimitProviderStatuses(repo, providers, { + ...input, + limit: LEGACY_RESOURCE_RECONCILE_REQUEST_LIMIT, + }).pipe(Effect.catchAll(() => Effect.void)); + yield* reconcileLegacyResourceReservations(repo, providers, { + ...input, + limit: LEGACY_RESOURCE_RECONCILE_REQUEST_LIMIT, + }); + }).pipe( + Effect.timeoutFail({ + duration: LEGACY_RESOURCE_RECONCILE_REQUEST_TIMEOUT, + onTimeout: () => new Error("shared resource repair timed out before create retry"), + }), + Effect.catchAll(() => Effect.void), + ); +} + +/** Defer one candidate with durable backoff when the provider cannot be read. */ +function deferLegacyResourceCandidate( + repo: VmRepositoryShape, + vm: CloudVmRow, + requestedAttemptAtMs?: number, +): Effect.Effect { + const defer = repo.deferResourceReservation; + if (!defer) return Effect.void; + const nowMs = Date.now(); + const nextAttemptAtMs = Math.max( + nowMs + LEGACY_RESOURCE_RECONCILE_RETRY_AFTER_MS, + requestedAttemptAtMs ?? 0, + ); + return defer({ + id: vm.id, + nextAttemptAt: new Date(nextAttemptAtMs), + }).pipe(Effect.catchAll(() => Effect.void)); +} + +type ResourceReservationWriter = NonNullable; +type ResizeUnconfirmedWriter = NonNullable; + +function reservationFromLegacyProviderStats( + stats: VMStats, + existing: VmResourceReservation, + diskMb: number, + minimumDiskMb: number, +): VmResourceReservation { + return { + ...existing, + vcpus: vmProviderResourceSize("vcpus", stats.cpus) ?? existing.vcpus, + memoryMb: vmProviderResourceSize("memoryMb", stats.memoryTotalMb) ?? existing.memoryMb, + diskMb: Math.max(minimumDiskMb, diskMb), + }; +} + +function reconcilePendingForkReservation(input: { + readonly setReservation: ResourceReservationWriter; + readonly vmId: string; + readonly stats: VMStats; + readonly existing: VmResourceReservation; + readonly minimum: VmResourceReservation; + readonly diskMb: number; +}) { + // The temporary headroom claim is larger than the copied VM by design. + // Replace each valid dimension with the measured shape while retaining the + // source claim as a floor. An invalid dimension keeps its full temporary + // hold so a partial provider response cannot undercount. + const observedVcpus = vmProviderResourceSize("vcpus", input.stats.cpus); + const observedMemoryMb = vmProviderResourceSize("memoryMb", input.stats.memoryTotalMb); + const reservation = { + vcpus: observedVcpus === null + ? input.existing.vcpus + : Math.max(input.minimum.vcpus, observedVcpus), + memoryMb: observedMemoryMb === null + ? input.existing.memoryMb + : Math.max(input.minimum.memoryMb, observedMemoryMb), + diskMb: Math.max(input.minimum.diskMb, input.diskMb), + }; + return input.setReservation({ + id: input.vmId, + reservation, + expectedReservation: input.existing, + }).pipe(Effect.asVoid); +} + +function recoverIncompletePendingResize(input: { + readonly repo: VmRepositoryShape; + readonly markUnconfirmed: ResizeUnconfirmedWriter | undefined; + readonly vm: CloudVmRow; + readonly existing: VmResourceReservation; + readonly pending: VmResourceResizePending; +}) { + // A worker can die after reserving headroom but before provider I/O. After + // the recovery window, retain a maximum claim until stats prove that the + // requested size exists. + if (!input.markUnconfirmed || !resizePendingHasExpired(input.pending)) { + return deferLegacyResourceCandidate(input.repo, input.vm); + } + return input.markUnconfirmed({ + id: input.vm.id, + expectedDiskMb: input.existing.diskMb, + minimumDiskMb: input.pending.requestedDiskMb, + previousDiskMb: input.pending.previousDiskMb, + operationId: input.pending.operationId, + }).pipe( + Effect.asVoid, + Effect.catchAll(() => deferLegacyResourceCandidate(input.repo, input.vm)), + ); +} + +function reconcileUnconfirmedResize(input: { + readonly repo: VmRepositoryShape; + readonly setReservation: ResourceReservationWriter; + readonly vm: CloudVmRow; + readonly stats: VMStats; + readonly existing: VmResourceReservation; + readonly unconfirmed: VmResourceResizeUnconfirmed; + readonly diskMb: number; +}) { + // Keep the maximum claim while the provider reports a stale size. Clearing + // the marker before the requested size is observed would undercount the pool. + if (input.diskMb < input.unconfirmed.requestedDiskMb) { + if (!unconfirmedResizeRecoveryHasExpired(input.vm, input.unconfirmed)) { + return deferLegacyResourceCandidate(input.repo, input.vm); + } + // The provider stayed below the requested size for the complete recovery + // window. Assume the resize never applied and release the temporary claim. + const reservation = reservationFromLegacyProviderStats( + input.stats, + input.existing, + input.diskMb, + Math.max( + DEFAULT_VM_RESOURCE_RESERVATION.diskMb, + input.unconfirmed.previousDiskMb ?? 0, + ), + ); + return input.setReservation({ + id: input.vm.id, + reservation, + expectedResizeUnconfirmedOperationId: input.unconfirmed.operationId, + }).pipe(Effect.asVoid); + } + const reservation = reservationFromLegacyProviderStats( + input.stats, + input.existing, + input.diskMb, + input.unconfirmed.requestedDiskMb, + ); + return input.setReservation({ + id: input.vm.id, + reservation, + expectedResizeUnconfirmedOperationId: input.unconfirmed.operationId, + }).pipe(Effect.asVoid); +} + +function reconcileMeasuredLegacyReservation(input: { + readonly setReservation: ResourceReservationWriter; + readonly vmId: string; + readonly stats: VMStats; + readonly existing: VmResourceReservation; + readonly pending: VmResourceResizePending | null; + readonly diskMb: number; +}) { + const minimumDiskMb = input.pending?.requestedDiskMb ?? DEFAULT_VM_RESOURCE_RESERVATION.diskMb; + const reservation = reservationFromLegacyProviderStats( + input.stats, + input.existing, + input.diskMb, + minimumDiskMb, + ); + return input.setReservation({ + id: input.vmId, + reservation, + ...(input.pending ? { expectedResizeOperationId: input.pending.operationId } : {}), + }).pipe(Effect.asVoid); +} + +/** Reconcile one legacy row, keeping provider work outside the request path. */ +function reconcileLegacyResourceCandidate( + repo: VmRepositoryShape, + providers: VmProviderGatewayShape, + vm: CloudVmRow, +): Effect.Effect { + const setReservation = repo.setResourceReservation; + const markUnconfirmed = repo.markVmResizeUnconfirmed; + const getStats = providers.getStats; + const providerVmId = vm.providerVmId; + if (!providerVmId || !setReservation || !getStats) return Effect.void; + + const metadata = vm.providerMetadata ?? {}; + const hasPendingMarker = Object.prototype.hasOwnProperty.call( + metadata, + VM_RESOURCE_RESIZE_PENDING_METADATA_KEY, + ); + const pending = vmResourceResizePendingFromMetadata(metadata); + const hasUnconfirmedMarker = Object.prototype.hasOwnProperty.call( + metadata, + VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY, + ); + const unconfirmed = vmResourceResizeUnconfirmedFromMetadata(metadata); + const hasForkPendingMarker = Object.prototype.hasOwnProperty.call( + metadata, + VM_RESOURCE_FORK_PENDING_METADATA_KEY, + ); + const forkMinimumReservation = vmResourceForkPendingFromMetadata(metadata); + const retry = vmResourceReconcileRetryFromMetadata(metadata); + // The repository filters these rows in SQL. Keep this second boundary for + // alternate adapters and stale replicas. + if (retry && retry.nextAttemptAtMs > Date.now()) return Effect.void; + // A malformed control marker has no safe generation to clear. Keep it and + // retry later instead of releasing a newer claim by accident. + if (hasPendingMarker && !pending || hasUnconfirmedMarker && !unconfirmed || hasForkPendingMarker && !forkMinimumReservation) { + return deferLegacyResourceCandidate(repo, vm); + } + // A live pending resize still has an owner that can confirm it. Do not read + // provider stats and clear the marker while that request may be in flight. + if (pending && !resizePendingHasExpired(pending)) { + const recoveryAtMs = pending.createdAtMs === undefined + ? undefined + : pending.createdAtMs + RESIZE_PENDING_RECOVERY_AFTER_MS; + return deferLegacyResourceCandidate(repo, vm, recoveryAtMs); + } + + const readStats = getStats(vm.provider, providerVmId).pipe( + // A hung provider read must not hold the cron worker or starve later rows. + Effect.timeoutFail({ + duration: LEGACY_RESOURCE_RECONCILE_PROVIDER_TIMEOUT, + onTimeout: () => new Error(`legacy resource stats timed out for ${providerVmId}`), + }), + ); + return readStats.pipe( + Effect.flatMap((stats) => { + const diskMb = vmProviderResourceSize("diskMb", stats.diskTotalMb); + if (diskMb === null) return deferLegacyResourceCandidate(repo, vm); + const existing = vmResourceReservationFromMetadata(metadata); + if (hasForkPendingMarker && forkMinimumReservation) { + return reconcilePendingForkReservation({ + setReservation, + vmId: vm.id, + stats, + existing, + minimum: forkMinimumReservation, + diskMb, + }); + } + if (pending && diskMb < pending.requestedDiskMb) { + return recoverIncompletePendingResize({ + repo, + markUnconfirmed, + vm, + existing, + pending, + }); + } + if (unconfirmed) { + return reconcileUnconfirmedResize({ + repo, + setReservation, + vm, + stats, + existing, + unconfirmed, + diskMb, + }); + } + return reconcileMeasuredLegacyReservation({ + setReservation, + vmId: vm.id, + stats, + existing, + pending, + diskMb, }); }), + // An unavailable provider leaves the claim conservative and retries later. + Effect.catchAll(() => deferLegacyResourceCandidate(repo, vm)), ); } -/** Refresh live provider state before retrying an active-VM limit conflict. */ +function resizePendingHasExpired( + pending: VmResourceResizePending, +): boolean { + // Old markers have no reliable start time. Treat them as recoverable so a + // migration cannot remain blocked forever; new markers carry their own + // generation timestamp and get the full recovery window. + if (pending.createdAtMs === undefined) return true; + const startedAtMs = pending.createdAtMs; + return Date.now() - startedAtMs >= RESIZE_PENDING_RECOVERY_AFTER_MS; +} + +function unconfirmedResizeRecoveryHasExpired( + vm: Pick, + unconfirmed: { readonly markedAtMs?: number }, +): boolean { + const markedAtMs = unconfirmed.markedAtMs ?? vm.updatedAt.getTime(); + return Date.now() - markedAtMs >= RESIZE_UNCONFIRMED_RECOVERY_AFTER_MS; +} + +/** Refresh live provider state before retrying a count or shared-resource limit conflict. */ function refreshActiveLimitProviderStatuses( repo: VmRepositoryShape, providers: VmProviderGatewayShape, @@ -1224,11 +1934,13 @@ function refreshActiveLimitProviderStatuses( readonly userId: string; readonly billingTeamId: string; readonly modelPlane?: VmModelPlaneRevoker; + readonly limit?: number; }, ): Effect.Effect { return Effect.gen(function* () { const getStatus = providers.getStatus; - if (!getStatus) return; + if (!getStatus || !repo.activeLimitCandidates) return; + const limit = input.limit ?? VM_STATUS_RECONCILE_BATCH_LIMIT; const candidates = yield* repo.activeLimitCandidates({ userId: input.userId, @@ -1236,12 +1948,12 @@ function refreshActiveLimitProviderStatuses( // Keep the synchronous retry bounded. If an account has more rows than // this, the database remains conservative until the background reconcile // catches up; we never create above the recorded active limit. - limit: VM_STATUS_RECONCILE_BATCH_LIMIT, + limit, }); // The repository applies the limit in SQL. Keep a second boundary here so // alternate repository implementations cannot turn this request path into // an unbounded provider sweep. - yield* Effect.forEach(candidates.slice(0, VM_STATUS_RECONCILE_BATCH_LIMIT), (vm) => { + yield* Effect.forEach(candidates.slice(0, limit), (vm) => { const providerVmId = vm.providerVmId; if (!providerVmId) return Effect.void; // Provider-agnostic on purpose: the cron reconcile path already refreshes @@ -1941,7 +2653,12 @@ export function resizeVm(input: { readonly teamIds?: readonly string[]; readonly providerVmId: string; readonly storageMb: number; + /** Current caller/VM plan. Shared capacity applies to paid plans only. */ + readonly billingPlanId?: string | null; + /** Team allowance used to scale the shared resource pool. */ + readonly maxActiveVms?: number | null; }) { + // oxlint-disable-next-line complexity -- Resize orchestration must keep reservation, provider, rollback, and confirmation order explicit. return Effect.gen(function* () { const repo = yield* VmRepository; const providers = yield* VmProviderGateway; @@ -1953,8 +2670,8 @@ export function resizeVm(input: { forceProviderProbe: true, }); const current = yield* providers.getStats(vm.provider, input.providerVmId); - const currentMb = current.diskTotalMb; - if (currentMb === undefined || currentMb === null || currentMb <= 0) { + const currentMb = vmProviderResourceSize("diskMb", current.diskTotalMb); + if (currentMb === null) { return yield* Effect.fail(new VmOperationUnsupportedError({ provider: vm.provider, operation: "resize" })); } if (input.storageMb < currentMb) { @@ -1975,9 +2692,84 @@ export function resizeVm(input: { reason: "above_max", })); } + // Claim the new disk size under the same billing-team lock used by create. + // The live repository always provides this method; test doubles from + // before shared-pool accounting may omit it and exercise provider behavior + // without a database. + let reservation: VmResizeReservation | null = null; + if (repo.reserveVmResize && isPaidVmPlan(input.billingPlanId ?? vm.billingPlanId ?? "")) { + reservation = yield* repo.reserveVmResize({ + id: vm.id, + userId: input.userId, + billingTeamId: vm.billingTeamId ?? input.billingTeamId, + providerVmId: input.providerVmId, + currentDiskMb: currentMb, + storageMb: input.storageMb, + maxActiveVms: input.maxActiveVms ?? maxActiveVmsForPlan(vm.billingPlanId), + sharedResourceCapacity: sharedResourceCapacityForMaxActiveVms( + input.maxActiveVms ?? maxActiveVmsForPlan(vm.billingPlanId), + ), + }); + if (!reservation) { + return yield* Effect.fail(new VmNotFoundError({ vmId: input.providerVmId })); + } + } + // A no-op request still backfills the durable reservation for legacy rows + // whose provider metadata predates the shared-pool policy. if (input.storageMb === currentMb) return current; - yield* providers.resize(vm.provider, input.providerVmId, { storageMb: input.storageMb }); - const updated = yield* providers.getStats(vm.provider, input.providerVmId); + const rollbackReservation = () => reservation && repo.restoreVmResize + ? repo.restoreVmResize({ + id: vm.id, + expectedDiskMb: reservation.reservedDiskMb, + previousDiskMb: reservation.previousDiskMb, + operationId: reservation.operationId, + }).pipe(Effect.catchAll(() => Effect.void)) + : Effect.void; + const rollbackIfProviderDidNotGrow = ( + exit: Exit.Exit, + ) => { + if (!reservation || !repo.restoreVmResize || Exit.isSuccess(exit)) return Effect.void; + // A provider request can complete and lose its response before the + // caller observes success. Release the claim only when a fresh provider + // read proves that the disk is still at its pre-resize size. If the read + // fails or reports growth, keep the larger claim as a safe upper bound. + return providers.getStats!(vm.provider, input.providerVmId).pipe( + Effect.flatMap((stats) => { + const observedDiskMb = vmProviderResourceSize("diskMb", stats.diskTotalMb); + return observedDiskMb !== null && observedDiskMb <= currentMb + ? rollbackReservation() + : Effect.void; + }), + Effect.catchAll(() => Effect.void), + ); + }; + yield* providers.resize(vm.provider, input.providerVmId, { storageMb: input.storageMb }).pipe( + Effect.onExit(rollbackIfProviderDidNotGrow), + ); + const updated = yield* providers.getStats(vm.provider, input.providerVmId).pipe( + Effect.tapError(() => finalizeUnobservedResize(repo, vm.id, reservation)), + ); + // The provider can round a requested disk up. Persist the observed claim + // before returning so the next shared-pool check cannot undercount it. + // Missing or malformed stats fail closed at the per-VM maximum. + const confirmedDiskMb = vmProviderResourceSize("diskMb", updated.diskTotalMb) ?? VM_DISK_MB_MAX; + if (reservation && repo.confirmVmResize) { + const confirmed = yield* repo.confirmVmResize({ + id: vm.id, + expectedDiskMb: reservation.reservedDiskMb, + ...(reservation.requestedDiskMb === undefined + ? {} + : { minimumDiskMb: reservation.requestedDiskMb }), + confirmedDiskMb, + operationId: reservation.operationId, + }); + if (!confirmed) { + return yield* Effect.fail(new VmDatabaseError({ + operation: "confirmVmResize", + cause: new Error("resize confirmation no longer owns the pending generation"), + })); + } + } yield* repo.recordUsageEvent({ userId: input.userId, billingTeamId: vm.billingTeamId, @@ -1986,12 +2778,45 @@ export function resizeVm(input: { eventType: "vm.resize", provider: vm.provider, imageId: vm.imageId, - metadata: { storageMb: input.storageMb, previousStorageMb: currentMb }, + metadata: { + storageMb: input.storageMb, + confirmedStorageMb: confirmedDiskMb, + previousStorageMb: currentMb, + }, }).pipe(Effect.catchAll(() => Effect.void)); return updated; }); } +/** + * A successful provider resize followed by a lost stats response still owns + * its reservation. Replace the active marker with an unconfirmed marker so a + * later reconcile can lower the conservative claim without blocking new work. + */ +function finalizeUnobservedResize( + repo: VmRepositoryShape, + vmId: string, + reservation: VmResizeReservation | null, +): Effect.Effect { + if (!reservation || !repo.markVmResizeUnconfirmed) return Effect.void; + return repo.markVmResizeUnconfirmed({ + id: vmId, + expectedDiskMb: reservation.reservedDiskMb, + ...(reservation.requestedDiskMb === undefined + ? {} + : { minimumDiskMb: reservation.requestedDiskMb }), + previousDiskMb: reservation.previousDiskMb, + operationId: reservation.operationId, + }).pipe( + Effect.asVoid, + Effect.catchAll((err) => + Effect.sync(() => { + console.error(`[vm] could not finalize unobserved resize for ${vmId}`, errorMessage(err)); + }), + ), + ); +} + export function openVmPort(input: { readonly userId: string; readonly billingTeamId?: string | null; diff --git a/web/tests/app-pricing-page.test.tsx b/web/tests/app-pricing-page.test.tsx index 888722ad7529..70d760301c52 100644 --- a/web/tests/app-pricing-page.test.tsx +++ b/web/tests/app-pricing-page.test.tsx @@ -101,6 +101,9 @@ describe("app pricing page", () => { expect(html).not.toContain("/mo."); expect(html).toContain("$50"); expect(html).toContain("$60/user/mo"); + expect(html).toContain( + "Up to 50 Cloud VMs, all sharing a total of 5 vCPU, 20 GB RAM, and 200 GB disk", + ); expect(html).toContain('

Includes:

'); expect(html).not.toContain('style="min-height:4rem"'); expect(html).toContain("text-3xl font-medium tabular-nums tracking-tight"); diff --git a/web/tests/pricing-page.test.tsx b/web/tests/pricing-page.test.tsx index 776c7096122a..81901d1c1736 100644 --- a/web/tests/pricing-page.test.tsx +++ b/web/tests/pricing-page.test.tsx @@ -42,6 +42,12 @@ mock.module("next-intl/server", () => ({ setRequestLocale: () => undefined, })); +mock.module("../i18n/navigation", () => ({ + Link: ({ href, children, ...props }: { href: string; children: React.ReactNode }) => ( + {children} + ), +})); + mock.module("../app/[locale]/components/site-header", () => ({ SiteHeader: () =>
, })); @@ -153,10 +159,10 @@ describe("localized pricing page", () => { "/api/billing/checkout?plan=pro&cmux_external_browser=1&interval=year", ); expect(html).toMatch( - /href="\/api\/billing\/checkout\?plan=pro[^"]*interval=year"[^>]*class="[^"]*px-5 py-2\.5 text-\[15px\][^"]*"[^>]*>Get Pro/, + /href="\/api\/billing\/checkout\?plan=pro[^"]*interval=year"[^>]*class="[^"]*min-h-12 px-5 py-3 text-\[15px\][^"]*"[^>]*>Get Pro/, ); expect(html).toMatch( - /href="\/api\/billing\/checkout\?plan=team[^"]*interval=year"[^>]*class="[^"]*px-5 py-2\.5 text-\[15px\][^"]*"[^>]*>Get Teams/, + /href="\/api\/billing\/checkout\?plan=team[^"]*interval=year"[^>]*class="[^"]*min-h-12 px-5 py-3 text-\[15px\][^"]*"[^>]*>Get Teams/, ); expect(html).toContain('

Includes:

'); expect(html).not.toContain('style="min-height:4rem"'); @@ -241,7 +247,9 @@ describe("localized pricing page", () => { expect(html).toContain("$50"); expect(html).toContain("$60"); - expect(html).toContain("Up to 50 Cloud VMs, each with 8 GB RAM and 32 GB disk by default; sizes from 4 to 64 GB RAM are available"); + expect(html).toContain( + "Up to 50 Cloud VMs, all sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk; each VM starts at 8 GB RAM and 32 GB disk, with sizes from 4 to 64 GB RAM available as capacity allows", + ); expect(html).toContain("Unlimited workspaces"); expect(html).not.toContain("Unlimited active Cloud VMs"); expect(html).toContain( diff --git a/web/tests/pro-pricing.test.ts b/web/tests/pro-pricing.test.ts index ef318a9bd769..395133434027 100644 --- a/web/tests/pro-pricing.test.ts +++ b/web/tests/pro-pricing.test.ts @@ -2,6 +2,8 @@ import { describe, expect, test } from "bun:test"; import enMessages from "../messages/en.json"; import jaMessages from "../messages/ja.json"; +import { loadMessages } from "../i18n/messages"; +import { locales } from "../i18n/routing"; import { LEGACY_PRICE_LOOKUP_KEYS, PRO_PRICING_USD, @@ -9,8 +11,15 @@ import { proBillingInterval, } from "../services/billing/plans"; import { - PAID_MAX_ACTIVE_VMS_DEFAULT, + PLAN_SHARED_DISK_MB, + PLAN_SHARED_MEMORY_MB, + PLAN_SHARED_VCPU, PLAN_MACHINE_MEMORY_MB, + PAID_MAX_ACTIVE_VMS_DEFAULT, + firstExceededSharedResource, + sharedResourceUsage, + sharedResourceCapacityForMaxActiveVms, + vmResourceReservationForCreate, VM_DISK_MB_DEFAULT, } from "../services/vms/entitlements"; @@ -90,39 +99,185 @@ describe("pricing plans", () => { }); describe("pricing copy matches the plan policy", () => { - // The public pricing copy states the machine allowance as prose, so pin the - // numbers to the entitlement constants that enforce them. A price or spec - // change that forgets the copy (or the copy that forgets the policy) fails - // here instead of on the live page. + // The public pricing copy states both the shared pool and the selectable + // machine default. Pin each number to the policy constants so copy and + // enforcement cannot drift independently. + const sharedVcpus = PLAN_SHARED_VCPU; + const sharedMemoryGb = PLAN_SHARED_MEMORY_MB / 1024; + const sharedDiskGb = PLAN_SHARED_DISK_MB / 1024; const memoryGb = PLAN_MACHINE_MEMORY_MB / 1024; - const diskGb = VM_DISK_MB_DEFAULT / 1024; + const startingDiskGb = VM_DISK_MB_DEFAULT / 1024; - test("the default machine is 8 GB RAM, 32 GB disk, up to 50 machines", () => { + test("the default machine and shared Cloud VM capacity match the plan", () => { + expect(sharedVcpus).toBe(5); + expect(sharedMemoryGb).toBe(20); + expect(sharedDiskGb).toBe(200); expect(memoryGb).toBe(8); - expect(diskGb).toBe(32); + expect(startingDiskGb).toBe(32); expect(PAID_MAX_ACTIVE_VMS_DEFAULT).toBe(50); }); - for (const [locale, messages] of [ - ["en", enMessages], - ["ja", jaMessages], + test("new Cloud VM disks start at 32 GB", () => { + expect(startingDiskGb).toBe(32); + }); + + test("the shared pool scales by paid seat and sums every resource claim", () => { + expect(sharedResourceCapacityForMaxActiveVms(PAID_MAX_ACTIVE_VMS_DEFAULT)).toEqual({ + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: PLAN_SHARED_DISK_MB, + }); + expect(sharedResourceCapacityForMaxActiveVms(PAID_MAX_ACTIVE_VMS_DEFAULT * 2)).toEqual({ + vcpus: PLAN_SHARED_VCPU * 2, + memoryMb: PLAN_SHARED_MEMORY_MB * 2, + diskMb: PLAN_SHARED_DISK_MB * 2, + }); + expect(firstExceededSharedResource({ + used: { + vcpus: PLAN_SHARED_VCPU - 1, + memoryMb: PLAN_SHARED_MEMORY_MB - 1, + diskMb: PLAN_SHARED_DISK_MB - 1, + }, + requested: { vcpus: 1, memoryMb: 1, diskMb: 2 }, + capacity: { + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: PLAN_SHARED_DISK_MB, + }, + })).toEqual({ + resource: "diskMb", + used: PLAN_SHARED_DISK_MB - 1, + requested: 2, + limit: PLAN_SHARED_DISK_MB, + }); + expect(sharedResourceUsage("vcpus", PLAN_SHARED_VCPU, 1)).toBe(PLAN_SHARED_VCPU + 1); + expect(sharedResourceUsage("memoryMb", PLAN_SHARED_MEMORY_MB, 1)).toBe(PLAN_SHARED_MEMORY_MB + 1); + expect(sharedResourceUsage("diskMb", PLAN_SHARED_DISK_MB - 1, 2)).toBe(PLAN_SHARED_DISK_MB + 1); + expect(firstExceededSharedResource({ + used: { + vcpus: PLAN_SHARED_VCPU - 1, + memoryMb: PLAN_SHARED_MEMORY_MB - 1, + diskMb: VM_DISK_MB_DEFAULT, + }, + requested: { + vcpus: 1, + memoryMb: 1, + diskMb: VM_DISK_MB_DEFAULT, + }, + capacity: { + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: PLAN_SHARED_DISK_MB, + }, + })).toBeNull(); + expect(firstExceededSharedResource({ + used: { vcpus: PLAN_SHARED_VCPU - 1, memoryMb: 0, diskMb: 0 }, + requested: { vcpus: 2, memoryMb: 1, diskMb: 1 }, + capacity: { + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: PLAN_SHARED_DISK_MB, + }, + })).toEqual({ + resource: "vcpus", + used: PLAN_SHARED_VCPU - 1, + requested: 2, + limit: PLAN_SHARED_VCPU, + }); + }); + + test("a size-less plan reservation follows requested memory", () => { + expect(vmResourceReservationForCreate({ + memoryMb: PLAN_SHARED_MEMORY_MB, + env: {}, + })).toEqual({ + vcpus: PLAN_SHARED_VCPU, + memoryMb: PLAN_SHARED_MEMORY_MB, + diskMb: VM_DISK_MB_DEFAULT, + }); + }); + + test("a sized image reserves its complete provider shape", () => { + expect(vmResourceReservationForCreate({ + imageSize: { cpu: 8, memoryMb: 32768, storageMb: 65536 }, + })).toEqual({ vcpus: 8, memoryMb: 32768, diskMb: 65536 }); + }); + + test("a 4 GB image still reserves the documented 32 GB starting disk", () => { + expect(vmResourceReservationForCreate({ + imageSize: { cpu: 1, memoryMb: 4096, storageMb: 16384 }, + })).toEqual({ vcpus: 1, memoryMb: 4096, diskMb: VM_DISK_MB_DEFAULT }); + }); + + test("an image reservation includes an operator disk override", () => { + expect(vmResourceReservationForCreate({ + imageSize: { cpu: 1, memoryMb: 4096, storageMb: 16384 }, + env: { CMUX_VM_DISK_MB: "65536" }, + })).toEqual({ vcpus: 1, memoryMb: 4096, diskMb: 65536 }); + }); + + for (const [ + locale, + messages, + sharedPhrase, + stalePerVmPhrase, + capacityLabel, + faqSharedPhrase, + faqPerUserPhrase, + startingDiskPhrase, + ] of [ + [ + "en", + enMessages, + "all sharing a total pool of", + "each with", + "Resources shared across Cloud VMs", + "Pro includes up to 50 machines sharing a total pool of 5 vCPU, 20 GB of RAM, and 200 GB of disk", + "same shared pool for each user", + "New VM disks start at 32 GB", + ], + [ + "ja", + jaMessages, + "すべての VM で合計", + "各 5 vCPU", + "Cloud VM 間で共有するリソース", + "Pro では最大 50 台のマシンで合計 5 vCPU、20 GB の RAM、200 GB のディスクを共有", + "Team では、ユーザーごとに最大 50 台のマシンと同じ共有プール", + "新しい VM のディスクは 32 GB で開始", + ], ] as const) { - test(`${locale} pricing copy states the current prices and machine spec`, () => { + test(`${locale} pricing copy states the current prices and shared capacity`, () => { const pricing = messages.pricing; const proFeatures = pricing.pro.features.join("\n"); expect(proFeatures).toContain(`${PAID_MAX_ACTIVE_VMS_DEFAULT} `); + expect(proFeatures).toContain(`${sharedVcpus} vCPU`); + expect(proFeatures).toContain(`${sharedMemoryGb} GB`); + expect(proFeatures).toContain(`${sharedDiskGb} GB`); + expect(proFeatures).toContain(sharedPhrase); expect(proFeatures).toContain(`${memoryGb} GB`); - expect(proFeatures).toContain(`${diskGb} GB`); + expect(proFeatures).toContain(`${startingDiskGb} GB`); + expect(proFeatures).not.toContain(stalePerVmPhrase); const vmRow = pricing.compare.rows.find((row) => row.label.includes("Cloud VM") && row.pro === String(PAID_MAX_ACTIVE_VMS_DEFAULT), ); expect(vmRow).toBeDefined(); - const sizeRow = pricing.compare.rows.find((row) => - row.label.includes("Cloud VM") && typeof row.pro === "string" && row.pro.includes(`${memoryGb} GB`), + const capacityRow = pricing.compare.rows.find((row) => row.label === capacityLabel); + expect(capacityRow?.pro).toContain(`${sharedVcpus} vCPU`); + expect(capacityRow?.pro).toContain(`${sharedMemoryGb} GB`); + expect(capacityRow?.pro).toContain(`${sharedDiskGb} GB`); + expect(capacityRow?.pro).toContain(`${memoryGb} GB`); + expect(capacityRow?.pro).toContain(`${startingDiskGb} GB`); + expect(capacityRow?.team).toContain(`${sharedVcpus} vCPU`); + expect(capacityRow?.team).toContain(`${sharedMemoryGb} GB`); + expect(capacityRow?.team).toContain(`${sharedDiskGb} GB`); + expect(capacityRow?.team).toContain(`${memoryGb} GB`); + expect(capacityRow?.team).toContain(`${startingDiskGb} GB`); + const teamCapacity = capacityRow?.team ?? ""; + expect(locale === "en" ? teamCapacity.toLowerCase() : teamCapacity).toContain( + locale === "en" ? "per user" : "ユーザーごとに", ); - expect(sizeRow?.pro).toContain(`${memoryGb} GB`); - expect(sizeRow?.pro).toContain(`${diskGb} GB`); const faq = pricing.faq.items.map((item) => item.a).join("\n"); expect(faq).toContain(`$${PRO_PRICING_USD.month.billedAmount}/`); @@ -134,6 +289,31 @@ describe("pricing copy matches the plan policy", () => { } expect(faq.toLowerCase()).not.toContain("unlimited active"); expect(faq).not.toContain("無制限に利用"); + expect(faq).not.toContain(stalePerVmPhrase); + expect(faq).toContain(faqSharedPhrase); + expect(faq).toContain(faqPerUserPhrase); + expect(faq).toContain(startingDiskPhrase); + if (locale === "en") { + expect(faq).toContain("Team includes up to 50 machines per user"); + } else { + expect(faq).toContain("Team では、ユーザーごとに最大 50 台のマシン"); + } }); } + + test("fallback locales inherit the shared English VM policy", async () => { + for (const locale of locales) { + if (locale === "en" || locale === "ja") continue; + const messages = await loadMessages(locale) as unknown as typeof enMessages; + const pricing = messages.pricing; + const proFeatures = pricing.pro.features.join("\n"); + expect(proFeatures).toContain("all sharing a total pool of 5 vCPU, 20 GB RAM, and 200 GB disk"); + expect(proFeatures).not.toContain("each with 5 vCPU"); + const capacityRow = pricing.compare.rows.find((row) => row.label === "Resources shared across Cloud VMs"); + expect(capacityRow?.pro).toBe("5 vCPU, 20 GB RAM, 200 GB disk total; default VM size 8 GB RAM and 32 GB disk, with sizes from 4 to 64 GB RAM available as capacity allows"); + expect(capacityRow?.team).toBe("Per user: 5 vCPU, 20 GB RAM, and 200 GB disk total; default VM size 8 GB RAM and 32 GB disk, with sizes from 4 to 64 GB RAM available as capacity allows"); + const faq = pricing.faq.items.map((item) => item.a).join("\n"); + expect(faq).toContain("Pro includes up to 50 machines sharing a total pool of 5 vCPU, 20 GB of RAM, and 200 GB of disk"); + } + }); }); diff --git a/web/tests/vm-freestyle-provider.test.ts b/web/tests/vm-freestyle-provider.test.ts index 7ac201d7327e..8ad30f6b4261 100644 --- a/web/tests/vm-freestyle-provider.test.ts +++ b/web/tests/vm-freestyle-provider.test.ts @@ -40,6 +40,7 @@ const EDGE_RULE: VmEdgeRule = { // platform. `probeExit` is what the edge readiness probe returns. function fakeFreestyle(input: { readonly probeExit: number }) { const creates: unknown[] = []; + const resizes: unknown[] = []; const execs: string[] = []; const writes: Array<{ path: string; content: string }> = []; const deletes: string[] = []; @@ -60,7 +61,9 @@ function fakeFreestyle(input: { readonly probeExit: number }) { data: async () => ({ publicIpv6: "2602:f75c:0:1::2a" }), // Every VM boots at its snapshot's resources; create grows it to the plan // machine before bootstrap (see growToRequestedSize). - resize: async () => {}, + resize: async (options: unknown) => { + resizes.push(options); + }, }; const client = { vms: { @@ -72,7 +75,7 @@ function fakeFreestyle(input: { readonly probeExit: number }) { ref: () => vm, }, } as unknown as Freestyle; - return { client, creates, execs, writes, deletes }; + return { client, creates, resizes, execs, writes, deletes }; } function providerWith(fake: ReturnType): FreestyleProvider { @@ -327,6 +330,16 @@ describe("FreestyleProvider create with edge rules", () => { expect(fake.writes).toEqual([]); }); + test("grows a 4 GB image to the documented 32 GB starting disk", async () => { + const fake = fakeFreestyle({ probeExit: 0 }); + await providerWith(fake).create({ + image: "sh-devbox-4gb", + imageSize: { name: "sm", cpu: 1, memoryMb: 4096, storageMb: 16384 }, + }); + + expect(fake.resizes).toEqual([{ storage: 32768 }]); + }); + test("restore passes the rule inline and writes nothing into the guest", async () => { diff --git a/web/tests/vm-limit-refresh.test.ts b/web/tests/vm-limit-refresh.test.ts index ee30c8f9d2bf..2483c3ecf63c 100644 --- a/web/tests/vm-limit-refresh.test.ts +++ b/web/tests/vm-limit-refresh.test.ts @@ -3,10 +3,10 @@ import * as Effect from "effect/Effect"; import * as Layer from "effect/Layer"; import { VmBillingGateway, noOpVmBillingGateway } from "../services/vms/billingGateway"; -import { VmLimitExceededError } from "../services/vms/errors"; +import { VmLimitExceededError, VmSharedResourceLimitExceededError } from "../services/vms/errors"; import { VmProviderGateway, type VmProviderGatewayShape } from "../services/vms/providerGateway"; import { VmRepository, type CloudVmRow, type VmRepositoryShape } from "../services/vms/repository"; -import { createVm } from "../services/vms/workflows"; +import { createVm, reconcileVmProviderStatuses } from "../services/vms/workflows"; type ObservedStatusUpdate = Parameters[0]; const FIXTURE_NOW = new Date("2026-01-01T00:00:00.000Z"); @@ -41,6 +41,133 @@ function row(overrides: Partial): CloudVmRow { // status read. The lazy refresh on limit-exceeded must reconcile every // provider the gateway can report on, exactly like the cron path. describe("lazy active-limit provider refresh", () => { + test("does not fan out legacy provider reads on a successful create", async () => { + const requested = row({ status: "provisioning", providerVmId: null }); + const running = row({ status: "running", providerVmId: "provider-vm-new" }); + let legacyCandidateCalls = 0; + let statsCalls = 0; + let beginReservation: unknown; + const repo = { + beginCreate: (input: { resourceReservation?: unknown }) => { + beginReservation = input.resourceReservation; + return Effect.succeed({ inserted: true, vm: requested }); + }, + legacyResourceReservationCandidates: () => Effect.sync(() => { + legacyCandidateCalls += 1; + return []; + }), + claimBillingGrant: () => Effect.succeed({ kind: "already_claimed" as const }), + markBillingGrantApplied: () => Effect.void, + deleteBillingGrant: () => Effect.void, + markCreateRunning: () => Effect.succeed(running), + markCreateFailed: () => Effect.void, + recordUsageEvent: () => Effect.void, + recordUsageEvents: () => Effect.void, + } as unknown as VmRepositoryShape; + const providers = { + create: () => Effect.succeed({ + provider: "freestyle" as const, + providerVmId: "provider-vm-new", + status: "running" as const, + image: "snapshot-test", + createdAt: FIXTURE_NOW.getTime(), + }), + destroy: () => Effect.void, + getStats: () => Effect.sync(() => { + statsCalls += 1; + return { state: "awake" as const, sampledAt: FIXTURE_NOW.getTime(), diskTotalMb: 65536 }; + }), + exec: () => Effect.succeed({ exitCode: 0, stdout: "", stderr: "" }), + openAttach: () => Effect.fail(new Error("unused") as never), + openSSH: () => Effect.fail(new Error("unused") as never), + } as unknown as VmProviderGatewayShape; + const layer = Layer.mergeAll( + Layer.succeed(VmRepository, repo), + Layer.succeed(VmProviderGateway, providers), + Layer.succeed(VmBillingGateway, noOpVmBillingGateway()), + ); + + await Effect.runPromise( + createVm({ + userId: requested.userId, + billingCustomerType: "team", + billingTeamId: requested.billingTeamId!, + billingPlanId: "pro", + maxActiveVms: 50, + provider: "freestyle", + image: "snapshot-test", + imageSize: { name: "xl", cpu: 16, memoryMb: 32768, storageMb: 131072 }, + }).pipe(Effect.provide(layer)), + ); + expect(legacyCandidateCalls).toBe(0); + expect(statsCalls).toBe(0); + expect(beginReservation).toEqual({ vcpus: 16, memoryMb: 32768, diskMb: 131072 }); + }); + + test("keeps the baked image disk in a memory-sized paid reservation", async () => { + const requested = row({ + id: "00000000-0000-4000-8000-000000000107", + status: "provisioning", + providerVmId: null, + }); + const running = row({ + id: "00000000-0000-4000-8000-000000000108", + status: "running", + providerVmId: "provider-vm-memory-image", + }); + let beginReservation: unknown; + const repo = { + beginCreate: (input: { resourceReservation?: unknown }) => { + beginReservation = input.resourceReservation; + return Effect.succeed({ inserted: true, vm: requested }); + }, + claimBillingGrant: () => Effect.succeed({ kind: "already_claimed" as const }), + markBillingGrantApplied: () => Effect.void, + deleteBillingGrant: () => Effect.void, + markCreateRunning: () => Effect.succeed(running), + markCreateFailed: () => Effect.void, + recordUsageEvent: () => Effect.void, + recordUsageEvents: () => Effect.void, + } as unknown as VmRepositoryShape; + const provider = { + create: () => Effect.succeed({ + provider: "freestyle" as const, + providerVmId: running.providerVmId!, + status: "running" as const, + image: "snapshot-test", + createdAt: FIXTURE_NOW.getTime(), + }), + destroy: () => Effect.void, + exec: () => Effect.succeed({ exitCode: 0, stdout: "", stderr: "" }), + openAttach: () => Effect.fail(new Error("unused") as never), + openSSH: () => Effect.fail(new Error("unused") as never), + } as unknown as VmProviderGatewayShape; + + await Effect.runPromise( + createVm({ + userId: requested.userId, + billingCustomerType: "team", + billingTeamId: requested.billingTeamId!, + billingPlanId: "pro", + maxActiveVms: 50, + provider: "freestyle", + image: "snapshot-test", + memoryMb: 16 * 1024, + imageSize: { name: "xl", cpu: 16, memoryMb: 32 * 1024, storageMb: 128 * 1024 }, + }).pipe(Effect.provide(Layer.mergeAll( + Layer.succeed(VmRepository, repo), + Layer.succeed(VmProviderGateway, provider), + Layer.succeed(VmBillingGateway, noOpVmBillingGateway()), + ))), + ); + + expect(beginReservation).toEqual({ + vcpus: 4, + memoryMb: 16 * 1024, + diskMb: 128 * 1024, + }); + }); + test("refreshes stale rows for every provider with a status read, not just freestyle", async () => { const requested = row({ status: "provisioning", providerVmId: null }); const running = row({ @@ -153,4 +280,336 @@ describe("lazy active-limit provider refresh", () => { expect(observedIds).toHaveLength(200); expect(new Set(observed.map((u) => u.status))).toEqual(new Set(["destroyed"])); }); + + test("repairs scoped legacy resource claims before retrying a shared-pool create", async () => { + const requested = row({ + id: "00000000-0000-4000-8000-000000000109", + status: "provisioning", + providerVmId: null, + }); + const legacy = row({ + id: "00000000-0000-4000-8000-000000000110", + status: "running", + providerVmId: "provider-vm-legacy-resource-repair", + providerMetadata: {}, + }); + const staleValid = row({ + id: "00000000-0000-4000-8000-000000000111", + status: "running", + providerVmId: "provider-vm-stale-valid-resource", + providerMetadata: { + cmuxResourceReservation: { vcpus: 1, memoryMb: 4096, diskMb: 32768 }, + }, + }); + let beginCalls = 0; + let candidateInput: { userId?: string; billingTeamId?: string | null; limit: number } | undefined; + let statsCalls = 0; + let statusCalls = 0; + const reservations: unknown[] = []; + const repo = { + beginCreate: () => { + beginCalls += 1; + return beginCalls === 1 + ? Effect.fail(new VmSharedResourceLimitExceededError({ + kind: "shared_resources", + billingTeamId: requested.billingTeamId!, + phase: "create", + resource: "diskMb", + used: 200 * 1024, + requested: 32 * 1024, + limit: 200 * 1024, + })) + : Effect.succeed({ inserted: true, vm: requested }); + }, + legacyResourceReservationCandidates: (input: typeof candidateInput) => Effect.sync(() => { + candidateInput = input; + return [legacy]; + }), + activeLimitCandidates: () => Effect.succeed([staleValid]), + markProviderObservedStatus: () => Effect.sync(() => { + statusCalls += 1; + return true; + }), + setResourceReservation: (input: unknown) => Effect.sync(() => { + reservations.push(input); + return true; + }), + claimBillingGrant: () => Effect.succeed({ kind: "already_claimed" as const }), + markBillingGrantApplied: () => Effect.void, + deleteBillingGrant: () => Effect.void, + markCreateRunning: () => Effect.succeed({ + ...requested, + status: "running" as const, + providerVmId: "provider-vm-new-after-repair", + }), + markCreateFailed: () => Effect.void, + recordUsageEvent: () => Effect.void, + recordUsageEvents: () => Effect.void, + } as unknown as VmRepositoryShape; + const providers = { + create: () => Effect.succeed({ + provider: "freestyle" as const, + providerVmId: "provider-vm-new-after-repair", + status: "running" as const, + image: "snapshot-test", + createdAt: FIXTURE_NOW.getTime(), + }), + destroy: () => Effect.void, + getStats: (_provider: string, providerVmId: string) => { + expect(providerVmId).toBe(legacy.providerVmId); + statsCalls += 1; + return Effect.succeed({ + state: "awake" as const, + sampledAt: FIXTURE_NOW.getTime(), + cpus: 2, + memoryTotalMb: 8192, + diskTotalMb: 65536, + }); + }, + getStatus: (_provider: string, providerVmId: string) => { + expect(providerVmId).toBe(staleValid.providerVmId); + return Effect.succeed("destroyed" as const); + }, + exec: () => Effect.succeed({ exitCode: 0, stdout: "", stderr: "" }), + openAttach: () => Effect.fail(new Error("unused") as never), + openSSH: () => Effect.fail(new Error("unused") as never), + } as unknown as VmProviderGatewayShape; + + await Effect.runPromise( + createVm({ + userId: requested.userId, + billingCustomerType: "team", + billingTeamId: requested.billingTeamId!, + billingPlanId: "pro", + // A large team allowance must not turn the request-path repair into + // one provider call per VM. The remaining rows stay for the cron pass. + maxActiveVms: 5000, + provider: "freestyle", + image: "snapshot-test", + }).pipe(Effect.provide(Layer.mergeAll( + Layer.succeed(VmRepository, repo), + Layer.succeed(VmProviderGateway, providers), + Layer.succeed(VmBillingGateway, noOpVmBillingGateway()), + ))), + ); + + expect(beginCalls).toBe(2); + expect(candidateInput).toEqual({ + userId: requested.userId, + billingTeamId: requested.billingTeamId, + limit: 5, + }); + expect(statsCalls).toBe(1); + expect(statusCalls).toBe(1); + expect(reservations).toEqual([{ + id: legacy.id, + reservation: { vcpus: 2, memoryMb: 8192, diskMb: 65536 }, + }]); + }); + + test("returns an impossible shared-pool request without a provider refresh", async () => { + let beginCalls = 0; + let statusCalls = 0; + let candidateCalls = 0; + const requested = row({ status: "provisioning", providerVmId: null }); + const repo = { + beginCreate: () => { + beginCalls += 1; + return Effect.fail(new VmSharedResourceLimitExceededError({ + kind: "shared_resources", + billingTeamId: requested.billingTeamId!, + phase: "create", + resource: "memoryMb", + used: 0, + requested: 24 * 1024, + limit: 20 * 1024, + })); + }, + activeLimitCandidates: () => Effect.sync(() => { + statusCalls += 1; + return []; + }), + legacyResourceReservationCandidates: () => Effect.sync(() => { + candidateCalls += 1; + return []; + }), + } as unknown as VmRepositoryShape; + const providers = { + getStatus: () => Effect.sync(() => { + statusCalls += 1; + return "running" as const; + }), + getStats: () => Effect.sync(() => { + statusCalls += 1; + return { state: "awake" as const, sampledAt: FIXTURE_NOW.getTime(), diskTotalMb: 65536 }; + }), + } as unknown as VmProviderGatewayShape; + + const result = await Effect.runPromise( + createVm({ + userId: requested.userId, + billingCustomerType: "team", + billingTeamId: requested.billingTeamId!, + billingPlanId: "pro", + maxActiveVms: 50, + provider: "freestyle", + image: "snapshot-test", + }).pipe( + Effect.provide(Layer.mergeAll( + Layer.succeed(VmRepository, repo), + Layer.succeed(VmProviderGateway, providers), + Layer.succeed(VmBillingGateway, noOpVmBillingGateway()), + )), + Effect.flip, + ), + ); + + expect(result).toMatchObject({ + _tag: "VmSharedResourceLimitExceededError", + requested: 24 * 1024, + limit: 20 * 1024, + }); + expect(beginCalls).toBe(1); + expect(statusCalls).toBe(0); + expect(candidateCalls).toBe(0); + }); +}); + +describe("background resource reconciliation", () => { + test("uses a global bounded batch instead of a request owner scope", async () => { + const legacy = row({ + id: "00000000-0000-4000-8000-000000000106", + status: "running", + providerVmId: "provider-vm-background-legacy", + providerMetadata: {}, + }); + let candidateInput: { userId?: string; billingTeamId?: string | null; limit: number } | undefined; + const reservations: Array<{ id: string; reservation: { vcpus: number; memoryMb: number; diskMb: number } }> = []; + const repo = { + legacyResourceReservationCandidates: (input: typeof candidateInput) => + Effect.sync(() => { + candidateInput = input; + return [legacy]; + }), + setResourceReservation: (input: typeof reservations[number]) => + Effect.sync(() => { + reservations.push(input); + return true; + }), + reconciliationCandidates: () => Effect.succeed([]), + } as unknown as VmRepositoryShape; + const provider = { + getStatus: () => Effect.succeed("running" as const), + getStats: () => Effect.succeed({ + state: "awake" as const, + sampledAt: FIXTURE_NOW.getTime(), + diskTotalMb: 65536, + }), + } as unknown as VmProviderGatewayShape; + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe( + Effect.provide(Layer.mergeAll( + Layer.succeed(VmRepository, repo), + Layer.succeed(VmProviderGateway, provider), + )), + ), + ); + + expect(candidateInput).toEqual({ limit: 50 }); + expect(reservations).toEqual([{ + id: legacy.id, + reservation: { vcpus: 5, memoryMb: 20 * 1024, diskMb: 65536 }, + }]); + }); + + test("rotates past legacy providers that fail instead of starving newer rows", async () => { + const failed = Array.from({ length: 50 }, (_, index) => row({ + id: `legacy-failed-${index}`, + providerVmId: `provider-vm-failed-${index}`, + status: "running", + providerMetadata: {}, + })); + const newer = row({ + id: "legacy-newer", + providerVmId: "provider-vm-newer", + status: "running", + providerMetadata: {}, + }); + const deferred: string[] = []; + const reservations: string[] = []; + let candidateCalls = 0; + const repo = { + legacyResourceReservationCandidates: () => Effect.sync(() => { + candidateCalls += 1; + return [...failed, newer].filter((candidate) => !deferred.includes(candidate.id)).slice(0, 50); + }), + deferResourceReservation: (input: { id: string }) => Effect.sync(() => { + deferred.push(input.id); + }), + setResourceReservation: (input: { id: string }) => Effect.sync(() => { + reservations.push(input.id); + return true; + }), + reconciliationCandidates: () => Effect.succeed([]), + } as unknown as VmRepositoryShape; + const provider = { + getStats: (_provider: string, providerVmId: string) => failed.some( + (candidate) => candidate.providerVmId === providerVmId, + ) + ? Effect.fail(new Error("provider unavailable")) + : Effect.succeed({ + state: "awake" as const, + sampledAt: FIXTURE_NOW.getTime(), + cpus: 5, + memoryTotalMb: 20 * 1024, + diskTotalMb: 32768, + }), + getStatus: () => Effect.succeed("running" as const), + } as unknown as VmProviderGatewayShape; + const layer = Layer.mergeAll( + Layer.succeed(VmRepository, repo), + Layer.succeed(VmProviderGateway, provider), + ); + + await Effect.runPromise(reconcileVmProviderStatuses().pipe(Effect.provide(layer))); + await Effect.runPromise(reconcileVmProviderStatuses().pipe(Effect.provide(layer))); + + expect(candidateCalls).toBe(2); + expect(deferred).toHaveLength(50); + expect(reservations).toEqual([newer.id]); + }); + + test("times out a hung legacy provider stats read and defers the row", async () => { + const legacy = row({ + id: "legacy-hung", + providerVmId: "provider-vm-hung", + status: "running", + providerMetadata: {}, + }); + const deferred: string[] = []; + const repo = { + legacyResourceReservationCandidates: () => Effect.succeed([legacy]), + deferResourceReservation: (input: { id: string }) => Effect.sync(() => { + deferred.push(input.id); + }), + setResourceReservation: () => Effect.succeed(true), + reconciliationCandidates: () => Effect.succeed([]), + } as unknown as VmRepositoryShape; + const provider = { + getStats: () => Effect.never, + getStatus: () => Effect.succeed("running" as const), + } as unknown as VmProviderGatewayShape; + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe( + Effect.provide(Layer.mergeAll( + Layer.succeed(VmRepository, repo), + Layer.succeed(VmProviderGateway, provider), + )), + ), + ); + + expect(deferred).toEqual([legacy.id]); + }); }); diff --git a/web/tests/vm-model-plane-workflow.test.ts b/web/tests/vm-model-plane-workflow.test.ts index 5f30abb7abf3..bc72676647dc 100644 --- a/web/tests/vm-model-plane-workflow.test.ts +++ b/web/tests/vm-model-plane-workflow.test.ts @@ -443,7 +443,7 @@ describe("model-plane error responses", () => { }); expect(JSON.stringify(createPayload)).not.toContain("db down"); - const restore = vmCreateLikeErrorResponse(err, { operation: "restore", planId: "pro", retryAction: "unused" }); + const restore = await vmCreateLikeErrorResponse(err, { operation: "restore", planId: "pro", retryAction: "unused" }); expect(restore?.status).toBe(503); expect(await restore!.json()).toMatchObject({ error: "vm_model_plane_unavailable", phase: "restore" }); }); diff --git a/web/tests/vm-route-input.test.ts b/web/tests/vm-route-input.test.ts index 2e3e8ef52dbb..18406df108d2 100644 --- a/web/tests/vm-route-input.test.ts +++ b/web/tests/vm-route-input.test.ts @@ -10,10 +10,17 @@ import { providerField, stringField, } from "../services/vms/routeInput"; -import { VmFreeAccessExpiredError, VmNotFoundError } from "../services/vms/errors"; +import { + VmFreeAccessExpiredError, + VmNotFoundError, + VmResizeInProgressError, + VmSharedResourceLimitExceededError, +} from "../services/vms/errors"; import { vmCreateLikeErrorResponse, vmResourceErrorResponse, + vmSharedResourceLimitExceededResponse, + vmWorkflowErrorResponse, } from "../services/vms/routeHelpers"; async function responseBody(response: Response): Promise> { @@ -123,14 +130,79 @@ describe("Cloud VM route error adapters", () => { expect((await responseBody(expired!)).error).toBe("vm_access_requires_pro"); }); + test("maps shared pool exhaustion to a non-retryable conflict", async () => { + const error = new VmSharedResourceLimitExceededError({ + kind: "shared_resources", + billingTeamId: "team-1", + resource: "diskMb", + used: 196608, + requested: 32768, + limit: 204800, + }); + const direct = await vmSharedResourceLimitExceededResponse(error, "resize"); + expect(direct.status).toBe(409); + const directPayload = await responseBody(direct); + expect(directPayload.error).toBe("vm_shared_resource_limit_exceeded"); + expect(directPayload.phase).toBe("resize"); + expect(directPayload.retryable).toBe(false); + expect((directPayload.details as Record).shared).toBe(true); + + const oversized = await vmSharedResourceLimitExceededResponse({ + ...error, + used: 0, + requested: 262144, + limit: 204800, + }, "resize", "en"); + const oversizedPayload = await responseBody(oversized); + expect(oversizedPayload.action).toContain("smaller"); + expect(oversizedPayload.action).not.toContain("Delete"); + + const japanese = await vmSharedResourceLimitExceededResponse(error, "resize", "ja"); + const japanesePayload = await responseBody(japanese); + expect(japanesePayload.message).toContain("共有 Cloud VM"); + + const workflow = await vmWorkflowErrorResponse(error); + expect(workflow?.status).toBe(409); + expect((await responseBody(workflow!)).error).toBe("vm_shared_resource_limit_exceeded"); + + const fork = await vmCreateLikeErrorResponse(error, { + operation: "fork", + planId: "pro", + retryAction: "fork retry", + }); + expect(fork?.status).toBe(409); + const forkPayload = await responseBody(fork!); + expect(forkPayload.error).toBe("vm_shared_resource_limit_exceeded"); + expect(forkPayload.phase).toBe("fork"); + const restore = await vmCreateLikeErrorResponse(error, { + operation: "restore", + planId: "pro", + retryAction: "restore retry", + }); + expect((await responseBody(restore!)).phase).toBe("restore"); + }); + + test("maps a concurrent disk resize to a retryable conflict", async () => { + const response = await vmWorkflowErrorResponse(new VmResizeInProgressError({ + vmId: "vm-resize-in-progress", + })); + expect(response?.status).toBe(409); + expect(response?.headers.get("retry-after")).toBe("5"); + expect(await responseBody(response!)).toMatchObject({ + error: "vm_resize_in_progress", + retryable: true, + retryAfterSeconds: 5, + }); + }); + test("keeps fork and restore provisioning guidance distinct", async () => { const error = { _tag: "VmCreateFailedError", idempotencyKey: "key" }; - const fork = vmCreateLikeErrorResponse(error, { + const fork = await vmCreateLikeErrorResponse(error, { operation: "fork", planId: "free", retryAction: "fork retry", }); - const restore = vmCreateLikeErrorResponse(error, { + const restore = await vmCreateLikeErrorResponse(error, { operation: "restore", planId: "free", retryAction: "restore retry", diff --git a/web/tests/vm-snapshot-not-found-dispatch.test.ts b/web/tests/vm-snapshot-not-found-dispatch.test.ts index 2134cbbf0bdc..2d705ea07c7a 100644 --- a/web/tests/vm-snapshot-not-found-dispatch.test.ts +++ b/web/tests/vm-snapshot-not-found-dispatch.test.ts @@ -62,7 +62,7 @@ describe("snapshot-not-found error dispatch", () => { const thrown = await restoreUnknownSnapshot(); // Exactly what runVmWorkflow rethrows to the restore route's catch block. const routeError = vmWorkflowErrorCause(thrown) ?? thrown; - const response = vmCreateLikeErrorResponse(routeError, { + const response = await vmCreateLikeErrorResponse(routeError, { operation: "restore", planId: "pro", retryAction: "unused in this test", diff --git a/web/tests/vm-workflows.test.ts b/web/tests/vm-workflows.test.ts index c1074a019a03..89318ed46917 100644 --- a/web/tests/vm-workflows.test.ts +++ b/web/tests/vm-workflows.test.ts @@ -17,6 +17,8 @@ import { VmRepositoryLive, type CloudVmIdentityLeaseRow, type CloudVmLeaseRow, + type CloudVmBaseGenerationRow, + type CloudVmBaseRow, type CloudVmSessionRow, type CloudVmRow, type VmRepositoryShape, @@ -29,6 +31,7 @@ import { VmCreateInProgressError, VmDatabaseError, VmLimitExceededError, + VmSharedResourceLimitExceededError, VmNotFoundError, VmProviderOperationError, VmSnapshotNotFoundError, @@ -37,10 +40,17 @@ import { } from "../services/vms/errors"; import { accountDeletionUserHash } from "../services/account/deletionLock"; import { isVmAttachTransportUnsupportedError } from "../services/vms/errors"; +import { + PLAN_SHARED_DISK_MB, + VM_DISK_MB_MAX, + VM_RESOURCE_RESIZE_PENDING_METADATA_KEY, + VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY, +} from "../services/vms/machineSpec"; import { createVm, destroyVm, execVm, + forkVm, homeVolumeNameForUser, listUserVms, approveVmCmuxRemoteEnrollment, @@ -55,6 +65,7 @@ import { restoreVm, reconcileVmProviderStatuses, resizeVm, + snapshotVm, } from "../services/vms/workflows"; const runDbTests = process.env.CMUX_DB_TEST === "1"; @@ -134,6 +145,508 @@ afterAll(async () => { }); describe("VM Effect workflows", () => { + test("repairs a legacy fork claim from provider CPU and memory stats", async () => { + const source = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000151", + userId: "user-workflow-legacy-fork-shape", + billingTeamId: "team-workflow-legacy-fork-shape", + billingPlanId: "pro", + providerVmId: "provider-vm-legacy-fork-source", + status: "running", + providerMetadata: {}, + }); + const pendingFork = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000152", + userId: source.userId, + billingTeamId: source.billingTeamId, + billingPlanId: "pro", + providerVmId: null, + status: "provisioning", + providerMetadata: {}, + }); + let reservation: unknown; + let beginInput: { resourceReservation?: unknown; reserveSharedResourceHeadroom?: boolean; forkMinimumResourceReservation?: unknown } | undefined; + let finalizedReservation: unknown; + const repo = { + ...testWorkflowRepo({ vm: source }), + beginCreate: (input: { resourceReservation?: unknown; reserveSharedResourceHeadroom?: boolean; forkMinimumResourceReservation?: unknown }) => { + beginInput = input; + reservation = input.resourceReservation; + return Effect.succeed({ + inserted: true, + vm: { + ...pendingFork, + providerMetadata: { + cmuxResourceReservation: input.resourceReservation, + ...(input.forkMinimumResourceReservation === undefined + ? {} + : { cmuxResourceForkPending: input.forkMinimumResourceReservation }), + }, + }, + }); + }, + setResourceReservation: (input: { reservation: unknown }) => { + finalizedReservation = input.reservation; + return Effect.succeed(true); + }, + markCreateRunning: () => Effect.succeed({ + ...pendingFork, + providerVmId: "provider-vm-legacy-fork-copy", + status: "running" as const, + }), + } as unknown as VmRepositoryShape; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + resume: () => Effect.succeed(testVmHandle({ providerVmId: source.providerVmId! })), + getStats: (_provider: string, providerVmId: string) => { + expect(providerVmId).toBe("provider-vm-legacy-fork-copy"); + return Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 16, + memoryTotalMb: 32768, + diskTotalMb: 65536, + }); + }, + fork: () => Effect.succeed(testVmHandle({ providerVmId: "provider-vm-legacy-fork-copy" })), + }; + + await Effect.runPromise( + forkVm({ + userId: source.userId, + billingCustomerType: "team", + billingTeamId: source.billingTeamId!, + teamIds: [source.billingTeamId!], + billingPlanId: "pro", + maxActiveVms: 50, + providerVmId: source.providerVmId!, + }).pipe(Effect.provide(workflowLayer(repo, provider))), + ); + + expect(reservation).toEqual({ vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }); + expect(beginInput?.reserveSharedResourceHeadroom).toBe(true); + expect(beginInput?.forkMinimumResourceReservation).toEqual({ vcpus: 1, memoryMb: 4 * 1024, diskMb: 16 * 1024 }); + expect(finalizedReservation).toEqual({ vcpus: 16, memoryMb: 32768, diskMb: 65536 }); + }); + + test("uses the shared-pool fallback for implausible legacy fork stats", async () => { + const source = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000155", + userId: "user-workflow-legacy-fork-invalid-shape", + billingTeamId: "team-workflow-legacy-fork-invalid-shape", + billingPlanId: "pro", + providerVmId: "provider-vm-legacy-fork-invalid-source", + status: "running", + providerMetadata: {}, + }); + const pendingFork = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000156", + userId: source.userId, + billingTeamId: source.billingTeamId, + billingPlanId: "pro", + providerVmId: null, + status: "provisioning", + providerMetadata: {}, + }); + let reservation: unknown; + let beginInput: { resourceReservation?: unknown; reserveSharedResourceHeadroom?: boolean; forkMinimumResourceReservation?: unknown } | undefined; + let finalizedReservation: unknown; + const repo = { + ...testWorkflowRepo({ vm: source }), + beginCreate: (input: { resourceReservation?: unknown; reserveSharedResourceHeadroom?: boolean; forkMinimumResourceReservation?: unknown }) => { + beginInput = input; + reservation = input.resourceReservation; + return Effect.succeed({ + inserted: true, + vm: { + ...pendingFork, + providerMetadata: { + cmuxResourceReservation: input.resourceReservation, + ...(input.forkMinimumResourceReservation === undefined + ? {} + : { cmuxResourceForkPending: input.forkMinimumResourceReservation }), + }, + }, + }); + }, + setResourceReservation: (input: { reservation: unknown }) => { + finalizedReservation = input.reservation; + return Effect.succeed(true); + }, + markCreateRunning: () => Effect.succeed({ + ...pendingFork, + providerVmId: "provider-vm-legacy-fork-invalid-copy", + status: "running" as const, + }), + } as unknown as VmRepositoryShape; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + resume: () => Effect.succeed(testVmHandle({ providerVmId: source.providerVmId! })), + getStats: (_provider: string, providerVmId: string) => { + expect(providerVmId).toBe("provider-vm-legacy-fork-invalid-copy"); + return Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 0, + memoryTotalMb: 512, + diskTotalMb: 1, + }); + }, + fork: () => Effect.succeed(testVmHandle({ providerVmId: "provider-vm-legacy-fork-invalid-copy" })), + }; + + await Effect.runPromise( + forkVm({ + userId: source.userId, + billingCustomerType: "team", + billingTeamId: source.billingTeamId!, + teamIds: [source.billingTeamId!], + billingPlanId: "pro", + maxActiveVms: 50, + providerVmId: source.providerVmId!, + }).pipe(Effect.provide(workflowLayer(repo, provider))), + ); + + expect(reservation).toEqual({ vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }); + expect(beginInput?.reserveSharedResourceHeadroom).toBe(true); + expect(beginInput?.forkMinimumResourceReservation).toEqual({ vcpus: 1, memoryMb: 4 * 1024, diskMb: 16 * 1024 }); + expect(finalizedReservation).toEqual({ vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }); + }); + + test("keeps the supported 1-vCPU legacy fork shape", async () => { + const source = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000157", + userId: "user-workflow-legacy-fork-one-vcpu", + billingTeamId: "team-workflow-legacy-fork-one-vcpu", + billingPlanId: "pro", + providerVmId: "provider-vm-legacy-fork-one-vcpu-source", + status: "running", + providerMetadata: {}, + }); + const pendingFork = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000158", + userId: source.userId, + billingTeamId: source.billingTeamId, + billingPlanId: "pro", + providerVmId: null, + status: "provisioning", + providerMetadata: {}, + }); + let reservation: unknown; + let beginInput: { resourceReservation?: unknown; reserveSharedResourceHeadroom?: boolean; forkMinimumResourceReservation?: unknown } | undefined; + let finalizedReservation: unknown; + const repo = { + ...testWorkflowRepo({ vm: source }), + beginCreate: (input: { resourceReservation?: unknown; reserveSharedResourceHeadroom?: boolean; forkMinimumResourceReservation?: unknown }) => { + beginInput = input; + reservation = input.resourceReservation; + return Effect.succeed({ + inserted: true, + vm: { + ...pendingFork, + providerMetadata: { + cmuxResourceReservation: input.resourceReservation, + ...(input.forkMinimumResourceReservation === undefined + ? {} + : { cmuxResourceForkPending: input.forkMinimumResourceReservation }), + }, + }, + }); + }, + setResourceReservation: (input: { reservation: unknown }) => { + finalizedReservation = input.reservation; + return Effect.succeed(true); + }, + markCreateRunning: () => Effect.succeed({ + ...pendingFork, + providerVmId: "provider-vm-legacy-fork-one-vcpu-copy", + status: "running" as const, + }), + } as unknown as VmRepositoryShape; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + resume: () => Effect.succeed(testVmHandle({ providerVmId: source.providerVmId! })), + getStats: (_provider: string, providerVmId: string) => { + expect(providerVmId).toBe("provider-vm-legacy-fork-one-vcpu-copy"); + return Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 1, + memoryTotalMb: 4096, + diskTotalMb: 16384, + }); + }, + fork: () => Effect.succeed(testVmHandle({ providerVmId: "provider-vm-legacy-fork-one-vcpu-copy" })), + }; + + await Effect.runPromise( + forkVm({ + userId: source.userId, + billingCustomerType: "team", + billingTeamId: source.billingTeamId!, + teamIds: [source.billingTeamId!], + billingPlanId: "pro", + maxActiveVms: 50, + providerVmId: source.providerVmId!, + }).pipe(Effect.provide(workflowLayer(repo, provider))), + ); + + expect(reservation).toEqual({ vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }); + expect(beginInput?.reserveSharedResourceHeadroom).toBe(true); + expect(beginInput?.forkMinimumResourceReservation).toEqual({ vcpus: 1, memoryMb: 4 * 1024, diskMb: 16 * 1024 }); + expect(finalizedReservation).toEqual({ vcpus: 1, memoryMb: 4096, diskMb: 16384 }); + }); + + test("recomputes a legacy native fork claim after scoped repair", async () => { + const source = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000162", + userId: "user-workflow-fork-retry", + billingTeamId: "team-workflow-fork-retry", + billingPlanId: "pro", + providerVmId: "provider-vm-fork-retry-source", + status: "running", + providerMetadata: {}, + }); + const pendingFork = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000163", + userId: source.userId, + billingTeamId: source.billingTeamId, + billingPlanId: "pro", + providerVmId: null, + status: "provisioning", + }); + let currentSource = source; + const beginInputs: Array<{ resourceReservation?: unknown }> = []; + const reservations: unknown[] = []; + const repo = { + ...testWorkflowRepo({ vm: source }), + findUserVm: () => Effect.succeed(currentSource), + beginCreate: (input: { resourceReservation?: unknown; forkMinimumResourceReservation?: unknown }) => { + beginInputs.push(input); + if (beginInputs.length === 1) { + return Effect.fail(new VmSharedResourceLimitExceededError({ + kind: "shared_resources", + billingTeamId: source.billingTeamId!, + phase: "create", + resource: "diskMb", + used: 200 * 1024, + requested: 200 * 1024, + limit: 200 * 1024, + })); + } + return Effect.succeed({ + inserted: true, + vm: { + ...pendingFork, + providerMetadata: { + cmuxResourceReservation: input.resourceReservation, + cmuxResourceForkPending: input.forkMinimumResourceReservation, + }, + }, + }); + }, + legacyResourceReservationCandidates: () => Effect.succeed([currentSource]), + setResourceReservation: (input: { id: string; reservation: unknown }) => Effect.sync(() => { + reservations.push(input.reservation); + if (input.id === source.id) { + currentSource = { + ...currentSource, + providerMetadata: { cmuxResourceReservation: input.reservation }, + }; + } + return true; + }), + markCreateRunning: ({ providerVmId }: { providerVmId: string }) => Effect.succeed({ + ...pendingFork, + providerVmId, + status: "running" as const, + }), + } as unknown as VmRepositoryShape; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: (_provider, providerVmId) => Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 2, + memoryTotalMb: 8192, + diskTotalMb: providerVmId === source.providerVmId ? 65536 : 65536, + }), + fork: () => Effect.succeed(testVmHandle({ providerVmId: "provider-vm-fork-retry-copy" })), + }; + + await Effect.runPromise( + forkVm({ + userId: source.userId, + billingCustomerType: "team", + billingTeamId: source.billingTeamId!, + teamIds: [source.billingTeamId!], + billingPlanId: "pro", + maxActiveVms: 50, + providerVmId: source.providerVmId!, + }).pipe(Effect.provide(workflowLayer(repo, provider))), + ); + + expect(beginInputs).toHaveLength(2); + expect(beginInputs[0]?.resourceReservation).toEqual({ + vcpus: 5, + memoryMb: 20 * 1024, + diskMb: 200 * 1024, + }); + expect(beginInputs[1]?.resourceReservation).toEqual({ + vcpus: 2, + memoryMb: 8192, + diskMb: 65536, + }); + expect(reservations[0]).toEqual({ vcpus: 2, memoryMb: 8192, diskMb: 65536 }); + }); + + test("records CPU and memory in new snapshot claims", async () => { + const source = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000153", + userId: "user-workflow-snapshot-shape", + billingTeamId: "team-workflow-snapshot-shape", + billingPlanId: "pro", + providerVmId: "provider-vm-snapshot-shape", + status: "running", + providerMetadata: {}, + }); + const usageEvents: RecordedUsageEvent[] = []; + const repo = testWorkflowRepo({ vm: source, usageEvents }); + let snapshotFinished = false; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStats: () => Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 16, + memoryTotalMb: 32768, + // A grow-only resize can finish while the provider snapshot runs. The + // event must capture the copy point, not the earlier read. + diskTotalMb: snapshotFinished ? 65536 : 32768, + }), + snapshot: () => Effect.sync(() => { + snapshotFinished = true; + return { + id: "snapshot-with-resource-claim", + createdAt: Date.now(), + }; + }), + }; + + await Effect.runPromise( + snapshotVm({ + userId: source.userId, + teamIds: [source.billingTeamId!], + providerVmId: source.providerVmId!, + }).pipe(Effect.provide(workflowLayer(repo, provider))), + ); + + const event = usageEvents.find((candidate) => candidate.eventType === "vm.snapshot.created"); + expect(event?.metadata).toMatchObject({ + vcpus: 16, + memoryMb: 32768, + diskMb: 65536, + }); + }); + + test("keeps snapshot creation bounded when provider stats hang", async () => { + const source = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000159", + userId: "user-workflow-snapshot-timeout", + billingTeamId: "team-workflow-snapshot-timeout", + billingPlanId: "pro", + providerVmId: "provider-vm-snapshot-timeout", + status: "running", + providerMetadata: {}, + }); + const usageEvents: RecordedUsageEvent[] = []; + const repo = testWorkflowRepo({ vm: source, usageEvents }); + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStats: () => Effect.never, + snapshot: () => Effect.succeed({ + id: "snapshot-after-stats-timeout", + createdAt: Date.now(), + }), + }; + + await Effect.runPromise( + snapshotVm({ + userId: source.userId, + teamIds: [source.billingTeamId!], + providerVmId: source.providerVmId!, + }).pipe(Effect.provide(workflowLayer(repo, provider))), + ); + + const event = usageEvents.find((candidate) => candidate.eventType === "vm.snapshot.created"); + expect(event?.metadata).toMatchObject({ + vcpus: 5, + memoryMb: 20 * 1024, + diskMb: 200 * 1024, + }); + }); + + test("restores a captured small snapshot at the provider's effective target", async () => { + const provisioning = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000161", + userId: "user-workflow-restore-shape", + billingTeamId: "team-workflow-restore-shape", + billingPlanId: "pro", + providerVmId: null, + status: "provisioning", + }); + let beginInput: { resourceReservation?: unknown } | undefined; + let createOptions: { memoryMb?: number } | undefined; + const repo = { + ...testWorkflowRepo({ vm: provisioning }), + hasOwnedSnapshot: () => Effect.succeed(true), + ownedSnapshotResourceReservation: () => Effect.succeed({ + vcpus: 1, + memoryMb: 4096, + diskMb: 16384, + }), + beginCreate: (input: { resourceReservation?: unknown }) => { + beginInput = input; + return Effect.succeed({ inserted: true, vm: provisioning }); + }, + markCreateRunning: ({ providerVmId }: { providerVmId: string }) => Effect.succeed({ + ...provisioning, + providerVmId, + status: "running" as const, + }), + } as unknown as VmRepositoryShape; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + create: (_provider, options) => { + createOptions = options; + return Effect.succeed(testVmHandle({ providerVmId: "provider-vm-restore-shape" })); + }, + }; + + await Effect.runPromise( + restoreVm({ + userId: provisioning.userId, + billingCustomerType: "team", + billingTeamId: provisioning.billingTeamId!, + billingPlanId: "pro", + maxActiveVms: 50, + provider: "freestyle", + snapshotId: "snapshot-small-shape", + }).pipe(Effect.provide(workflowLayer(repo, provider))), + ); + + expect(beginInput?.resourceReservation).toEqual({ + vcpus: 1, + memoryMb: 4096, + diskMb: 32 * 1024, + }); + expect(createOptions?.memoryMb).toBe(4096); + }); + test("resizes a running VM disk, records the change, and returns provider-confirmed stats", async () => { const vm = testCloudVmRow({ id: "00000000-0000-4000-8000-000000000140", @@ -175,87 +688,256 @@ describe("VM Effect workflows", () => { expect(usageEvents[0]?.eventType).toBe("vm.resize"); }); - test("rejects a disk shrink before calling the provider", async () => { + test("persists a provider-rounded disk claim after a paid resize", async () => { const vm = testCloudVmRow({ - id: "00000000-0000-4000-8000-000000000141", - userId: "user-workflow-resize-shrink", - providerVmId: "provider-vm-resize-shrink", + id: "00000000-0000-4000-8000-000000000142", + userId: "user-workflow-resize-confirmed", + billingTeamId: "team-workflow-resize-confirmed", + billingPlanId: "pro", + providerVmId: "provider-vm-resize-confirmed", status: "running", }); - let resizeCalls = 0; + const confirmations: Array<{ + id: string; + expectedDiskMb: number; + confirmedDiskMb: number; + operationId: string; + }> = []; + let statsCalls = 0; + const repo = { + ...testWorkflowRepo({ vm }), + reserveVmResize: () => Effect.succeed({ + previousDiskMb: 32768, + reservedDiskMb: 65536, + operationId: "resize-operation-confirmed", + }), + confirmVmResize: (confirmation: typeof confirmations[number]) => + Effect.sync(() => { + confirmations.push(confirmation); + return true; + }), + } as unknown as VmRepositoryShape; const provider: VmProviderGatewayShape = { ...unusedProviderGateway(), getStatus: () => Effect.succeed("running"), - getStats: () => Effect.succeed({ state: "awake", sampledAt: Date.now(), diskTotalMb: 65536 }), - resize: () => Effect.sync(() => { resizeCalls += 1; }), + getStats: () => Effect.sync(() => ({ + state: "awake" as const, + sampledAt: Date.now(), + // Freestyle can round a requested disk up to its allocation step. + diskTotalMb: ++statsCalls === 1 ? 32768 : 73728, + })), + resize: () => Effect.void, }; - const error = await Effect.runPromise( + + await Effect.runPromise( resizeVm({ userId: vm.userId, - teamIds: [vm.billingTeamId ?? vm.userId], + teamIds: [vm.billingTeamId!], providerVmId: vm.providerVmId!, - storageMb: 32768, - }).pipe(Effect.flip, Effect.provide(workflowLayer(testWorkflowRepo({ vm }), provider))), + storageMb: 65536, + billingPlanId: "pro", + maxActiveVms: 50, + }).pipe(Effect.provide(workflowLayer(repo, provider))), ); - expect(error).toMatchObject({ _tag: "VmResizeInvalidError", reason: "below_current" }); - expect(resizeCalls).toBe(0); + + expect(confirmations).toEqual([{ + id: vm.id, + expectedDiskMb: 65536, + confirmedDiskMb: 73728, + operationId: "resize-operation-confirmed", + }]); }); - test("rejects an unsupported port before attempting to resume a paused VM", async () => { + test("fails a paid resize when its reservation confirmation loses the race", async () => { const vm = testCloudVmRow({ - id: "00000000-0000-4000-8000-000000000130", - userId: "user-workflow-port-unsupported", - providerVmId: "provider-vm-port-unsupported", - status: "paused", + id: "00000000-0000-4000-8000-000000000143", + userId: "user-workflow-resize-confirmation-race", + billingTeamId: "team-workflow-resize-confirmation-race", + billingPlanId: "pro", + providerVmId: "provider-vm-resize-confirmation-race", + status: "running", }); const usageEvents: RecordedUsageEvent[] = []; - const observedStatuses: ObservedStatusUpdate[] = []; - const repo = testWorkflowRepo({ vm, usageEvents, observedStatuses }); - let resumeCalls = 0; + let statsCalls = 0; + const repo = { + ...testWorkflowRepo({ vm, usageEvents }), + reserveVmResize: () => Effect.succeed({ + previousDiskMb: 32768, + reservedDiskMb: 204800, + requestedDiskMb: 65536, + operationId: "resize-operation-one", + }), + confirmVmResize: () => Effect.succeed(false), + } as unknown as VmRepositoryShape; const provider: VmProviderGatewayShape = { ...unusedProviderGateway(), - resume: () => - Effect.sync(() => { - resumeCalls += 1; - return testVmHandle({ providerVmId: vm.providerVmId! }); - }), + getStatus: () => Effect.succeed("running"), + getStats: () => Effect.sync(() => ({ + state: "awake" as const, + sampledAt: Date.now(), + diskTotalMb: ++statsCalls === 1 ? 32768 : 73728, + })), + resize: () => Effect.void, }; const error = await Effect.runPromise( - openVmPort({ + resizeVm({ userId: vm.userId, + teamIds: [vm.billingTeamId!], providerVmId: vm.providerVmId!, - port: 8000, + storageMb: 65536, + billingPlanId: "pro", + maxActiveVms: 50, }).pipe(Effect.flip, Effect.provide(workflowLayer(repo, provider))), ); - expect(error).toMatchObject({ - _tag: "VmOperationUnsupportedError", - operation: "openPort", - }); - expect(resumeCalls).toBe(0); - expect(observedStatuses).toHaveLength(0); + expect(error).toBeInstanceOf(VmDatabaseError); expect(usageEvents).toHaveLength(0); }); - test("exec resumes a paused VM, retries once, and records one usage event", async () => { + test("finalizes a paid resize conservatively when the post-resize stats read fails", async () => { const vm = testCloudVmRow({ - id: "00000000-0000-4000-8000-000000000101", - userId: "user-workflow-exec-resume", - billingTeamId: "team-workflow-exec-resume", - providerVmId: "provider-vm-exec-resume", - status: "paused", + id: "00000000-0000-4000-8000-000000000144", + userId: "user-workflow-resize-stats-failure", + billingTeamId: "team-workflow-resize-stats-failure", + billingPlanId: "pro", + providerVmId: "provider-vm-resize-stats-failure", + status: "running", }); - const usageEvents: RecordedUsageEvent[] = []; - const observedStatuses: ObservedStatusUpdate[] = []; - const repo = testWorkflowRepo({ vm, usageEvents, observedStatuses }); - const callOrder: string[] = []; - let execCalls = 0; - let statusCalls = 0; - let resumeCalls = 0; - const provider: VmProviderGatewayShape = { - ...unusedProviderGateway(), - exec: () => + const unconfirmed: Array<{ + id: string; + expectedDiskMb: number; + minimumDiskMb?: number; + operationId: string; + }> = []; + let statsCalls = 0; + const repo = { + ...testWorkflowRepo({ vm }), + reserveVmResize: () => Effect.succeed({ + previousDiskMb: 32768, + reservedDiskMb: 204800, + requestedDiskMb: 65536, + operationId: "resize-operation-stats-failure", + }), + markVmResizeUnconfirmed: (confirmation: typeof unconfirmed[number]) => + Effect.sync(() => { + unconfirmed.push(confirmation); + return true; + }), + } as unknown as VmRepositoryShape; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: () => { + statsCalls += 1; + return statsCalls === 1 + ? Effect.succeed({ state: "awake" as const, sampledAt: Date.now(), diskTotalMb: 32768 }) + : Effect.fail(providerOperationError("getStats", "stats response was lost")); + }, + resize: () => Effect.void, + }; + + const error = await Effect.runPromise( + resizeVm({ + userId: vm.userId, + teamIds: [vm.billingTeamId!], + providerVmId: vm.providerVmId!, + storageMb: 65536, + billingPlanId: "pro", + maxActiveVms: 50, + }).pipe(Effect.flip, Effect.provide(workflowLayer(repo, provider))), + ); + + expect(error).toMatchObject({ _tag: "VmProviderOperationError", operation: "getStats" }); + expect(unconfirmed).toHaveLength(1); + expect(unconfirmed[0]).toMatchObject({ + expectedDiskMb: 204800, + minimumDiskMb: 65536, + operationId: "resize-operation-stats-failure", + }); + }); + + test("rejects a disk shrink before calling the provider", async () => { + const vm = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000141", + userId: "user-workflow-resize-shrink", + providerVmId: "provider-vm-resize-shrink", + status: "running", + }); + let resizeCalls = 0; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: () => Effect.succeed({ state: "awake", sampledAt: Date.now(), diskTotalMb: 65536 }), + resize: () => Effect.sync(() => { resizeCalls += 1; }), + }; + const error = await Effect.runPromise( + resizeVm({ + userId: vm.userId, + teamIds: [vm.billingTeamId ?? vm.userId], + providerVmId: vm.providerVmId!, + storageMb: 32768, + }).pipe(Effect.flip, Effect.provide(workflowLayer(testWorkflowRepo({ vm }), provider))), + ); + expect(error).toMatchObject({ _tag: "VmResizeInvalidError", reason: "below_current" }); + expect(resizeCalls).toBe(0); + }); + + test("rejects an unsupported port before attempting to resume a paused VM", async () => { + const vm = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000130", + userId: "user-workflow-port-unsupported", + providerVmId: "provider-vm-port-unsupported", + status: "paused", + }); + const usageEvents: RecordedUsageEvent[] = []; + const observedStatuses: ObservedStatusUpdate[] = []; + const repo = testWorkflowRepo({ vm, usageEvents, observedStatuses }); + let resumeCalls = 0; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + resume: () => + Effect.sync(() => { + resumeCalls += 1; + return testVmHandle({ providerVmId: vm.providerVmId! }); + }), + }; + + const error = await Effect.runPromise( + openVmPort({ + userId: vm.userId, + providerVmId: vm.providerVmId!, + port: 8000, + }).pipe(Effect.flip, Effect.provide(workflowLayer(repo, provider))), + ); + + expect(error).toMatchObject({ + _tag: "VmOperationUnsupportedError", + operation: "openPort", + }); + expect(resumeCalls).toBe(0); + expect(observedStatuses).toHaveLength(0); + expect(usageEvents).toHaveLength(0); + }); + + test("exec resumes a paused VM, retries once, and records one usage event", async () => { + const vm = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000101", + userId: "user-workflow-exec-resume", + billingTeamId: "team-workflow-exec-resume", + providerVmId: "provider-vm-exec-resume", + status: "paused", + }); + const usageEvents: RecordedUsageEvent[] = []; + const observedStatuses: ObservedStatusUpdate[] = []; + const repo = testWorkflowRepo({ vm, usageEvents, observedStatuses }); + const callOrder: string[] = []; + let execCalls = 0; + let statusCalls = 0; + let resumeCalls = 0; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + exec: () => Effect.suspend(() => { execCalls += 1; callOrder.push("exec"); @@ -2416,6 +3098,912 @@ describe("VM Effect workflows", () => { expect(destroyedUsageCount).toBe("1"); }); + test("keeps paid Base recovery free of synchronous legacy provider fanout", async () => { + const now = new Date(); + const existing = testCloudVmRow({ + id: "00000000-0000-4000-8000-000000000143", + userId: "user-workflow-base-reconcile", + billingTeamId: "team-workflow-base-reconcile", + billingPlanId: "pro", + providerVmId: "provider-vm-base-reconcile-old", + status: "running", + }); + const base = { + id: "00000000-0000-4000-8000-000000000144", + scopeType: "team", + scopeId: existing.billingTeamId!, + name: "default", + activeGeneration: 1, + activeVmId: existing.id, + activeProvider: "freestyle", + activeProviderVmId: existing.providerVmId, + state: "ready", + createdByUserId: existing.userId, + lastOpenedByUserId: existing.userId, + createdAt: now, + updatedAt: now, + } as CloudVmBaseRow; + const generation = { + id: "00000000-0000-4000-8000-000000000145", + baseId: base.id, + generation: 1, + vmId: existing.id, + provider: "freestyle", + providerVmId: existing.providerVmId, + state: "active", + createdByUserId: existing.userId, + retainedAt: null, + deletedAt: null, + createdAt: now, + updatedAt: now, + } as CloudVmBaseGenerationRow; + const events: string[] = []; + let beginCalls = 0; + const legacy = { ...existing, id: "00000000-0000-4000-8000-000000000146" }; + const repo = { + ...testWorkflowRepo({ + vm: existing, + markProviderObservedStatus: () => { + events.push("mark-destroyed"); + return Effect.succeed(true); + }, + }), + legacyResourceReservationCandidates: () => { + events.push("legacy-candidates"); + return Effect.succeed([legacy]); + }, + setResourceReservation: () => { + events.push("legacy-write"); + return Effect.succeed(true); + }, + beginBaseOpen: () => { + events.push("begin"); + beginCalls += 1; + return beginCalls === 1 + ? Effect.succeed({ kind: "existing" as const, base, generation, vm: existing }) + : Effect.fail(new Error("stop after recovery reservation check") as never); + }, + } as unknown as VmRepositoryShape; + const deleted = new Error("VM_DELETED: provider VM is gone"); + deleted.name = "VmDeletedError"; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => { + events.push("provider-status"); + return Effect.fail(new VmProviderOperationError({ + provider: "freestyle", + operation: "getStatus", + cause: deleted, + })); + }, + getStats: () => { + events.push("provider-stats"); + return Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + diskTotalMb: 65536, + }); + }, + }; + + await Effect.runPromise( + openBaseVm({ + userId: existing.userId, + billingCustomerType: "team", + billingTeamId: existing.billingTeamId!, + billingPlanId: "pro", + maxActiveVms: 50, + provider: "freestyle", + image: existing.imageId, + baseName: "default", + }).pipe(Effect.provide(workflowLayer(repo, provider))).pipe(Effect.flip), + ); + + const firstBegin = events.indexOf("begin"); + const secondBegin = events.lastIndexOf("begin"); + expect(beginCalls).toBe(2); + expect(events).not.toContain("legacy-candidates"); + expect(events).not.toContain("legacy-write"); + expect(firstBegin).toBeGreaterThanOrEqual(0); + expect(secondBegin).toBeGreaterThan(firstBegin); + }); + + dbTest("does not stamp an unmeasured reservation on a free VM row", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + + const result = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.beginCreate({ + userId: "user-workflow-free-unmeasured", + billingTeamId: "team-workflow-free-unmeasured", + billingPlanId: "free", + provider: "freestyle", + image: "snapshot-test", + maxActiveVms: 3, + idempotencyKey: "free-unmeasured", + }); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + + expect(result.inserted).toBe(true); + expect(result.vm.providerMetadata).toEqual({}); + const [row] = await sql<{ providerMetadata: Record }[]>` + select provider_metadata as "providerMetadata" + from cloud_vms + where id = ${result.vm.id} + `; + expect(row?.providerMetadata).toEqual({}); + }); + + dbTest("holds shared disk headroom until a provider resize is confirmed", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000147"; + const teamId = "team-workflow-resize-headroom"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-headroom', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-headroom', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 }, + })} + ) + `; + + const runRepo = (operation: (repo: VmRepositoryShape) => Effect.Effect) => + Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* operation(repo); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + const reservation = await runRepo((repo) => repo.reserveVmResize!({ + id: vmId, + userId: "user-workflow-resize-headroom", + billingTeamId: teamId, + providerVmId: "provider-vm-resize-headroom", + currentDiskMb: 32768, + storageMb: 65536, + maxActiveVms: 50, + })); + + expect(reservation).toMatchObject({ + previousDiskMb: 32768, + reservedDiskMb: 200 * 1024, + requestedDiskMb: 65536, + }); + expect(typeof reservation?.operationId).toBe("string"); + const [pending] = await sql<{ pending: boolean }[]>` + select provider_metadata ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY} as pending + from cloud_vms + where id = ${vmId} + `; + expect(pending?.pending).toBe(true); + const blocked = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.beginCreate({ + userId: "user-workflow-resize-headroom", + billingTeamId: teamId, + billingPlanId: "pro", + provider: "freestyle", + image: "snapshot-test", + maxActiveVms: 50, + idempotencyKey: "blocked-while-resizing", + resourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 }, + sharedResourceCapacity: { vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }, + }); + }).pipe(Effect.flip, Effect.provide(VmRepositoryLive)), + ); + const blockedFailure = Array.isArray(blocked) ? blocked[0] : blocked; + expect(blockedFailure).toMatchObject({ _tag: "VmSharedResourceLimitExceededError", resource: "diskMb" }); + + const confirmed = await runRepo((repo) => repo.confirmVmResize!({ + id: vmId, + expectedDiskMb: reservation!.reservedDiskMb, + minimumDiskMb: reservation!.requestedDiskMb, + confirmedDiskMb: 73728, + operationId: reservation!.operationId, + })); + expect(confirmed).toBe(true); + const [cleared] = await sql<{ pending: boolean }[]>` + select provider_metadata ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY} as pending + from cloud_vms + where id = ${vmId} + `; + expect(cleared?.pending).toBe(false); + + const created = await runRepo((repo) => repo.beginCreate({ + userId: "user-workflow-resize-headroom", + billingTeamId: teamId, + billingPlanId: "pro", + provider: "freestyle", + image: "snapshot-test", + maxActiveVms: 50, + idempotencyKey: "after-resize-confirmed", + resourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 }, + sharedResourceCapacity: { vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }, + })); + expect(created.inserted).toBe(true); + + const [stored] = await sql<{ diskMb: number }[]>` + select (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb" + from cloud_vms + where id = ${vmId} + `; + expect(stored?.diskMb).toBe(73728); + }); + + dbTest("reserves remaining shared-pool headroom while a native fork runs", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const sourceId = "00000000-0000-4000-8000-000000000160"; + const teamId = "team-workflow-fork-headroom"; + const sourceProviderId = "provider-vm-fork-headroom-source"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${sourceId}, 'user-workflow-fork-headroom', ${teamId}, 'pro', 'freestyle', + ${sourceProviderId}, 'snapshot-test', 'running', + ${sql.json({ cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 } })} + ) + `; + + const runRepo = (operation: (repo: VmRepositoryShape) => Effect.Effect) => + Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* operation(repo); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + const created = await runRepo((repo) => repo.beginCreate({ + userId: "user-workflow-fork-headroom", + billingTeamId: teamId, + billingPlanId: "pro", + provider: "freestyle", + image: "snapshot-test", + maxActiveVms: 50, + idempotencyKey: "native-fork-headroom", + resourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 }, + sharedResourceCapacity: { vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }, + reserveSharedResourceHeadroom: true, + })); + expect(created.inserted).toBe(true); + + const [stored] = await sql<{ vcpus: number; memoryMb: number; diskMb: number }[]>` + select + (provider_metadata->'cmuxResourceReservation'->>'vcpus')::integer as vcpus, + (provider_metadata->'cmuxResourceReservation'->>'memoryMb')::integer as "memoryMb", + (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb" + from cloud_vms + where id = ${created.vm.id} + `; + expect(stored).toEqual({ vcpus: 3, memoryMb: 12 * 1024, diskMb: 168 * 1024 }); + + const blockedResize = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.reserveVmResize!({ + id: sourceId, + userId: "user-workflow-fork-headroom", + billingTeamId: teamId, + providerVmId: sourceProviderId, + currentDiskMb: 32768, + storageMb: 65536, + maxActiveVms: 50, + }); + }).pipe(Effect.flip, Effect.provide(VmRepositoryLive)), + ); + expect(blockedResize).toMatchObject({ _tag: "VmSharedResourceLimitExceededError", resource: "diskMb" }); + + const replaced = await runRepo((repo) => repo.setResourceReservation!({ + id: created.vm.id, + expectedReservation: { vcpus: 3, memoryMb: 12 * 1024, diskMb: 168 * 1024 }, + reservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 }, + sharedResourceCapacity: { vcpus: 5, memoryMb: 20 * 1024, diskMb: 200 * 1024 }, + })); + expect(replaced).toBe(true); + }); + + dbTest("rejects a second resize while the first resize marker is pending", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000148"; + const teamId = "team-workflow-resize-in-progress"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-in-progress', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-in-progress', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 }, + cmuxResourceResizePending: { + operationId: "resize-operation-existing", + requestedDiskMb: 65536, + previousDiskMb: 32768, + }, + })} + ) + `; + + const error = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.reserveVmResize!({ + id: vmId, + userId: "user-workflow-resize-in-progress", + billingTeamId: teamId, + providerVmId: "provider-vm-resize-in-progress", + currentDiskMb: 32768, + storageMb: 73728, + maxActiveVms: 50, + }); + }).pipe(Effect.flip, Effect.provide(VmRepositoryLive)), + ); + + expect(error).toMatchObject({ _tag: "VmResizeInProgressError", vmId }); + }); + + dbTest("confirms only the resize generation that owns the pending marker", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000149"; + const teamId = "team-workflow-resize-generation"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-generation', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-generation', 'snapshot-test', 'running', + ${sql.json({ cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: 32768 } })} + ) + `; + + const runRepo = (operation: (repo: VmRepositoryShape) => Effect.Effect) => + Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* operation(repo); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + const first = await runRepo((repo) => repo.reserveVmResize!({ + id: vmId, + userId: "user-workflow-resize-generation", + billingTeamId: teamId, + providerVmId: "provider-vm-resize-generation", + currentDiskMb: 32768, + storageMb: 65536, + maxActiveVms: 50, + })); + expect(typeof first?.operationId).toBe("string"); + + await runRepo((repo) => repo.mergeProviderMetadata!({ + id: vmId, + patch: { + networkId: "provider-network", + [VM_RESOURCE_RESIZE_PENDING_METADATA_KEY]: { + operationId: "provider-cannot-replace-generation", + requestedDiskMb: 131072, + previousDiskMb: 65536, + }, + }, + })); + const [protectedMarker] = await sql<{ operationId: string; networkId: string }[]>` + select + provider_metadata->'cmuxResourceResizePending'->>'operationId' as "operationId", + provider_metadata->>'networkId' as "networkId" + from cloud_vms + where id = ${vmId} + `; + expect(protectedMarker).toEqual({ + operationId: first!.operationId, + networkId: "provider-network", + }); + + await sql` + update cloud_vms + set provider_metadata = jsonb_set( + provider_metadata, + '{cmuxResourceResizePending,operationId}', + '"resize-operation-newer"'::jsonb, + true + ) + where id = ${vmId} + `; + const stale = await runRepo((repo) => repo.confirmVmResize!({ + id: vmId, + expectedDiskMb: first!.reservedDiskMb, + minimumDiskMb: first!.requestedDiskMb, + confirmedDiskMb: 73728, + operationId: first!.operationId, + })); + expect(stale).toBe(false); + const [stillPending] = await sql<{ operationId: string }[]>` + select provider_metadata->'cmuxResourceResizePending'->>'operationId' as "operationId" + from cloud_vms + where id = ${vmId} + `; + expect(stillPending?.operationId).toBe("resize-operation-newer"); + + const current = await runRepo((repo) => repo.confirmVmResize!({ + id: vmId, + expectedDiskMb: first!.reservedDiskMb, + minimumDiskMb: first!.requestedDiskMb, + confirmedDiskMb: 73728, + operationId: "resize-operation-newer", + })); + expect(current).toBe(true); + }); + + dbTest("background reconciliation recovers a completed pending resize before a paid create", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const oldVmId = "00000000-0000-4000-8000-000000000150"; + const teamId = "team-workflow-resize-recovery"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${oldVmId}, 'user-workflow-resize-recovery-old', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-recovery-old', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: PLAN_SHARED_DISK_MB }, + cmuxResourceResizePending: { + operationId: "resize-operation-recovery", + requestedDiskMb: 65536, + previousDiskMb: 32768, + }, + })} + ) + `; + + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: (_provider, providerVmId) => { + expect(providerVmId).toBe("provider-vm-resize-recovery-old"); + return Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 2, + memoryTotalMb: 8192, + diskTotalMb: 73728, + }); + }, + create: (_provider, options) => Effect.succeed({ + provider: "freestyle" as const, + providerVmId: "provider-vm-resize-recovery-new", + status: "running" as const, + image: options.image, + createdAt: Date.now(), + }), + }; + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe(Effect.provide(providerLayer(provider))), + ); + + const created = await Effect.runPromise( + createVm({ + userId: "user-workflow-resize-recovery-new", + billingCustomerType: "team", + billingTeamId: teamId, + billingPlanId: "pro", + maxActiveVms: 50, + provider: "freestyle", + image: "snapshot-test", + idempotencyKey: "resize-recovery-create", + }).pipe(Effect.provide(providerLayer(provider))), + ); + + expect(created.providerVmId).toBe("provider-vm-resize-recovery-new"); + const [oldRow] = await sql<{ diskMb: number; pending: boolean }[]>` + select + (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb", + provider_metadata ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY} as pending + from cloud_vms + where id = ${oldVmId} + `; + expect(oldRow).toEqual({ diskMb: 73728, pending: false }); + }); + + dbTest("background reconciliation lowers an unconfirmed resize claim after stats return", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000151"; + const teamId = "team-workflow-resize-unconfirmed"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-unconfirmed', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-unconfirmed', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: PLAN_SHARED_DISK_MB }, + [VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY]: { + operationId: "resize-operation-unconfirmed", + requestedDiskMb: 65536, + }, + })} + ) + `; + + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: (_provider, providerVmId) => { + expect(providerVmId).toBe("provider-vm-resize-unconfirmed"); + return Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 2, + memoryTotalMb: 8192, + diskTotalMb: 73728, + }); + }, + }; + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe(Effect.provide(providerLayer(provider))), + ); + + const [row] = await sql<{ diskMb: number; unconfirmed: boolean }[]>` + select + (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb", + provider_metadata ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY} as unconfirmed + from cloud_vms + where id = ${vmId} + `; + expect(row).toEqual({ diskMb: 73728, unconfirmed: false }); + }); + + dbTest("background reconciliation releases an abandoned resize marker into an unconfirmed claim", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000152"; + const teamId = "team-workflow-resize-abandoned"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-abandoned', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-abandoned', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: PLAN_SHARED_DISK_MB }, + [VM_RESOURCE_RESIZE_PENDING_METADATA_KEY]: { + operationId: "resize-operation-abandoned", + requestedDiskMb: 65536, + previousDiskMb: 32768, + createdAtMs: Date.now() - (60 * 60 * 1000), + }, + })} + ) + `; + + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: () => Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 2, + memoryTotalMb: 8192, + diskTotalMb: 32768, + }), + }; + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe(Effect.provide(providerLayer(provider))), + ); + + const [row] = await sql<{ + diskMb: number; + pending: boolean; + unconfirmed: boolean; + }[]>` + select + (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb", + provider_metadata ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY} as pending, + provider_metadata ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY} as unconfirmed + from cloud_vms + where id = ${vmId} + `; + expect(row).toEqual({ diskMb: VM_DISK_MB_MAX, pending: false, unconfirmed: true }); + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe(Effect.provide(providerLayer(provider))), + ); + const [stillUnconfirmed] = await sql<{ + diskMb: number; + pending: boolean; + unconfirmed: boolean; + }[]>` + select + (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb", + provider_metadata ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY} as pending, + provider_metadata ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY} as unconfirmed + from cloud_vms + where id = ${vmId} + `; + expect(stillUnconfirmed).toEqual({ diskMb: VM_DISK_MB_MAX, pending: false, unconfirmed: true }); + }); + + dbTest("does not reconcile a fresh pending resize while its request can still confirm", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000153"; + const teamId = "team-workflow-resize-fresh-pending"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-fresh-pending', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-fresh-pending', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: PLAN_SHARED_DISK_MB }, + [VM_RESOURCE_RESIZE_PENDING_METADATA_KEY]: { + operationId: "resize-operation-fresh-pending", + requestedDiskMb: 65536, + previousDiskMb: 32768, + createdAtMs: Date.now(), + }, + })} + ) + `; + + let statsCalls = 0; + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: () => Effect.sync(() => { + statsCalls += 1; + return { + state: "awake" as const, + sampledAt: Date.now(), + cpus: 2, + memoryTotalMb: 8192, + diskTotalMb: 73728, + }; + }), + }; + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe(Effect.provide(providerLayer(provider))), + ); + + expect(statsCalls).toBe(0); + const [row] = await sql<{ pending: boolean; unconfirmed: boolean }[]>` + select + provider_metadata ? ${VM_RESOURCE_RESIZE_PENDING_METADATA_KEY} as pending, + provider_metadata ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY} as unconfirmed + from cloud_vms + where id = ${vmId} + `; + expect(row).toEqual({ pending: true, unconfirmed: false }); + }); + + dbTest("rolls back an unconfirmed resize claim after bounded recovery", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000154"; + const teamId = "team-workflow-resize-unconfirmed-timeout"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-unconfirmed-timeout', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-unconfirmed-timeout', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: VM_DISK_MB_MAX }, + [VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY]: { + operationId: "resize-operation-unconfirmed-timeout", + requestedDiskMb: 65536, + previousDiskMb: 32768, + markedAtMs: Date.now() - (60 * 60 * 1000), + }, + })} + ) + `; + + const provider: VmProviderGatewayShape = { + ...unusedProviderGateway(), + getStatus: () => Effect.succeed("running"), + getStats: () => Effect.succeed({ + state: "awake" as const, + sampledAt: Date.now(), + cpus: 2, + memoryTotalMb: 8192, + diskTotalMb: 32768, + }), + }; + + await Effect.runPromise( + reconcileVmProviderStatuses().pipe(Effect.provide(providerLayer(provider))), + ); + + const [row] = await sql<{ diskMb: number; unconfirmed: boolean }[]>` + select + (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb", + provider_metadata ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY} as unconfirmed + from cloud_vms + where id = ${vmId} + `; + expect(row).toEqual({ diskMb: 32768, unconfirmed: false }); + }); + + dbTest("repairs an unconfirmed resize on a confirmed no-op retry", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const vmId = "00000000-0000-4000-8000-000000000155"; + const teamId = "team-workflow-resize-noop-retry"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${vmId}, 'user-workflow-resize-noop-retry', ${teamId}, 'pro', 'freestyle', + 'provider-vm-resize-noop-retry', 'snapshot-test', 'running', + ${sql.json({ + cmuxResourceReservation: { vcpus: 2, memoryMb: 8192, diskMb: VM_DISK_MB_MAX }, + [VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY]: { + operationId: "resize-operation-noop-retry", + requestedDiskMb: 65536, + previousDiskMb: 32768, + markedAtMs: Date.now(), + }, + })} + ) + `; + + const reservation = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.reserveVmResize!({ + id: vmId, + userId: "user-workflow-resize-noop-retry", + billingTeamId: teamId, + providerVmId: "provider-vm-resize-noop-retry", + currentDiskMb: 65536, + storageMb: 65536, + maxActiveVms: 50, + }); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + + expect(reservation).toMatchObject({ + previousDiskMb: 65536, + reservedDiskMb: 65536, + requestedDiskMb: 65536, + }); + const [row] = await sql<{ diskMb: number; unconfirmed: boolean }[]>` + select + (provider_metadata->'cmuxResourceReservation'->>'diskMb')::integer as "diskMb", + provider_metadata ? ${VM_RESOURCE_RESIZE_UNCONFIRMED_METADATA_KEY} as unconfirmed + from cloud_vms + where id = ${vmId} + `; + expect(row).toEqual({ diskMb: 65536, unconfirmed: false }); + }); + + dbTest("uses the shared disk pool for snapshot events without a recorded size", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + await sql` + insert into cloud_vm_usage_events ( + user_id, billing_team_id, billing_plan_id, event_type, provider, image_id, metadata + ) values ( + 'user-workflow-legacy-snapshot', 'team-workflow-legacy-snapshot', 'pro', + 'vm.snapshot.created', 'freestyle', 'snapshot-test', + ${sql.json({ snapshotId: "legacy-snapshot-without-size" })} + ) + `; + + const reservation = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.ownedSnapshotResourceReservation!({ + userId: "user-workflow-legacy-snapshot", + billingTeamId: "team-workflow-legacy-snapshot", + provider: "freestyle", + snapshotId: "legacy-snapshot-without-size", + }); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + + expect(reservation).toEqual({ + vcpus: 5, + memoryMb: 20 * 1024, + diskMb: PLAN_SHARED_DISK_MB, + }); + }); + + dbTest("uses recorded provider shape for a legacy snapshot source", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + await sql` + insert into cloud_vm_usage_events ( + user_id, billing_team_id, billing_plan_id, event_type, provider, image_id, metadata + ) values ( + 'user-workflow-recorded-snapshot', 'team-workflow-recorded-snapshot', 'pro', + 'vm.snapshot.created', 'freestyle', 'snapshot-test', + ${sql.json({ + snapshotId: "recorded-legacy-snapshot", + vcpus: 2, + memoryMb: 8192, + diskMb: 65536, + })} + ) + `; + + const reservation = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.ownedSnapshotResourceReservation!({ + userId: "user-workflow-recorded-snapshot", + billingTeamId: "team-workflow-recorded-snapshot", + provider: "freestyle", + snapshotId: "recorded-legacy-snapshot", + }); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + + expect(reservation).toEqual({ vcpus: 2, memoryMb: 8192, diskMb: 65536 }); + }); + + dbTest("uses snapshot-time dimensions when the source later grows", async () => { + if (!sql) throw new Error("test database not initialized"); + await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`; + const sourceVmId = "00000000-0000-4000-8000-000000000164"; + await sql` + insert into cloud_vms ( + id, user_id, billing_team_id, billing_plan_id, provider, provider_vm_id, + image_id, status, provider_metadata + ) values ( + ${sourceVmId}, 'user-workflow-snapshot-grown', 'team-workflow-snapshot-grown', + 'pro', 'freestyle', 'provider-vm-snapshot-grown', 'snapshot-test', 'running', + ${sql.json({ cmuxResourceReservation: { vcpus: 16, memoryMb: 32768, diskMb: 160 * 1024 } })} + ) + `; + await sql` + insert into cloud_vm_usage_events ( + user_id, billing_team_id, vm_id, billing_plan_id, event_type, provider, image_id, metadata + ) values ( + 'user-workflow-snapshot-grown', 'team-workflow-snapshot-grown', ${sourceVmId}, + 'pro', 'vm.snapshot.created', 'freestyle', 'snapshot-test', + ${sql.json({ snapshotId: "snapshot-before-growth", vcpus: 2, memoryMb: 8192, diskMb: 65536 })} + ) + `; + + const reservation = await Effect.runPromise( + Effect.gen(function* () { + const repo = yield* VmRepository; + return yield* repo.ownedSnapshotResourceReservation!({ + userId: "user-workflow-snapshot-grown", + billingTeamId: "team-workflow-snapshot-grown", + provider: "freestyle", + snapshotId: "snapshot-before-growth", + }); + }).pipe(Effect.provide(VmRepositoryLive)), + ); + + expect(reservation).toEqual({ vcpus: 2, memoryMb: 8192, diskMb: 65536 }); + }); + dbTest("resets Base by retaining the previous generation when capacity allows", async () => { if (!sql) throw new Error("test database not initialized"); await sql`truncate cloud_vm_billing_grants, cloud_vm_usage_events, cloud_vm_leases, cloud_vms restart identity cascade`;