Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 5 additions & 2 deletions service/src/hosted-app/control-plane.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -52,9 +52,12 @@ const runtimeConfig: HostedAppMicrovmConfig = {
tokenTps: 8,
};

/* The seed is salted per call, so fixtures share one allocation. */
const seededGeneration = hostedAppLaunchGenerationSeed(runtimeConfig);

class MemoryRegistry implements HostedAppRegistry {
record: RuntimeSessionRecord | null = null;
generation = hostedAppLaunchGenerationSeed(runtimeConfig);
generation = seededGeneration;
writes: RuntimeSessionRecord[] = [];
allocations = 0;

Expand Down Expand Up @@ -260,7 +263,7 @@ test('does not report an expired pending intent as starting', () => {
});

function pendingRecord(): RuntimeSessionRecord {
const generation = hostedAppLaunchGenerationSeed(runtimeConfig);
const generation = seededGeneration;
return {
runtime_session_id: input.hostedAppRuntimeId,
tenant_id: input.tenantId,
Expand Down
4 changes: 2 additions & 2 deletions service/src/hosted-app/microvm-runtime.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -162,14 +162,14 @@ describe('HostedAppMicrovmRuntime', () => {
expect(error).toBeInstanceOf(HostedAppMicrovmError);
expect(error.transient).toBe(true);
});
test('seeds idempotency from exact wire inputs while keeping semantic matching order-independent', () => {
test('fingerprints exact wire inputs, keeps semantic matching order-independent, and salts seeds', () => {
const first = { ...config(), ingressConnectorArns: ['arn:b', 'arn:a'] };
const reordered = { ...config(), ingressConnectorArns: ['arn:a', 'arn:b'] };
expect(hostedAppLaunchFingerprint(first)).toBe(hostedAppLaunchFingerprint(reordered));
expect(hostedAppLaunchRequestFingerprint(first)).not.toBe(
hostedAppLaunchRequestFingerprint(reordered),
);
expect(hostedAppLaunchGenerationSeed(first)).not.toBe(hostedAppLaunchGenerationSeed(reordered));
expect(hostedAppLaunchGenerationSeed(first)).not.toBe(hostedAppLaunchGenerationSeed(first));
});

test('launches the dedicated image with bounded idle policy and no egress connector', async () => {
Expand Down
7 changes: 5 additions & 2 deletions service/src/hosted-app/microvm-runtime.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ import {
type ThrottledOp,
} from '../runtime-session/throttle';
import type { ResidentHostedAppSpec } from './spec';
import { createHash } from 'node:crypto';
import { createHash, randomBytes } from 'node:crypto';

export interface HostedAppMicrovmConfig {
imageArn: string;
Expand Down Expand Up @@ -73,11 +73,14 @@ export function hostedAppLaunchRequestFingerprint(config: HostedAppMicrovmConfig
});
}

/** Keep reset Redis counters in an image/config-specific safe-integer range. */
/** Keep reset Redis counters in a safe-integer range. The random salt makes
* every reset counter start somewhere new, because AWS rejects a reused token
* it still remembers even when the request body is identical. */
export function hostedAppLaunchGenerationSeed(config: HostedAppMicrovmConfig): number {
const offset = Number.parseInt(
createHash('sha256')
.update(hostedAppLaunchRequestFingerprint(config), 'utf8')
.update(randomBytes(16))
.digest('hex')
.slice(0, 13),
16,
Expand Down
57 changes: 51 additions & 6 deletions service/src/sandbox-backend/lambda-microvm.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -289,11 +289,11 @@ describe('normalizeMicrovmEndpoint', () => {
});

describe('runtime session launch tokens', () => {
test('uses a deterministic launch namespace and stays within the AWS limit', () => {
test('salts every launch namespace seed and stays within the AWS limit', () => {
const cfg = config();
const seed = runtimeSessionLaunchGenerationSeed(cfg);
expect(seed).toBeGreaterThanOrEqual(RUNTIME_SESSION_NAMESPACED_GENERATION_MIN);
expect(runtimeSessionLaunchGenerationSeed({ ...cfg, imageVersion: '4' })).not.toBe(seed);
expect(runtimeSessionLaunchGenerationSeed(cfg)).not.toBe(seed);

const runtimeSessionId = `rt_${'a'.repeat(40)}`;
const token = runtimeSessionLaunchClientToken(runtimeSessionId, Number.MAX_SAFE_INTEGER);
Expand Down Expand Up @@ -690,10 +690,7 @@ describe('LambdaMicrovmSandboxBackend session execution', () => {
* (image builds stay hookless), so RunMicrovm carries no runHookPayload. */
expect(runArgs.runHookPayload).toBeUndefined();
expect(runArgs.idlePolicy?.autoResume).toBe(true);
expect(runArgs.clientToken).toBe(runtimeSessionLaunchClientToken(
'rt_session_1',
runtimeSessionLaunchGenerationSeed(config()),
));
expect(runArgs.clientToken).toMatch(/^sess-rt_session_1-\d+$/);
expect(runArgs.maximumDurationSeconds).toBe(28_800);

const executeReq = captured.find((c) => c.path === '/api/v2/execute');
Expand All @@ -704,6 +701,7 @@ describe('LambdaMicrovmSandboxBackend session execution', () => {
expect(record?.microvm_id).toBe([...fake.vms.keys()][0]);
expect(record?.generation).toBeGreaterThanOrEqual(RUNTIME_SESSION_NAMESPACED_GENERATION_MIN);
expect(record?.launch_client_token).toBe(runArgs.clientToken);
expect(runArgs.clientToken).toBe(runtimeSessionLaunchClientToken('rt_session_1', record?.generation as number));
});

test('replays a legacy recorded launch intent after the RunMicrovm response is lost', async () => {
Expand Down Expand Up @@ -875,6 +873,53 @@ describe('LambdaMicrovmSandboxBackend session execution', () => {
expect((await readRuntimeSessionRecord('rt_session_1'))?.image_version).toBe('4');
});

test('never reissues a launch token after registry loss on an unchanged config', async () => {
const fake = fakeClient();
await makeBackend(fake).execute(request(), sessionContext());
const firstToken = (fake.callsFor('runMicrovm')[0].args as { clientToken?: string }).clientToken;

/* The rtsx:gen counter expires after an idle night while AWS still
* remembers the token and rejects its reuse ("The provided clientToken was
* used with different request parameters"), even for an identical body.
* A relaunch on the same config must therefore never resend the token. */
await fake.terminateMicrovm([...fake.vms.keys()][0]);
await mock.del('rtsx:sess:{rt_session_1}', 'rtsx:gen:{rt_session_1}');

const second = await makeBackend(fake).execute(request(), sessionContext())
.catch((error: unknown) => error);
const secondToken = (fake.callsFor('runMicrovm')[1].args as { clientToken?: string }).clientToken;
expect(secondToken).not.toBe(firstToken);
expect(second).toEqual(EXECUTE_RESPONSE);
});

test('stops replaying a PENDING token that AWS rejected as reused', async () => {
const fake = fakeClient();
/* An ambiguous failure persists the PENDING intent and its token. */
fake.failNext('runMicrovm', new LambdaMicrovmApiError('other', 'RunMicrovm', 'connection lost'));
await expect(makeBackend(fake).execute(request(), sessionContext())).rejects.toMatchObject({
code: 'MICROVM_LAUNCH_FAILED',
});
const poisoned = (await readRuntimeSessionRecord('rt_session_1'))?.launch_client_token;
expect(poisoned).toBeDefined();

fake.failNext('runMicrovm', new LambdaMicrovmApiError(
'validation',
'RunMicrovm',
'The provided clientToken was used with different request parameters.',
));
await expect(makeBackend(fake).execute(request(), sessionContext())).rejects.toMatchObject({
code: 'MICROVM_LAUNCH_FAILED',
});

await expect(makeBackend(fake).execute(request(), sessionContext()))
.resolves.toEqual(EXECUTE_RESPONSE);
const tokens = fake.callsFor('runMicrovm')
.map(call => (call.args as { clientToken?: string }).clientToken);
expect(tokens).toHaveLength(3);
expect(tokens.slice(0, 2)).toEqual([poisoned, poisoned]);
expect(tokens[2]).not.toBe(poisoned);
});

test('does not replay a persisted token when the exact connector request changed', async () => {
const fake = fakeClient();
const firstRun = fake.runMicrovm.bind(fake);
Expand Down
32 changes: 22 additions & 10 deletions service/src/sandbox-backend/lambda-microvm.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
import axios from 'axios';
import { nanoid } from 'nanoid';
import * as fs from 'fs';
import { createHash } from 'crypto';
import { createHash, randomBytes } from 'crypto';
import type { LambdaMicrovmClient, MicrovmAuthToken, MicrovmDescription, MicrovmIdlePolicy } from '../runtime-session/lambda-client';
import type { SandboxBackend, SandboxExecuteContext, SandboxRawResponse, SandboxTransportRequest } from './types';
import type { RuntimeSessionRecord } from '../runtime-session/registry';
Expand Down Expand Up @@ -96,8 +96,8 @@ export function runtimeSessionLaunchFingerprint(config: LambdaMicrovmBackendConf
}

/** Generations below this boundary were allocated by the original INCR-only
* scheme. New launches start in a fingerprint-seeded namespace so losing the
* Redis counter cannot reuse an AWS clientToken after launch inputs change.
* scheme. New launches start in a randomly seeded namespace so losing the
* Redis counter cannot reuse an AWS clientToken.
*
* Keep the provider token itself in the legacy `sess-<id>-<generation>` shape:
* an older worker taking over a PENDING record during a rolling deployment
Expand All @@ -124,13 +124,17 @@ function runtimeSessionLaunchRequestFingerprint(config: LambdaMicrovmBackendConf
});
}

/** A 52-bit digest leaves ample safe-integer headroom for later INCRs while
* making a reset counter's first generation depend on the exact launch
* request. Token collisions are scoped to one runtime session. */
/** A 52-bit digest leaves ample safe-integer headroom for later INCRs. The
* random salt makes every reset counter start somewhere new: the rtsx:gen key
* expires after an idle night while AWS still remembers the old tokens, and AWS
* rejects a reused token ("The provided clientToken was used with different
* request parameters") even when the request body is identical. Token
* collisions are scoped to one runtime session. */
export function runtimeSessionLaunchGenerationSeed(config: LambdaMicrovmBackendConfig): number {
const offset = Number.parseInt(
createHash('sha256')
.update(runtimeSessionLaunchRequestFingerprint(config), 'utf8')
.update(randomBytes(16))
.digest('hex')
.slice(0, 13),
16,
Expand Down Expand Up @@ -730,10 +734,18 @@ export class LambdaMicrovmSandboxBackend implements SandboxBackend {
&& error.cause.kind === 'validation';
}

/** AWS reports a reused idempotency token only as a generic
* ValidationException, so the message text is the only discriminator. */
private isClientTokenReuseRejection(error: unknown): boolean {
return this.isValidationLaunchFailure(error)
&& /client ?token\b.*\bdifferent (request )?parameters/i.test((error as Error).message);
}

/** Both the base token and its single retry reached a terminal state and
* were successfully terminated. Keeping that PENDING intent would replay
* two known-dead tokens forever, so let the next request allocate a new
* generation. Ambiguous provider failures remain persisted for recovery. */
* were successfully terminated, or AWS refused the token as already used and
* launched nothing. Keeping that PENDING intent would replay a known-dead
* token forever, so let the next request allocate a new generation.
* Ambiguous provider failures remain persisted for recovery. */
private async retireExhaustedLaunchIntent(
launchIntent: RuntimeSessionRecord,
lockToken: string,
Expand All @@ -742,7 +754,7 @@ export class LambdaMicrovmSandboxBackend implements SandboxBackend {
const cleanlyExhausted =
error instanceof SandboxBackendError
&& error.code === 'MICROVM_LAUNCH_FAILED'
&& error.transient;
&& (error.transient || this.isClientTokenReuseRejection(error));
if (!cleanlyExhausted) return;
try {
const retired = await writeRuntimeSessionRecord({
Expand Down
Loading