Repository navigation
Expand file tree
/
Copy pathlauncher.ts
More file actions
405 lines (377 loc) · 16.8 KB
/
Copy pathlauncher.ts
File metadata and controls
405 lines (377 loc) · 16.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
import { promises as fs } from 'fs';
import * as path from 'path';
import type { CloudHypervisorLandlockRule } from './api-client';
/**
* Secure host launch-argv construction and resource-confinement helpers for
* Cloud Hypervisor.
*
* Cloud Hypervisor has no native process that joins a prepared network namespace,
* chroots, drops capabilities, and execs the VMM as one atomic operation.
* This module documents and implements the exact replacement boundary AWF
* uses instead:
*
* 1. **Network namespace join** — `ip netns exec <namespace> ...` (the
* already-trusted, already-required `ip` tool) execs directly into the
* per-run namespace {@link https://man7.org/linux/man-pages/man8/ip-netns.8.html}
* without an intermediate fork, so the resulting process keeps the PID
* the host process observes.
* 2. **Privilege drop** — `setpriv --reuid --regid --clear-groups
* --no-new-privs --inh-caps=-all --bounding-set=-all` execs the Cloud
* Hypervisor binary as a dedicated per-run system uid/gid with no
* supplementary groups and `no_new_privs` set, before any guest code runs.
* 3. **Filesystem confinement** — Cloud Hypervisor has no chroot of its
* own, and jailer's userspace chroot+pivot_root cannot be replicated
* for a foreign static binary without reimplementing jailer itself.
* Instead AWF combines:
* - a **private run directory** (mode `0700`, owned by the target
* uid/gid) holding only the staged kernel/rootfs, virtio-fs sockets,
* and API/vsock sockets — see `CloudHypervisorManager`;
* - **Landlock** (`landlock_enable`/`landlock_rules` in the
* `vm.create` payload — see {@link computeCloudHypervisorLandlockRules})
* restricting the VMM process's own filesystem access to exactly
* those paths, enforced by the kernel LSM rather than a userspace
* boundary;
* - Cloud Hypervisor's own **default seccomp** filter
* (`--seccomp true`, its default "kill on violation" mode).
* This is a different (kernel-LSM-based) boundary than jailer's chroot,
* not a weaker one — it is intentionally documented and covered by
* tests instead of silently degrading to "no filesystem confinement".
* 4. **Resource limits** — a dedicated cgroup (v1 or v2, matching
* preflight's detected version) bounding memory and CPU time, created
* before launch and assigned by PID immediately after spawn (see
* {@link CloudHypervisorCgroup}).
*
* No shell is ever invoked: every argv below is passed as a plain array to
* `execa`, never interpolated into a shell string.
*/
export const CLOUD_HYPERVISOR_GUEST_CID = 3;
export interface CloudHypervisorLaunchPaths {
readonly kernelPath: string;
readonly rootfsPath: string;
readonly runDirectory: string;
readonly apiSocketPath: string;
readonly vsockSocketPath: string;
/** Host TAP interface name (e.g. `vmt<token>`), for the
* `/sys/class/net/<tapName>/tun_flags` Landlock rule — see
* {@link computeCloudHypervisorLandlockRules}. */
readonly tapName?: string;
}
export interface CloudHypervisorLaunchIdentity {
readonly uid: number;
readonly gid: number;
}
export interface CloudHypervisorLaunchCommand {
readonly command: string;
readonly args: readonly string[];
readonly confinementPolicy: CloudHypervisorLaunchConfinementPolicy;
}
export interface CloudHypervisorLaunchToolPaths {
readonly ip: string;
readonly setpriv: string;
}
export interface CloudHypervisorLaunchConfinementPolicy {
readonly supplementaryGroups: readonly number[];
readonly capabilities: {
readonly inheritable: string;
readonly permitted: string;
readonly effective: string;
readonly bounding: string;
readonly ambient: string;
};
readonly noNewPrivs: 1;
}
const CLOUD_HYPERVISOR_ALLOWED_CAPABILITIES: readonly {
readonly setprivName: string;
readonly bit: number;
}[] = [];
function buildCloudHypervisorConfinementPolicy(): CloudHypervisorLaunchConfinementPolicy {
const capabilityMask = CLOUD_HYPERVISOR_ALLOWED_CAPABILITIES
.reduce((mask, capability) => mask | (1n << BigInt(capability.bit)), 0n)
.toString(16)
.padStart(16, '0');
return {
supplementaryGroups: [],
capabilities: {
inheritable: capabilityMask,
permitted: capabilityMask,
effective: capabilityMask,
bounding: capabilityMask,
ambient: capabilityMask,
},
noNewPrivs: 1,
};
}
/**
* Builds the argv AWF spawns to launch Cloud Hypervisor: join the prepared
* network namespace, drop to the dedicated per-run VMM identity with no
* capabilities, then exec the pinned Cloud Hypervisor binary with only its API
* socket configured; the VM itself is created and booted afterwards over that
* socket.
*
* The launched process has no supplementary groups. A serialized, temporary
* per-run ACL lease grants its uid access to `/dev/kvm` and `/dev/net/tun`
* only through VM creation, avoiding both persistent host-group membership and
* long-lived access to shared host devices.
*
* The network manager creates, configures, and brings up the TAP before this
* process starts, with the target uid/gid recorded as its owner. Cloud
* Hypervisor therefore only needs ordinary `/dev/net/tun` access to reopen that
* TAP and read-only Landlock access to its `tun_flags` sysfs attribute. The
* earlier `CAP_NET_ADMIN` requirement was a Landlock denial misdiagnosed as a
* TAP ownership failure; retaining it would let a compromised VMM reconfigure
* the namespace firewall and interfaces.
*/
export function buildCloudHypervisorLaunchCommand(options: {
readonly tools: CloudHypervisorLaunchToolPaths;
readonly namespaceName: string;
readonly identity: CloudHypervisorLaunchIdentity;
readonly cloudHypervisorBinary: string;
readonly apiSocketPath: string;
readonly logFilePath: string;
}): CloudHypervisorLaunchCommand {
assertSafeNamespaceName(options.namespaceName);
assertPositiveIdentity(options.identity.uid, 'uid');
assertPositiveIdentity(options.identity.gid, 'gid');
if (!path.isAbsolute(options.cloudHypervisorBinary)) {
throw new Error(`Cloud Hypervisor binary path must be absolute: ${options.cloudHypervisorBinary}`);
}
if (!path.isAbsolute(options.apiSocketPath)) {
throw new Error(`Cloud Hypervisor API socket path must be absolute: ${options.apiSocketPath}`);
}
const setprivCapabilities = CLOUD_HYPERVISOR_ALLOWED_CAPABILITIES
.map((capability) => `+${capability.setprivName}`)
.join(',');
const resetCapabilitySet = setprivCapabilities
? `-all,${setprivCapabilities}`
: '-all';
return {
command: options.tools.ip,
args: [
'netns', 'exec', options.namespaceName,
options.tools.setpriv,
`--reuid=${options.identity.uid}`,
`--regid=${options.identity.gid}`,
'--clear-groups',
'--no-new-privs',
`--inh-caps=${resetCapabilitySet}`,
`--bounding-set=${resetCapabilitySet}`,
`--ambient-caps=${setprivCapabilities || '-all'}`,
'--',
options.cloudHypervisorBinary,
'--api-socket', `path=${options.apiSocketPath}`,
'--log-file', options.logFilePath,
'-v',
'--seccomp', 'true',
],
confinementPolicy: buildCloudHypervisorConfinementPolicy(),
};
}
/**
* Computes the minimal set of Landlock filesystem rules Cloud Hypervisor's
* own process needs after `vm.create`: read access to the kernel image,
* read-write access to the rootfs,
* read-write access to the private run directory (for the API and vsock
* UNIX domain sockets it creates there), read-write access to the device
* nodes it must reopen for virtio-net TAP attachment and KVM ioctls, and
* read access to the TAP's own sysfs flags attribute. Cloud Hypervisor's
* virtio-net setup reads `/sys/class/net/<tapName>/tun_flags` (a
* world-readable, `0444` file with no capability requirement of its own)
* to detect multi-queue support; without a Landlock rule for it, that read
* fails with the kernel LSM's own EACCES — observed live as `vm.boot`
* failing with "Failed to read the TAP flags from sysfs: Permission
* denied" even though ordinary Unix file permissions would have allowed
* the read. Any path not listed here becomes inaccessible to the Cloud
* Hypervisor process the instant Landlock is enabled, even to a
* hypothetical guest-escape.
*/
export function computeCloudHypervisorLandlockRules(
paths: CloudHypervisorLaunchPaths,
): CloudHypervisorLandlockRule[] {
const rules: CloudHypervisorLandlockRule[] = [
{ path: paths.kernelPath, access: 'r' },
{ path: paths.rootfsPath, access: 'rw' },
{ path: paths.runDirectory, access: 'rw' },
{ path: '/dev/kvm', access: 'rw' },
];
if (paths.tapName) {
rules.push(
{ path: '/dev/net/tun', access: 'rw' },
{ path: `/sys/class/net/${paths.tapName}/tun_flags`, access: 'r' },
);
}
return rules;
}
export interface CloudHypervisorResourceLimits {
readonly memoryMib: number;
readonly vcpuCount: number;
readonly cpuQuotaMilli?: number;
/** Host tmpfs pages charged to virtio-fs, separate from guest RAM. */
readonly writableStorageBytes?: number;
}
export interface CloudHypervisorCgroupLimits {
readonly memoryMax: string;
readonly cpuMax: string;
readonly pidsMax: string;
}
/** Fixed VMM/guest-overhead headroom added on top of configured guest memory. */
const CGROUP_MEMORY_HEADROOM_MIB = 256;
/**
* Fixed CPU headroom (in the same units as `CGROUP_V2_PERIOD_US`) added on
* top of the per-vCPU quota. Cloud Hypervisor's own I/O, virtio device
* emulation (including the tap fd read/write loop for the guest's
* network device), and API threads all run in this *same* cgroup as the
* vCPU thread(s) and compete for the *same* CPU quota -- a quota sized
* for "1 CPU per vCPU" alone left no dedicated room for that VMM-side
* work. Live-KVM validation on GitHub-hosted runners (nested KVM, so
* vCPU exits are unusually expensive) showed guest network I/O
* essentially stalled (a handful of packets relayed regardless of how
* long the test waited) even after ruling out tap/vnet_hdr negotiation,
* conntrack/offload, and Docker bridge-isolation causes -- consistent
* with the VMM's own non-vCPU threads being starved of their share of an
* already-tight, vCPU-only-sized quota.
*/
const CGROUP_CPU_HEADROOM_QUOTA_US = 100_000;
/** Bounds the number of Cloud Hypervisor host threads/tasks (defense in depth; it is a single process). */
const CGROUP_MAX_PIDS = 256;
const CGROUP_V2_PERIOD_US = 100_000;
const CGROUP_V2_CONTROLLERS = '+cpu +memory +pids';
/**
* cgroup v2 rejects `rmdir()` on a non-empty cgroup (`EBUSY`) not only
* while a process is still a live member, but also for a short window
* after that process has fully exited: charge migration/accounting
* teardown for the memory controller can lag process-exit by a handful of
* milliseconds under load. `stop()` only calls `cleanup()` once process
* termination is already confirmed, so any EBUSY here is this teardown
* race, not a leaked process -- retry briefly instead of leaving residue.
* Observed live: "Cloud Hypervisor cgroup residue remains after cleanup"
* following a guest connectivity-probe failure and immediate teardown.
*/
const CGROUP_REMOVAL_RETRY_INTERVAL_MS = 100;
const CGROUP_REMOVAL_MAX_WAIT_MS = 5_000;
export interface CloudHypervisorCgroupDependencies {
mkdir(directory: string): Promise<unknown>;
writeFile(filePath: string, contents: string): Promise<void>;
/** Removes exactly the (now-empty) leaf cgroup directory. cgroupfs's
* controller/interface files are virtual and cannot be `unlink()`ed, so
* this must be a plain `rmdir`, not a recursive tree removal. */
rmdir(directory: string): Promise<void>;
sleep(milliseconds: number): Promise<void>;
}
const defaultCgroupDependencies: CloudHypervisorCgroupDependencies = {
mkdir: (directory) => fs.mkdir(directory, { recursive: true, mode: 0o700 }),
writeFile: (filePath, contents) => fs.writeFile(filePath, contents),
rmdir: (directory) => fs.rmdir(directory),
sleep: (milliseconds) => new Promise((resolve) => setTimeout(resolve, milliseconds)),
};
export function computeCloudHypervisorCgroupLimits(
limits: CloudHypervisorResourceLimits,
): CloudHypervisorCgroupLimits {
if (
!Number.isSafeInteger(limits.memoryMib) ||
limits.memoryMib < 1 ||
!Number.isSafeInteger(limits.vcpuCount) ||
limits.vcpuCount < 1 ||
(limits.writableStorageBytes !== undefined &&
(!Number.isSafeInteger(limits.writableStorageBytes) || limits.writableStorageBytes < 1)) ||
(limits.cpuQuotaMilli !== undefined &&
(!Number.isSafeInteger(limits.cpuQuotaMilli) ||
limits.cpuQuotaMilli < 1 ||
limits.cpuQuotaMilli > limits.vcpuCount * 1000))
) {
throw new Error('Cloud Hypervisor cgroup resource limits are invalid');
}
const memoryMaxBytes = (limits.memoryMib + CGROUP_MEMORY_HEADROOM_MIB) * 1024 * 1024 +
(limits.writableStorageBytes ?? 0);
if (!Number.isSafeInteger(memoryMaxBytes)) {
throw new Error('Cloud Hypervisor cgroup resource limits are invalid');
}
const cpuQuotaUs = limits.cpuQuotaMilli === undefined
? limits.vcpuCount * CGROUP_V2_PERIOD_US + CGROUP_CPU_HEADROOM_QUOTA_US
: limits.cpuQuotaMilli * CGROUP_V2_PERIOD_US / 1000;
return {
memoryMax: String(memoryMaxBytes),
cpuMax: `${cpuQuotaUs} ${CGROUP_V2_PERIOD_US}`,
pidsMax: String(CGROUP_MAX_PIDS),
};
}
/**
* Places one Cloud Hypervisor run under an explicit memory/CPU/PID cgroup,
* created before launch and assigned by PID immediately after spawn (moving
* a PID into `cgroup.procs` requires only host-root write access to that
* file, not any privilege from the moved process itself).
*
* Cgroup v2 only: `runCloudHypervisorPreflight` rejects cgroup v1-only
* hosts explicitly rather than falling back to a v1 hierarchy this class
* does not manage (v1's memory/cpu/pids controllers live under separate
* per-controller mount points, not a single directory).
*/
export class CloudHypervisorCgroup {
private created = false;
constructor(
readonly cgroupPath: string,
private readonly limits: CloudHypervisorResourceLimits,
private readonly dependencies: CloudHypervisorCgroupDependencies = defaultCgroupDependencies,
) {}
async setup(): Promise<void> {
// cgroup v2 only materializes a controller's interface files
// (memory.max, cpu.max, pids.max, ...) in a directory once that
// controller is enabled in the *parent's* `cgroup.subtree_control`.
// That delegation has to happen at every level from the cgroup root
// down to (but excluding) the leaf we actually place limits on.
const parentDir = path.dirname(this.cgroupPath);
const rootDir = path.dirname(parentDir);
await this.enableControllers(rootDir);
await this.dependencies.mkdir(parentDir);
await this.enableControllers(parentDir);
await this.dependencies.mkdir(this.cgroupPath);
this.created = true;
const expected = this.expectedLimits();
await this.dependencies.writeFile(path.join(this.cgroupPath, 'memory.max'), expected.memoryMax);
await this.dependencies.writeFile(path.join(this.cgroupPath, 'cpu.max'), expected.cpuMax);
await this.dependencies.writeFile(path.join(this.cgroupPath, 'pids.max'), expected.pidsMax);
}
expectedLimits(): CloudHypervisorCgroupLimits {
return computeCloudHypervisorCgroupLimits(this.limits);
}
async assign(pid: number): Promise<void> {
if (!Number.isInteger(pid) || pid <= 0) {
throw new Error(`Cannot assign an invalid PID to the Cloud Hypervisor cgroup: ${pid}`);
}
await this.dependencies.writeFile(path.join(this.cgroupPath, 'cgroup.procs'), String(pid));
}
async cleanup(): Promise<void> {
if (!this.created) return;
const deadline = Date.now() + CGROUP_REMOVAL_MAX_WAIT_MS;
for (;;) {
try {
await this.dependencies.rmdir(this.cgroupPath);
this.created = false;
return;
} catch (error) {
const code = (error as NodeJS.ErrnoException | undefined)?.code;
// Retry both EBUSY and ENOTEMPTY: different kernel/cgroup-v2
// versions have been observed to report either errno for this
// same "a process only just exited, controller teardown hasn't
// fully settled yet" race.
if ((code !== 'EBUSY' && code !== 'ENOTEMPTY') || Date.now() >= deadline) throw error;
await this.dependencies.sleep(CGROUP_REMOVAL_RETRY_INTERVAL_MS);
}
}
}
private async enableControllers(directory: string): Promise<void> {
await this.dependencies.writeFile(
path.join(directory, 'cgroup.subtree_control'),
CGROUP_V2_CONTROLLERS,
);
}
}
function assertSafeNamespaceName(value: string): void {
if (!/^[A-Za-z0-9_.-]+$/.test(value)) {
throw new Error(`Unsafe Cloud Hypervisor network namespace name: ${value}`);
}
}
function assertPositiveIdentity(value: number, label: string): void {
if (!Number.isSafeInteger(value) || value <= 0) {
throw new Error(`Cloud Hypervisor launch ${label} must be a positive integer`);
}
}