microvm: drop the kata-config asset (#1704)

Today, `ateom` fetches the `kata-config` asset (configuration-clh.toml)
to read only 3 values `default_memory`, `default_vcpus` and
`kernel_params`. The first 2 are redundant because (1) the values never
change and (2) ateom has the same defaults. Only `kernel_params` is
relevant for the `kata-agent`; its value also changed once in the Kata
project history.

`ateom` now owns all three values, which also allows us to tune them
specifically for Substrate.

Verified e2e with a GKE cluster.

Fixes #1693

> It's a good idea to open an issue first for discussion.

- [x] Tests pass
- [x] Appropriate changes to documentation are included in the PR
This commit is contained in:
Huy Pham
2026-09-22 00:18:59 +00:00
committed by GitHub
parent dc1f263076
commit 8062bbafb4
26 changed files with 83 additions and 215 deletions
@@ -38,9 +38,9 @@ const agentVsockPort = 1024
const debugConsoleVsockPort = 1026
// DebugConsoleDump connects to the guest's kata debug console (vsock 1026) and
// runs cmd, returning its combined output. Diagnostic only (requires
// debug_console_enabled=true in the kata config). Best-effort: returns the error
// text on failure rather than failing the caller.
// runs cmd, returning its combined output. Diagnostic only (requires the guest to
// have booted with the WithDebugConsole kernel params). Best-effort: returns the
// error text on failure rather than failing the caller.
func DebugConsoleDump(ctx context.Context, vsockPath, cmd string) string {
d := net.Dialer{}
dctx, cancel := context.WithTimeout(ctx, 8*time.Second)
+21 -61
View File
@@ -15,72 +15,32 @@
package kata
import (
"fmt"
"strings"
toml "github.com/pelletier/go-toml/v2"
)
// KataConfig holds the values ateom reads from a kata configuration.toml. ateom
// owns the cloud-hypervisor boot and points it at the runtime-fetched asset paths
// directly, so the only things it needs from the config are the guest sizing and
// the agent kernel command line.
type KataConfig struct {
// MemoryMiB is the guest RAM size ([hypervisor.clh] default_memory).
MemoryMiB int
// VCPUs is the guest vCPU count ([hypervisor.clh] default_vcpus).
VCPUs int
// KernelParams is the guest kernel command line ([hypervisor.clh]
// kernel_params): the kata-agent parameters (agent.log, the systemd target,
// etc.). ateom appends these to the cloud-hypervisor payload cmdline, since
const (
// TODO(#1724): Tune the following values for Substrate actors.
// DefaultMemoryMiB is the default guest memory size (MiB).
DefaultMemoryMiB = 2048
// DefaultVCPUs is the default guest vCPU count.
DefaultVCPUs = 1
)
const (
// baseKernelParams is the guest kernel command line parameters ateom boots with;
// there is no kata shim to inject them.
KernelParams string
}
baseKernelParams = "cgroup_no_v1=all systemd.unified_cgroup_hierarchy=1"
debugConsoleKernelParams = baseKernelParams + " agent.debug_console agent.debug_console_vport=1026"
)
// clhConfigTOML mirrors the subset of a kata configuration.toml ateom reads.
// Unmarshalling ignores every other key, so it stays valid across kata releases.
type clhConfigTOML struct {
Hypervisor struct {
CLH struct {
DefaultMemory int `toml:"default_memory"`
DefaultVCPUs int `toml:"default_vcpus"`
KernelParams string `toml:"kernel_params"`
} `toml:"clh"`
} `toml:"hypervisor"`
}
// ParseConfig reads the guest sizing and kernel_params from a kata
// configuration.toml. memDefault/vcpuDefault are substituted when the key is
// absent or non-positive (kata also accepts default_vcpus = -1 meaning "all host
// CPUs", which ateom does not support).
func ParseConfig(base []byte, memDefault, vcpuDefault int) (KataConfig, error) {
var c clhConfigTOML
if err := toml.Unmarshal(base, &c); err != nil {
return KataConfig{}, fmt.Errorf("parsing kata config: %w", err)
}
cfg := KataConfig{
MemoryMiB: c.Hypervisor.CLH.DefaultMemory,
VCPUs: c.Hypervisor.CLH.DefaultVCPUs,
KernelParams: c.Hypervisor.CLH.KernelParams,
}
if cfg.MemoryMiB <= 0 {
cfg.MemoryMiB = memDefault
}
if cfg.VCPUs <= 0 {
cfg.VCPUs = vcpuDefault
}
return cfg, nil
}
// WithDebugConsole appends the kata-agent debug-console kernel parameters so the
// guest agent binds a root debug shell on vsock port 1026, which DebugConsoleDump
// connects to for in-guest diagnostics. Both params are required: agent.debug_console
// enables the console and agent.debug_console_vport=1026 makes the agent bind it on
// the vsock port (the agent only binds a vsock listener when the vport is > 0).
// Idempotent.
func WithDebugConsole(kernelParams string) string {
return appendKernelParams(kernelParams, "agent.debug_console_vport",
"agent.debug_console agent.debug_console_vport=1026")
// WithDebugConsole returns the kernel params with the kata-agent debug console
// enabled, so the guest agent binds a root debug shell on vsock port 1026, which
// DebugConsoleDump connects to for in-guest diagnostics. Both params are required:
// agent.debug_console enables the console and agent.debug_console_vport=1026 makes
// the agent bind it on the vsock port (the agent only binds a vsock listener when
// the vport is > 0).
func WithDebugConsole() string {
return debugConsoleKernelParams
}
// WithAgentDebug appends agent.log=debug so the guest kata-agent emits
@@ -19,71 +19,6 @@ import (
"testing"
)
// stockConfig mirrors the [hypervisor.clh] keys ateom reads from a kata
// configuration.toml (every other key is ignored by ParseConfig).
const stockConfig = `[hypervisor.clh]
path = "/usr/local/bin/cloud-hypervisor"
kernel = "/opt/kata/share/kata-containers/vmlinux.container"
image = "/opt/kata/share/kata-containers/kata-containers.img"
default_memory = 512
default_vcpus = 2
kernel_params = "agent.foo=bar systemd.unit=kata-containers.target"
shared_fs = "virtio-fs"
`
func TestParseConfig(t *testing.T) {
cfg, err := ParseConfig([]byte(stockConfig), 2048, 1)
if err != nil {
t.Fatalf("ParseConfig: %v", err)
}
if cfg.MemoryMiB != 512 {
t.Errorf("MemoryMiB = %d, want 512", cfg.MemoryMiB)
}
if cfg.VCPUs != 2 {
t.Errorf("VCPUs = %d, want 2", cfg.VCPUs)
}
if want := "agent.foo=bar systemd.unit=kata-containers.target"; cfg.KernelParams != want {
t.Errorf("KernelParams = %q, want %q", cfg.KernelParams, want)
}
}
// TestParseConfigDefaults asserts the mem/vcpu defaults kick in when the keys are
// absent or non-positive (kata also accepts default_vcpus = -1 meaning "all host
// CPUs", which ateom does not support).
func TestParseConfigDefaults(t *testing.T) {
for _, tc := range []struct {
name string
toml string
}{
{"absent", "[hypervisor.clh]\nkernel_params = \"x\"\n"},
{"nonpositive", "[hypervisor.clh]\ndefault_memory = 0\ndefault_vcpus = -1\n"},
} {
t.Run(tc.name, func(t *testing.T) {
cfg, err := ParseConfig([]byte(tc.toml), 2048, 1)
if err != nil {
t.Fatalf("ParseConfig: %v", err)
}
if cfg.MemoryMiB != 2048 {
t.Errorf("MemoryMiB = %d, want default 2048", cfg.MemoryMiB)
}
if cfg.VCPUs != 1 {
t.Errorf("VCPUs = %d, want default 1", cfg.VCPUs)
}
})
}
}
func TestWithDebugConsole(t *testing.T) {
got := WithDebugConsole("root=/dev/vda1")
if !strings.Contains(got, "agent.debug_console") || !strings.Contains(got, "agent.debug_console_vport=1026") {
t.Errorf("WithDebugConsole did not append the debug-console params: %q", got)
}
// Idempotent: a second call must not append the params again.
if again := WithDebugConsole(got); again != got {
t.Errorf("WithDebugConsole not idempotent:\n first = %q\nsecond = %q", got, again)
}
}
func TestWithAgentDebug(t *testing.T) {
got := WithAgentDebug("root=/dev/vda1")
if !strings.Contains(got, "agent.log=debug") {
+2 -3
View File
@@ -18,9 +18,8 @@
// ttrpc API (DialAgent / AgentClient) to create the sandbox and run each container
// on its host-merged rootfs (overlay_linux.go).
//
// It also renders the kata configuration.toml (for the agent kernel_params + guest
// sizing) from runtime-fetched assets (config.go) and sweeps leftover per-sandbox
// host-side state (CleanupSandboxState).
// It also owns the guest kernel_params and default guest sizing (config.go) and
// sweeps leftover per-sandbox host-side state (CleanupSandboxState).
package kata
import (
+5 -8
View File
@@ -63,7 +63,6 @@ import (
var (
podUID = flag.String("pod-uid", "", "The UID of the current pod")
chBinary = flag.String("cloud-hypervisor-binary", "cloud-hypervisor", "Path to the cloud-hypervisor binary (used to relaunch on restore).")
kataConfig = flag.String("kata-config", "", "Path to a kata configuration.toml (passed to the shim as KATA_CONF_FILE). Empty uses kata's default. atelet generates one pointing at runtime-fetched assets.")
kataDebug = flag.Bool("kata-debug", false, "Verbose kata-agent debugging: raise the guest agent log level and forward the guest console (incl. agent logs) into the pod logs.")
vmmMemReserve = flag.Int("vmm-mem-reserve-mib", vmmMemReserveMiB, "Guest RAM (MiB) held back from the pod's memory limit for the cloud-hypervisor VMM + virtiofsd, which run as host processes in the pod cgroup alongside the guest RAM. Prevents the pod OOMing when the VM is sized to the pod's memory limit.")
showVersion = flag.Bool("version", false, "Print version and exit.")
@@ -266,7 +265,7 @@ func do(ctx context.Context) error {
}()
slog.InfoContext(ctx, "atunnel egress serving", slog.String("address", *atunnelEgressListenAddress))
ateomService := NewService(*podUID, *chBinary, *kataConfig, *kataDebug, *vmmMemReserve, interiorNetNS, actorLogger, atunnelIngress, atunnelEgress, atunnelEgressPort, *workerCredentialBundle, *podIdentityTrustBundle, *egressGatewayTrustBundle)
ateomService := NewService(*podUID, *chBinary, *kataDebug, *vmmMemReserve, interiorNetNS, actorLogger, atunnelIngress, atunnelEgress, atunnelEgressPort, *workerCredentialBundle, *podIdentityTrustBundle, *egressGatewayTrustBundle)
svr := grpc.NewServer(
grpc.StatsHandler(otelgrpc.NewServerHandler()),
@@ -418,10 +417,9 @@ type AteomService struct {
activeRPCMu sync.Mutex
activeRPC *activeRPCInfo
podUID string
chBinary string
kataConfig string
kataDebug bool
podUID string
chBinary string
kataDebug bool
// memReserveMiB is guest RAM (MiB) held back from the pod's memory limit for
// the cloud-hypervisor VMM + virtiofsd (host processes sharing the pod cgroup
@@ -502,12 +500,11 @@ type AteomService struct {
var _ ateompb.AteomServer = (*AteomService)(nil)
// NewService creates a new AteomService.
func NewService(podUID, chBinary, kataConfig string, kataDebug bool, memReserveMiB int, interiorNetNS netns.NsHandle, actorLogger *actorlog.ActorLogger, atunnelIngress *atunnel.Server, atunnelEgress *atunnel.Egress, atunnelEgressPort uint16, workerCredentialBundlePath, podIdentityTrustBundlePath, egressGatewayTrustBundlePath string) *AteomService {
func NewService(podUID, chBinary string, kataDebug bool, memReserveMiB int, interiorNetNS netns.NsHandle, actorLogger *actorlog.ActorLogger, atunnelIngress *atunnel.Server, atunnelEgress *atunnel.Egress, atunnelEgressPort uint16, workerCredentialBundlePath, podIdentityTrustBundlePath, egressGatewayTrustBundlePath string) *AteomService {
return &AteomService{
lock: newCancelableMutex(),
podUID: podUID,
chBinary: chBinary,
kataConfig: kataConfig,
kataDebug: kataDebug,
memReserveMiB: memReserveMiB,
interiorNetNS: interiorNetNS,
+21 -35
View File
@@ -113,7 +113,6 @@ const (
assetCH = "cloud-hypervisor"
assetKernel = "kata-kernel"
assetImage = "kata-image"
assetConfig = "kata-config"
assetVirtiofsd = "virtiofsd"
)
@@ -185,12 +184,11 @@ type actorContainer struct {
imageMounts []*ateompb.ImageVolumeMount
}
// resolvedRuntime holds the concrete binary/config paths for a request, taken
// from fetched runtime assets when present, else the process flags.
// resolvedRuntime holds the concrete binary paths for a request, taken from fetched
// runtime assets when present, else the process flags.
type resolvedRuntime struct {
chBinary string // path to the cloud-hypervisor binary
configFile string // path to the kata configuration.toml
virtiofsd string // path to virtiofsd (overlay RO lower); "" => "virtiofsd" on PATH
chBinary string // path to the cloud-hypervisor binary
virtiofsd string // path to virtiofsd (overlay RO lower); "" => "virtiofsd" on PATH
}
// firstNonEmpty returns the first non-empty string, or "" if all are empty.
@@ -203,13 +201,12 @@ func firstNonEmpty(vals ...string) string {
return ""
}
// resolveRuntime resolves the cloud-hypervisor binary + the kata config path from
// fetched assets, falling back to flags.
// resolveRuntime resolves the cloud-hypervisor binary from fetched assets, falling
// back to the flag.
func (s *AteomService) resolveRuntime(paths map[string]string) resolvedRuntime {
return resolvedRuntime{
chBinary: firstNonEmpty(paths[assetCH], s.chBinary),
configFile: firstNonEmpty(paths[assetConfig], s.kataConfig),
virtiofsd: paths[assetVirtiofsd],
chBinary: firstNonEmpty(paths[assetCH], s.chBinary),
virtiofsd: paths[assetVirtiofsd],
}
}
@@ -264,8 +261,8 @@ func writeGuestResolvConf(rootfs string) error {
// start each container.
//
// Contract with atelet:
// - The runtime assets (guest kernel, guest OS image, cloud-hypervisor, virtiofsd,
// base kata config) are on disk and passed as runtime asset paths.
// - The runtime assets (guest kernel, guest OS image, cloud-hypervisor, virtiofsd)
// are on disk and passed as runtime asset paths.
// - The OCI bundle (config.json + populated rootfs/) is prepared per container.
func (s *AteomService) RunWorkload(ctx context.Context, req *ateompb.RunWorkloadRequest) (resp *ateompb.RunWorkloadResponse, retErr error) {
s.lock.Lock()
@@ -437,14 +434,11 @@ func (s *AteomService) coldBootActor(ctx context.Context, p actorBootParams) (re
}
}()
// Guest sizing + agent kernel params from the kata config.
memMiB, vcpus, kparams, err := s.guestConfig(rr)
if err != nil {
return err
}
// Guest sizing + agent kernel params.
memMiB, vcpus, kparams := s.guestConfig()
// Right-size the VM to the actor's declared limits (see internal/sizing),
// keeping the kata-config values above as the fallback when a limit is unset.
// keeping the defaults above as the fallback when a limit is unset.
// vCPUs round up; VM RAM reserves a fixed margin for the VMM + virtiofsd, which
// share the pod cgroup with the guest RAM. A declared memory limit the reserve
// leaves too small to boot is rejected (resolveGuestMemMiB) rather than silently
@@ -729,28 +723,20 @@ func (s *AteomService) stageMergedRootfs(ctx context.Context, rr resolvedRuntime
return vfsdCmd, nil
}
// guestConfig reads guest sizing + agent kernel params from the resolved kata
// config, enabling the debug console (vsock 1026) for in-guest diagnostics and,
// with kataDebug, raising the agent log level.
func (s *AteomService) guestConfig(rr resolvedRuntime) (memMiB, vcpus int, kparams string, err error) {
var cfgBytes []byte
if rr.configFile != "" {
cfgBytes, _ = os.ReadFile(rr.configFile)
}
cfg, err := kata.ParseConfig(cfgBytes, 2048, 1)
if err != nil {
return 0, 0, "", fmt.Errorf("while parsing kata config: %w", err)
}
kparams = kata.WithDebugConsole(cfg.KernelParams)
// guestConfig returns the default guest sizing and the agent kernel params, enabling
// the debug console (vsock 1026) for in-guest diagnostics and, with kataDebug, raising
// the agent log level.
func (s *AteomService) guestConfig() (memMiB, vcpus int, kparams string) {
kparams = kata.WithDebugConsole()
if s.kataDebug {
kparams = kata.WithAgentDebug(kparams)
}
return cfg.MemoryMiB, cfg.VCPUs, kparams, nil
return kata.DefaultMemoryMiB, kata.DefaultVCPUs, kparams
}
// resolveGuestMemMiB returns the micro-VM guest RAM (MiB) for an actor's declared
// memory limit. declaredBytes == 0 means "unset" and returns fallbackMiB (the
// kata-config default). Otherwise the guest gets the declared memory minus the VMM
// memory limit. declaredBytes == 0 means "unset" and returns fallbackMiB
// (kata.DefaultMemoryMiB). Otherwise the guest gets the declared memory minus the VMM
// reserve; if that leaves less than a bootable minimum it errors — naming the limit,
// the reserve, and the minimum — instead of silently reverting to the (larger)
// fallback, which would boot the actor bigger than the worker was sized for and OOM
+1 -1
View File
@@ -171,7 +171,7 @@ func TestResolveGuestMemMiB(t *testing.T) {
const (
mib = 1024 * 1024
reserve = 256 // vmmMemReserveMiB
fallback = 2048 // kata-config default
fallback = 2048 // kata.DefaultMemoryMiB
)
tests := []struct {
name string
@@ -65,7 +65,7 @@ volumes:
- trustBundle:
name: egress-mitm.ate.dev
path: trust-bundle.pem
# Sandbox size: without these the guest boots at the kata config's default
# Sandbox size: without these the guest boots at ateom's default size
# (2GiB). ateom applies them to the VM (see internal/sizing). Quantities are
# strings in the proto.
resources:
@@ -38,7 +38,7 @@ containers:
# See ../counter/counter-microvm-template.yaml.tmpl: a micro-VM guest needs
# more than readyz's 30s default once CI's single kind node is under load.
timeoutSeconds: 120
# Sandbox size: without these the guest boots at the kata config's default
# Sandbox size: without these the guest boots at ateom's default size
# (2GiB). ateom applies them to the VM (see internal/sizing). Quantities are
# strings in the proto.
resources:
+2 -2
View File
@@ -135,7 +135,7 @@ Unlike a Pod, an actor is sized by its **`limits`** (CPU and Memory): the size i
- **gVisor (`ateom-gvisor`)** — applied to the container OCI spec: `limits.cpu` sets the cgroup v2 CPU quota (`cpu.max`) and the Sentry vCPU count (`--cpu-num-from-quota`); `limits.memory` sets the cgroup v2 memory limit (`memory.max`) and bounds the virtual total memory the sandbox reports (so JVM/Go do not over-allocate from host RAM).
- **Micro-VM (`ateom-microvm`)** — `limits.cpu` sets Cloud Hypervisor `BootVcpus` / `MaxVcpus` (rounded up to whole vCPUs); `limits.memory` sets guest RAM, reserving a small configurable margin (default 256 MiB, `--vmm-mem-reserve-mib`) for the VMM and virtiofsd so the pod cgroup does not OOM.
2. **Gate scheduling.** An actor is only placed on a `WorkerPool` whose [worker capacity](#worker-capacity-spectemplateresources) is `>=` these limits.
3. **Fall back to runtime defaults.** A zero or absent limit leaves that dimension at the runtime default — unlimited for gVisor, the kata config for the micro-VM.
3. **Fall back to runtime defaults.** A zero or absent limit leaves that dimension at the runtime default: unlimited for gVisor, and 2 GiB / 1 vCPU for the micro-VM.
`requests` are not consulted today (an actor occupies its whole worker). Because the size is baked into snapshots, a **micro-VM FULL-scope restore reuses the size in the snapshot**; changing an actor's limits takes effect on its next cold boot.
@@ -424,7 +424,7 @@ spec:
### Micro-VM SandboxConfig
A `microvm` `SandboxConfig` supplies the [Kata Containers](https://katacontainers.io/) + [Cloud Hypervisor](https://www.cloudhypervisor.org/) toolchain instead of `runsc`. Each architecture must define the full asset set — `kata-shim`, `cloud-hypervisor`, `virtiofsd`, `kata-kernel`, `kata-image`, and `kata-config` — which a `ValidatingAdmissionPolicy` enforces at apply time. Worker pods for a micro-VM pool require `/dev/kvm` and nested-virtualization-capable nodes. The controller requests those devices on the pod automatically, and atelet advertises them only where they exist, so placement follows the hardware rather than a node label. Clusters that reserve nested-virt nodes with an `ate.dev/sandboxClass=microvm` taint are still tolerated: advertising a device attracts these pods to capable nodes but repels nothing else from them.
A `microvm` `SandboxConfig` supplies the [Kata Containers](https://katacontainers.io/) + [Cloud Hypervisor](https://www.cloudhypervisor.org/) toolchain instead of `runsc`. Each architecture must define the full asset set — `cloud-hypervisor`, `virtiofsd`, `kata-kernel`, and `kata-image` — which a `ValidatingAdmissionPolicy` enforces at apply time. Worker pods for a micro-VM pool require `/dev/kvm` and nested-virtualization-capable nodes. The controller requests those devices on the pod automatically, and atelet advertises them only where they exist, so placement follows the hardware rather than a node label. Clusters that reserve nested-virt nodes with an `ate.dev/sandboxClass=microvm` taint are still tolerated: advertising a device attracts these pods to capable nodes but repels nothing else from them.
See [`hack/microvm-assets/`](../hack/microvm-assets/) for scripts that assemble and stage these assets, plus a worked counter demo (`demos/counter/counter-microvm.yaml.tmpl`) that suspends and resumes an in-RAM counter across worker pods.
+1 -1
View File
@@ -36,7 +36,6 @@ require (
github.com/openfga/api/proto v0.0.0-20260319214821-f153694bfc20
github.com/openfga/language/pkg/go v0.3.2-0.20260730144454-83fedf8a4e70
github.com/openfga/openfga v1.20.0
github.com/pelletier/go-toml/v2 v2.4.0
github.com/pressly/goose/v3 v3.27.3
github.com/prometheus/client_golang v1.24.1
github.com/spf13/cobra v1.10.2
@@ -203,6 +202,7 @@ require (
github.com/olekukonko/tablewriter v1.0.8 // indirect
github.com/opencontainers/go-digest v1.0.0 // indirect
github.com/opencontainers/image-spec v1.1.1 // indirect
github.com/pelletier/go-toml/v2 v2.4.0 // indirect
github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 // indirect
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect
github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55 // indirect
+4 -4
View File
@@ -142,13 +142,13 @@ OUT="${OUT:-${ROOT}/bin/microvm-assets/$ARCH}"
# --- 1. assets: assemble (if missing or stale) -----------------------------
need_assemble=false
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml; do
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img; do
if [[ ! -f "${OUT}/${f}" ]]; then
need_assemble=true
break
fi
done
# Presence alone is not enough. The five filenames don't change when a version pin
# Presence alone is not enough. The filenames don't change when a version pin
# moves, so an asset dir assembled before a bump looks complete while holding the old
# bytes — we'd then stage those against a SandboxConfig pinning the new shas, and the
# mismatch would only surface at runtime as an actor wedged in STATUS_RESUMING while
@@ -170,7 +170,7 @@ else
fi
# --- 2. stage assets to rustfs (kind) / GCS (GKE) --------------------------
# Upload the five assets under kata-assets/, where atelet fetches them: the
# Upload the four assets under kata-assets/, where atelet fetches them: the
# in-cluster rustfs (S3 API) on kind, or the GCS bucket on GKE.
if [[ "${ATE_INSTALL_KIND}" == "true" ]]; then
log "Staging assets to in-cluster rustfs bucket ${BUCKET_NAME} (kata-assets/)..."
@@ -185,7 +185,7 @@ fi
# its binary bytes are not reproducible across toolchains and its sha can't
# be a fixed pin in the manifest. Compute it from the freshly-staged binary
# and inject it, so the deployed SandboxConfig always matches whatever was
# staged. The downloaded assets (cloud-hypervisor/kernel/rootfs/config, plus
# staged. The downloaded assets (cloud-hypervisor/kernel/rootfs, plus
# virtiofsd on amd64 where upstream publishes a prebuilt) keep their
# committed, reproducible per-arch shas.
log "Applying microvm SandboxConfig from ${MANIFEST_TEMPLATE}..."
+1 -2
View File
@@ -5,13 +5,12 @@ toolchain at runtime — nothing kata-specific is baked into the worker image. a
the kata-agent directly (no kata shim, no containerd). Each actor container's rootfs is an
overlay of a read-only lower (the OCI image, served into the guest over virtio-fs by
`virtiofsd`) and a writable upper on a guest tmpfs, so `virtiofsd` is part of the asset
set. The asset set is five files:
set. The asset set is four files:
- `cloud-hypervisor` — the VMM binary (fetched from its release)
- `virtiofsd` — the virtio-fs daemon serving the RO lower (built from source; see `assemble.sh`)
- `vmlinux` — the guest kernel (from kata-static)
- `rootfs.img` — the guest rootfs image (from kata-static)
- `configuration-clh.toml` — the base kata config (from kata-static)
These helpers assemble the asset set for your node arch, stage it into the cluster's rustfs
S3 bucket, and the demo manifest's `SandboxConfig` points at it. When `/dev/kvm` is
+6 -7
View File
@@ -18,8 +18,8 @@
# ateom-microvm fetches at runtime (fetch-not-bake). Run this on a Linux
# host of the TARGET arch.
#
# Produces, under $OUT, the five assets named as the SandboxConfig expects:
# cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml
# Produces, under $OUT, the four assets named as the SandboxConfig expects:
# cloud-hypervisor virtiofsd vmlinux rootfs.img
# The DOWNLOADED assets are reproducible, so paste their sha256 sums into the
# manifest (demos/counter/counter-microvm.yaml.tmpl). That now includes virtiofsd on
# amd64 (upstream prebuilt); on arm64 virtiofsd is still built from source
@@ -64,7 +64,7 @@ esac
# Identifies the asset set this script produces. Cleared before the first write into
# $OUT and re-written to $OUT/$STAMP_FILE on success; install-microvm-deps.sh compares
# it against what the current checkout would build, because the five filenames stay the
# it against what the current checkout would build, because the filenames stay the
# same when a pin moves and an asset dir from an older checkout is otherwise
# indistinguishable from a current one.
STAMP_FILE=".asset-versions"
@@ -98,7 +98,6 @@ KROOT="kata/opt/kata"
cp "$(readlink -f "${KROOT}/share/kata-containers/vmlinux.container")" "${OUT}/vmlinux"
cp "$(readlink -f "${KROOT}/share/kata-containers/kata-containers.img")" "${OUT}/rootfs.img"
cp "${KROOT}/share/defaults/kata-containers/configuration-clh.toml" "${OUT}/configuration-clh.toml"
echo ">> Downloading cloud-hypervisor ${CH_VER} (${CH_ASSET})..."
curl -fSL -o "${OUT}/cloud-hypervisor" \
@@ -145,10 +144,10 @@ chmod +x "${OUT}/virtiofsd"
echo
echo ">> Assets assembled in ${OUT}:"
cd "${OUT}"
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml; do
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img; do
[ -f "$f" ] || { echo "MISSING: $f" >&2; exit 1; }
done
# Written only once all five are present, and only after the up-front rm, so the stamp
# Written only once all four are present, and only after the up-front rm, so the stamp
# exists exactly when this dir was assembled end-to-end by these pins.
asset_stamp > "${OUT}/${STAMP_FILE}"
"${OUT}/virtiofsd" --version 2>/dev/null | head -1 || true
@@ -156,4 +155,4 @@ echo
echo ">> sha256 (paste the DOWNLOADED assets into counter-microvm.yaml.tmpl; that"
echo ">> includes virtiofsd on amd64 (prebuilt). The arm64 virtiofsd is built from"
echo ">> source, so its sha is injected at deploy by run-microvm-demo.sh, not pinned):"
sha256sum cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml
sha256sum cloud-hypervisor virtiofsd vmlinux rootfs.img
+1 -1
View File
@@ -33,7 +33,7 @@ BUCKET="${BUCKET:-ate-snapshots}"
# gcloud uses its active config project. ${PROJECT_ID:+...} elides the flag entirely
# when unset (same idiom as KUBECTL_CONTEXT in hack/run-microvm-demo.sh).
echo ">> Uploading assets to gs://${BUCKET}/kata-assets/ ..."
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml; do
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img; do
echo " $f"
gcloud storage cp ${PROJECT_ID:+--project="${PROJECT_ID}"} "${OUT}/${f}" "gs://${BUCKET}/kata-assets/${f}"
done
+1 -1
View File
@@ -48,7 +48,7 @@ KIND_CLUSTER_NAME="${KIND_CLUSTER_NAME:-kind}"
# manifests/ate-install/kind/rustfs.yaml, which creates the bucket we upload into.
AWS_CLI_IMAGE="amazon/aws-cli:2.17.0@sha256:643507c10ada7964ca6157b3d799f030b90577643da9955d319a77399ed80d73"
ASSETS=(cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml)
ASSETS=(cloud-hypervisor virtiofsd vmlinux rootfs.img)
run_kubectl() {
kubectl ${KUBECTL_CONTEXT:+--context="${KUBECTL_CONTEXT}"} -n "${NAMESPACE}" "$@"
+1 -1
View File
@@ -60,7 +60,7 @@ func substrateTemplateSubstitutions(bucket, name string, trustBundle bool) (inli
// a missing or stale one fails loudly at template creation.
blocks["${TEMPLATE_SANDBOX_CONFIG}"] = "sandboxConfig:\n sandboxClass: SANDBOX_CLASS_MICROVM\n configName: microvm"
// Only for fixtures that declare no limits of their own. Without them the
// guest boots at the kata config's default (2GiB), and several of those
// guest boots at ateom's default size (2GiB), and several of those
// do not fit beside the demo pools on CI's single kind node. These size
// the VM itself — see internal/sizing. Quantities are strings.
blocks["${TEMPLATE_RESOURCES}"] = "resources:\n limits:\n - name: cpu\n quantity: \"1\"\n - name: memory\n quantity: 512Mi"
+1 -1
View File
@@ -203,7 +203,7 @@ func fixtureSubstitutions(bucket, name string) (inline, blocks map[string]string
// classes, so only same-class pools are eligible to run these actors.
blocks["${TEMPLATE_SANDBOX_CLASS}"] = " sandboxClass: microvm"
// Only for fixtures that declare no limits of their own. Without them the
// guest boots at the kata config's default (2GiB), and several of those do
// guest boots at ateom's default size (2GiB), and several of those do
// not fit beside the demo pools on CI's single kind node. These size the VM
// itself — see internal/sizing.
blocks["${TEMPLATE_RESOURCES}"] = " resources:\n limits:\n cpu: \"1\"\n memory: 512Mi"
+1 -1
View File
@@ -195,7 +195,7 @@ func TestRenderSubstrateFixtures_MicroVM(t *testing.T) {
if got := tmpl.GetSandboxConfig().GetConfigName(); got != "microvm" {
t.Errorf("template %s configName = %q, want microvm", name, got)
}
// Undeclared limits boot the guest at the kata config default
// Undeclared limits boot the guest at ateom's default size
// (2GiB), which does not fit beside the demo pools on one kind
// node.
if memoryLimit(tmpl) == "" {
+1 -1
View File
@@ -109,7 +109,7 @@ func CreateSubstrateTemplateFrom(ctx context.Context, t *testing.T, clients *Cli
Containers: srcTmpl.GetContainers(),
// The source's limits size the sandbox. Copying them matters most on
// micro-VM, where an ActorTemplate that declares none boots the guest
// at the kata config default (2GiB) instead of the demo's 512Mi.
// at ateom's default guest size (2GiB) instead of the demo's 512Mi.
Resources: srcTmpl.GetResources(),
// The source carries the sandbox_class/config_name pair for the
// class under test.
+2 -2
View File
@@ -350,7 +350,7 @@ type RunWorkloadRequest struct {
RunscPath string `protobuf:"bytes,6,opt,name=runsc_path,json=runscPath,proto3" json:"runsc_path,omitempty"`
Spec *WorkloadSpec `protobuf:"bytes,7,opt,name=spec,proto3" json:"spec,omitempty"`
// runtime_asset_paths maps a runtime asset name (e.g. "cloud-hypervisor",
// "virtiofsd", "kata-kernel", "kata-image", "kata-config")
// "virtiofsd", "kata-kernel", "kata-image")
// to the local on-disk path atelet fetched it to (content-addressed, like
// runsc_path). Empty for the gVisor runtime, which uses runsc_path.
RuntimeAssetPaths map[string]string `protobuf:"bytes,8,rep,name=runtime_asset_paths,json=runtimeAssetPaths,proto3" json:"runtime_asset_paths,omitempty" protobuf_key:"bytes,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"`
@@ -359,7 +359,7 @@ type RunWorkloadRequest struct {
// The actor's declared size, from the ActorTemplate's resource limits. ateom
// sizes the sandbox to these (cgroup caps via the OCI spec, and for the
// micro-VM the VM's vCPU count and memory). Zero means "unset": keep the
// runtime default (unlimited for gVisor, the kata config for the micro-VM).
// runtime default (unlimited for gVisor, ateom's own default for the micro-VM).
CpuMilli int64 `protobuf:"varint,11,opt,name=cpu_milli,json=cpuMilli,proto3" json:"cpu_milli,omitempty"` // CPU limit in millicores (1000 = one core).
MemoryBytes int64 `protobuf:"varint,12,opt,name=memory_bytes,json=memoryBytes,proto3" json:"memory_bytes,omitempty"` // Memory limit in bytes.
unknownFields protoimpl.UnknownFields
+2 -2
View File
@@ -131,7 +131,7 @@ message RunWorkloadRequest {
WorkloadSpec spec = 7;
// runtime_asset_paths maps a runtime asset name (e.g. "cloud-hypervisor",
// "virtiofsd", "kata-kernel", "kata-image", "kata-config")
// "virtiofsd", "kata-kernel", "kata-image")
// to the local on-disk path atelet fetched it to (content-addressed, like
// runsc_path). Empty for the gVisor runtime, which uses runsc_path.
map<string, string> runtime_asset_paths = 8;
@@ -142,7 +142,7 @@ message RunWorkloadRequest {
// The actor's declared size, from the ActorTemplate's resource limits. ateom
// sizes the sandbox to these (cgroup caps via the OCI spec, and for the
// micro-VM the VM's vCPU count and memory). Zero means "unset": keep the
// runtime default (unlimited for gVisor, the kata config for the micro-VM).
// runtime default (unlimited for gVisor, ateom's own default for the micro-VM).
int64 cpu_milli = 11; // CPU limit in millicores (1000 = one core).
int64 memory_bytes = 12; // Memory limit in bytes.
}
+1 -1
View File
@@ -34,7 +34,7 @@ const (
// SandboxSize is the sandbox's target size, derived from the actor's declared
// resource limits. A zero field means "unset": the caller keeps its own default
// (the kata config for the micro-VM, unlimited for gVisor).
// (kata.DefaultMemoryMiB / kata.DefaultVCPUs for the micro-VM, unlimited for gVisor).
type SandboxSize struct {
// MilliCPU is the CPU limit in millicores (1000 = one core), or 0 if unset.
MilliCPU int64
@@ -45,9 +45,9 @@ spec:
object.spec.sandboxClass != 'microvm' ||
(has(object.spec.assets) && size(object.spec.assets) > 0 &&
object.spec.assets.all(arch,
['cloud-hypervisor', 'virtiofsd', 'kata-kernel', 'kata-image', 'kata-config']
['cloud-hypervisor', 'virtiofsd', 'kata-kernel', 'kata-image']
.all(name, name in object.spec.assets[arch])))
message: "a microvm SandboxConfig must define cloud-hypervisor, virtiofsd, kata-kernel, kata-image, and kata-config assets for every architecture under spec.assets"
message: "a microvm SandboxConfig must define cloud-hypervisor, virtiofsd, kata-kernel, and kata-image assets for every architecture under spec.assets"
---
apiVersion: admissionregistration.k8s.io/v1
kind: ValidatingAdmissionPolicyBinding
@@ -64,9 +64,6 @@ spec:
kata-image:
url: "gs://${BUCKET_NAME}/kata-assets/rootfs.img"
sha256: "24d77700749846355f3be25b87974c1fdea9fd2a9534256b4e4bbdde5df23588"
kata-config:
url: "gs://${BUCKET_NAME}/kata-assets/configuration-clh.toml"
sha256: "20f5de4f705423ddd1ee93318c784c6aede30c07c78dc6b311a771cc2f6859c7"
# amd64 assets are kata 4.0.0 + virtiofsd v1.14.0 (assemble.sh ARCH=amd64).
amd64:
cloud-hypervisor:
@@ -84,6 +81,3 @@ spec:
kata-image:
url: "gs://${BUCKET_NAME}/kata-assets/rootfs.img"
sha256: "fbbd40de605ab2f99dc82c777dc5b41f9a00dad3a0f699f2247c306ac9c1204d"
kata-config:
url: "gs://${BUCKET_NAME}/kata-assets/configuration-clh.toml"
sha256: "1e67eabdfb9d900bf95a00edb52cacc1efe64f23ea095f8b49ee81c7434bce90"
@@ -61,7 +61,7 @@ func gvisorAsset() AssetFile {
}
// microVMAssets returns a full, valid micro-VM asset set for one architecture:
// the five assets the policy requires. The overlay rootfs serves the OCI image
// the four assets the policy requires. The overlay rootfs serves the OCI image
// over virtio-fs, so virtiofsd is part of the set.
func microVMAssets() map[string]AssetFile {
a := AssetFile{URL: "gs://bucket/asset", SHA256: validSHA256}
@@ -70,7 +70,6 @@ func microVMAssets() map[string]AssetFile {
"virtiofsd": a,
"kata-kernel": a,
"kata-image": a,
"kata-config": a,
}
}