mirror of
https://github.com/agent-substrate/substrate.git
synced 2026-10-02 03:24:42 +08:00
microvm: drop the kata-config asset (#1704)
Today, `ateom` fetches the `kata-config` asset (configuration-clh.toml) to read only 3 values `default_memory`, `default_vcpus` and `kernel_params`. The first 2 are redundant because (1) the values never change and (2) ateom has the same defaults. Only `kernel_params` is relevant for the `kata-agent`; its value also changed once in the Kata project history. `ateom` now owns all three values, which also allows us to tune them specifically for Substrate. Verified e2e with a GKE cluster. Fixes #1693 > It's a good idea to open an issue first for discussion. - [x] Tests pass - [x] Appropriate changes to documentation are included in the PR
This commit is contained in:
@@ -38,9 +38,9 @@ const agentVsockPort = 1024
|
||||
const debugConsoleVsockPort = 1026
|
||||
|
||||
// DebugConsoleDump connects to the guest's kata debug console (vsock 1026) and
|
||||
// runs cmd, returning its combined output. Diagnostic only (requires
|
||||
// debug_console_enabled=true in the kata config). Best-effort: returns the error
|
||||
// text on failure rather than failing the caller.
|
||||
// runs cmd, returning its combined output. Diagnostic only (requires the guest to
|
||||
// have booted with the WithDebugConsole kernel params). Best-effort: returns the
|
||||
// error text on failure rather than failing the caller.
|
||||
func DebugConsoleDump(ctx context.Context, vsockPath, cmd string) string {
|
||||
d := net.Dialer{}
|
||||
dctx, cancel := context.WithTimeout(ctx, 8*time.Second)
|
||||
|
||||
@@ -15,72 +15,32 @@
|
||||
package kata
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
toml "github.com/pelletier/go-toml/v2"
|
||||
)
|
||||
|
||||
// KataConfig holds the values ateom reads from a kata configuration.toml. ateom
|
||||
// owns the cloud-hypervisor boot and points it at the runtime-fetched asset paths
|
||||
// directly, so the only things it needs from the config are the guest sizing and
|
||||
// the agent kernel command line.
|
||||
type KataConfig struct {
|
||||
// MemoryMiB is the guest RAM size ([hypervisor.clh] default_memory).
|
||||
MemoryMiB int
|
||||
// VCPUs is the guest vCPU count ([hypervisor.clh] default_vcpus).
|
||||
VCPUs int
|
||||
// KernelParams is the guest kernel command line ([hypervisor.clh]
|
||||
// kernel_params): the kata-agent parameters (agent.log, the systemd target,
|
||||
// etc.). ateom appends these to the cloud-hypervisor payload cmdline, since
|
||||
const (
|
||||
// TODO(#1724): Tune the following values for Substrate actors.
|
||||
// DefaultMemoryMiB is the default guest memory size (MiB).
|
||||
DefaultMemoryMiB = 2048
|
||||
// DefaultVCPUs is the default guest vCPU count.
|
||||
DefaultVCPUs = 1
|
||||
)
|
||||
|
||||
const (
|
||||
// baseKernelParams is the guest kernel command line parameters ateom boots with;
|
||||
// there is no kata shim to inject them.
|
||||
KernelParams string
|
||||
}
|
||||
baseKernelParams = "cgroup_no_v1=all systemd.unified_cgroup_hierarchy=1"
|
||||
debugConsoleKernelParams = baseKernelParams + " agent.debug_console agent.debug_console_vport=1026"
|
||||
)
|
||||
|
||||
// clhConfigTOML mirrors the subset of a kata configuration.toml ateom reads.
|
||||
// Unmarshalling ignores every other key, so it stays valid across kata releases.
|
||||
type clhConfigTOML struct {
|
||||
Hypervisor struct {
|
||||
CLH struct {
|
||||
DefaultMemory int `toml:"default_memory"`
|
||||
DefaultVCPUs int `toml:"default_vcpus"`
|
||||
KernelParams string `toml:"kernel_params"`
|
||||
} `toml:"clh"`
|
||||
} `toml:"hypervisor"`
|
||||
}
|
||||
|
||||
// ParseConfig reads the guest sizing and kernel_params from a kata
|
||||
// configuration.toml. memDefault/vcpuDefault are substituted when the key is
|
||||
// absent or non-positive (kata also accepts default_vcpus = -1 meaning "all host
|
||||
// CPUs", which ateom does not support).
|
||||
func ParseConfig(base []byte, memDefault, vcpuDefault int) (KataConfig, error) {
|
||||
var c clhConfigTOML
|
||||
if err := toml.Unmarshal(base, &c); err != nil {
|
||||
return KataConfig{}, fmt.Errorf("parsing kata config: %w", err)
|
||||
}
|
||||
cfg := KataConfig{
|
||||
MemoryMiB: c.Hypervisor.CLH.DefaultMemory,
|
||||
VCPUs: c.Hypervisor.CLH.DefaultVCPUs,
|
||||
KernelParams: c.Hypervisor.CLH.KernelParams,
|
||||
}
|
||||
if cfg.MemoryMiB <= 0 {
|
||||
cfg.MemoryMiB = memDefault
|
||||
}
|
||||
if cfg.VCPUs <= 0 {
|
||||
cfg.VCPUs = vcpuDefault
|
||||
}
|
||||
return cfg, nil
|
||||
}
|
||||
|
||||
// WithDebugConsole appends the kata-agent debug-console kernel parameters so the
|
||||
// guest agent binds a root debug shell on vsock port 1026, which DebugConsoleDump
|
||||
// connects to for in-guest diagnostics. Both params are required: agent.debug_console
|
||||
// enables the console and agent.debug_console_vport=1026 makes the agent bind it on
|
||||
// the vsock port (the agent only binds a vsock listener when the vport is > 0).
|
||||
// Idempotent.
|
||||
func WithDebugConsole(kernelParams string) string {
|
||||
return appendKernelParams(kernelParams, "agent.debug_console_vport",
|
||||
"agent.debug_console agent.debug_console_vport=1026")
|
||||
// WithDebugConsole returns the kernel params with the kata-agent debug console
|
||||
// enabled, so the guest agent binds a root debug shell on vsock port 1026, which
|
||||
// DebugConsoleDump connects to for in-guest diagnostics. Both params are required:
|
||||
// agent.debug_console enables the console and agent.debug_console_vport=1026 makes
|
||||
// the agent bind it on the vsock port (the agent only binds a vsock listener when
|
||||
// the vport is > 0).
|
||||
func WithDebugConsole() string {
|
||||
return debugConsoleKernelParams
|
||||
}
|
||||
|
||||
// WithAgentDebug appends agent.log=debug so the guest kata-agent emits
|
||||
|
||||
@@ -19,71 +19,6 @@ import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
// stockConfig mirrors the [hypervisor.clh] keys ateom reads from a kata
|
||||
// configuration.toml (every other key is ignored by ParseConfig).
|
||||
const stockConfig = `[hypervisor.clh]
|
||||
path = "/usr/local/bin/cloud-hypervisor"
|
||||
kernel = "/opt/kata/share/kata-containers/vmlinux.container"
|
||||
image = "/opt/kata/share/kata-containers/kata-containers.img"
|
||||
default_memory = 512
|
||||
default_vcpus = 2
|
||||
kernel_params = "agent.foo=bar systemd.unit=kata-containers.target"
|
||||
shared_fs = "virtio-fs"
|
||||
`
|
||||
|
||||
func TestParseConfig(t *testing.T) {
|
||||
cfg, err := ParseConfig([]byte(stockConfig), 2048, 1)
|
||||
if err != nil {
|
||||
t.Fatalf("ParseConfig: %v", err)
|
||||
}
|
||||
if cfg.MemoryMiB != 512 {
|
||||
t.Errorf("MemoryMiB = %d, want 512", cfg.MemoryMiB)
|
||||
}
|
||||
if cfg.VCPUs != 2 {
|
||||
t.Errorf("VCPUs = %d, want 2", cfg.VCPUs)
|
||||
}
|
||||
if want := "agent.foo=bar systemd.unit=kata-containers.target"; cfg.KernelParams != want {
|
||||
t.Errorf("KernelParams = %q, want %q", cfg.KernelParams, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseConfigDefaults asserts the mem/vcpu defaults kick in when the keys are
|
||||
// absent or non-positive (kata also accepts default_vcpus = -1 meaning "all host
|
||||
// CPUs", which ateom does not support).
|
||||
func TestParseConfigDefaults(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
toml string
|
||||
}{
|
||||
{"absent", "[hypervisor.clh]\nkernel_params = \"x\"\n"},
|
||||
{"nonpositive", "[hypervisor.clh]\ndefault_memory = 0\ndefault_vcpus = -1\n"},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
cfg, err := ParseConfig([]byte(tc.toml), 2048, 1)
|
||||
if err != nil {
|
||||
t.Fatalf("ParseConfig: %v", err)
|
||||
}
|
||||
if cfg.MemoryMiB != 2048 {
|
||||
t.Errorf("MemoryMiB = %d, want default 2048", cfg.MemoryMiB)
|
||||
}
|
||||
if cfg.VCPUs != 1 {
|
||||
t.Errorf("VCPUs = %d, want default 1", cfg.VCPUs)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithDebugConsole(t *testing.T) {
|
||||
got := WithDebugConsole("root=/dev/vda1")
|
||||
if !strings.Contains(got, "agent.debug_console") || !strings.Contains(got, "agent.debug_console_vport=1026") {
|
||||
t.Errorf("WithDebugConsole did not append the debug-console params: %q", got)
|
||||
}
|
||||
// Idempotent: a second call must not append the params again.
|
||||
if again := WithDebugConsole(got); again != got {
|
||||
t.Errorf("WithDebugConsole not idempotent:\n first = %q\nsecond = %q", got, again)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithAgentDebug(t *testing.T) {
|
||||
got := WithAgentDebug("root=/dev/vda1")
|
||||
if !strings.Contains(got, "agent.log=debug") {
|
||||
|
||||
@@ -18,9 +18,8 @@
|
||||
// ttrpc API (DialAgent / AgentClient) to create the sandbox and run each container
|
||||
// on its host-merged rootfs (overlay_linux.go).
|
||||
//
|
||||
// It also renders the kata configuration.toml (for the agent kernel_params + guest
|
||||
// sizing) from runtime-fetched assets (config.go) and sweeps leftover per-sandbox
|
||||
// host-side state (CleanupSandboxState).
|
||||
// It also owns the guest kernel_params and default guest sizing (config.go) and
|
||||
// sweeps leftover per-sandbox host-side state (CleanupSandboxState).
|
||||
package kata
|
||||
|
||||
import (
|
||||
|
||||
@@ -63,7 +63,6 @@ import (
|
||||
var (
|
||||
podUID = flag.String("pod-uid", "", "The UID of the current pod")
|
||||
chBinary = flag.String("cloud-hypervisor-binary", "cloud-hypervisor", "Path to the cloud-hypervisor binary (used to relaunch on restore).")
|
||||
kataConfig = flag.String("kata-config", "", "Path to a kata configuration.toml (passed to the shim as KATA_CONF_FILE). Empty uses kata's default. atelet generates one pointing at runtime-fetched assets.")
|
||||
kataDebug = flag.Bool("kata-debug", false, "Verbose kata-agent debugging: raise the guest agent log level and forward the guest console (incl. agent logs) into the pod logs.")
|
||||
vmmMemReserve = flag.Int("vmm-mem-reserve-mib", vmmMemReserveMiB, "Guest RAM (MiB) held back from the pod's memory limit for the cloud-hypervisor VMM + virtiofsd, which run as host processes in the pod cgroup alongside the guest RAM. Prevents the pod OOMing when the VM is sized to the pod's memory limit.")
|
||||
showVersion = flag.Bool("version", false, "Print version and exit.")
|
||||
@@ -266,7 +265,7 @@ func do(ctx context.Context) error {
|
||||
}()
|
||||
slog.InfoContext(ctx, "atunnel egress serving", slog.String("address", *atunnelEgressListenAddress))
|
||||
|
||||
ateomService := NewService(*podUID, *chBinary, *kataConfig, *kataDebug, *vmmMemReserve, interiorNetNS, actorLogger, atunnelIngress, atunnelEgress, atunnelEgressPort, *workerCredentialBundle, *podIdentityTrustBundle, *egressGatewayTrustBundle)
|
||||
ateomService := NewService(*podUID, *chBinary, *kataDebug, *vmmMemReserve, interiorNetNS, actorLogger, atunnelIngress, atunnelEgress, atunnelEgressPort, *workerCredentialBundle, *podIdentityTrustBundle, *egressGatewayTrustBundle)
|
||||
|
||||
svr := grpc.NewServer(
|
||||
grpc.StatsHandler(otelgrpc.NewServerHandler()),
|
||||
@@ -418,10 +417,9 @@ type AteomService struct {
|
||||
activeRPCMu sync.Mutex
|
||||
activeRPC *activeRPCInfo
|
||||
|
||||
podUID string
|
||||
chBinary string
|
||||
kataConfig string
|
||||
kataDebug bool
|
||||
podUID string
|
||||
chBinary string
|
||||
kataDebug bool
|
||||
|
||||
// memReserveMiB is guest RAM (MiB) held back from the pod's memory limit for
|
||||
// the cloud-hypervisor VMM + virtiofsd (host processes sharing the pod cgroup
|
||||
@@ -502,12 +500,11 @@ type AteomService struct {
|
||||
var _ ateompb.AteomServer = (*AteomService)(nil)
|
||||
|
||||
// NewService creates a new AteomService.
|
||||
func NewService(podUID, chBinary, kataConfig string, kataDebug bool, memReserveMiB int, interiorNetNS netns.NsHandle, actorLogger *actorlog.ActorLogger, atunnelIngress *atunnel.Server, atunnelEgress *atunnel.Egress, atunnelEgressPort uint16, workerCredentialBundlePath, podIdentityTrustBundlePath, egressGatewayTrustBundlePath string) *AteomService {
|
||||
func NewService(podUID, chBinary string, kataDebug bool, memReserveMiB int, interiorNetNS netns.NsHandle, actorLogger *actorlog.ActorLogger, atunnelIngress *atunnel.Server, atunnelEgress *atunnel.Egress, atunnelEgressPort uint16, workerCredentialBundlePath, podIdentityTrustBundlePath, egressGatewayTrustBundlePath string) *AteomService {
|
||||
return &AteomService{
|
||||
lock: newCancelableMutex(),
|
||||
podUID: podUID,
|
||||
chBinary: chBinary,
|
||||
kataConfig: kataConfig,
|
||||
kataDebug: kataDebug,
|
||||
memReserveMiB: memReserveMiB,
|
||||
interiorNetNS: interiorNetNS,
|
||||
|
||||
+21
-35
@@ -113,7 +113,6 @@ const (
|
||||
assetCH = "cloud-hypervisor"
|
||||
assetKernel = "kata-kernel"
|
||||
assetImage = "kata-image"
|
||||
assetConfig = "kata-config"
|
||||
assetVirtiofsd = "virtiofsd"
|
||||
)
|
||||
|
||||
@@ -185,12 +184,11 @@ type actorContainer struct {
|
||||
imageMounts []*ateompb.ImageVolumeMount
|
||||
}
|
||||
|
||||
// resolvedRuntime holds the concrete binary/config paths for a request, taken
|
||||
// from fetched runtime assets when present, else the process flags.
|
||||
// resolvedRuntime holds the concrete binary paths for a request, taken from fetched
|
||||
// runtime assets when present, else the process flags.
|
||||
type resolvedRuntime struct {
|
||||
chBinary string // path to the cloud-hypervisor binary
|
||||
configFile string // path to the kata configuration.toml
|
||||
virtiofsd string // path to virtiofsd (overlay RO lower); "" => "virtiofsd" on PATH
|
||||
chBinary string // path to the cloud-hypervisor binary
|
||||
virtiofsd string // path to virtiofsd (overlay RO lower); "" => "virtiofsd" on PATH
|
||||
}
|
||||
|
||||
// firstNonEmpty returns the first non-empty string, or "" if all are empty.
|
||||
@@ -203,13 +201,12 @@ func firstNonEmpty(vals ...string) string {
|
||||
return ""
|
||||
}
|
||||
|
||||
// resolveRuntime resolves the cloud-hypervisor binary + the kata config path from
|
||||
// fetched assets, falling back to flags.
|
||||
// resolveRuntime resolves the cloud-hypervisor binary from fetched assets, falling
|
||||
// back to the flag.
|
||||
func (s *AteomService) resolveRuntime(paths map[string]string) resolvedRuntime {
|
||||
return resolvedRuntime{
|
||||
chBinary: firstNonEmpty(paths[assetCH], s.chBinary),
|
||||
configFile: firstNonEmpty(paths[assetConfig], s.kataConfig),
|
||||
virtiofsd: paths[assetVirtiofsd],
|
||||
chBinary: firstNonEmpty(paths[assetCH], s.chBinary),
|
||||
virtiofsd: paths[assetVirtiofsd],
|
||||
}
|
||||
}
|
||||
|
||||
@@ -264,8 +261,8 @@ func writeGuestResolvConf(rootfs string) error {
|
||||
// start each container.
|
||||
//
|
||||
// Contract with atelet:
|
||||
// - The runtime assets (guest kernel, guest OS image, cloud-hypervisor, virtiofsd,
|
||||
// base kata config) are on disk and passed as runtime asset paths.
|
||||
// - The runtime assets (guest kernel, guest OS image, cloud-hypervisor, virtiofsd)
|
||||
// are on disk and passed as runtime asset paths.
|
||||
// - The OCI bundle (config.json + populated rootfs/) is prepared per container.
|
||||
func (s *AteomService) RunWorkload(ctx context.Context, req *ateompb.RunWorkloadRequest) (resp *ateompb.RunWorkloadResponse, retErr error) {
|
||||
s.lock.Lock()
|
||||
@@ -437,14 +434,11 @@ func (s *AteomService) coldBootActor(ctx context.Context, p actorBootParams) (re
|
||||
}
|
||||
}()
|
||||
|
||||
// Guest sizing + agent kernel params from the kata config.
|
||||
memMiB, vcpus, kparams, err := s.guestConfig(rr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// Guest sizing + agent kernel params.
|
||||
memMiB, vcpus, kparams := s.guestConfig()
|
||||
|
||||
// Right-size the VM to the actor's declared limits (see internal/sizing),
|
||||
// keeping the kata-config values above as the fallback when a limit is unset.
|
||||
// keeping the defaults above as the fallback when a limit is unset.
|
||||
// vCPUs round up; VM RAM reserves a fixed margin for the VMM + virtiofsd, which
|
||||
// share the pod cgroup with the guest RAM. A declared memory limit the reserve
|
||||
// leaves too small to boot is rejected (resolveGuestMemMiB) rather than silently
|
||||
@@ -729,28 +723,20 @@ func (s *AteomService) stageMergedRootfs(ctx context.Context, rr resolvedRuntime
|
||||
return vfsdCmd, nil
|
||||
}
|
||||
|
||||
// guestConfig reads guest sizing + agent kernel params from the resolved kata
|
||||
// config, enabling the debug console (vsock 1026) for in-guest diagnostics and,
|
||||
// with kataDebug, raising the agent log level.
|
||||
func (s *AteomService) guestConfig(rr resolvedRuntime) (memMiB, vcpus int, kparams string, err error) {
|
||||
var cfgBytes []byte
|
||||
if rr.configFile != "" {
|
||||
cfgBytes, _ = os.ReadFile(rr.configFile)
|
||||
}
|
||||
cfg, err := kata.ParseConfig(cfgBytes, 2048, 1)
|
||||
if err != nil {
|
||||
return 0, 0, "", fmt.Errorf("while parsing kata config: %w", err)
|
||||
}
|
||||
kparams = kata.WithDebugConsole(cfg.KernelParams)
|
||||
// guestConfig returns the default guest sizing and the agent kernel params, enabling
|
||||
// the debug console (vsock 1026) for in-guest diagnostics and, with kataDebug, raising
|
||||
// the agent log level.
|
||||
func (s *AteomService) guestConfig() (memMiB, vcpus int, kparams string) {
|
||||
kparams = kata.WithDebugConsole()
|
||||
if s.kataDebug {
|
||||
kparams = kata.WithAgentDebug(kparams)
|
||||
}
|
||||
return cfg.MemoryMiB, cfg.VCPUs, kparams, nil
|
||||
return kata.DefaultMemoryMiB, kata.DefaultVCPUs, kparams
|
||||
}
|
||||
|
||||
// resolveGuestMemMiB returns the micro-VM guest RAM (MiB) for an actor's declared
|
||||
// memory limit. declaredBytes == 0 means "unset" and returns fallbackMiB (the
|
||||
// kata-config default). Otherwise the guest gets the declared memory minus the VMM
|
||||
// memory limit. declaredBytes == 0 means "unset" and returns fallbackMiB
|
||||
// (kata.DefaultMemoryMiB). Otherwise the guest gets the declared memory minus the VMM
|
||||
// reserve; if that leaves less than a bootable minimum it errors — naming the limit,
|
||||
// the reserve, and the minimum — instead of silently reverting to the (larger)
|
||||
// fallback, which would boot the actor bigger than the worker was sized for and OOM
|
||||
|
||||
@@ -171,7 +171,7 @@ func TestResolveGuestMemMiB(t *testing.T) {
|
||||
const (
|
||||
mib = 1024 * 1024
|
||||
reserve = 256 // vmmMemReserveMiB
|
||||
fallback = 2048 // kata-config default
|
||||
fallback = 2048 // kata.DefaultMemoryMiB
|
||||
)
|
||||
tests := []struct {
|
||||
name string
|
||||
|
||||
@@ -65,7 +65,7 @@ volumes:
|
||||
- trustBundle:
|
||||
name: egress-mitm.ate.dev
|
||||
path: trust-bundle.pem
|
||||
# Sandbox size: without these the guest boots at the kata config's default
|
||||
# Sandbox size: without these the guest boots at ateom's default size
|
||||
# (2GiB). ateom applies them to the VM (see internal/sizing). Quantities are
|
||||
# strings in the proto.
|
||||
resources:
|
||||
|
||||
@@ -38,7 +38,7 @@ containers:
|
||||
# See ../counter/counter-microvm-template.yaml.tmpl: a micro-VM guest needs
|
||||
# more than readyz's 30s default once CI's single kind node is under load.
|
||||
timeoutSeconds: 120
|
||||
# Sandbox size: without these the guest boots at the kata config's default
|
||||
# Sandbox size: without these the guest boots at ateom's default size
|
||||
# (2GiB). ateom applies them to the VM (see internal/sizing). Quantities are
|
||||
# strings in the proto.
|
||||
resources:
|
||||
|
||||
+2
-2
@@ -135,7 +135,7 @@ Unlike a Pod, an actor is sized by its **`limits`** (CPU and Memory): the size i
|
||||
- **gVisor (`ateom-gvisor`)** — applied to the container OCI spec: `limits.cpu` sets the cgroup v2 CPU quota (`cpu.max`) and the Sentry vCPU count (`--cpu-num-from-quota`); `limits.memory` sets the cgroup v2 memory limit (`memory.max`) and bounds the virtual total memory the sandbox reports (so JVM/Go do not over-allocate from host RAM).
|
||||
- **Micro-VM (`ateom-microvm`)** — `limits.cpu` sets Cloud Hypervisor `BootVcpus` / `MaxVcpus` (rounded up to whole vCPUs); `limits.memory` sets guest RAM, reserving a small configurable margin (default 256 MiB, `--vmm-mem-reserve-mib`) for the VMM and virtiofsd so the pod cgroup does not OOM.
|
||||
2. **Gate scheduling.** An actor is only placed on a `WorkerPool` whose [worker capacity](#worker-capacity-spectemplateresources) is `>=` these limits.
|
||||
3. **Fall back to runtime defaults.** A zero or absent limit leaves that dimension at the runtime default — unlimited for gVisor, the kata config for the micro-VM.
|
||||
3. **Fall back to runtime defaults.** A zero or absent limit leaves that dimension at the runtime default: unlimited for gVisor, and 2 GiB / 1 vCPU for the micro-VM.
|
||||
|
||||
`requests` are not consulted today (an actor occupies its whole worker). Because the size is baked into snapshots, a **micro-VM FULL-scope restore reuses the size in the snapshot**; changing an actor's limits takes effect on its next cold boot.
|
||||
|
||||
@@ -424,7 +424,7 @@ spec:
|
||||
|
||||
### Micro-VM SandboxConfig
|
||||
|
||||
A `microvm` `SandboxConfig` supplies the [Kata Containers](https://katacontainers.io/) + [Cloud Hypervisor](https://www.cloudhypervisor.org/) toolchain instead of `runsc`. Each architecture must define the full asset set — `kata-shim`, `cloud-hypervisor`, `virtiofsd`, `kata-kernel`, `kata-image`, and `kata-config` — which a `ValidatingAdmissionPolicy` enforces at apply time. Worker pods for a micro-VM pool require `/dev/kvm` and nested-virtualization-capable nodes. The controller requests those devices on the pod automatically, and atelet advertises them only where they exist, so placement follows the hardware rather than a node label. Clusters that reserve nested-virt nodes with an `ate.dev/sandboxClass=microvm` taint are still tolerated: advertising a device attracts these pods to capable nodes but repels nothing else from them.
|
||||
A `microvm` `SandboxConfig` supplies the [Kata Containers](https://katacontainers.io/) + [Cloud Hypervisor](https://www.cloudhypervisor.org/) toolchain instead of `runsc`. Each architecture must define the full asset set — `cloud-hypervisor`, `virtiofsd`, `kata-kernel`, and `kata-image` — which a `ValidatingAdmissionPolicy` enforces at apply time. Worker pods for a micro-VM pool require `/dev/kvm` and nested-virtualization-capable nodes. The controller requests those devices on the pod automatically, and atelet advertises them only where they exist, so placement follows the hardware rather than a node label. Clusters that reserve nested-virt nodes with an `ate.dev/sandboxClass=microvm` taint are still tolerated: advertising a device attracts these pods to capable nodes but repels nothing else from them.
|
||||
|
||||
See [`hack/microvm-assets/`](../hack/microvm-assets/) for scripts that assemble and stage these assets, plus a worked counter demo (`demos/counter/counter-microvm.yaml.tmpl`) that suspends and resumes an in-RAM counter across worker pods.
|
||||
|
||||
|
||||
@@ -36,7 +36,6 @@ require (
|
||||
github.com/openfga/api/proto v0.0.0-20260319214821-f153694bfc20
|
||||
github.com/openfga/language/pkg/go v0.3.2-0.20260730144454-83fedf8a4e70
|
||||
github.com/openfga/openfga v1.20.0
|
||||
github.com/pelletier/go-toml/v2 v2.4.0
|
||||
github.com/pressly/goose/v3 v3.27.3
|
||||
github.com/prometheus/client_golang v1.24.1
|
||||
github.com/spf13/cobra v1.10.2
|
||||
@@ -203,6 +202,7 @@ require (
|
||||
github.com/olekukonko/tablewriter v1.0.8 // indirect
|
||||
github.com/opencontainers/go-digest v1.0.0 // indirect
|
||||
github.com/opencontainers/image-spec v1.1.1 // indirect
|
||||
github.com/pelletier/go-toml/v2 v2.4.0 // indirect
|
||||
github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 // indirect
|
||||
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect
|
||||
github.com/power-devops/perfstat v0.0.0-20240221224432-82ca36839d55 // indirect
|
||||
|
||||
@@ -142,13 +142,13 @@ OUT="${OUT:-${ROOT}/bin/microvm-assets/$ARCH}"
|
||||
|
||||
# --- 1. assets: assemble (if missing or stale) -----------------------------
|
||||
need_assemble=false
|
||||
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml; do
|
||||
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img; do
|
||||
if [[ ! -f "${OUT}/${f}" ]]; then
|
||||
need_assemble=true
|
||||
break
|
||||
fi
|
||||
done
|
||||
# Presence alone is not enough. The five filenames don't change when a version pin
|
||||
# Presence alone is not enough. The filenames don't change when a version pin
|
||||
# moves, so an asset dir assembled before a bump looks complete while holding the old
|
||||
# bytes — we'd then stage those against a SandboxConfig pinning the new shas, and the
|
||||
# mismatch would only surface at runtime as an actor wedged in STATUS_RESUMING while
|
||||
@@ -170,7 +170,7 @@ else
|
||||
fi
|
||||
|
||||
# --- 2. stage assets to rustfs (kind) / GCS (GKE) --------------------------
|
||||
# Upload the five assets under kata-assets/, where atelet fetches them: the
|
||||
# Upload the four assets under kata-assets/, where atelet fetches them: the
|
||||
# in-cluster rustfs (S3 API) on kind, or the GCS bucket on GKE.
|
||||
if [[ "${ATE_INSTALL_KIND}" == "true" ]]; then
|
||||
log "Staging assets to in-cluster rustfs bucket ${BUCKET_NAME} (kata-assets/)..."
|
||||
@@ -185,7 +185,7 @@ fi
|
||||
# its binary bytes are not reproducible across toolchains and its sha can't
|
||||
# be a fixed pin in the manifest. Compute it from the freshly-staged binary
|
||||
# and inject it, so the deployed SandboxConfig always matches whatever was
|
||||
# staged. The downloaded assets (cloud-hypervisor/kernel/rootfs/config, plus
|
||||
# staged. The downloaded assets (cloud-hypervisor/kernel/rootfs, plus
|
||||
# virtiofsd on amd64 where upstream publishes a prebuilt) keep their
|
||||
# committed, reproducible per-arch shas.
|
||||
log "Applying microvm SandboxConfig from ${MANIFEST_TEMPLATE}..."
|
||||
|
||||
@@ -5,13 +5,12 @@ toolchain at runtime — nothing kata-specific is baked into the worker image. a
|
||||
the kata-agent directly (no kata shim, no containerd). Each actor container's rootfs is an
|
||||
overlay of a read-only lower (the OCI image, served into the guest over virtio-fs by
|
||||
`virtiofsd`) and a writable upper on a guest tmpfs, so `virtiofsd` is part of the asset
|
||||
set. The asset set is five files:
|
||||
set. The asset set is four files:
|
||||
|
||||
- `cloud-hypervisor` — the VMM binary (fetched from its release)
|
||||
- `virtiofsd` — the virtio-fs daemon serving the RO lower (built from source; see `assemble.sh`)
|
||||
- `vmlinux` — the guest kernel (from kata-static)
|
||||
- `rootfs.img` — the guest rootfs image (from kata-static)
|
||||
- `configuration-clh.toml` — the base kata config (from kata-static)
|
||||
|
||||
These helpers assemble the asset set for your node arch, stage it into the cluster's rustfs
|
||||
S3 bucket, and the demo manifest's `SandboxConfig` points at it. When `/dev/kvm` is
|
||||
|
||||
@@ -18,8 +18,8 @@
|
||||
# ateom-microvm fetches at runtime (fetch-not-bake). Run this on a Linux
|
||||
# host of the TARGET arch.
|
||||
#
|
||||
# Produces, under $OUT, the five assets named as the SandboxConfig expects:
|
||||
# cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml
|
||||
# Produces, under $OUT, the four assets named as the SandboxConfig expects:
|
||||
# cloud-hypervisor virtiofsd vmlinux rootfs.img
|
||||
# The DOWNLOADED assets are reproducible, so paste their sha256 sums into the
|
||||
# manifest (demos/counter/counter-microvm.yaml.tmpl). That now includes virtiofsd on
|
||||
# amd64 (upstream prebuilt); on arm64 virtiofsd is still built from source
|
||||
@@ -64,7 +64,7 @@ esac
|
||||
|
||||
# Identifies the asset set this script produces. Cleared before the first write into
|
||||
# $OUT and re-written to $OUT/$STAMP_FILE on success; install-microvm-deps.sh compares
|
||||
# it against what the current checkout would build, because the five filenames stay the
|
||||
# it against what the current checkout would build, because the filenames stay the
|
||||
# same when a pin moves and an asset dir from an older checkout is otherwise
|
||||
# indistinguishable from a current one.
|
||||
STAMP_FILE=".asset-versions"
|
||||
@@ -98,7 +98,6 @@ KROOT="kata/opt/kata"
|
||||
|
||||
cp "$(readlink -f "${KROOT}/share/kata-containers/vmlinux.container")" "${OUT}/vmlinux"
|
||||
cp "$(readlink -f "${KROOT}/share/kata-containers/kata-containers.img")" "${OUT}/rootfs.img"
|
||||
cp "${KROOT}/share/defaults/kata-containers/configuration-clh.toml" "${OUT}/configuration-clh.toml"
|
||||
|
||||
echo ">> Downloading cloud-hypervisor ${CH_VER} (${CH_ASSET})..."
|
||||
curl -fSL -o "${OUT}/cloud-hypervisor" \
|
||||
@@ -145,10 +144,10 @@ chmod +x "${OUT}/virtiofsd"
|
||||
echo
|
||||
echo ">> Assets assembled in ${OUT}:"
|
||||
cd "${OUT}"
|
||||
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml; do
|
||||
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img; do
|
||||
[ -f "$f" ] || { echo "MISSING: $f" >&2; exit 1; }
|
||||
done
|
||||
# Written only once all five are present, and only after the up-front rm, so the stamp
|
||||
# Written only once all four are present, and only after the up-front rm, so the stamp
|
||||
# exists exactly when this dir was assembled end-to-end by these pins.
|
||||
asset_stamp > "${OUT}/${STAMP_FILE}"
|
||||
"${OUT}/virtiofsd" --version 2>/dev/null | head -1 || true
|
||||
@@ -156,4 +155,4 @@ echo
|
||||
echo ">> sha256 (paste the DOWNLOADED assets into counter-microvm.yaml.tmpl; that"
|
||||
echo ">> includes virtiofsd on amd64 (prebuilt). The arm64 virtiofsd is built from"
|
||||
echo ">> source, so its sha is injected at deploy by run-microvm-demo.sh, not pinned):"
|
||||
sha256sum cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml
|
||||
sha256sum cloud-hypervisor virtiofsd vmlinux rootfs.img
|
||||
|
||||
@@ -33,7 +33,7 @@ BUCKET="${BUCKET:-ate-snapshots}"
|
||||
# gcloud uses its active config project. ${PROJECT_ID:+...} elides the flag entirely
|
||||
# when unset (same idiom as KUBECTL_CONTEXT in hack/run-microvm-demo.sh).
|
||||
echo ">> Uploading assets to gs://${BUCKET}/kata-assets/ ..."
|
||||
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml; do
|
||||
for f in cloud-hypervisor virtiofsd vmlinux rootfs.img; do
|
||||
echo " $f"
|
||||
gcloud storage cp ${PROJECT_ID:+--project="${PROJECT_ID}"} "${OUT}/${f}" "gs://${BUCKET}/kata-assets/${f}"
|
||||
done
|
||||
|
||||
@@ -48,7 +48,7 @@ KIND_CLUSTER_NAME="${KIND_CLUSTER_NAME:-kind}"
|
||||
# manifests/ate-install/kind/rustfs.yaml, which creates the bucket we upload into.
|
||||
AWS_CLI_IMAGE="amazon/aws-cli:2.17.0@sha256:643507c10ada7964ca6157b3d799f030b90577643da9955d319a77399ed80d73"
|
||||
|
||||
ASSETS=(cloud-hypervisor virtiofsd vmlinux rootfs.img configuration-clh.toml)
|
||||
ASSETS=(cloud-hypervisor virtiofsd vmlinux rootfs.img)
|
||||
|
||||
run_kubectl() {
|
||||
kubectl ${KUBECTL_CONTEXT:+--context="${KUBECTL_CONTEXT}"} -n "${NAMESPACE}" "$@"
|
||||
|
||||
@@ -60,7 +60,7 @@ func substrateTemplateSubstitutions(bucket, name string, trustBundle bool) (inli
|
||||
// a missing or stale one fails loudly at template creation.
|
||||
blocks["${TEMPLATE_SANDBOX_CONFIG}"] = "sandboxConfig:\n sandboxClass: SANDBOX_CLASS_MICROVM\n configName: microvm"
|
||||
// Only for fixtures that declare no limits of their own. Without them the
|
||||
// guest boots at the kata config's default (2GiB), and several of those
|
||||
// guest boots at ateom's default size (2GiB), and several of those
|
||||
// do not fit beside the demo pools on CI's single kind node. These size
|
||||
// the VM itself — see internal/sizing. Quantities are strings.
|
||||
blocks["${TEMPLATE_RESOURCES}"] = "resources:\n limits:\n - name: cpu\n quantity: \"1\"\n - name: memory\n quantity: 512Mi"
|
||||
|
||||
@@ -203,7 +203,7 @@ func fixtureSubstitutions(bucket, name string) (inline, blocks map[string]string
|
||||
// classes, so only same-class pools are eligible to run these actors.
|
||||
blocks["${TEMPLATE_SANDBOX_CLASS}"] = " sandboxClass: microvm"
|
||||
// Only for fixtures that declare no limits of their own. Without them the
|
||||
// guest boots at the kata config's default (2GiB), and several of those do
|
||||
// guest boots at ateom's default size (2GiB), and several of those do
|
||||
// not fit beside the demo pools on CI's single kind node. These size the VM
|
||||
// itself — see internal/sizing.
|
||||
blocks["${TEMPLATE_RESOURCES}"] = " resources:\n limits:\n cpu: \"1\"\n memory: 512Mi"
|
||||
|
||||
@@ -195,7 +195,7 @@ func TestRenderSubstrateFixtures_MicroVM(t *testing.T) {
|
||||
if got := tmpl.GetSandboxConfig().GetConfigName(); got != "microvm" {
|
||||
t.Errorf("template %s configName = %q, want microvm", name, got)
|
||||
}
|
||||
// Undeclared limits boot the guest at the kata config default
|
||||
// Undeclared limits boot the guest at ateom's default size
|
||||
// (2GiB), which does not fit beside the demo pools on one kind
|
||||
// node.
|
||||
if memoryLimit(tmpl) == "" {
|
||||
|
||||
@@ -109,7 +109,7 @@ func CreateSubstrateTemplateFrom(ctx context.Context, t *testing.T, clients *Cli
|
||||
Containers: srcTmpl.GetContainers(),
|
||||
// The source's limits size the sandbox. Copying them matters most on
|
||||
// micro-VM, where an ActorTemplate that declares none boots the guest
|
||||
// at the kata config default (2GiB) instead of the demo's 512Mi.
|
||||
// at ateom's default guest size (2GiB) instead of the demo's 512Mi.
|
||||
Resources: srcTmpl.GetResources(),
|
||||
// The source carries the sandbox_class/config_name pair for the
|
||||
// class under test.
|
||||
|
||||
@@ -350,7 +350,7 @@ type RunWorkloadRequest struct {
|
||||
RunscPath string `protobuf:"bytes,6,opt,name=runsc_path,json=runscPath,proto3" json:"runsc_path,omitempty"`
|
||||
Spec *WorkloadSpec `protobuf:"bytes,7,opt,name=spec,proto3" json:"spec,omitempty"`
|
||||
// runtime_asset_paths maps a runtime asset name (e.g. "cloud-hypervisor",
|
||||
// "virtiofsd", "kata-kernel", "kata-image", "kata-config")
|
||||
// "virtiofsd", "kata-kernel", "kata-image")
|
||||
// to the local on-disk path atelet fetched it to (content-addressed, like
|
||||
// runsc_path). Empty for the gVisor runtime, which uses runsc_path.
|
||||
RuntimeAssetPaths map[string]string `protobuf:"bytes,8,rep,name=runtime_asset_paths,json=runtimeAssetPaths,proto3" json:"runtime_asset_paths,omitempty" protobuf_key:"bytes,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"`
|
||||
@@ -359,7 +359,7 @@ type RunWorkloadRequest struct {
|
||||
// The actor's declared size, from the ActorTemplate's resource limits. ateom
|
||||
// sizes the sandbox to these (cgroup caps via the OCI spec, and for the
|
||||
// micro-VM the VM's vCPU count and memory). Zero means "unset": keep the
|
||||
// runtime default (unlimited for gVisor, the kata config for the micro-VM).
|
||||
// runtime default (unlimited for gVisor, ateom's own default for the micro-VM).
|
||||
CpuMilli int64 `protobuf:"varint,11,opt,name=cpu_milli,json=cpuMilli,proto3" json:"cpu_milli,omitempty"` // CPU limit in millicores (1000 = one core).
|
||||
MemoryBytes int64 `protobuf:"varint,12,opt,name=memory_bytes,json=memoryBytes,proto3" json:"memory_bytes,omitempty"` // Memory limit in bytes.
|
||||
unknownFields protoimpl.UnknownFields
|
||||
|
||||
@@ -131,7 +131,7 @@ message RunWorkloadRequest {
|
||||
WorkloadSpec spec = 7;
|
||||
|
||||
// runtime_asset_paths maps a runtime asset name (e.g. "cloud-hypervisor",
|
||||
// "virtiofsd", "kata-kernel", "kata-image", "kata-config")
|
||||
// "virtiofsd", "kata-kernel", "kata-image")
|
||||
// to the local on-disk path atelet fetched it to (content-addressed, like
|
||||
// runsc_path). Empty for the gVisor runtime, which uses runsc_path.
|
||||
map<string, string> runtime_asset_paths = 8;
|
||||
@@ -142,7 +142,7 @@ message RunWorkloadRequest {
|
||||
// The actor's declared size, from the ActorTemplate's resource limits. ateom
|
||||
// sizes the sandbox to these (cgroup caps via the OCI spec, and for the
|
||||
// micro-VM the VM's vCPU count and memory). Zero means "unset": keep the
|
||||
// runtime default (unlimited for gVisor, the kata config for the micro-VM).
|
||||
// runtime default (unlimited for gVisor, ateom's own default for the micro-VM).
|
||||
int64 cpu_milli = 11; // CPU limit in millicores (1000 = one core).
|
||||
int64 memory_bytes = 12; // Memory limit in bytes.
|
||||
}
|
||||
|
||||
@@ -34,7 +34,7 @@ const (
|
||||
|
||||
// SandboxSize is the sandbox's target size, derived from the actor's declared
|
||||
// resource limits. A zero field means "unset": the caller keeps its own default
|
||||
// (the kata config for the micro-VM, unlimited for gVisor).
|
||||
// (kata.DefaultMemoryMiB / kata.DefaultVCPUs for the micro-VM, unlimited for gVisor).
|
||||
type SandboxSize struct {
|
||||
// MilliCPU is the CPU limit in millicores (1000 = one core), or 0 if unset.
|
||||
MilliCPU int64
|
||||
|
||||
@@ -45,9 +45,9 @@ spec:
|
||||
object.spec.sandboxClass != 'microvm' ||
|
||||
(has(object.spec.assets) && size(object.spec.assets) > 0 &&
|
||||
object.spec.assets.all(arch,
|
||||
['cloud-hypervisor', 'virtiofsd', 'kata-kernel', 'kata-image', 'kata-config']
|
||||
['cloud-hypervisor', 'virtiofsd', 'kata-kernel', 'kata-image']
|
||||
.all(name, name in object.spec.assets[arch])))
|
||||
message: "a microvm SandboxConfig must define cloud-hypervisor, virtiofsd, kata-kernel, kata-image, and kata-config assets for every architecture under spec.assets"
|
||||
message: "a microvm SandboxConfig must define cloud-hypervisor, virtiofsd, kata-kernel, and kata-image assets for every architecture under spec.assets"
|
||||
---
|
||||
apiVersion: admissionregistration.k8s.io/v1
|
||||
kind: ValidatingAdmissionPolicyBinding
|
||||
|
||||
@@ -64,9 +64,6 @@ spec:
|
||||
kata-image:
|
||||
url: "gs://${BUCKET_NAME}/kata-assets/rootfs.img"
|
||||
sha256: "24d77700749846355f3be25b87974c1fdea9fd2a9534256b4e4bbdde5df23588"
|
||||
kata-config:
|
||||
url: "gs://${BUCKET_NAME}/kata-assets/configuration-clh.toml"
|
||||
sha256: "20f5de4f705423ddd1ee93318c784c6aede30c07c78dc6b311a771cc2f6859c7"
|
||||
# amd64 assets are kata 4.0.0 + virtiofsd v1.14.0 (assemble.sh ARCH=amd64).
|
||||
amd64:
|
||||
cloud-hypervisor:
|
||||
@@ -84,6 +81,3 @@ spec:
|
||||
kata-image:
|
||||
url: "gs://${BUCKET_NAME}/kata-assets/rootfs.img"
|
||||
sha256: "fbbd40de605ab2f99dc82c777dc5b41f9a00dad3a0f699f2247c306ac9c1204d"
|
||||
kata-config:
|
||||
url: "gs://${BUCKET_NAME}/kata-assets/configuration-clh.toml"
|
||||
sha256: "1e67eabdfb9d900bf95a00edb52cacc1efe64f23ea095f8b49ee81c7434bce90"
|
||||
|
||||
@@ -61,7 +61,7 @@ func gvisorAsset() AssetFile {
|
||||
}
|
||||
|
||||
// microVMAssets returns a full, valid micro-VM asset set for one architecture:
|
||||
// the five assets the policy requires. The overlay rootfs serves the OCI image
|
||||
// the four assets the policy requires. The overlay rootfs serves the OCI image
|
||||
// over virtio-fs, so virtiofsd is part of the set.
|
||||
func microVMAssets() map[string]AssetFile {
|
||||
a := AssetFile{URL: "gs://bucket/asset", SHA256: validSHA256}
|
||||
@@ -70,7 +70,6 @@ func microVMAssets() map[string]AssetFile {
|
||||
"virtiofsd": a,
|
||||
"kata-kernel": a,
|
||||
"kata-image": a,
|
||||
"kata-config": a,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user